Fit 0.5b training on 32GB: SDPA MLA, block checkpoint, chunked CE
Whole-mixer checkpoint plus T×T MLA scores OOM'd a 31GB GPU on backward. Checkpoint each AttnRes block, run absorbed MLA through SDPA, and compute CE in vocab chunks so [B,T,V] logits are never materialized. --max-tokens is now the training budget; default --steps 2000 no longer caps a 1B-token run at 250 optimizer steps.
This commit is contained in:
@@ -34,14 +34,16 @@ def total_opt_steps(
|
||||
seq_len: int,
|
||||
grad_acc: int,
|
||||
) -> int:
|
||||
"""Optimizer-step horizon used by cosine. At least 1."""
|
||||
"""Optimizer-step horizon used by cosine. At least 1.
|
||||
|
||||
``max_tokens`` is the training budget when set; ``max_micro`` is only used
|
||||
when ``max_tokens`` is None. Otherwise a default ``--steps 2000`` would
|
||||
shrink a 1B-token cosine to 250 opt steps.
|
||||
"""
|
||||
acc = max(grad_acc, 1)
|
||||
candidates: list[int] = []
|
||||
if max_tokens is not None and max_tokens > 0:
|
||||
tpm = max(tokens_per_micro(batch, seq_len), 1)
|
||||
candidates.append(math.ceil(max_tokens / (tpm * acc)))
|
||||
return max(math.ceil(max_tokens / (tpm * acc)), 1)
|
||||
if max_micro is not None and max_micro > 0:
|
||||
candidates.append(math.ceil(max_micro / acc))
|
||||
if not candidates:
|
||||
return 1
|
||||
return max(min(candidates), 1)
|
||||
return max(math.ceil(max_micro / acc), 1)
|
||||
return 1
|
||||
|
||||
Reference in New Issue
Block a user