Whole-mixer checkpoint plus T×T MLA scores OOM'd a 31GB GPU on backward. Checkpoint each AttnRes block, run absorbed MLA through SDPA, and compute CE in vocab chunks so [B,T,V] logits are never materialized. --max-tokens is now the training budget; default --steps 2000 no longer caps a 1B-token run at 250 optimizer steps.
30 lines
1.1 KiB
Python
30 lines
1.1 KiB
Python
from kda.training.schedule import lr_scale, total_opt_steps
|
|
|
|
|
|
def test_warmup_then_cosine_floor():
|
|
assert abs(lr_scale(0, warmup=10, total_opt=100) - 0.1) < 1e-9
|
|
assert abs(lr_scale(9, warmup=10, total_opt=100) - 1.0) < 1e-9
|
|
assert abs(lr_scale(10, warmup=10, total_opt=100) - 1.0) < 1e-6
|
|
end = lr_scale(99, warmup=10, total_opt=100)
|
|
assert abs(end - 0.1) < 1e-6
|
|
|
|
|
|
def test_horizon_max_tokens_overrides_micro_cap():
|
|
# 8.2M tokens @ batch 2 seq 2048 acc 8 -> 250 opt even if --steps is larger
|
|
opt_from_tokens = total_opt_steps(
|
|
max_tokens=8_192_000, max_micro=10_000, batch=2, seq_len=2048, grad_acc=8
|
|
)
|
|
assert opt_from_tokens == 250
|
|
# 1B-token run must not inherit the default --steps 2000 cap (250 opt)
|
|
opt_1b = total_opt_steps(
|
|
max_tokens=10**9, max_micro=2000, batch=2, seq_len=2048, grad_acc=8
|
|
)
|
|
assert opt_1b == 30518
|
|
|
|
|
|
def test_horizon_micro_when_tokens_unset():
|
|
opt_from_micro = total_opt_steps(
|
|
max_tokens=None, max_micro=2000, batch=2, seq_len=2048, grad_acc=8
|
|
)
|
|
assert opt_from_micro == 250
|