Initial K3 snapshot: 0.5B KDA/MLA/MoE train path
Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
This commit is contained in:
@@ -0,0 +1,21 @@
|
||||
from kda.training.schedule import lr_scale, total_opt_steps
|
||||
|
||||
|
||||
def test_warmup_then_cosine_floor():
|
||||
assert abs(lr_scale(0, warmup=10, total_opt=100) - 0.1) < 1e-9
|
||||
assert abs(lr_scale(9, warmup=10, total_opt=100) - 1.0) < 1e-9
|
||||
assert abs(lr_scale(10, warmup=10, total_opt=100) - 1.0) < 1e-6
|
||||
end = lr_scale(99, warmup=10, total_opt=100)
|
||||
assert abs(end - 0.1) < 1e-6
|
||||
|
||||
|
||||
def test_horizon_prefers_the_earlier_stop():
|
||||
# 8.2M tokens @ batch 2 seq 2048 acc 8 -> 250 opt
|
||||
opt_from_tokens = total_opt_steps(
|
||||
max_tokens=8_192_000, max_micro=10_000, batch=2, seq_len=2048, grad_acc=8
|
||||
)
|
||||
assert opt_from_tokens == 250
|
||||
opt_from_micro = total_opt_steps(
|
||||
max_tokens=10**12, max_micro=2000, batch=2, seq_len=2048, grad_acc=8
|
||||
)
|
||||
assert opt_from_micro == 250
|
||||
Reference in New Issue
Block a user