from kda.training.schedule import lr_scale, total_opt_steps def test_warmup_then_cosine_floor(): assert abs(lr_scale(0, warmup=10, total_opt=100) - 0.1) < 1e-9 assert abs(lr_scale(9, warmup=10, total_opt=100) - 1.0) < 1e-9 assert abs(lr_scale(10, warmup=10, total_opt=100) - 1.0) < 1e-6 end = lr_scale(99, warmup=10, total_opt=100) assert abs(end - 0.1) < 1e-6 def test_horizon_max_tokens_overrides_micro_cap(): # 8.2M tokens @ batch 2 seq 2048 acc 8 -> 250 opt even if --steps is larger opt_from_tokens = total_opt_steps( max_tokens=8_192_000, max_micro=10_000, batch=2, seq_len=2048, grad_acc=8 ) assert opt_from_tokens == 250 # 1B-token run must not inherit the default --steps 2000 cap (250 opt) opt_1b = total_opt_steps( max_tokens=10**9, max_micro=2000, batch=2, seq_len=2048, grad_acc=8 ) assert opt_1b == 30518 def test_horizon_micro_when_tokens_unset(): opt_from_micro = total_opt_steps( max_tokens=None, max_micro=2000, batch=2, seq_len=2048, grad_acc=8 ) assert opt_from_micro == 250