From 5a7d949b01cc5566f36fcd5ec1c7ee70ac21a657 Mon Sep 17 00:00:00 2001 From: dela Date: Wed, 26 Aug 2026 10:08:31 +0800 Subject: [PATCH] Skip the cold-start SFT best ckpt and free CUDA cache after eval Step 0 generate was writing a 5GB success=0 snapshot and leaving the 32GB card fragmented, so the next Adam step OOM'd after batch-32 eval. Only promote _best after step 0 and empty_cache when eval returns. --- train_sft.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/train_sft.py b/train_sft.py index 656eca5..56618bc 100644 --- a/train_sft.py +++ b/train_sft.py @@ -348,7 +348,7 @@ def main() -> None: }, step=step, ) - if _is_better_eval( + if step > 0 and _is_better_eval( printable["success_rate"], printable["chrf"], best_success, @@ -362,6 +362,8 @@ def main() -> None: f"chrf {best_chrf:.2f} -> {best_path}" ) model.train() + if device == "cuda": + torch.cuda.empty_cache() if ckpt_now: _save(last_path, dump()) print(f" last -> {last_path}")