Skip the cold-start SFT best ckpt and free CUDA cache after eval

Step 0 generate was writing a 5GB success=0 snapshot and leaving the
32GB card fragmented, so the next Adam step OOM'd after batch-32 eval.
Only promote _best after step 0 and empty_cache when eval returns.
This commit is contained in:
dela
2026-08-26 10:08:31 +08:00
parent 9652a9a7eb
commit 5a7d949b01
+3 -1
View File
@@ -348,7 +348,7 @@ def main() -> None:
}, },
step=step, step=step,
) )
if _is_better_eval( if step > 0 and _is_better_eval(
printable["success_rate"], printable["success_rate"],
printable["chrf"], printable["chrf"],
best_success, best_success,
@@ -362,6 +362,8 @@ def main() -> None:
f"chrf {best_chrf:.2f} -> {best_path}" f"chrf {best_chrf:.2f} -> {best_path}"
) )
model.train() model.train()
if device == "cuda":
torch.cuda.empty_cache()
if ckpt_now: if ckpt_now:
_save(last_path, dump()) _save(last_path, dump())
print(f" last -> {last_path}") print(f" last -> {last_path}")