Skip the cold-start SFT best ckpt and free CUDA cache after eval
Step 0 generate was writing a 5GB success=0 snapshot and leaving the 32GB card fragmented, so the next Adam step OOM'd after batch-32 eval. Only promote _best after step 0 and empty_cache when eval returns.
This commit is contained in:
+3
-1
@@ -348,7 +348,7 @@ def main() -> None:
|
||||
},
|
||||
step=step,
|
||||
)
|
||||
if _is_better_eval(
|
||||
if step > 0 and _is_better_eval(
|
||||
printable["success_rate"],
|
||||
printable["chrf"],
|
||||
best_success,
|
||||
@@ -362,6 +362,8 @@ def main() -> None:
|
||||
f"chrf {best_chrf:.2f} -> {best_path}"
|
||||
)
|
||||
model.train()
|
||||
if device == "cuda":
|
||||
torch.cuda.empty_cache()
|
||||
if ckpt_now:
|
||||
_save(last_path, dump())
|
||||
print(f" last -> {last_path}")
|
||||
|
||||
Reference in New Issue
Block a user