Skip the cold-start SFT best ckpt and free CUDA cache after eval
Step 0 generate was writing a 5GB success=0 snapshot and leaving the 32GB card fragmented, so the next Adam step OOM'd after batch-32 eval. Only promote _best after step 0 and empty_cache when eval returns.
This commit is contained in:
+3
-1
@@ -348,7 +348,7 @@ def main() -> None:
|
|||||||
},
|
},
|
||||||
step=step,
|
step=step,
|
||||||
)
|
)
|
||||||
if _is_better_eval(
|
if step > 0 and _is_better_eval(
|
||||||
printable["success_rate"],
|
printable["success_rate"],
|
||||||
printable["chrf"],
|
printable["chrf"],
|
||||||
best_success,
|
best_success,
|
||||||
@@ -362,6 +362,8 @@ def main() -> None:
|
|||||||
f"chrf {best_chrf:.2f} -> {best_path}"
|
f"chrf {best_chrf:.2f} -> {best_path}"
|
||||||
)
|
)
|
||||||
model.train()
|
model.train()
|
||||||
|
if device == "cuda":
|
||||||
|
torch.cuda.empty_cache()
|
||||||
if ckpt_now:
|
if ckpt_now:
|
||||||
_save(last_path, dump())
|
_save(last_path, dump())
|
||||||
print(f" last -> {last_path}")
|
print(f" last -> {last_path}")
|
||||||
|
|||||||
Reference in New Issue
Block a user