Keep attnres on resume, fix final chunk_index, default 0.5b to Yi-6B

CLI default attnres=off was overwriting block checkpoints on resume
so later loads hit Unexpected key(s). Only apply flags the user
passed. Track next_chunk so a budget-exit save does not skip the
untrained yield. 0.5b now uses 01-ai/Yi-6B (64k); refuse resume
when the ckpt tokenizer does not match.
This commit is contained in:
dela
2026-08-26 14:23:29 +08:00
parent 5cc0555563
commit 24c9d56b72
5 changed files with 130 additions and 30 deletions
+7 -5
View File
@@ -9,13 +9,15 @@ Hybrid Attention (K3): 每 4 层 1 次 Gated MLA, 末层强制 MLA.
Presets:
toy — ~8M, 自训 8k SP, 本地过拟合
0.5b — ~482M, Qwen3 词表, 32–40GB bf16;默认 step 是冒烟,翻译前置用 --max-tokens
0.5b — ~415M, Yi-6B 词表 (64k), 32–40GB bf16;默认 step 是冒烟,翻译前置用 --max-tokens
"""
from __future__ import annotations
from dataclasses import dataclass
# Qwen3 config.json; train_k3 overrides with len(tokenizer).
# 01-ai/Yi-6B config.json; train_k3 overrides with len(tokenizer).
YI6B_VOCAB_SIZE = 64000
# Kept for old Qwen3 checkpoints / docs.
QWEN3_VOCAB_SIZE = 151936
@@ -24,7 +26,7 @@ class K3Config:
# 主干
hidden_size: int = 256
num_hidden_layers: int = 4
vocab_size: int = 8192 # toy: data/spm_4k; 0.5b: Qwen3
vocab_size: int = 8192 # toy: data/spm_4k; 0.5b: Yi-6B 64k
initializer_range: float = 0.02
norm_eps: float = 1e-6
tie_word_embeddings: bool = False
@@ -76,11 +78,11 @@ class K3Config:
return cls()
if name in {"0.5b", "500m"}:
# H * head_dim == hidden. Routed 16 Top-2; LatentMoE padded bmm.
# ~482M with tied Qwen3 embeddings. 6×(3 KDA + 1 MLA).
# ~415M with tied Yi-6B embeddings. 6×(3 KDA + 1 MLA).
return cls(
hidden_size=768,
num_hidden_layers=24,
vocab_size=QWEN3_VOCAB_SIZE,
vocab_size=YI6B_VOCAB_SIZE,
tie_word_embeddings=True,
max_position_embeddings=2048,
num_heads=12,