Keep attnres on resume, fix final chunk_index, default 0.5b to Yi-6B
CLI default attnres=off was overwriting block checkpoints on resume so later loads hit Unexpected key(s). Only apply flags the user passed. Track next_chunk so a budget-exit save does not skip the untrained yield. 0.5b now uses 01-ai/Yi-6B (64k); refuse resume when the ckpt tokenizer does not match.
This commit is contained in:
@@ -9,13 +9,15 @@ Hybrid Attention (K3): 每 4 层 1 次 Gated MLA, 末层强制 MLA.
|
||||
|
||||
Presets:
|
||||
toy — ~8M, 自训 8k SP, 本地过拟合
|
||||
0.5b — ~482M, Qwen3 词表, 32–40GB bf16;默认 step 是冒烟,翻译前置用 --max-tokens
|
||||
0.5b — ~415M, Yi-6B 词表 (64k), 32–40GB bf16;默认 step 是冒烟,翻译前置用 --max-tokens
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
# Qwen3 config.json; train_k3 overrides with len(tokenizer).
|
||||
# 01-ai/Yi-6B config.json; train_k3 overrides with len(tokenizer).
|
||||
YI6B_VOCAB_SIZE = 64000
|
||||
# Kept for old Qwen3 checkpoints / docs.
|
||||
QWEN3_VOCAB_SIZE = 151936
|
||||
|
||||
|
||||
@@ -24,7 +26,7 @@ class K3Config:
|
||||
# 主干
|
||||
hidden_size: int = 256
|
||||
num_hidden_layers: int = 4
|
||||
vocab_size: int = 8192 # toy: data/spm_4k; 0.5b: Qwen3
|
||||
vocab_size: int = 8192 # toy: data/spm_4k; 0.5b: Yi-6B 64k
|
||||
initializer_range: float = 0.02
|
||||
norm_eps: float = 1e-6
|
||||
tie_word_embeddings: bool = False
|
||||
@@ -76,11 +78,11 @@ class K3Config:
|
||||
return cls()
|
||||
if name in {"0.5b", "500m"}:
|
||||
# H * head_dim == hidden. Routed 16 Top-2; LatentMoE padded bmm.
|
||||
# ~482M with tied Qwen3 embeddings. 6×(3 KDA + 1 MLA).
|
||||
# ~415M with tied Yi-6B embeddings. 6×(3 KDA + 1 MLA).
|
||||
return cls(
|
||||
hidden_size=768,
|
||||
num_hidden_layers=24,
|
||||
vocab_size=QWEN3_VOCAB_SIZE,
|
||||
vocab_size=YI6B_VOCAB_SIZE,
|
||||
tie_word_embeddings=True,
|
||||
max_position_embeddings=2048,
|
||||
num_heads=12,
|
||||
|
||||
Reference in New Issue
Block a user