From 94d0f2ff6a3c768d5941248fc6811581d8028f18 Mon Sep 17 00:00:00 2001 From: dela Date: Wed, 26 Aug 2026 14:31:07 +0800 Subject: [PATCH] Do not let HF tokenizers truncate wiki articles at 4096 Yi-6B sets model_max_length=4096. encode() would clip long Wikipedia pages before we pack seq_len chunks. Raise the cap so only our chunker limits context. --- kda/training/data.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/kda/training/data.py b/kda/training/data.py index 3d8f1fe..e23e3dc 100644 --- a/kda/training/data.py +++ b/kda/training/data.py @@ -75,6 +75,8 @@ def load_tokenizer(source: str) -> Tokenizer: from transformers import AutoTokenizer tok = AutoTokenizer.from_pretrained(source, trust_remote_code=True) + # 只借词表分词, 语料随后按 seq_len 切块, 不受原模型 4096 上限约束 + tok.model_max_length = 10**9 return HuggingFaceTokenizer(tok)