Do not let HF tokenizers truncate wiki articles at 4096

Yi-6B sets model_max_length=4096. encode() would clip long Wikipedia
pages before we pack seq_len chunks. Raise the cap so only our
chunker limits context.
This commit is contained in:
dela
2026-08-26 14:31:07 +08:00
parent 24c9d56b72
commit 94d0f2ff6a
+2
View File
@@ -75,6 +75,8 @@ def load_tokenizer(source: str) -> Tokenizer:
from transformers import AutoTokenizer
tok = AutoTokenizer.from_pretrained(source, trust_remote_code=True)
# 只借词表分词, 语料随后按 seq_len 切块, 不受原模型 4096 上限约束
tok.model_max_length = 10**9
return HuggingFaceTokenizer(tok)