Do not let HF tokenizers truncate wiki articles at 4096
Yi-6B sets model_max_length=4096. encode() would clip long Wikipedia pages before we pack seq_len chunks. Raise the cap so only our chunker limits context.
This commit is contained in:
@@ -75,6 +75,8 @@ def load_tokenizer(source: str) -> Tokenizer:
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained(source, trust_remote_code=True)
|
||||
# 只借词表分词, 语料随后按 seq_len 切块, 不受原模型 4096 上限约束
|
||||
tok.model_max_length = 10**9
|
||||
return HuggingFaceTokenizer(tok)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user