From d83c2cc1df641e0bac77b198de7efe771407fa55 Mon Sep 17 00:00:00 2001 From: sora <2075279110@qq.com> Date: Tue, 15 Sep 2026 02:49:34 +0000 Subject: [PATCH] Serialize tokenizer first load; one-shot degradation warning 96 worker threads racing transformers 5.x lazy imports on the FIRST _get_tokenizer call raised ImportError and degraded that whole first batch to the char approximation (the old single-threaded path never raced). First load now holds a threading.Lock; the transformers '>model_max_length' logging is silenced inside truncation (counting a 2M-token doc before trimming it is the point), and the per-sample degradation print becomes a one-shot warning. Co-Authored-By: Claude --- evalharness/model/truncation.py | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/evalharness/model/truncation.py b/evalharness/model/truncation.py index ac7fac2..9635cd5 100644 --- a/evalharness/model/truncation.py +++ b/evalharness/model/truncation.py @@ -9,21 +9,34 @@ Requires a tokenizer (transformers) at tokenizer_path or auto from the model. """ import os +import threading from functools import lru_cache from typing import Optional DEFAULT_TRUNCATION_TOKENS = 32768 * 4 # 131072, mirrors evalside run.py +_TOK_LOCK = threading.Lock() + + @lru_cache(maxsize=4) def _get_tokenizer(tokenizer_path: str): if not tokenizer_path or not os.path.exists(tokenizer_path): raise FileNotFoundError( f'tokenizer not found at {tokenizer_path!r} -- token-level truncation ' 'needs a local tokenizer dir (e.g. /data1/models/DeepSeek-V4-Flash-INT8)') - from transformers import AutoTokenizer + # serialize the FIRST load: 96 worker threads racing transformers 5.x's + # lazy imports raised ImportError and silently degraded batches to the + # char approximation; after one success lru_cache serves the rest + with _TOK_LOCK: + from transformers import AutoTokenizer - return AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True) + # the '> model_max_length' warnings are EXPECTED here -- counting a + # 2M-token doc before trimming it is the whole point of truncation + import logging + + logging.getLogger('transformers').setLevel(logging.ERROR) + return AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True) def truncate_middle_tokens(text: str, max_tokens: int, tokenizer_path: str) -> str: