From b1c88bef256805e8324092a8a8026ba0cf2c22cd Mon Sep 17 00:00:00 2001 From: sora <2075279110@qq.com> Date: Wed, 16 Sep 2026 04:02:34 +0000 Subject: [PATCH] --thinking: unified switch (off/low/medium/high/max/full) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One flag for the whole ladder: off == --disable-thinking, low..max map to reasoning_effort, full = plain default. Takes precedence over the two older flags. Run Plan shows the active mode ('disabled' / 'enabled · effort=low' / 'enabled'). Co-Authored-By: Claude --- evalharness/cli.py | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/evalharness/cli.py b/evalharness/cli.py index 4ba298a..c8fced3 100644 --- a/evalharness/cli.py +++ b/evalharness/cli.py @@ -287,6 +287,9 @@ def _print_run_plan(console, args, model_spec): samples = '? (unknown benchmarks)' if n_uncached: samples += f' · {n_uncached} bench(es) not cached yet' + _eff = getattr(args, 'reasoning_effort', '') or '' + _think_txt = ('disabled' if getattr(args, 'disable_thinking', False) + else (f'enabled · effort={_eff}' if _eff else 'enabled')) _auto = getattr(args, 'auto_concurrency', False) _conc = (f'auto (start {args.concurrency}, gate decides)' if _auto else str(args.concurrency)) @@ -297,8 +300,7 @@ def _print_run_plan(console, args, model_spec): print(f'Model: {model_name}') print(f'Benchmarks: {len(args.datasets)} -> {", ".join(args.datasets)}') print(f'Samples: {samples}') - print(f'Concurrency: {_conc} | Thinking: ' - f'{"enabled" if not args.disable_thinking else "disabled"} | ' + print(f'Concurrency: {_conc} | Thinking: {_think_txt} | ' f'Performance: {"on" if args.perf else "off"}') print(f'Resume: {"on" if args.resume else "off"} | Output: {args.out_dir or "(none)"}') return @@ -315,7 +317,10 @@ def _print_run_plan(console, args, model_spec): table.add_row('Benchmarks', f'{len(args.datasets)} · {", ".join(args.datasets)}') table.add_row('Samples', samples) table.add_row('Concurrency', f'[magenta]{_conc}[/magenta]' if _auto else _conc) - table.add_row('Thinking', '[red]disabled[/red]' if args.disable_thinking else '[green]enabled[/green]') + table.add_row('Thinking', + '[red]disabled[/red]' if getattr(args, 'disable_thinking', False) + else (f'[green]enabled · effort={_eff}[/green]' if _eff + else '[green]enabled[/green]')) table.add_row('Performance', '[green]enabled[/green]' if args.perf else '[dim]disabled[/dim]') table.add_row('Checkpoint', '[green]resume[/green]' if args.resume else '[dim]new run[/dim]') table.add_row('Output', args.out_dir or '[dim](not specified)[/dim]') @@ -1210,6 +1215,13 @@ def build_parser() -> argparse.ArgumentParser: "profile.default < profile[''] < explicit kwargs") p.add_argument('--disable-thinking', action='store_true', help='send enable_thinking=false to the OpenAI-compatible model') + p.add_argument('--thinking', default='', + choices=('', 'off', 'low', 'medium', 'high', 'max', 'full'), + help='unified thinking switch: off = --disable-thinking; ' + 'low/medium/high/max = reasoning_effort (verified on ' + 'this endpoint: low = ~1/9 tokens); full = default ' + 'thinking. Takes precedence over --disable-thinking ' + 'and --reasoning-effort') p.add_argument('--reasoning-effort', default='', choices=('', 'minimal', 'low', 'medium', 'high', 'max'), help="thinking intensity (GLM/Anthropic-style; verified " @@ -1292,6 +1304,13 @@ def main(argv=None) -> int: # '--concurrency auto' == '--auto-concurrency': normalize once, here, # so every downstream site (plan display, run_eval, gate initial) sees # an int + the flag + _th = str(getattr(args, 'thinking', '') or '').strip().lower() + if _th: + if _th == 'off': + args.disable_thinking = True + else: + args.disable_thinking = False + args.reasoning_effort = '' if _th == 'full' else _th if str(getattr(args, 'concurrency', '32')).strip().lower() == 'auto': args.auto_concurrency = True args.concurrency = 1 # gate starts at 1: probe x2, bisect to capacity