Pluginize the run shell: @register_progress (rich/plain), @register_theme, @register_hook (on_benchmark_failed/done), @register_prober; CLI gains --progress-plugin/--theme
Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
98007cccd7
commit
5af8e2a7fd
@ -182,6 +182,15 @@ def mybench():
|
|||||||
|
|
||||||
其余插件点同构:`@register_prompt_renderer`、`@register_extractor/scorer/aggregator`、`@register_adapter`、`@register_sandbox`、`@register_env`、`@register_renderer`。
|
其余插件点同构:`@register_prompt_renderer`、`@register_extractor/scorer/aggregator`、`@register_adapter`、`@register_sandbox`、`@register_env`、`@register_renderer`。
|
||||||
|
|
||||||
|
运行外壳同样是插件:
|
||||||
|
|
||||||
|
| 插件点 | 注册 | 说明 |
|
||||||
|
|---|---|---|
|
||||||
|
| 进度报告 | `@register_progress('rich'/'plain'/...)` | `--progress-plugin` 选择;rich 是终端进度条,plain 是纯叙事行(CI/日志) |
|
||||||
|
| 叙事主题 | `@register_theme('default'/...)` | `--theme` 选择;图标/配色/句子高亮的映射表 |
|
||||||
|
| 生命周期钩子 | `@register_hook('on_benchmark_failed'/'on_benchmark_done')` | 观察/扩展运行(webhook 通知、失败重试策略),钩子报错不影响主流程 |
|
||||||
|
| 端点探针 | `@register_prober('ping'/...)` | 运行前的端点可用性验证策略(`EVALHARNESS_PROBER` 环境变量选择) |
|
||||||
|
|
||||||
## 7. 架构
|
## 7. 架构
|
||||||
|
|
||||||
```
|
```
|
||||||
|
|||||||
@ -199,9 +199,12 @@ def _print_run_plan(console, args, model_spec):
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
def _narration(msg: str) -> str:
|
def _narration(msg: str) -> str: # theme-plugin facade
|
||||||
"""Icon + path-highlighting for narration lines (visual separation at
|
"""Facade over the active narration THEME plugin (--theme, default
|
||||||
a glance; paths in cyan)."""
|
'default'). The mapping lives in evalharness/themes/."""
|
||||||
|
from evalharness.themes import get_theme
|
||||||
|
|
||||||
|
return get_theme(getattr(_narration, 'active', 'default'))(msg)
|
||||||
icon = ''
|
icon = ''
|
||||||
m = msg.lower()
|
m = msg.lower()
|
||||||
if m.startswith('loading/'):
|
if m.startswith('loading/'):
|
||||||
@ -573,19 +576,27 @@ def _cmd_eval_run(args) -> int:
|
|||||||
|
|
||||||
progress_reporter = None
|
progress_reporter = None
|
||||||
if args.progress:
|
if args.progress:
|
||||||
from evalharness.progress import RichTerminalProgress
|
from evalharness.progress import PROGRESS_REGISTRY
|
||||||
|
|
||||||
if RichTerminalProgress is not None:
|
_pname = getattr(args, 'progress_plugin', 'rich') \
|
||||||
|
if args.progress is True else args.progress
|
||||||
|
try:
|
||||||
|
_P = PROGRESS_REGISTRY.get(_pname)
|
||||||
|
except KeyError:
|
||||||
|
raise SystemExit(f'unknown progress plugin {_pname!r}; '
|
||||||
|
f'available: {", ".join(PROGRESS_REGISTRY.names())}')
|
||||||
|
if _P is not None:
|
||||||
# share ONE console: phase lines printed by another
|
# share ONE console: phase lines printed by another
|
||||||
# writer during the live bar interleave incorrectly.
|
# writer during the live bar interleave incorrectly.
|
||||||
# One reporter for the WHOLE run: overall bar (which
|
# One reporter for the WHOLE run: overall bar (which
|
||||||
# benchmark) + sample bar (which sample), reused per
|
# benchmark) + sample bar (which sample), reused per
|
||||||
# benchmark via reset_samples().
|
# benchmark via reset_samples().
|
||||||
if _shared_reporter is None and console.is_terminal:
|
if _shared_reporter is None and (
|
||||||
|
_pname == 'plain' or console.is_terminal):
|
||||||
# live bars only on a real terminal: through pipes
|
# live bars only on a real terminal: through pipes
|
||||||
# (| grep, > log) rich's refresh thread misbehaves
|
# (| grep, > log) rich's refresh thread misbehaves
|
||||||
# and stalls the run -- plain phases instead
|
# and stalls the run -- plain phases instead
|
||||||
_shared_reporter = RichTerminalProgress(console=console)
|
_shared_reporter = _P(console=console)
|
||||||
_shared_reporter.owned_externally = True
|
_shared_reporter.owned_externally = True
|
||||||
progress_reporter = _shared_reporter
|
progress_reporter = _shared_reporter
|
||||||
|
|
||||||
@ -672,6 +683,10 @@ def _cmd_eval_run(args) -> int:
|
|||||||
'lat_p50': _pct(0.50), 'lat_p90': _pct(0.90),
|
'lat_p50': _pct(0.50), 'lat_p90': _pct(0.90),
|
||||||
'trunc': sum(1 for f in fins if f == 'length'),
|
'trunc': sum(1 for f in fins if f == 'length'),
|
||||||
'groups': groups, 'ok': True})
|
'groups': groups, 'ok': True})
|
||||||
|
from evalharness.hooks import fire as _fire2
|
||||||
|
|
||||||
|
_fire2('on_benchmark_done', name=name, metrics=dict(report.metrics),
|
||||||
|
num_samples=report.num_samples, out_dir=out_dir)
|
||||||
if progress_reporter is not None:
|
if progress_reporter is not None:
|
||||||
progress_reporter.advance_overall()
|
progress_reporter.advance_overall()
|
||||||
_print_benchmark_result(console, i + 1, total_runs, name,
|
_print_benchmark_result(console, i + 1, total_runs, name,
|
||||||
@ -681,6 +696,9 @@ def _cmd_eval_run(args) -> int:
|
|||||||
'secs': round(_time.time() - t0, 1), 'ok': False,
|
'secs': round(_time.time() - t0, 1), 'ok': False,
|
||||||
'err': f'{type(e).__name__}: {str(e)[:100]}'})
|
'err': f'{type(e).__name__}: {str(e)[:100]}'})
|
||||||
print(f'{name}: FAILED {type(e).__name__}: {str(e)[:160]}', file=sys.stderr)
|
print(f'{name}: FAILED {type(e).__name__}: {str(e)[:160]}', file=sys.stderr)
|
||||||
|
from evalharness.hooks import fire as _fire
|
||||||
|
|
||||||
|
_fire('on_benchmark_failed', name=name, error=e, dataset=name)
|
||||||
if console is not None:
|
if console is not None:
|
||||||
from rich.panel import Panel
|
from rich.panel import Panel
|
||||||
|
|
||||||
@ -911,9 +929,14 @@ def build_parser() -> argparse.ArgumentParser:
|
|||||||
p.add_argument('--env', default='', help="agent environment (e.g. 'bfcl_mock') -> message pump")
|
p.add_argument('--env', default='', help="agent environment (e.g. 'bfcl_mock') -> message pump")
|
||||||
p.add_argument('--concurrency', type=int, default=32, help='parallel model calls (default 32)')
|
p.add_argument('--concurrency', type=int, default=32, help='parallel model calls (default 32)')
|
||||||
p.add_argument('--progress', action='store_true', default=True,
|
p.add_argument('--progress', action='store_true', default=True,
|
||||||
help='show per-sample Rich terminal progress (default: on)')
|
help='show per-sample progress (default: on)')
|
||||||
p.add_argument('--no-progress', dest='progress', action='store_false',
|
p.add_argument('--no-progress', dest='progress', action='store_false',
|
||||||
help='disable per-sample Rich terminal progress')
|
help='disable per-sample progress')
|
||||||
|
p.add_argument('--progress-plugin', default='rich',
|
||||||
|
help='progress reporter plugin (rich | plain | any '
|
||||||
|
'@register_progress name)')
|
||||||
|
p.add_argument('--theme', default='default',
|
||||||
|
help='narration theme plugin (default | any @register_theme name)')
|
||||||
p.add_argument('--limit', type=int, help='evaluate only the first N samples total')
|
p.add_argument('--limit', type=int, help='evaluate only the first N samples total')
|
||||||
p.add_argument('--resume', nargs='?', const=True, default=False,
|
p.add_argument('--resume', nargs='?', const=True, default=False,
|
||||||
help='resume from per-sample checkpoint (default path auto-derived; '
|
help='resume from per-sample checkpoint (default path auto-derived; '
|
||||||
|
|||||||
31
evalharness/hooks.py
Normal file
31
evalharness/hooks.py
Normal file
@ -0,0 +1,31 @@
|
|||||||
|
"""Run-lifecycle hooks: plugins that observe/extend a run without touching
|
||||||
|
its logic. Register with @register_hook('on_benchmark_failed') etc.; the CLI
|
||||||
|
fires them at the matching points. Multiple hooks per event all run.
|
||||||
|
|
||||||
|
from evalharness.hooks import register_hook
|
||||||
|
|
||||||
|
@register_hook('on_benchmark_failed')
|
||||||
|
def notify(name, error, **ctx):
|
||||||
|
requests.post(webhook, json={'bench': name, 'error': str(error)})
|
||||||
|
"""
|
||||||
|
from typing import Callable, Dict, List
|
||||||
|
|
||||||
|
_HOOKS: Dict[str, List[Callable]] = {}
|
||||||
|
|
||||||
|
|
||||||
|
def register_hook(event: str):
|
||||||
|
def decorator(fn):
|
||||||
|
_HOOKS.setdefault(event, []).append(fn)
|
||||||
|
return fn
|
||||||
|
|
||||||
|
return decorator
|
||||||
|
|
||||||
|
|
||||||
|
def fire(event: str, **ctx) -> None:
|
||||||
|
"""Invoke every hook registered for the event; hook errors are printed
|
||||||
|
and swallowed -- observability must never break the run."""
|
||||||
|
for fn in _HOOKS.get(event, []):
|
||||||
|
try:
|
||||||
|
fn(**ctx)
|
||||||
|
except Exception as e:
|
||||||
|
print(f'hook {fn.__name__!r} failed: {type(e).__name__}: {e}', flush=True)
|
||||||
24
evalharness/model/probers.py
Normal file
24
evalharness/model/probers.py
Normal file
@ -0,0 +1,24 @@
|
|||||||
|
"""Endpoint prober plugins: how the runner verifies a model endpoint is
|
||||||
|
usable before burning samples. Default 'ping' sends one 1-token request."""
|
||||||
|
from ..eval.registry import EvalRegistry
|
||||||
|
|
||||||
|
PROBER_REGISTRY = EvalRegistry('prober')
|
||||||
|
|
||||||
|
|
||||||
|
def register_prober(name: str):
|
||||||
|
def decorator(fn):
|
||||||
|
PROBER_REGISTRY.register(name, fn)
|
||||||
|
return fn
|
||||||
|
|
||||||
|
return decorator
|
||||||
|
|
||||||
|
|
||||||
|
def get_prober(name: str = 'ping'):
|
||||||
|
return PROBER_REGISTRY.get(name)
|
||||||
|
|
||||||
|
|
||||||
|
@register_prober('ping')
|
||||||
|
async def ping(adapter):
|
||||||
|
from .runner import _default_ping_probe
|
||||||
|
|
||||||
|
await _default_ping_probe(adapter)
|
||||||
@ -510,6 +510,18 @@ _PROBED_SPECS = set()
|
|||||||
|
|
||||||
|
|
||||||
async def _probe_model(adapter, model_spec: str) -> None:
|
async def _probe_model(adapter, model_spec: str) -> None:
|
||||||
|
"""Delegate to the registered PROBER plugin ('ping' by default).
|
||||||
|
|
||||||
|
@register_prober('custom') async def probe(adapter): ... replaces the
|
||||||
|
whole reachability strategy without touching the runner."""
|
||||||
|
from .probers import get_prober
|
||||||
|
|
||||||
|
if getattr(adapter, 'name', '') != 'mock':
|
||||||
|
await get_prober(os.environ.get('EVALHARNESS_PROBER', 'ping'))(adapter)
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
async def _default_ping_probe(adapter, model_spec=None):
|
||||||
"""Fail fast on an unreachable model endpoint.
|
"""Fail fast on an unreachable model endpoint.
|
||||||
|
|
||||||
One 1-token request before any dataset work: a wrong api-url/model
|
One 1-token request before any dataset work: a wrong api-url/model
|
||||||
|
|||||||
@ -4,9 +4,13 @@ Degrades to None when rich is absent so callers fall back to plain-text
|
|||||||
progress (the CLI stays dependency-free in its fallback path).
|
progress (the CLI stays dependency-free in its fallback path).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from .plain import PROGRESS_REGISTRY, register_progress # noqa: F401 (re-export)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
from .rich_terminal import RichTerminalProgress
|
from .rich_terminal import RichTerminalProgress
|
||||||
except ImportError: # rich not installed
|
except ImportError: # rich not installed
|
||||||
RichTerminalProgress = None
|
RichTerminalProgress = None
|
||||||
|
else:
|
||||||
|
PROGRESS_REGISTRY.register('rich', RichTerminalProgress)
|
||||||
|
|
||||||
__all__ = ["RichTerminalProgress"]
|
__all__ = ["RichTerminalProgress", "PROGRESS_REGISTRY", "register_progress"]
|
||||||
|
|||||||
61
evalharness/progress/plain.py
Normal file
61
evalharness/progress/plain.py
Normal file
@ -0,0 +1,61 @@
|
|||||||
|
"""Bar-less progress reporter: narration lines only (terminals without
|
||||||
|
rich, CI logs, JSON-adjacent consumers)."""
|
||||||
|
from ..eval.registry import EvalRegistry
|
||||||
|
|
||||||
|
PROGRESS_REGISTRY = EvalRegistry('progress reporter')
|
||||||
|
|
||||||
|
|
||||||
|
def register_progress(name: str):
|
||||||
|
def decorator(cls):
|
||||||
|
PROGRESS_REGISTRY.register(name, cls)
|
||||||
|
return cls
|
||||||
|
|
||||||
|
return decorator
|
||||||
|
|
||||||
|
|
||||||
|
@register_progress('plain')
|
||||||
|
class PlainProgress:
|
||||||
|
name = 'plain'
|
||||||
|
|
||||||
|
def __init__(self, console=None):
|
||||||
|
self.console = console
|
||||||
|
|
||||||
|
def log(self, message):
|
||||||
|
print(message, flush=True)
|
||||||
|
|
||||||
|
# the rest are no-ops: no bars, no per-sample accounting
|
||||||
|
def set_overall(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def advance_overall(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def start(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def reset_samples(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def set_bench_tag(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def set_phase(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def begin_sample(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def rollback(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def advance(self, *a, **k):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def pause(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def resume(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
pass
|
||||||
35
evalharness/themes/__init__.py
Normal file
35
evalharness/themes/__init__.py
Normal file
@ -0,0 +1,35 @@
|
|||||||
|
"""Narration themes: the icon/word/color mapping for progress lines.
|
||||||
|
|
||||||
|
A theme is a narrate(msg) -> rich-markup-string function plus its name;
|
||||||
|
the CLI picks one with --theme (default 'default'). Drop a module in
|
||||||
|
evalharness/themes/ and it registers itself.
|
||||||
|
"""
|
||||||
|
from typing import Callable
|
||||||
|
|
||||||
|
from ..eval.registry import EvalRegistry
|
||||||
|
|
||||||
|
THEME_REGISTRY = EvalRegistry('narration theme')
|
||||||
|
|
||||||
|
|
||||||
|
def register_theme(name: str):
|
||||||
|
def decorator(fn: Callable[[str], str]):
|
||||||
|
THEME_REGISTRY.register(name, fn)
|
||||||
|
return fn
|
||||||
|
|
||||||
|
return decorator
|
||||||
|
|
||||||
|
|
||||||
|
def get_theme(name: str = 'default'):
|
||||||
|
return THEME_REGISTRY.get(name)
|
||||||
|
|
||||||
|
|
||||||
|
def _discover():
|
||||||
|
import importlib
|
||||||
|
import pkgutil
|
||||||
|
|
||||||
|
for m in pkgutil.iter_modules(__path__):
|
||||||
|
if m.name != '__init__':
|
||||||
|
importlib.import_module(f'{__name__}.{m.name}')
|
||||||
|
|
||||||
|
|
||||||
|
_discover()
|
||||||
41
evalharness/themes/default.py
Normal file
41
evalharness/themes/default.py
Normal file
@ -0,0 +1,41 @@
|
|||||||
|
"""Default narration theme: icons per stage, green facts, blue paths."""
|
||||||
|
|
||||||
|
from . import register_theme
|
||||||
|
|
||||||
|
ICONS = [
|
||||||
|
('loading/', '⬇ '),
|
||||||
|
('dataset ready', '📦 '),
|
||||||
|
('few-shot', '✳ '),
|
||||||
|
('checkpoint', '◷ '),
|
||||||
|
('generation skipped', '⏭ '),
|
||||||
|
('generating', '🤖 '),
|
||||||
|
('generation complete', '✓ '),
|
||||||
|
('scoring', '★ '),
|
||||||
|
('writing', '📝 '),
|
||||||
|
('endpoint', '🔗 '),
|
||||||
|
]
|
||||||
|
|
||||||
|
FACT_COLOR = 'green' # numbers/phrases/scores
|
||||||
|
PATH_COLOR = 'blue' # filesystem locations
|
||||||
|
|
||||||
|
|
||||||
|
@register_theme('default')
|
||||||
|
def narrate(msg: str) -> str:
|
||||||
|
import re
|
||||||
|
|
||||||
|
low = msg.lower()
|
||||||
|
icon = next((i for k, i in ICONS if k in low), '')
|
||||||
|
|
||||||
|
def fact(text):
|
||||||
|
return re.sub(r'(?<![\w/%.])(\d+(?:/\d+)?(?:\s+[a-z-]+){0,4})(?=[\s,.]|$)',
|
||||||
|
rf'[{FACT_COLOR}]\1[/{FACT_COLOR}]', text)
|
||||||
|
|
||||||
|
m = re.search(r'[:·] ([a-zA-Z_@]+ [0-9.]+%)(?=\s|$)', msg)
|
||||||
|
if m:
|
||||||
|
head = f'{icon}{msg[:m.start()]}: [bold {FACT_COLOR}]{m.group(1)}[/{FACT_COLOR}]'
|
||||||
|
return head + fact(msg[m.end():])
|
||||||
|
for sep in ('-> ', 'to '):
|
||||||
|
head, _, tail = msg.rpartition(sep)
|
||||||
|
if head and tail.startswith('/'):
|
||||||
|
return f'{icon}{head}{sep}[{PATH_COLOR}]{tail}[/{PATH_COLOR}]'
|
||||||
|
return f'{icon}{fact(msg)}'
|
||||||
Loading…
x
Reference in New Issue
Block a user