ruoxi_sun 09b2add673 fingerprint: integrate fp_fusion model fingerprint benchmark
New evalharness/fingerprint/ package (from evalstone fp_fusion v1.1,
2026-09-07 pruning final): probe battery -> concurrent collection ->
five scoring views (verify/attribution/variant/adversarial/robustness),
bundled family aliases + 27 reference fingerprints (12 fp_fusion schema).

- CLI: 'evalharness fingerprint run ...' (REMAINDER passthrough, single
  source of arg definitions) + 'fingerprint list' for bundled references
- imports rewritten package-relative; direct 'python3 run_fp_fusion.py'
  execution kept working via package bootstrap
- offline analysis/collection scripts made path-independent (previously
  pinned to a /opt/evalscope path absent on this host)
- shell scripts: hardcoded API key -> FP_API_KEY/OPENAI_API_KEY env vars
- --reference accepts short names resolved against bundled references/
- pyproject: +httpx dependency, package-data references/*.json
- tests/test_fingerprint.py: 10 offline tests (battery definitions,
  assembly counts, normalization, signals, verdict ladder, CLI wiring)
- README: fingerprint section + architecture entry

Verified on H20-1: tests 10/10, installed CLI OK, full-protocol run vs
vectron GLM-5.3 reproduces baseline (score 0.9451, s_idn 0.846).
2026-09-11 03:52:21 +00:00

97 lines
4.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""slim 电池端到端验收(纯离线):
1. 四次 glm_53 运行信号分解对比(全量 verify / rerun / conc2 / slim——定位 score 差来源
2. slim 样本 top-1 归因 vs 9 参考(主验收判据)
3. 错误普查 + 与历次样本的 meanJSD
"""
import json
import sys
from collections import Counter
from pathlib import Path # noqa: E402
sys.path.insert(0, str(Path(__file__).resolve().parents[2])) # 仓库根/site-packages, 使 evalharness 包可导入
from evalharness.fingerprint.engine import (build_d_normalized, distributions_by_cell, # noqa: E402
jsd_bits, load_reference)
BFD = "/tmp/bfd"
R = str(Path(__file__).resolve().parent / 'references')
MODELS = [
("deepseek_v4_flash", "deepseek_v4_flash_fusion_reference.json"),
("deepseek_v4_flash_0731", "deepseek_v4_flash_0731_fusion_reference.json"),
("deepseek_v4_pro", "deepseek_v4_pro_fusion_reference.json"),
("glm_51", "glm_51_fusion_reference.json"),
("glm_52", "glm52_vectron_fusion_reference.json"),
("glm_53", "glm53_fusion_reference.json"),
("kimi_k2_6", "kimi_k2_6_fusion_reference.json"),
("kimi_k2_7code", "kimi_k2_7code_fusion_reference.json"),
("kimi_k3", "kimi_k3_fusion_reference.json"),
]
NAMES = [m[0] for m in MODELS]
# ---------- 1. 四次运行信号分解 ----------
RUNS = [("全量verify", f"{BFD}/glm_53/verify.json"),
("全量rerun", f"{BFD}/glm_53/rerun/verify_rerun.json"),
("全量conc2", f"{BFD}/glm_53/conc2/verify_conc2.json"),
("slim并发2", f"{BFD}/glm_53/slim/verify_slim.json")]
print("【信号分解】s_dist 由 ratio=meanJSD/max(split_half,0.02) 阶梯量化")
print(f"{'运行':12s} {'score':>7s} {'verdict':>15s} {'s_dist':>7s} {'meanJSD':>8s} "
f"{'splitH':>7s} {'ratio':>6s} {'cells':>5s} {'s_idn':>6s} {'s_meta':>6s}")
for name, path in RUNS:
r = json.load(open(path))
dist = r["signals"]["dist"]
idn = r["signals"].get("identity", {})
met = r["signals"].get("meta", {})
print(f"{name:12s} {r['score']:7.4f} {r['verdict']:>15s} "
f"{dist.get('s_dist', 0):7.3f} {dist.get('mean_jsd') or 0:8.4f} "
f"{(dist.get('split_half') or 0):7.4f} {dist.get('relative_ratio') or 0:6.2f} "
f"{dist.get('comparable_cells'):5d} {idn.get('s_idn', 0):6.3f} {met.get('s_meta', 0):6.3f}")
print()
# ---------- 2. slim 样本验收 ----------
recs = [json.loads(l) for l in open(f"{BFD}/glm_53/slim/raw_answers.jsonl")]
errs = [r for r in recs if r.get("error")]
ec = Counter()
for r in errs:
e = str(r["error"])
for code in ("400", "402", "429", "500", "502", "503", "504", "timeout"):
if code in e.lower():
ec[code] += 1
break
else:
ec[e[:30]] += 1
print(f"slim 记录 {len(recs)}期望434错误 {len(errs)}: {dict(ec) or ''}")
d_new = distributions_by_cell(build_d_normalized(recs))
refs = {n: load_reference(f"{R}/{rf}")["cells"] for n, rf in MODELS}
def score(dist, ref):
js = []
for c in set(dist) & set(ref):
a, b = dist[c], ref[c]
if sum(a.values()) >= 10 and sum(b.values()) >= 10:
js.append(jsd_bits(a, b))
return (sum(js) / len(js) if js else None), len(js)
vals = {n: score(d_new, refs[n])[0] for n in NAMES}
best = min(vals, key=vals.get)
sib = min(v for n, v in vals.items() if n.startswith("glm") and n != "glm_53")
print(f"top-1 归因: {best} {'✓ 正确' if best == 'glm_53' else '✗ 混淆!'} "
f"own={vals['glm_53']:.4f} 同族间距={sib - vals['glm_53']:+.4f} "
f"可比cell={score(d_new, refs['glm_53'])[1]}(期望16)")
prior = {}
for tag, p in [("样本1(全量verify)", f"{BFD}/glm_53/raw_answers.jsonl"),
("样本2(rerun)", f"{BFD}/glm_53/rerun/raw_answers.jsonl"),
("样本3(conc2)", f"{BFD}/glm_53/conc2/raw_answers.jsonl")]:
prior[tag] = distributions_by_cell(build_d_normalized(
[json.loads(l) for l in open(p)]))
for tag, d in prior.items():
js = [jsd_bits(d_new[c], d[c]) for c in set(d_new) & set(d)
if sum(d_new[c].values()) >= 10 and sum(d[c].values()) >= 10]
print(f" vs {tag}: meanJSD={sum(js) / len(js):.4f} ({len(js)} cell)")
tot_v = sum(1 for s in build_d_normalized(recs) if s["cat"] == "valid")
print(f"D 层 valid: {tot_v}/400")