ruoxi_sun 09b2add673 fingerprint: integrate fp_fusion model fingerprint benchmark
New evalharness/fingerprint/ package (from evalstone fp_fusion v1.1,
2026-09-07 pruning final): probe battery -> concurrent collection ->
five scoring views (verify/attribution/variant/adversarial/robustness),
bundled family aliases + 27 reference fingerprints (12 fp_fusion schema).

- CLI: 'evalharness fingerprint run ...' (REMAINDER passthrough, single
  source of arg definitions) + 'fingerprint list' for bundled references
- imports rewritten package-relative; direct 'python3 run_fp_fusion.py'
  execution kept working via package bootstrap
- offline analysis/collection scripts made path-independent (previously
  pinned to a /opt/evalscope path absent on this host)
- shell scripts: hardcoded API key -> FP_API_KEY/OPENAI_API_KEY env vars
- --reference accepts short names resolved against bundled references/
- pyproject: +httpx dependency, package-data references/*.json
- tests/test_fingerprint.py: 10 offline tests (battery definitions,
  assembly counts, normalization, signals, verdict ladder, CLI wiring)
- README: fingerprint section + architecture entry

Verified on H20-1: tests 10/10, installed CLI OK, full-protocol run vs
vectron GLM-5.3 reproduces baseline (score 0.9451, s_idn 0.846).
2026-09-11 03:52:21 +00:00

86 lines
3.6 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""并发2实验评估错误普查 + 指纹稳定性(纯离线,零 API
对照:
- 串行 rerun 基线: 2079s, 688/691, p50=2110ms, meanJSD=0.0836, score=0.7318
- 并发2 (本次): 677s, 689/691, p50=1404ms, meanJSD=0.0881, score=0.7167
稳定性锚点:串行样本间 test-retest meanJSD ≈ 0.068
"""
import json
import sys
from collections import Counter
from pathlib import Path # noqa: E402
sys.path.insert(0, str(Path(__file__).resolve().parents[2])) # 仓库根/site-packages, 使 evalharness 包可导入
from evalharness.fingerprint.engine import (build_d_normalized, distributions_by_cell, # noqa: E402
jsd_bits, load_reference)
BFD = "/tmp/bfd"
R = str(Path(__file__).resolve().parent / 'references')
MODELS = [
("deepseek_v4_flash", "deepseek_v4_flash_fusion_reference.json", "DS"),
("deepseek_v4_flash_0731", "deepseek_v4_flash_0731_fusion_reference.json", "DS"),
("deepseek_v4_pro", "deepseek_v4_pro_fusion_reference.json", "DS"),
("glm_51", "glm_51_fusion_reference.json", "GLM"),
("glm_52", "glm52_vectron_fusion_reference.json", "GLM"),
("glm_53", "glm53_fusion_reference.json", "GLM"),
("kimi_k2_6", "kimi_k2_6_fusion_reference.json", "Kimi"),
("kimi_k2_7code", "kimi_k2_7code_fusion_reference.json", "Kimi"),
("kimi_k3", "kimi_k3_fusion_reference.json", "Kimi"),
]
NAMES = [m[0] for m in MODELS]
FAMILY = {m[0]: m[2] for m in MODELS}
recs = [json.loads(l) for l in open(f"{BFD}/glm_53/conc2/raw_answers.jsonl")]
errs = [r for r in recs if r.get("error")]
print(f"记录 {len(recs)},错误 {len(errs)} ({len(errs) / len(recs) * 100:.1f}%)")
ec = Counter()
for r in errs:
e = str(r["error"])
for code in ("400", "401", "402", "403", "429", "500", "502", "503", "504", "timeout"):
if code in e.lower():
ec[code] += 1
break
else:
ec[e[:50]] += 1
for k, v in ec.items():
print(f" {k}: {v}")
conc2 = build_d_normalized(recs)
d2 = distributions_by_cell(conc2)
with open(f"{BFD}/glm_53/raw_answers.jsonl") as f:
s1 = distributions_by_cell(build_d_normalized([json.loads(l) for l in f]))
with open(f"{BFD}/glm_53/rerun/raw_answers.jsonl") as f:
s2 = distributions_by_cell(build_d_normalized([json.loads(l) for l in f]))
refs = {n: load_reference(f"{R}/{rf}")["cells"] for n, rf, _ in MODELS}
def score(dist, ref):
js = []
for c in set(dist) & set(ref):
a, b = dist[c], ref[c]
if sum(a.values()) >= 10 and sum(b.values()) >= 10:
js.append(jsd_bits(a, b))
return (sum(js) / len(js) if js else None), len(js)
vals = {r: score(d2, refs[r])[0] for r in NAMES}
own = vals["glm_53"]
best = min(vals, key=vals.get)
sib = min(v for r, v in vals.items() if FAMILY[r] == "GLM" and r != "glm_53")
n_cells = score(d2, refs["glm_53"])[1]
print(f"\ntop-1 归因: {best} {'✓ 正确' if best == 'glm_53' else '✗ 混淆!'} "
f"own={own:.4f} 同族间距={sib - own:+.4f} 可比cell={n_cells}")
j1 = [jsd_bits(d2[c], s1[c]) for c in set(d2) & set(s1)
if sum(d2[c].values()) >= 10 and sum(s1[c].values()) >= 10]
j2 = [jsd_bits(d2[c], s2[c]) for c in set(d2) & set(s2)
if sum(d2[c].values()) >= 10 and sum(s2[c].values()) >= 10]
print(f"vs 串行样本1(原verify): meanJSD={sum(j1) / len(j1):.4f} ({len(j1)} cell)")
print(f"vs 串行样本2(rerun): meanJSD={sum(j2) / len(j2):.4f} ({len(j2)} cell)")
print("锚点: 串行样本间 test-retest meanJSD≈0.068 —— 并发样本若同量级即不扰动指纹")
tot_v = sum(1 for s in conc2 if s["cat"] == "valid")
print(f"\nD 层 valid: {tot_v}/650串行 rerun 基线 549/650")