New evalharness/fingerprint/ package (from evalstone fp_fusion v1.1, 2026-09-07 pruning final): probe battery -> concurrent collection -> five scoring views (verify/attribution/variant/adversarial/robustness), bundled family aliases + 27 reference fingerprints (12 fp_fusion schema). - CLI: 'evalharness fingerprint run ...' (REMAINDER passthrough, single source of arg definitions) + 'fingerprint list' for bundled references - imports rewritten package-relative; direct 'python3 run_fp_fusion.py' execution kept working via package bootstrap - offline analysis/collection scripts made path-independent (previously pinned to a /opt/evalscope path absent on this host) - shell scripts: hardcoded API key -> FP_API_KEY/OPENAI_API_KEY env vars - --reference accepts short names resolved against bundled references/ - pyproject: +httpx dependency, package-data references/*.json - tests/test_fingerprint.py: 10 offline tests (battery definitions, assembly counts, normalization, signals, verdict ladder, CLI wiring) - README: fingerprint section + architecture entry Verified on H20-1: tests 10/10, installed CLI OK, full-protocol run vs vectron GLM-5.3 reproduces baseline (score 0.9451, s_idn 0.846).
173 lines
4.2 KiB
JSON
173 lines
4.2 KiB
JSON
{
|
|
"formatVersion": 1,
|
|
"protocol": "one-token/v1",
|
|
"model": "Qwen3-8B",
|
|
"collectedAt": "2026-08-28T05:58:28.724Z",
|
|
"samplesPerCell": 25,
|
|
"postReasoning": false,
|
|
"cells": {
|
|
"random-number-1-100:en": {
|
|
"cellId": "random-number-1-100:en",
|
|
"counts": {
|
|
"42": 25
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0,
|
|
"normalizedEntropy": 0,
|
|
"medianLatencyMs": 9036.364354999998,
|
|
"meanCompletionTokens": 2,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-number-1-100:zh": {
|
|
"cellId": "random-number-1-100:zh",
|
|
"counts": {
|
|
"42": 25
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0,
|
|
"normalizedEntropy": 0,
|
|
"medianLatencyMs": 9103.937199000007,
|
|
"meanCompletionTokens": 2,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-color:en": {
|
|
"cellId": "random-color:en",
|
|
"counts": {
|
|
"blue": 13,
|
|
"indigo": 4,
|
|
"orange": 1,
|
|
"azure": 3,
|
|
"teal": 2,
|
|
"cyan": 1,
|
|
"turquoise": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.1294320362548183,
|
|
"normalizedEntropy": 0.43396770210458313,
|
|
"medianLatencyMs": 8225.354339000012,
|
|
"meanCompletionTokens": 4.72,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-animal:en": {
|
|
"cellId": "random-animal:en",
|
|
"counts": {
|
|
"elephant": 11,
|
|
"seal": 1,
|
|
"platypus": 1,
|
|
"giraffe": 4,
|
|
"penguin": 5,
|
|
"zebra": 2,
|
|
"lion": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.257320658596841,
|
|
"normalizedEntropy": 0.3999606975611018,
|
|
"medianLatencyMs": 8505.359566999978,
|
|
"meanCompletionTokens": 7.08,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-number-1-10:en": {
|
|
"cellId": "random-number-1-10:en",
|
|
"counts": {
|
|
"7": 25
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0,
|
|
"normalizedEntropy": 0,
|
|
"medianLatencyMs": 8840.802993999998,
|
|
"meanCompletionTokens": 1,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-letter:en": {
|
|
"cellId": "random-letter:en",
|
|
"counts": {
|
|
"z": 3,
|
|
"q": 11,
|
|
"x": 6,
|
|
"t": 1,
|
|
"m": 1,
|
|
"y": 3
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.120924277228159,
|
|
"normalizedEntropy": 0.45121826986581004,
|
|
"medianLatencyMs": 8260.609531000024,
|
|
"meanCompletionTokens": 1,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"random-color:zh": {
|
|
"cellId": "random-color:zh",
|
|
"counts": {
|
|
"蓝": 18,
|
|
"蓝紫": 1,
|
|
"靛蓝": 2,
|
|
"天蓝": 2,
|
|
"钴蓝": 1,
|
|
"珊瑚橙": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.4815101887362598,
|
|
"normalizedEntropy": 0.3019244386785708,
|
|
"medianLatencyMs": 8469.920075000031,
|
|
"meanCompletionTokens": 2.08,
|
|
"meanReasoningTokens": null
|
|
},
|
|
"coin-flip:en": {
|
|
"cellId": "coin-flip:en",
|
|
"counts": {
|
|
"heads": 17,
|
|
"tails": 2
|
|
},
|
|
"validCount": 19,
|
|
"invalidCount": 6,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.4854607607459134,
|
|
"normalizedEntropy": 0.4854607607459134,
|
|
"medianLatencyMs": 8714.726423000015,
|
|
"meanCompletionTokens": 5.24,
|
|
"meanReasoningTokens": null
|
|
}
|
|
},
|
|
"meta": {
|
|
"tool": "llm-fingerprint-detector"
|
|
}
|
|
}
|