- perf_stats aggregator lives in eval/, not model/: the import failed silently and EVERY perf column was empty (not just ttft). Now warns on stderr instead of swallowing. - repeats > 1 get their own checkpoint key (:rep2, :rep3, ...): repeat 2 previously restored repeat 1's predictions and finished instantly with identical scores. rep1 keeps the legacy key (existing checkpoints still resume). - repeats summary: report the MEAN score and aggregate time/tokens over ALL runs (was: last run only). - README: six-benchmark command as the primary example. Co-Authored-By: Claude <noreply@anthropic.com>
341 lines
8.5 KiB
JSON
341 lines
8.5 KiB
JSON
{
|
|
"formatVersion": 1,
|
|
"protocol": "one-token/v1",
|
|
"model": "DeepSeek/DeepSeek-V4-Pro",
|
|
"collectedAt": "2026-09-01T09:34:18.748Z",
|
|
"samplesPerCell": 25,
|
|
"postReasoning": false,
|
|
"cells": {
|
|
"random-number-1-100:en": {
|
|
"cellId": "random-number-1-100:en",
|
|
"counts": {
|
|
"7": 1,
|
|
"42": 20,
|
|
"50": 2,
|
|
"60": 1,
|
|
"73": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.1063137138648347,
|
|
"normalizedEntropy": 0.16651680624386705,
|
|
"medianLatencyMs": 2347.750417000003,
|
|
"meanCompletionTokens": 168.52,
|
|
"meanReasoningTokens": 165.4
|
|
},
|
|
"random-number-1-100:zh": {
|
|
"cellId": "random-number-1-100:zh",
|
|
"counts": {
|
|
"37": 5,
|
|
"38": 1,
|
|
"42": 13,
|
|
"64": 1,
|
|
"67": 2,
|
|
"73": 1,
|
|
"74": 1,
|
|
"77": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.175241917363884,
|
|
"normalizedEntropy": 0.3274065324760801,
|
|
"medianLatencyMs": 1496.2746009999973,
|
|
"meanCompletionTokens": 38.36,
|
|
"meanReasoningTokens": 35.36
|
|
},
|
|
"random-color:en": {
|
|
"cellId": "random-color:en",
|
|
"counts": {
|
|
"blue": 23,
|
|
"turquoise": 2
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.4021791902022728,
|
|
"normalizedEntropy": 0.08196212700609383,
|
|
"medianLatencyMs": 2145.143300000025,
|
|
"meanCompletionTokens": 62.64,
|
|
"meanReasoningTokens": 59.56
|
|
},
|
|
"random-animal:en": {
|
|
"cellId": "random-animal:en",
|
|
"counts": {
|
|
"elephant": 16,
|
|
"cat": 4,
|
|
"dog": 4,
|
|
"giraffe": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.4438561897747249,
|
|
"normalizedEntropy": 0.25582795543065684,
|
|
"medianLatencyMs": 2106.3286720000033,
|
|
"meanCompletionTokens": 63.4,
|
|
"meanReasoningTokens": 59.68
|
|
},
|
|
"random-number-1-10:en": {
|
|
"cellId": "random-number-1-10:en",
|
|
"counts": {
|
|
"4": 1,
|
|
"5": 1,
|
|
"7": 23
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.48217919020227284,
|
|
"normalizedEntropy": 0.14515039953585215,
|
|
"medianLatencyMs": 2558.256677999976,
|
|
"meanCompletionTokens": 106.72,
|
|
"meanReasoningTokens": 103.72
|
|
},
|
|
"random-letter:en": {
|
|
"cellId": "random-letter:en",
|
|
"counts": {
|
|
"k": 10,
|
|
"m": 6,
|
|
"a": 1,
|
|
"q": 4,
|
|
"x": 2,
|
|
"g": 1,
|
|
"r": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.294693951646702,
|
|
"normalizedEntropy": 0.4881870823256078,
|
|
"medianLatencyMs": 2110.3464540000423,
|
|
"meanCompletionTokens": 63.32,
|
|
"meanReasoningTokens": 60.32
|
|
},
|
|
"random-color:zh": {
|
|
"cellId": "random-color:zh",
|
|
"counts": {
|
|
"蓝": 14,
|
|
"紫": 5,
|
|
"靛蓝": 3,
|
|
"蔚蓝": 2,
|
|
"橙": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.7771563143584552,
|
|
"normalizedEntropy": 0.3621756547718718,
|
|
"medianLatencyMs": 1476.1146930000104,
|
|
"meanCompletionTokens": 29.08,
|
|
"meanReasoningTokens": 25.76
|
|
},
|
|
"coin-flip:en": {
|
|
"cellId": "coin-flip:en",
|
|
"counts": {
|
|
"heads": 23,
|
|
"tails": 2
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.4021791902022728,
|
|
"normalizedEntropy": 0.4021791902022728,
|
|
"medianLatencyMs": 2447.511105999991,
|
|
"meanCompletionTokens": 79.92,
|
|
"meanReasoningTokens": 76.92
|
|
},
|
|
"favorite-number:en": {
|
|
"cellId": "favorite-number:en",
|
|
"counts": {
|
|
"7": 22,
|
|
"42": 3
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.5293608652873644,
|
|
"normalizedEntropy": 0.039837942232197415,
|
|
"medianLatencyMs": 3366.7504310000077,
|
|
"meanCompletionTokens": 128.48,
|
|
"meanReasoningTokens": 125.48
|
|
},
|
|
"random-city:en": {
|
|
"cellId": "random-city:en",
|
|
"counts": {
|
|
"paris": 8,
|
|
"tokyo": 16,
|
|
"kyoto": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.1238561897747246,
|
|
"normalizedEntropy": 0.19912913298727825,
|
|
"medianLatencyMs": 2154.4987719999917,
|
|
"meanCompletionTokens": 58.68,
|
|
"meanReasoningTokens": 55.64
|
|
},
|
|
"random-number-1-10:zh": {
|
|
"cellId": "random-number-1-10:zh",
|
|
"counts": {
|
|
"2": 1,
|
|
"4": 2,
|
|
"7": 22
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.6395563653739031,
|
|
"normalizedEntropy": 0.19252564989537765,
|
|
"medianLatencyMs": 1566.4150939999963,
|
|
"meanCompletionTokens": 30.76,
|
|
"meanReasoningTokens": 27.76
|
|
},
|
|
"coin-flip:zh": {
|
|
"cellId": "coin-flip:zh",
|
|
"counts": {
|
|
"tails": 9,
|
|
"heads": 16
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.9426831892554922,
|
|
"normalizedEntropy": 0.9426831892554922,
|
|
"medianLatencyMs": 1722.1478929999867,
|
|
"meanCompletionTokens": 39.64,
|
|
"meanReasoningTokens": 36.64
|
|
},
|
|
"random-letter:zh": {
|
|
"cellId": "random-letter:zh",
|
|
"counts": {
|
|
"g": 3,
|
|
"r": 2,
|
|
"q": 2,
|
|
"e": 2,
|
|
"k": 2,
|
|
"x": 4,
|
|
"z": 4,
|
|
"b": 3,
|
|
"a": 1,
|
|
"m": 1,
|
|
"s": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 3.303465189601647,
|
|
"normalizedEntropy": 0.702799182138663,
|
|
"medianLatencyMs": 1494.7614950000134,
|
|
"meanCompletionTokens": 35.8,
|
|
"meanReasoningTokens": 32.8
|
|
},
|
|
"random-animal:zh": {
|
|
"cellId": "random-animal:zh",
|
|
"counts": {
|
|
"海豚": 3,
|
|
"猫": 17,
|
|
"大象": 1,
|
|
"斑马": 1,
|
|
"企鹅": 1,
|
|
"长颈鹿": 1,
|
|
"狗": 1
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 1.6741859576379552,
|
|
"normalizedEntropy": 0.29663866359160024,
|
|
"medianLatencyMs": 1486.468074000033,
|
|
"meanCompletionTokens": 31.4,
|
|
"meanReasoningTokens": 28.12
|
|
},
|
|
"random-city:zh": {
|
|
"cellId": "random-city:zh",
|
|
"counts": {
|
|
"巴黎": 7,
|
|
"北京": 4,
|
|
"上海": 3,
|
|
"东京": 9,
|
|
"伦敦": 2
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 2.1264283109928246,
|
|
"normalizedEntropy": 0.37676869138611085,
|
|
"medianLatencyMs": 1559.4182869999786,
|
|
"meanCompletionTokens": 38.12,
|
|
"meanReasoningTokens": 35.12
|
|
},
|
|
"favorite-number:zh": {
|
|
"cellId": "favorite-number:zh",
|
|
"counts": {
|
|
"7": 23,
|
|
"42": 2
|
|
},
|
|
"validCount": 25,
|
|
"invalidCount": 0,
|
|
"refusalCount": 0,
|
|
"emptyCount": 0,
|
|
"errorCount": 0,
|
|
"totalCount": 25,
|
|
"entropyBits": 0.4021791902022728,
|
|
"normalizedEntropy": 0.030266671370904073,
|
|
"medianLatencyMs": 1677.2399570000125,
|
|
"meanCompletionTokens": 64.88,
|
|
"meanReasoningTokens": 61.88
|
|
}
|
|
},
|
|
"meta": {
|
|
"tool": "llm-fingerprint-detector"
|
|
}
|
|
}
|