fm-bench 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +88 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +34 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +43 -0
- package/package.json +11 -6
- package/src/bench.js +129 -37
- package/src/capabilities.js +196 -0
- package/src/cli.js +192 -76
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +110 -106
- package/src/history.js +2 -1
- package/src/macos.js +72 -0
- package/src/metrics.js +212 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/src/metrics.js
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
// Metric provenance catalog.
|
|
2
|
+
//
|
|
3
|
+
// fm-bench measures `fm` from the outside, so not every named metric can be
|
|
4
|
+
// observed directly. Each metric declares how it is obtained:
|
|
5
|
+
//
|
|
6
|
+
// measured — observed directly (process timings, exit codes, token counts)
|
|
7
|
+
// proxy — observed at a coarser granularity than the ideal metric
|
|
8
|
+
// derived — computed from other measured values
|
|
9
|
+
// controlled — an input setting, not a measurement
|
|
10
|
+
//
|
|
11
|
+
// A metric whose requirement is missing from the detected `fm` build is
|
|
12
|
+
// reported as unavailable with a reason instead of a blank or invented value.
|
|
13
|
+
|
|
14
|
+
const DEFINITIONS = [
|
|
15
|
+
{
|
|
16
|
+
key: 'ttft',
|
|
17
|
+
label: 'TTFT',
|
|
18
|
+
kind: 'proxy',
|
|
19
|
+
source: 'arrival time of the first streamed stdout chunk',
|
|
20
|
+
requires: ['streaming'],
|
|
21
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
key: 'e2eLatency',
|
|
25
|
+
label: 'E2E latency',
|
|
26
|
+
kind: 'measured',
|
|
27
|
+
source: 'wall clock from spawning fm until it exits',
|
|
28
|
+
requires: []
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
key: 'generationMs',
|
|
32
|
+
label: 'generation time',
|
|
33
|
+
kind: 'derived',
|
|
34
|
+
source: 'E2E latency minus TTFT',
|
|
35
|
+
requires: ['streaming'],
|
|
36
|
+
reason: 'requires at least two streamed output chunks to separate prefill from decode'
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
key: 'tpot',
|
|
40
|
+
label: 'TPOT',
|
|
41
|
+
kind: 'derived',
|
|
42
|
+
source: '(E2E - TTFT) / (output tokens - 1)',
|
|
43
|
+
requires: ['streaming', 'tokenCounting'],
|
|
44
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
key: 'promptTokens',
|
|
48
|
+
label: 'prompt tokens',
|
|
49
|
+
kind: 'measured',
|
|
50
|
+
source: 'fm count-tokens on the prompt',
|
|
51
|
+
requires: ['tokenCounting'],
|
|
52
|
+
reason: 'this fm build exposes no token-counting command'
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
key: 'outputTokens',
|
|
56
|
+
label: 'output tokens',
|
|
57
|
+
kind: 'measured',
|
|
58
|
+
source: 'fm count-tokens on the captured output',
|
|
59
|
+
requires: ['tokenCounting'],
|
|
60
|
+
reason: 'this fm build exposes no token-counting command'
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
key: 'tokensPerSecond',
|
|
64
|
+
label: 'per-request output tokens/s',
|
|
65
|
+
kind: 'derived',
|
|
66
|
+
source: 'output tokens / E2E seconds',
|
|
67
|
+
requires: ['tokenCounting'],
|
|
68
|
+
reason: 'requires a token-counting fm command'
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
key: 'decodeTokensPerSecond',
|
|
72
|
+
label: 'decode tokens/s',
|
|
73
|
+
kind: 'derived',
|
|
74
|
+
source: '(output tokens - 1) / generation seconds',
|
|
75
|
+
requires: ['streaming', 'tokenCounting'],
|
|
76
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
key: 'prefillTokensPerSecond',
|
|
80
|
+
label: 'prefill tokens/s',
|
|
81
|
+
kind: 'proxy',
|
|
82
|
+
source: 'prompt tokens / TTFT seconds',
|
|
83
|
+
requires: ['streaming', 'tokenCounting'],
|
|
84
|
+
reason: 'requires streaming plus a token-counting fm command'
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
key: 'outputTokenThroughput',
|
|
88
|
+
label: 'aggregate output token throughput',
|
|
89
|
+
kind: 'derived',
|
|
90
|
+
source: 'successful output tokens / measured wall-clock window',
|
|
91
|
+
requires: ['tokenCounting'],
|
|
92
|
+
reason: 'requires a token-counting fm command'
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
key: 'rps',
|
|
96
|
+
label: 'request throughput',
|
|
97
|
+
kind: 'measured',
|
|
98
|
+
source: 'successful requests / measured wall-clock window',
|
|
99
|
+
requires: []
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
key: 'chunkGaps',
|
|
103
|
+
label: 'chunk gaps and second-chunk delay',
|
|
104
|
+
kind: 'proxy',
|
|
105
|
+
source: 'gaps between consecutive streamed stdout chunks',
|
|
106
|
+
requires: ['streaming'],
|
|
107
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
key: 'percentiles',
|
|
111
|
+
label: 'percentiles',
|
|
112
|
+
kind: 'derived',
|
|
113
|
+
source: 'percentile interpolation over successful samples',
|
|
114
|
+
requires: []
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
key: 'successRate',
|
|
118
|
+
label: 'success rate',
|
|
119
|
+
kind: 'measured',
|
|
120
|
+
source: 'successful runs / attempted runs',
|
|
121
|
+
requires: []
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
key: 'goodput',
|
|
125
|
+
label: 'goodput',
|
|
126
|
+
kind: 'derived',
|
|
127
|
+
source: 'successful runs meeting every configured SLO / runs with an SLO verdict',
|
|
128
|
+
requires: ['slo'],
|
|
129
|
+
reason: 'set --slo-ttft-ms, --slo-e2e-ms, or --slo-tpot-ms to enable goodput'
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
key: 'repeatability',
|
|
133
|
+
label: 'repeatability',
|
|
134
|
+
kind: 'derived',
|
|
135
|
+
source: 'most common normalized output hash share across repeated runs',
|
|
136
|
+
requires: []
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
key: 'variability',
|
|
140
|
+
label: 'CV and 95% confidence interval',
|
|
141
|
+
kind: 'derived',
|
|
142
|
+
source: 'sample standard deviation over successful samples',
|
|
143
|
+
requires: [],
|
|
144
|
+
reason: 'needs at least two successful samples'
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
key: 'quota',
|
|
148
|
+
label: 'quota',
|
|
149
|
+
kind: 'measured',
|
|
150
|
+
source: 'fm quota-usage',
|
|
151
|
+
requires: ['quota'],
|
|
152
|
+
reason: 'this fm build exposes no quota command'
|
|
153
|
+
}
|
|
154
|
+
];
|
|
155
|
+
|
|
156
|
+
function requirementsMet(requires = [], context = {}) {
|
|
157
|
+
return requires.every((name) => {
|
|
158
|
+
if (name === 'streaming') return Boolean(context.streaming);
|
|
159
|
+
if (name === 'tokenCounting') return Boolean(context.tokenCounting);
|
|
160
|
+
if (name === 'quota') return Boolean(context.quota);
|
|
161
|
+
if (name === 'slo') return Boolean(context.slo);
|
|
162
|
+
return true;
|
|
163
|
+
});
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Describe which metrics the detected `fm` build can support for a run.
|
|
168
|
+
*
|
|
169
|
+
* @param {{ features?: Record<string, any> }} [capabilities]
|
|
170
|
+
* @param {{ stream?: boolean, slo?: boolean }} [options]
|
|
171
|
+
*/
|
|
172
|
+
export function metricAvailability(capabilities = {}, options = {}) {
|
|
173
|
+
const features = capabilities.features ?? {};
|
|
174
|
+
const context = {
|
|
175
|
+
streaming: options.stream !== false && features.streaming !== false,
|
|
176
|
+
tokenCounting: features.tokenCounting !== false,
|
|
177
|
+
quota: features.quota === true,
|
|
178
|
+
slo: options.slo === true
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
const metrics = {};
|
|
182
|
+
for (const definition of DEFINITIONS) {
|
|
183
|
+
const available = requirementsMet(definition.requires, context);
|
|
184
|
+
metrics[definition.key] = {
|
|
185
|
+
label: definition.label,
|
|
186
|
+
kind: definition.kind,
|
|
187
|
+
source: definition.source,
|
|
188
|
+
available,
|
|
189
|
+
unavailableReason: available ? '' : (definition.reason ?? 'unavailable in this environment')
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
return metrics;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
export function metricDefinition(key) {
|
|
196
|
+
const definition = DEFINITIONS.find((item) => item.key === key);
|
|
197
|
+
return definition ? { ...definition, requires: [...definition.requires] } : null;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
export function metricDefinitions() {
|
|
201
|
+
return DEFINITIONS.map((definition) => ({ ...definition, requires: [...definition.requires] }));
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** Short human line for CLI output, e.g. "token counting: yes, quota: no". */
|
|
205
|
+
export function formatCapabilitySummary(capabilities = {}) {
|
|
206
|
+
const features = capabilities.features ?? {};
|
|
207
|
+
return [
|
|
208
|
+
`token counting ${features.tokenCounting ? 'yes' : 'no'}`,
|
|
209
|
+
`streaming ${features.streaming ? 'yes' : 'no'}`,
|
|
210
|
+
`quota ${features.quota ? 'yes' : 'no'}`
|
|
211
|
+
].join(', ');
|
|
212
|
+
}
|
package/src/process.js
CHANGED
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
import { spawn } from 'node:child_process';
|
|
2
2
|
|
|
3
|
+
// Children currently alive. The CLI registers signal handlers that terminate
|
|
4
|
+
// these so Ctrl+C cannot leave `fm` processes behind.
|
|
5
|
+
const activeChildren = new Set();
|
|
6
|
+
|
|
7
|
+
export function activeChildCount() {
|
|
8
|
+
return activeChildren.size;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Terminate every running child process. Returns how many were signalled.
|
|
13
|
+
* @param {NodeJS.Signals} signal
|
|
14
|
+
*/
|
|
15
|
+
export function killActiveChildren(signal = 'SIGTERM') {
|
|
16
|
+
let killed = 0;
|
|
17
|
+
for (const child of activeChildren) {
|
|
18
|
+
if (child.exitCode != null || child.signalCode != null) continue;
|
|
19
|
+
try {
|
|
20
|
+
child.kill(signal);
|
|
21
|
+
killed += 1;
|
|
22
|
+
} catch {
|
|
23
|
+
// Process already gone.
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
return killed;
|
|
27
|
+
}
|
|
28
|
+
|
|
3
29
|
export function runProcess(command, args = [], options = {}) {
|
|
4
30
|
const {
|
|
5
31
|
input,
|
|
@@ -15,6 +41,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
15
41
|
env,
|
|
16
42
|
stdio: ['pipe', 'pipe', 'pipe']
|
|
17
43
|
});
|
|
44
|
+
activeChildren.add(child);
|
|
18
45
|
|
|
19
46
|
let stdout = '';
|
|
20
47
|
let stderr = '';
|
|
@@ -55,11 +82,16 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
55
82
|
stderr += chunk;
|
|
56
83
|
});
|
|
57
84
|
|
|
58
|
-
|
|
85
|
+
const finish = (result) => {
|
|
86
|
+
if (settled) return;
|
|
59
87
|
settled = true;
|
|
88
|
+
activeChildren.delete(child);
|
|
60
89
|
if (timer) clearTimeout(timer);
|
|
61
|
-
|
|
62
|
-
|
|
90
|
+
resolve(result);
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
child.on('error', (error) => {
|
|
94
|
+
finish({
|
|
63
95
|
command,
|
|
64
96
|
args,
|
|
65
97
|
code: null,
|
|
@@ -73,15 +105,12 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
73
105
|
firstStderrMs,
|
|
74
106
|
error,
|
|
75
107
|
timedOut,
|
|
76
|
-
durationMs: Number(
|
|
108
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
77
109
|
});
|
|
78
110
|
});
|
|
79
111
|
|
|
80
112
|
child.on('close', (code, signal) => {
|
|
81
|
-
|
|
82
|
-
if (timer) clearTimeout(timer);
|
|
83
|
-
const endedAt = process.hrtime.bigint();
|
|
84
|
-
resolve({
|
|
113
|
+
finish({
|
|
85
114
|
command,
|
|
86
115
|
args,
|
|
87
116
|
code,
|
|
@@ -93,8 +122,9 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
93
122
|
stdoutChunkTimesMs,
|
|
94
123
|
firstStdoutMs,
|
|
95
124
|
firstStderrMs,
|
|
125
|
+
error: null,
|
|
96
126
|
timedOut,
|
|
97
|
-
durationMs: Number(
|
|
127
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
98
128
|
});
|
|
99
129
|
});
|
|
100
130
|
|
package/src/prompts.js
CHANGED
|
@@ -195,16 +195,28 @@ export async function loadPrompts(options = {}) {
|
|
|
195
195
|
|
|
196
196
|
async function loadPromptFile(filePath) {
|
|
197
197
|
const absolutePath = path.resolve(filePath);
|
|
198
|
-
|
|
198
|
+
let content;
|
|
199
|
+
try {
|
|
200
|
+
content = await fs.readFile(absolutePath, 'utf8');
|
|
201
|
+
} catch (error) {
|
|
202
|
+
throw new Error(error.code === 'ENOENT'
|
|
203
|
+
? `Prompt file not found: ${absolutePath}`
|
|
204
|
+
: `Cannot read prompt file ${absolutePath}: ${error.message}`);
|
|
205
|
+
}
|
|
199
206
|
const trimmed = content.trim();
|
|
200
207
|
|
|
201
208
|
if (!trimmed) return [];
|
|
202
209
|
|
|
203
210
|
if (absolutePath.endsWith('.json')) {
|
|
204
|
-
|
|
211
|
+
let parsed;
|
|
212
|
+
try {
|
|
213
|
+
parsed = JSON.parse(trimmed);
|
|
214
|
+
} catch (error) {
|
|
215
|
+
throw new Error(`Cannot parse ${absolutePath} as JSON: ${error.message}`);
|
|
216
|
+
}
|
|
205
217
|
const items = Array.isArray(parsed) ? parsed : parsed.prompts;
|
|
206
218
|
if (!Array.isArray(items)) {
|
|
207
|
-
throw new Error(
|
|
219
|
+
throw new Error(`${absolutePath} must be a JSON array of prompts or an object with a prompts array`);
|
|
208
220
|
}
|
|
209
221
|
return items.map((item, index) => normalizePromptItem(item, index));
|
|
210
222
|
}
|
|
@@ -212,7 +224,13 @@ async function loadPromptFile(filePath) {
|
|
|
212
224
|
if (absolutePath.endsWith('.jsonl')) {
|
|
213
225
|
return trimmed.split(/\r?\n/)
|
|
214
226
|
.filter(Boolean)
|
|
215
|
-
.map((line, index) =>
|
|
227
|
+
.map((line, index) => {
|
|
228
|
+
try {
|
|
229
|
+
return normalizePromptItem(JSON.parse(line), index);
|
|
230
|
+
} catch (error) {
|
|
231
|
+
throw new Error(`Cannot parse line ${index + 1} of ${absolutePath} as JSON: ${error.message}`);
|
|
232
|
+
}
|
|
233
|
+
});
|
|
216
234
|
}
|
|
217
235
|
|
|
218
236
|
return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({
|
package/src/report.js
CHANGED
|
@@ -16,6 +16,7 @@ export function flattenResults(results) {
|
|
|
16
16
|
concurrency: result.concurrency ?? '',
|
|
17
17
|
prompt_id: result.promptId,
|
|
18
18
|
run: result.run,
|
|
19
|
+
attempts: result.attempts ?? 1,
|
|
19
20
|
ok: result.ok,
|
|
20
21
|
duration_ms: round(result.durationMs),
|
|
21
22
|
ttft_ms: round(result.firstTokenMs),
|
|
@@ -58,9 +59,19 @@ export async function writeReport(filePath, payload, format) {
|
|
|
58
59
|
|
|
59
60
|
function csvEscape(value) {
|
|
60
61
|
const text = String(value ?? '');
|
|
61
|
-
|
|
62
|
-
|
|
62
|
+
const safe = guardFormula(text);
|
|
63
|
+
if (/[",\n\r]/.test(safe)) {
|
|
64
|
+
return `"${safe.replaceAll('"', '""')}"`;
|
|
63
65
|
}
|
|
66
|
+
return safe;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// Spreadsheet programs execute cells that start with =, +, @, or a non-numeric
|
|
70
|
+
// leading -. Prompt text and captured output are untrusted input, so prefix
|
|
71
|
+
// those cells with a single quote to keep them literal text.
|
|
72
|
+
function guardFormula(text) {
|
|
73
|
+
if (/^[=+@\t\r]/.test(text)) return `'${text}`;
|
|
74
|
+
if (text.startsWith('-') && !/^-?\d+(\.\d+)?$/.test(text)) return `'${text}`;
|
|
64
75
|
return text;
|
|
65
76
|
}
|
|
66
77
|
|
package/src/schema.js
CHANGED
|
@@ -126,10 +126,26 @@ export function validateReport(value) {
|
|
|
126
126
|
for (const key of REQUIRED_TOP_LEVEL) {
|
|
127
127
|
if (!(key in report)) errors.push(`missing required field: ${key}`);
|
|
128
128
|
}
|
|
129
|
-
if (!Array.isArray(report.summary))
|
|
129
|
+
if (!Array.isArray(report.summary)) {
|
|
130
|
+
errors.push('summary must be an array');
|
|
131
|
+
} else {
|
|
132
|
+
for (const [index, item] of report.summary.entries()) {
|
|
133
|
+
if (!item || typeof item !== 'object') {
|
|
134
|
+
errors.push(`summary[${index}] must be an object`);
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
const row = /** @type {Record<string, unknown>} */ (item);
|
|
138
|
+
if (typeof row.model !== 'string' || row.model === '') {
|
|
139
|
+
errors.push(`summary[${index}].model must be a non-empty string`);
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
130
143
|
if (report.schemaVersion != null && report.schemaVersion !== REPORT_SCHEMA_VERSION) {
|
|
131
144
|
errors.push(`unsupported schemaVersion: ${report.schemaVersion} (expected ${REPORT_SCHEMA_VERSION})`);
|
|
132
145
|
}
|
|
146
|
+
if (report.metrics != null && (typeof report.metrics !== 'object' || Array.isArray(report.metrics))) {
|
|
147
|
+
errors.push('metrics must be an object when present');
|
|
148
|
+
}
|
|
133
149
|
if (errors.length > 0) return { ok: false, errors };
|
|
134
150
|
return { ok: true, report };
|
|
135
151
|
}
|
package/src/stats.js
CHANGED
|
@@ -1,3 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sample statistics for one metric.
|
|
3
|
+
*
|
|
4
|
+
* Spread metrics (`stddev`, `cv`, `ci95*`) need at least two samples. With a
|
|
5
|
+
* single sample they are `null` rather than `0`, because a "0% variation" or a
|
|
6
|
+
* zero-width confidence interval is invented precision, not a measurement.
|
|
7
|
+
*
|
|
8
|
+
* @param {number[]} values
|
|
9
|
+
*/
|
|
1
10
|
export function summarizeNumbers(values) {
|
|
2
11
|
const clean = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
|
|
3
12
|
if (clean.length === 0) {
|
|
@@ -20,11 +29,12 @@ export function summarizeNumbers(values) {
|
|
|
20
29
|
|
|
21
30
|
const total = clean.reduce((sum, value) => sum + value, 0);
|
|
22
31
|
const avg = total / clean.length;
|
|
23
|
-
const
|
|
32
|
+
const hasSpread = clean.length > 1;
|
|
33
|
+
const variance = hasSpread
|
|
24
34
|
? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
|
|
25
|
-
:
|
|
26
|
-
const stddev = Math.sqrt(variance);
|
|
27
|
-
const margin =
|
|
35
|
+
: null;
|
|
36
|
+
const stddev = variance == null ? null : Math.sqrt(variance);
|
|
37
|
+
const margin = hasSpread ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : null;
|
|
28
38
|
return {
|
|
29
39
|
count: clean.length,
|
|
30
40
|
min: clean[0],
|
|
@@ -32,9 +42,9 @@ export function summarizeNumbers(values) {
|
|
|
32
42
|
avg,
|
|
33
43
|
sum: total,
|
|
34
44
|
stddev,
|
|
35
|
-
cv: avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
36
|
-
ci95Low: avg - margin,
|
|
37
|
-
ci95High: avg + margin,
|
|
45
|
+
cv: hasSpread && avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
46
|
+
ci95Low: hasSpread ? avg - margin : null,
|
|
47
|
+
ci95High: hasSpread ? avg + margin : null,
|
|
38
48
|
p50: percentile(clean, 50),
|
|
39
49
|
p90: percentile(clean, 90),
|
|
40
50
|
p95: percentile(clean, 95),
|
|
@@ -42,16 +52,25 @@ export function summarizeNumbers(values) {
|
|
|
42
52
|
};
|
|
43
53
|
}
|
|
44
54
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
/**
|
|
56
|
+
* Percentile with linear interpolation between closest ranks (the same method
|
|
57
|
+
* as Excel's PERCENTILE.INC / NumPy's default). Accepts unsorted input so
|
|
58
|
+
* callers cannot silently get a wrong answer from an unsorted array.
|
|
59
|
+
*
|
|
60
|
+
* @param {number[]} values
|
|
61
|
+
* @param {number} percentileValue 0-100
|
|
62
|
+
*/
|
|
63
|
+
export function percentile(values, percentileValue) {
|
|
64
|
+
const sorted = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
|
|
65
|
+
if (sorted.length === 0) return null;
|
|
66
|
+
if (sorted.length === 1) return sorted[0];
|
|
48
67
|
|
|
49
|
-
const rank = (percentileValue / 100) * (
|
|
68
|
+
const rank = (percentileValue / 100) * (sorted.length - 1);
|
|
50
69
|
const low = Math.floor(rank);
|
|
51
70
|
const high = Math.ceil(rank);
|
|
52
|
-
if (low === high) return
|
|
71
|
+
if (low === high) return sorted[low];
|
|
53
72
|
const weight = rank - low;
|
|
54
|
-
return
|
|
73
|
+
return sorted[low] * (1 - weight) + sorted[high] * weight;
|
|
55
74
|
}
|
|
56
75
|
|
|
57
76
|
export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
@@ -66,6 +85,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
66
85
|
concurrency,
|
|
67
86
|
description: status.description,
|
|
68
87
|
available: status.available,
|
|
88
|
+
unsupported: Boolean(status.unsupported),
|
|
69
89
|
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
70
90
|
results: []
|
|
71
91
|
});
|
|
@@ -80,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
80
100
|
concurrency: result.concurrency,
|
|
81
101
|
description: '',
|
|
82
102
|
available: true,
|
|
103
|
+
unsupported: false,
|
|
83
104
|
skippedReason: '',
|
|
84
105
|
results: []
|
|
85
106
|
});
|
|
@@ -98,7 +119,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
98
119
|
const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
|
|
99
120
|
const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
|
|
100
121
|
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
101
|
-
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
122
|
+
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
|
|
102
123
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
103
124
|
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
104
125
|
const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
|
|
@@ -110,14 +131,18 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
110
131
|
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
111
132
|
const totalTokens = promptTokens.sum + outputTokens.sum;
|
|
112
133
|
const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
|
|
134
|
+
const attempts = entry.results.reduce((sum, result) => sum + (result.attempts ?? 1), 0);
|
|
113
135
|
|
|
114
136
|
return {
|
|
115
137
|
model: entry.model,
|
|
116
138
|
concurrency: entry.concurrency,
|
|
117
139
|
description: entry.description,
|
|
118
140
|
available: entry.available,
|
|
141
|
+
unsupported: entry.unsupported,
|
|
119
142
|
skippedReason: entry.skippedReason,
|
|
120
143
|
attempted: entry.results.length,
|
|
144
|
+
attempts,
|
|
145
|
+
retried: entry.results.length > 0 ? Math.max(0, attempts - entry.results.length) : 0,
|
|
121
146
|
successes: successes.length,
|
|
122
147
|
failures: failures.length,
|
|
123
148
|
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
@@ -147,6 +172,10 @@ function summaryKey(model, concurrency) {
|
|
|
147
172
|
return `${model}::${concurrency ?? 'default'}`;
|
|
148
173
|
}
|
|
149
174
|
|
|
175
|
+
// Two-sided 95% t critical values. Exact table entries up to 30 degrees of
|
|
176
|
+
// freedom, then the standard 2.0 / 1.96 approximations for larger samples.
|
|
177
|
+
// fm-bench uses this for a mean confidence interval, which is context for
|
|
178
|
+
// small samples rather than a hypothesis test.
|
|
150
179
|
function tCritical95(n) {
|
|
151
180
|
const df = Math.max(1, n - 1);
|
|
152
181
|
const table = {
|