fm-bench 0.7.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -24
- package/docs/compatibility.md +10 -5
- package/docs/methodology.md +14 -9
- package/docs/releasing.md +33 -16
- package/docs/report-format.md +23 -14
- package/docs/supported-platforms.md +9 -4
- package/package.json +5 -5
- package/src/bench.js +73 -23
- package/src/capabilities.js +26 -9
- package/src/cli.js +80 -8
- package/src/compare.js +10 -1
- package/src/export.js +6 -4
- package/src/fm-help.js +62 -7
- package/src/fm.js +123 -2
- package/src/metrics.js +15 -7
- package/src/process.js +7 -1
- package/src/report.js +3 -2
- package/src/schema.js +25 -0
- package/src/stats.js +36 -0
- package/src/table.js +60 -26
package/src/fm.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import os from 'node:os';
|
|
2
2
|
import { stripAnsi } from './ansi.js';
|
|
3
3
|
import { detectFmCapabilities } from './capabilities.js';
|
|
4
|
-
import { firstLine, isUnsupportedModelError, parseAvailabilityOutput } from './fm-help.js';
|
|
4
|
+
import { firstLine, isUnsupportedModelError, parseAvailabilityOutput, parseModelList, withoutWarnings } from './fm-help.js';
|
|
5
5
|
import { runProcess } from './process.js';
|
|
6
6
|
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
7
7
|
|
|
@@ -25,12 +25,33 @@ export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
|
25
25
|
};
|
|
26
26
|
}
|
|
27
27
|
|
|
28
|
+
/**
|
|
29
|
+
* Ask `fm models` once for every model's status. Returns `null` when the
|
|
30
|
+
* build has no `models` command (older builds use per-model `fm available`).
|
|
31
|
+
* @returns {Promise<{ entries: Map<string, { name: string, available: boolean, identity: string, reason: string }>, error: string } | null>}
|
|
32
|
+
*/
|
|
33
|
+
export async function listModelStatus(fmBin, options = {}) {
|
|
34
|
+
if (options.capabilities?.features?.modelListCommand !== 'models') return null;
|
|
35
|
+
const result = await runProcess(fmBin, ['models'], {
|
|
36
|
+
timeoutMs: options.timeoutMs ?? 15_000
|
|
37
|
+
});
|
|
38
|
+
const output = `${result.stdout}${result.stderr}`;
|
|
39
|
+
const entries = new Map(parseModelList(output).map((entry) => [entry.name, entry]));
|
|
40
|
+
const error = result.error
|
|
41
|
+
? (result.stderr || result.error.message)
|
|
42
|
+
: (entries.size === 0 ? firstLine(output) || `fm models exited with code ${result.code}` : '');
|
|
43
|
+
return { entries, error };
|
|
44
|
+
}
|
|
45
|
+
|
|
28
46
|
/**
|
|
29
47
|
* Check one model against the detected `fm` build.
|
|
30
48
|
*
|
|
31
49
|
* Models the build does not expose are reported as unsupported without
|
|
32
50
|
* spawning `fm` at all, so a raw argument-error blob from `fm` can never end
|
|
33
51
|
* up in a report or in `fm-bench models` output.
|
|
52
|
+
*
|
|
53
|
+
* Pass `options.modelList` (from `listModelStatus`) to reuse one `fm models`
|
|
54
|
+
* call across several models.
|
|
34
55
|
*/
|
|
35
56
|
export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
36
57
|
const capabilities = options.capabilities;
|
|
@@ -40,11 +61,33 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
|
40
61
|
model,
|
|
41
62
|
available: false,
|
|
42
63
|
unsupported: true,
|
|
64
|
+
identity: '',
|
|
43
65
|
raw: '',
|
|
44
66
|
reason: `not supported by this fm build (supported: ${known.join(', ')})`
|
|
45
67
|
};
|
|
46
68
|
}
|
|
47
69
|
|
|
70
|
+
const listCommand = capabilities ? capabilities.features?.modelListCommand : 'available';
|
|
71
|
+
if (listCommand === 'models') {
|
|
72
|
+
const list = options.modelList ?? await listModelStatus(fmBin, options);
|
|
73
|
+
const entry = list?.entries.get(model);
|
|
74
|
+
if (entry) {
|
|
75
|
+
return { model, available: entry.available, identity: entry.identity, raw: '', reason: entry.reason };
|
|
76
|
+
}
|
|
77
|
+
return {
|
|
78
|
+
model,
|
|
79
|
+
available: false,
|
|
80
|
+
identity: '',
|
|
81
|
+
raw: '',
|
|
82
|
+
reason: list?.error || 'not reported by fm models'
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
if (listCommand == null) {
|
|
86
|
+
// No availability command at all: the first fm respond call is the only
|
|
87
|
+
// way to find out, and a failure there is recorded per run.
|
|
88
|
+
return { model, available: true, identity: '', raw: '', reason: '' };
|
|
89
|
+
}
|
|
90
|
+
|
|
48
91
|
const result = await runProcess(fmBin, ['available', '--model', model], {
|
|
49
92
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
50
93
|
});
|
|
@@ -61,7 +104,27 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
|
61
104
|
const supported = known.length > 0 ? ` (supported: ${known.join(', ')})` : '';
|
|
62
105
|
parsed.reason = `not supported by this fm build${supported}`;
|
|
63
106
|
}
|
|
64
|
-
return parsed;
|
|
107
|
+
return { identity: '', ...parsed };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Whether the fm Legal Notice & Terms have been accepted. `fm respond` cannot
|
|
112
|
+
* run until they are, so `doctor` reports it explicitly.
|
|
113
|
+
* @returns {Promise<{ supported: boolean, agreed: boolean|null, detail: string }>}
|
|
114
|
+
*/
|
|
115
|
+
export async function getLicenseStatus(fmBin, options = {}) {
|
|
116
|
+
if (options.capabilities && !options.capabilities.features?.license) {
|
|
117
|
+
return { supported: false, agreed: null, detail: 'this fm build has no license command' };
|
|
118
|
+
}
|
|
119
|
+
const result = await runProcess(fmBin, ['license', '--status'], {
|
|
120
|
+
timeoutMs: options.timeoutMs ?? 10_000
|
|
121
|
+
});
|
|
122
|
+
const output = withoutWarnings(`${result.stdout}${result.stderr}`).replace(/\s+/g, ' ').trim();
|
|
123
|
+
if (result.error) {
|
|
124
|
+
return { supported: true, agreed: null, detail: result.stderr || result.error.message };
|
|
125
|
+
}
|
|
126
|
+
const agreed = result.code === 0 && /\bagreed\b/i.test(output) && !/\bnot\b|\bnever\b/i.test(output);
|
|
127
|
+
return { supported: true, agreed, detail: output || `fm license --status exited with code ${result.code}` };
|
|
65
128
|
}
|
|
66
129
|
|
|
67
130
|
/**
|
|
@@ -133,6 +196,61 @@ export async function countTokens(fmBin, text, options = {}) {
|
|
|
133
196
|
};
|
|
134
197
|
}
|
|
135
198
|
|
|
199
|
+
/**
|
|
200
|
+
* Measure the constant framing overhead `fm count-tokens` adds to every count.
|
|
201
|
+
*
|
|
202
|
+
* On macOS 27.2 `count-tokens` reports 2 for "a", 3 for "a a", and 5 for
|
|
203
|
+
* "a a a a": one token per word plus one framing token. Model output must not
|
|
204
|
+
* carry that extra token, so the overhead is derived from three counts whose
|
|
205
|
+
* per-word step must agree; anything inconsistent leaves counts uncorrected.
|
|
206
|
+
*
|
|
207
|
+
* @returns {Promise<{ overhead: number, calibrated: boolean }>}
|
|
208
|
+
*/
|
|
209
|
+
export async function calibrateTokenCounter(fmBin, options = {}) {
|
|
210
|
+
const counts = [];
|
|
211
|
+
for (const text of ['a', 'a a', 'a a a']) {
|
|
212
|
+
const counted = await countTokens(fmBin, text, options);
|
|
213
|
+
if (!counted.ok) return { overhead: 0, calibrated: false };
|
|
214
|
+
counts.push(counted.count);
|
|
215
|
+
}
|
|
216
|
+
const step = counts[1] - counts[0];
|
|
217
|
+
const overhead = counts[0] - step;
|
|
218
|
+
const consistent = step >= 1 && counts[2] - counts[1] === step && overhead >= 0 && overhead <= 16;
|
|
219
|
+
return consistent ? { overhead, calibrated: true } : { overhead: 0, calibrated: false };
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
// Stdout chunks closer together than this are one write burst from fm, not
|
|
223
|
+
// separate streaming steps. On macOS 27.2 a short answer's tail arrives as
|
|
224
|
+
// several writes well under 1 ms apart, while real deltas are 20 ms or more
|
|
225
|
+
// apart; counting the burst as decode steps reported tens of thousands of
|
|
226
|
+
// tokens per second.
|
|
227
|
+
export const DELIVERY_COALESCE_MS = 5;
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Group stdout chunk arrivals into deliveries: runs of chunks that each
|
|
231
|
+
* arrived less than `coalesceMs` after the previous one.
|
|
232
|
+
* @param {number[]} timesMs chunk arrival times
|
|
233
|
+
* @param {number[]} lengths chunk lengths in characters
|
|
234
|
+
* @returns {{ atMs: number, endChars: number }[]} delivery start time and the
|
|
235
|
+
* stdout length once the delivery is complete
|
|
236
|
+
*/
|
|
237
|
+
export function groupDeliveries(timesMs = [], lengths = [], coalesceMs = DELIVERY_COALESCE_MS) {
|
|
238
|
+
const deliveries = [];
|
|
239
|
+
let chars = 0;
|
|
240
|
+
let previousAtMs = null;
|
|
241
|
+
for (const [index, atMs] of timesMs.entries()) {
|
|
242
|
+
chars += lengths[index] ?? 0;
|
|
243
|
+
const current = deliveries.at(-1);
|
|
244
|
+
if (current && atMs - previousAtMs < coalesceMs) {
|
|
245
|
+
current.endChars = chars;
|
|
246
|
+
} else {
|
|
247
|
+
deliveries.push({ atMs, endChars: chars });
|
|
248
|
+
}
|
|
249
|
+
previousAtMs = atMs;
|
|
250
|
+
}
|
|
251
|
+
return deliveries;
|
|
252
|
+
}
|
|
253
|
+
|
|
136
254
|
export async function respond(fmBin, model, prompt, options = {}) {
|
|
137
255
|
const features = options.capabilities?.features;
|
|
138
256
|
const streamControl = features ? features.streaming : true;
|
|
@@ -155,6 +273,7 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
155
273
|
const output = stripAnsi(result.stdout).trim();
|
|
156
274
|
const errorText = stripAnsi(result.stderr).trim();
|
|
157
275
|
const failed = result.code !== 0 || result.timedOut;
|
|
276
|
+
const deliveries = streamed ? groupDeliveries(result.stdoutChunkTimesMs, result.stdoutChunkLengths) : [];
|
|
158
277
|
|
|
159
278
|
return {
|
|
160
279
|
ok: !failed,
|
|
@@ -167,9 +286,11 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
167
286
|
timedOut: result.timedOut,
|
|
168
287
|
durationMs: result.durationMs,
|
|
169
288
|
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
289
|
+
firstChunkText: deliveries.length > 0 ? stripAnsi(result.stdout.slice(0, deliveries[0].endChars)) : null,
|
|
170
290
|
streamed,
|
|
171
291
|
stdoutChunks: result.stdoutChunks,
|
|
172
292
|
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : [],
|
|
293
|
+
deliveryTimesMs: deliveries.map((delivery) => delivery.atMs),
|
|
173
294
|
error: failed
|
|
174
295
|
? (result.timedOut
|
|
175
296
|
? `timed out after ${options.timeoutMs ?? 60_000}ms`
|
package/src/metrics.js
CHANGED
|
@@ -31,7 +31,7 @@ const DEFINITIONS = [
|
|
|
31
31
|
key: 'generationMs',
|
|
32
32
|
label: 'generation time',
|
|
33
33
|
kind: 'derived',
|
|
34
|
-
source: '
|
|
34
|
+
source: 'last streamed stdout delivery minus the first (chunks under 5 ms apart are one delivery)',
|
|
35
35
|
requires: ['streaming'],
|
|
36
36
|
reason: 'requires at least two streamed output chunks to separate prefill from decode'
|
|
37
37
|
},
|
|
@@ -39,9 +39,17 @@ const DEFINITIONS = [
|
|
|
39
39
|
key: 'tpot',
|
|
40
40
|
label: 'TPOT',
|
|
41
41
|
kind: 'derived',
|
|
42
|
-
source: '
|
|
42
|
+
source: 'generation time / output tokens that arrived after the first chunk',
|
|
43
43
|
requires: ['streaming', 'tokenCounting'],
|
|
44
|
-
reason: 'requires streaming, a token-counting fm command, and at least
|
|
44
|
+
reason: 'requires streaming, a token-counting fm command, and at least two tokens after the first chunk'
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
key: 'firstChunkTokens',
|
|
48
|
+
label: 'first-chunk tokens',
|
|
49
|
+
kind: 'measured',
|
|
50
|
+
source: 'fm count-tokens on the first streamed stdout delivery, minus the counter framing overhead',
|
|
51
|
+
requires: ['streaming', 'tokenCounting'],
|
|
52
|
+
reason: 'requires streaming plus a token-counting fm command'
|
|
45
53
|
},
|
|
46
54
|
{
|
|
47
55
|
key: 'promptTokens',
|
|
@@ -55,7 +63,7 @@ const DEFINITIONS = [
|
|
|
55
63
|
key: 'outputTokens',
|
|
56
64
|
label: 'output tokens',
|
|
57
65
|
kind: 'measured',
|
|
58
|
-
source: 'fm count-tokens on the captured output',
|
|
66
|
+
source: 'fm count-tokens on the captured output, minus the calibrated counter framing overhead',
|
|
59
67
|
requires: ['tokenCounting'],
|
|
60
68
|
reason: 'this fm build exposes no token-counting command'
|
|
61
69
|
},
|
|
@@ -71,9 +79,9 @@ const DEFINITIONS = [
|
|
|
71
79
|
key: 'decodeTokensPerSecond',
|
|
72
80
|
label: 'decode tokens/s',
|
|
73
81
|
kind: 'derived',
|
|
74
|
-
source: '
|
|
82
|
+
source: 'output tokens after the first chunk / generation seconds',
|
|
75
83
|
requires: ['streaming', 'tokenCounting'],
|
|
76
|
-
reason: 'requires streaming, a token-counting fm command, and at least
|
|
84
|
+
reason: 'requires streaming, a token-counting fm command, and at least two tokens after the first chunk'
|
|
77
85
|
},
|
|
78
86
|
{
|
|
79
87
|
key: 'prefillTokensPerSecond',
|
|
@@ -102,7 +110,7 @@ const DEFINITIONS = [
|
|
|
102
110
|
key: 'chunkGaps',
|
|
103
111
|
label: 'chunk gaps and second-chunk delay',
|
|
104
112
|
kind: 'proxy',
|
|
105
|
-
source: 'gaps between consecutive streamed stdout chunks',
|
|
113
|
+
source: 'gaps between consecutive streamed stdout deliveries (chunks under 5 ms apart are one delivery)',
|
|
106
114
|
requires: ['streaming'],
|
|
107
115
|
reason: 'requires an fm build whose output can be streamed'
|
|
108
116
|
},
|
package/src/process.js
CHANGED
|
@@ -48,6 +48,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
48
48
|
let stdoutChunks = 0;
|
|
49
49
|
let stderrChunks = 0;
|
|
50
50
|
const stdoutChunkTimesMs = [];
|
|
51
|
+
const stdoutChunkLengths = [];
|
|
51
52
|
let firstStdoutMs = null;
|
|
52
53
|
let firstStderrMs = null;
|
|
53
54
|
let timedOut = false;
|
|
@@ -71,7 +72,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
71
72
|
if (firstStdoutMs == null && chunk.length > 0) {
|
|
72
73
|
firstStdoutMs = chunkAtMs;
|
|
73
74
|
}
|
|
74
|
-
if (chunk.length > 0)
|
|
75
|
+
if (chunk.length > 0) {
|
|
76
|
+
stdoutChunkTimesMs.push(chunkAtMs);
|
|
77
|
+
stdoutChunkLengths.push(chunk.length);
|
|
78
|
+
}
|
|
75
79
|
stdout += chunk;
|
|
76
80
|
});
|
|
77
81
|
child.stderr.on('data', (chunk) => {
|
|
@@ -101,6 +105,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
101
105
|
stdoutChunks,
|
|
102
106
|
stderrChunks,
|
|
103
107
|
stdoutChunkTimesMs,
|
|
108
|
+
stdoutChunkLengths,
|
|
104
109
|
firstStdoutMs,
|
|
105
110
|
firstStderrMs,
|
|
106
111
|
error,
|
|
@@ -120,6 +125,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
120
125
|
stdoutChunks,
|
|
121
126
|
stderrChunks,
|
|
122
127
|
stdoutChunkTimesMs,
|
|
128
|
+
stdoutChunkLengths,
|
|
123
129
|
firstStdoutMs,
|
|
124
130
|
firstStderrMs,
|
|
125
131
|
error: null,
|
package/src/report.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import fs from 'node:fs/promises';
|
|
2
2
|
import path from 'node:path';
|
|
3
|
+
import { renderHtmlReport } from './export.js';
|
|
3
4
|
|
|
4
5
|
export function toCsv(rows) {
|
|
5
6
|
if (rows.length === 0) return '';
|
|
@@ -37,7 +38,8 @@ export function flattenResults(results) {
|
|
|
37
38
|
chunk_gap_max_ms: round(result.chunkGapMaxMs),
|
|
38
39
|
output_hash: result.outputHash || '',
|
|
39
40
|
good: result.good == null ? '' : result.good,
|
|
40
|
-
error: result.error || ''
|
|
41
|
+
error: result.error || '',
|
|
42
|
+
first_chunk_tokens: result.firstChunkTokens ?? ''
|
|
41
43
|
}));
|
|
42
44
|
}
|
|
43
45
|
|
|
@@ -48,7 +50,6 @@ export async function writeReport(filePath, payload, format) {
|
|
|
48
50
|
if (format === 'csv') {
|
|
49
51
|
content = toCsv(flattenResults(payload.results));
|
|
50
52
|
} else if (format === 'html') {
|
|
51
|
-
const { renderHtmlReport } = await import('./export.js');
|
|
52
53
|
content = renderHtmlReport(payload);
|
|
53
54
|
} else {
|
|
54
55
|
content = `${JSON.stringify(payload, null, 2)}\n`;
|
package/src/schema.js
CHANGED
|
@@ -62,11 +62,30 @@ export function environmentFingerprint(report) {
|
|
|
62
62
|
macOSBuildVersion: build,
|
|
63
63
|
fmBin: env.fmBin ?? null,
|
|
64
64
|
fmHelpDigest: env.fmHelpDigest ?? null,
|
|
65
|
+
modelIdentities: modelIdentities(report),
|
|
65
66
|
thermal: env.thermal ?? null,
|
|
66
67
|
power: env.power ?? null
|
|
67
68
|
};
|
|
68
69
|
}
|
|
69
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Model name → identity reported by `fm models` (for example
|
|
73
|
+
* `{ system: 'AFM 3 Core Advanced' }`). Empty for reports from builds or
|
|
74
|
+
* fm-bench versions that do not expose it.
|
|
75
|
+
* @param {Record<string, unknown>} report
|
|
76
|
+
* @returns {Record<string, string>}
|
|
77
|
+
*/
|
|
78
|
+
export function modelIdentities(report) {
|
|
79
|
+
const identities = {};
|
|
80
|
+
const models = Array.isArray(report.models) ? report.models : [];
|
|
81
|
+
for (const model of models) {
|
|
82
|
+
if (model && typeof model.name === 'string' && typeof model.identity === 'string' && model.identity) {
|
|
83
|
+
identities[model.name] = model.identity;
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return identities;
|
|
87
|
+
}
|
|
88
|
+
|
|
70
89
|
/**
|
|
71
90
|
* @param {Record<string, unknown>} before
|
|
72
91
|
* @param {Record<string, unknown>} after
|
|
@@ -99,6 +118,12 @@ export function compareCompatibility(before, after) {
|
|
|
99
118
|
if (bFp.macOSBuildVersion && aFp.macOSBuildVersion && bFp.macOSBuildVersion !== aFp.macOSBuildVersion) {
|
|
100
119
|
warnings.push(`macOS build differs (${bFp.macOSBuildVersion} vs ${aFp.macOSBuildVersion})`);
|
|
101
120
|
}
|
|
121
|
+
for (const [name, identity] of Object.entries(bFp.modelIdentities)) {
|
|
122
|
+
const other = aFp.modelIdentities[name];
|
|
123
|
+
if (other && other !== identity) {
|
|
124
|
+
warnings.push(`model ${name} differs (${identity} vs ${other})`);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
102
127
|
|
|
103
128
|
return {
|
|
104
129
|
compatible: errors.length === 0,
|
package/src/stats.js
CHANGED
|
@@ -84,6 +84,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
84
84
|
model: status.name,
|
|
85
85
|
concurrency,
|
|
86
86
|
description: status.description,
|
|
87
|
+
identity: status.identity || '',
|
|
87
88
|
available: status.available,
|
|
88
89
|
unsupported: Boolean(status.unsupported),
|
|
89
90
|
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
@@ -99,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
99
100
|
model: result.model,
|
|
100
101
|
concurrency: result.concurrency,
|
|
101
102
|
description: '',
|
|
103
|
+
identity: '',
|
|
102
104
|
available: true,
|
|
103
105
|
unsupported: false,
|
|
104
106
|
skippedReason: '',
|
|
@@ -119,12 +121,19 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
119
121
|
const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
|
|
120
122
|
const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
|
|
121
123
|
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
124
|
+
const firstChunkTokens = summarizeNumbers(successes.map((result) => result.firstChunkTokens).filter((value) => value != null));
|
|
122
125
|
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
|
|
123
126
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
124
127
|
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
125
128
|
const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
|
|
126
129
|
const secondChunk = summarizeNumbers(successes.map((result) => result.secondChunkMs).filter((value) => value != null));
|
|
127
130
|
const chunkGap = summarizeNumbers(successes.flatMap((result) => result.chunkGapsMs || []));
|
|
131
|
+
const decodeRuns = successes.filter((result) => result.decodeTokens != null && result.generationMs > 0);
|
|
132
|
+
const decodeMs = decodeRuns.reduce((sum, result) => sum + result.generationMs, 0);
|
|
133
|
+
// Token-weighted: a 3-token tail cannot outweigh a 200-token generation.
|
|
134
|
+
const decodeThroughput = decodeMs > 0
|
|
135
|
+
? decodeRuns.reduce((sum, result) => sum + result.decodeTokens, 0) / (decodeMs / 1000)
|
|
136
|
+
: null;
|
|
128
137
|
const windowMs = modelWindowMs(successes);
|
|
129
138
|
const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
|
|
130
139
|
const goodputRps = goodMeasured.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
|
|
@@ -137,6 +146,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
137
146
|
model: entry.model,
|
|
138
147
|
concurrency: entry.concurrency,
|
|
139
148
|
description: entry.description,
|
|
149
|
+
identity: entry.identity,
|
|
140
150
|
available: entry.available,
|
|
141
151
|
unsupported: entry.unsupported,
|
|
142
152
|
skippedReason: entry.skippedReason,
|
|
@@ -147,6 +157,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
147
157
|
failures: failures.length,
|
|
148
158
|
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
149
159
|
goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
|
|
160
|
+
stabilityCv: summarizeStability(successes),
|
|
150
161
|
rps,
|
|
151
162
|
goodputRps,
|
|
152
163
|
outputTokenThroughput,
|
|
@@ -158,9 +169,11 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
158
169
|
tpot,
|
|
159
170
|
promptTokens,
|
|
160
171
|
outputTokens,
|
|
172
|
+
firstChunkTokens,
|
|
161
173
|
charsPerSecond,
|
|
162
174
|
tokensPerSecond,
|
|
163
175
|
decodeTokensPerSecond,
|
|
176
|
+
decodeThroughput,
|
|
164
177
|
prefillTokensPerSecond,
|
|
165
178
|
secondChunk,
|
|
166
179
|
chunkGap
|
|
@@ -222,6 +235,29 @@ function modelWindowMs(results) {
|
|
|
222
235
|
return Math.max(...ends) - Math.min(...starts);
|
|
223
236
|
}
|
|
224
237
|
|
|
238
|
+
/**
|
|
239
|
+
* Run-to-run latency stability: the E2E coefficient of variation of each
|
|
240
|
+
* prompt across its repeated runs, averaged over prompts.
|
|
241
|
+
*
|
|
242
|
+
* The pooled `latency.cv` mixes every prompt, so a suite with a one-line answer and a
|
|
243
|
+
* long generation reports high "variation" even when every prompt is
|
|
244
|
+
* perfectly steady. Grouping by prompt isolates repeat noise. `null` until at
|
|
245
|
+
* least one prompt has two successful runs.
|
|
246
|
+
*/
|
|
247
|
+
function summarizeStability(results) {
|
|
248
|
+
const byPrompt = new Map();
|
|
249
|
+
for (const result of results) {
|
|
250
|
+
if (!Number.isFinite(result.durationMs)) continue;
|
|
251
|
+
if (!byPrompt.has(result.promptId)) byPrompt.set(result.promptId, []);
|
|
252
|
+
byPrompt.get(result.promptId).push(result.durationMs);
|
|
253
|
+
}
|
|
254
|
+
const cvs = [...byPrompt.values()]
|
|
255
|
+
.map((durations) => summarizeNumbers(durations).cv)
|
|
256
|
+
.filter((cv) => cv != null);
|
|
257
|
+
if (cvs.length === 0) return null;
|
|
258
|
+
return cvs.reduce((sum, cv) => sum + cv, 0) / cvs.length;
|
|
259
|
+
}
|
|
260
|
+
|
|
225
261
|
function summarizeRepeatability(results) {
|
|
226
262
|
const byPrompt = new Map();
|
|
227
263
|
for (const result of results) {
|
package/src/table.js
CHANGED
|
@@ -55,6 +55,12 @@ export function renderBenchmarkReport(payload, options = {}) {
|
|
|
55
55
|
const note = payload.options?.note ?? null;
|
|
56
56
|
lines.push(...wrapText(title, width));
|
|
57
57
|
lines.push(...wrapText(meta, width));
|
|
58
|
+
const identities = (payload.models ?? [])
|
|
59
|
+
.filter((model) => model.available && model.identity)
|
|
60
|
+
.map((model) => `${model.name} = ${model.identity}`);
|
|
61
|
+
if (identities.length > 0) {
|
|
62
|
+
lines.push(...wrapText(`models: ${identities.join(', ')}`, width));
|
|
63
|
+
}
|
|
58
64
|
if (tags.length > 0) {
|
|
59
65
|
for (const line of wrapText(`tags: ${tags.join(', ')}`, width)) lines.push(line);
|
|
60
66
|
}
|
|
@@ -88,21 +94,22 @@ export function legendEntries() {
|
|
|
88
94
|
entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.', 'measured'),
|
|
89
95
|
entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Runs whose SLO metric is unmeasurable count as not good.', 'derived'),
|
|
90
96
|
entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.', 'derived'),
|
|
91
|
-
entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity,
|
|
97
|
+
entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity, and the first chunk can already hold many tokens (see 1ST CHUNK). Lower is better.', 'proxy'),
|
|
92
98
|
entry('summary', 'TTFT P95', '95th percentile time to the first streamed stdout chunk.', 'Lower is better.', 'proxy'),
|
|
93
99
|
entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better. This is a direct wall-clock measurement.', 'measured'),
|
|
94
100
|
entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.', 'measured'),
|
|
95
|
-
entry('summary', 'TPOT', 'Time per output token after the first
|
|
96
|
-
entry('summary', 'TPOT P95', '95th percentile time per output token after first
|
|
101
|
+
entry('summary', 'TPOT', 'Time per output token for tokens that arrived after the first streamed chunk, p50.', 'Derived from fm token counts and stream timings. Lower is better.', 'derived'),
|
|
102
|
+
entry('summary', 'TPOT P95', '95th percentile time per output token after the first streamed chunk.', 'Lower is better.', 'derived'),
|
|
97
103
|
entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Derived from fm token counts. Higher is better.', 'derived'),
|
|
98
104
|
entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better.', 'derived'),
|
|
99
105
|
entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Measured from process timings. Higher is better.', 'measured'),
|
|
100
|
-
entry('summary', 'CV', '
|
|
106
|
+
entry('summary', 'CV', 'Run-to-run E2E variation: coefficient of variation of each prompt across its repeated runs, averaged over prompts.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%. Blank until a prompt has two successful runs (use --runs 2 or more).', 'derived'),
|
|
101
107
|
entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', '', 'measured'),
|
|
102
108
|
entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
|
|
103
|
-
entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
|
|
109
|
+
entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens, minus its constant framing token.', 'Blank when the fm build cannot count tokens.', 'measured'),
|
|
110
|
+
entry('detail', '1ST CHUNK', 'Average output tokens carried by the first streamed stdout chunk.', 'Shows what TTFT really measures: fm streams coarse deltas (about 20 tokens in the first chunk on macOS 27.2). Wide layout only.', 'measured'),
|
|
104
111
|
entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Proxy: prefill is inferred from time to first chunk, not observed directly. Higher is better.', 'proxy'),
|
|
105
|
-
entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first
|
|
112
|
+
entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens that arrived after the first streamed delivery, summed over runs, divided by the summed generation time.', 'Derived and token-weighted, so long generations dominate short tails; requires streaming and token counts. Higher is better.', 'derived'),
|
|
106
113
|
entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.', 'proxy'),
|
|
107
114
|
entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Proxy for decode smoothness at chunk granularity. Lower is smoother.', 'proxy'),
|
|
108
115
|
entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.', 'measured'),
|
|
@@ -110,7 +117,8 @@ export function legendEntries() {
|
|
|
110
117
|
entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.', 'derived'),
|
|
111
118
|
entry('detail', 'ATTEMPTS', 'Total fm invocations including retries.', 'Shown in reports; a retried run is still one measured result.', 'measured'),
|
|
112
119
|
entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', '', 'measured'),
|
|
113
|
-
entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', '', 'measured'),
|
|
120
|
+
entry('models', 'AVAILABLE', 'Whether fm models (fm available on older builds) reports the model as usable on this machine right now.', '', 'measured'),
|
|
121
|
+
entry('models', 'IDENTITY', 'Model identity reported by fm models, for example AFM 3 Core Advanced.', 'Recorded in reports; compare warns when it changes between two runs.', 'measured'),
|
|
114
122
|
entry('models', 'QUOTA', 'Quota information when the fm build exposes a quota command.', 'Column is omitted entirely when the installed fm has no quota command.', 'measured'),
|
|
115
123
|
entry('metrics', 'SOURCE', 'How a metric is obtained: measured, proxy, derived, or controlled.', 'measured = observed directly; proxy = observed at coarser granularity; derived = computed from measured values.', 'measured'),
|
|
116
124
|
entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.', 'derived'),
|
|
@@ -220,7 +228,7 @@ export function renderSummaryTable(summary, options = {}) {
|
|
|
220
228
|
cell(formatNumber(item.tokensPerSecond.avg), tones.userTps),
|
|
221
229
|
cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
|
|
222
230
|
cell(formatNumber(item.rps), tones.rps),
|
|
223
|
-
cell(formatPercent(item
|
|
231
|
+
cell(formatPercent(stabilityCv(item)), cvTone(stabilityCv(item))),
|
|
224
232
|
cell(item.available ? '' : cleanReason(item.skippedReason), item.available ? null : 'yellow')
|
|
225
233
|
];
|
|
226
234
|
|
|
@@ -274,20 +282,21 @@ export function renderDetailTable(summary, options = {}) {
|
|
|
274
282
|
cell(formatNumber(item.promptTokens.avg, 0)),
|
|
275
283
|
cell(formatNumber(item.outputTokens.avg, 0)),
|
|
276
284
|
cell(formatNumber(item.prefillTokensPerSecond?.avg), tones.prefillTps),
|
|
277
|
-
cell(formatNumber(item
|
|
285
|
+
cell(formatNumber(decodeRate(item)), tones.decodeTps),
|
|
278
286
|
cell(formatMs(item.secondChunk?.p50), tones.secondChunk),
|
|
279
287
|
cell(formatMs(item.chunkGap?.p95), tones.chunkGapP95),
|
|
280
288
|
cell(formatMs(item.latency.p99), tones.e2eP99),
|
|
281
|
-
cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item
|
|
289
|
+
cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(stabilityCv(item))),
|
|
282
290
|
cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
|
|
283
|
-
cell(item.description || '-', 'muted')
|
|
291
|
+
cell(item.description || '-', 'muted'),
|
|
292
|
+
cell(formatNumber(item.firstChunkTokens?.avg, 0))
|
|
284
293
|
];
|
|
285
294
|
|
|
286
295
|
if (mode === 'medium') {
|
|
287
296
|
return [base[0], base[1], base[2], base[3], base[4], base[5], base[7], base[8], base[10]];
|
|
288
297
|
}
|
|
289
298
|
|
|
290
|
-
return base;
|
|
299
|
+
return [base[0], base[1], base[2], base[3], base[12], ...base.slice(4, 12)];
|
|
291
300
|
});
|
|
292
301
|
|
|
293
302
|
const headers = mode === 'medium'
|
|
@@ -297,6 +306,7 @@ export function renderDetailTable(summary, options = {}) {
|
|
|
297
306
|
'model',
|
|
298
307
|
'in tok avg',
|
|
299
308
|
'out tok avg',
|
|
309
|
+
'1st chunk',
|
|
300
310
|
'prefill tok/s',
|
|
301
311
|
'decode tok/s',
|
|
302
312
|
'2nd chunk',
|
|
@@ -325,7 +335,7 @@ export function renderCompactSummary(summary, options = {}) {
|
|
|
325
335
|
const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
|
|
326
336
|
const tones = metricTones(item, ranks, options.slo);
|
|
327
337
|
lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
|
|
328
|
-
lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item
|
|
338
|
+
lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(stabilityCv(item))}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(stabilityCv(item)), percentTone(item.goodputRate, 0.8, 1)), options));
|
|
329
339
|
lines.push(toneText(truncate(` prefill ${formatNumber(item.prefillTokensPerSecond?.avg)} tok/s | TPOT ${formatMs(item.tpot.p50)} | chunk p95 ${formatMs(item.chunkGap?.p95)}`, width), worstTone(tones.prefillTps, tones.tpot, tones.chunkGapP95), options));
|
|
330
340
|
lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | repeat ${formatPercent(item.repeatability)}`, width), percentTone(item.repeatability, 0.5, 0.9), options));
|
|
331
341
|
} else {
|
|
@@ -344,29 +354,37 @@ export function renderModelsTable(models, options = {}) {
|
|
|
344
354
|
if (compact) {
|
|
345
355
|
return models.map((model) => {
|
|
346
356
|
const status = model.available ? 'yes' : 'no';
|
|
347
|
-
|
|
357
|
+
const detail = model.available ? (model.identity || model.description) : model.reason;
|
|
358
|
+
return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(detail || model.description || '-')}`;
|
|
348
359
|
}).join('\n');
|
|
349
360
|
}
|
|
350
361
|
|
|
351
362
|
const quotaSupported = options.capabilities
|
|
352
363
|
? Boolean(options.capabilities.features?.quota)
|
|
353
364
|
: models.some((model) => model.quotaSupported === true || (model.quota != null && model.quota !== ''));
|
|
365
|
+
const hasIdentity = models.some((model) => model.identity);
|
|
366
|
+
const base = (model) => {
|
|
367
|
+
const cells = [
|
|
368
|
+
cell(model.name),
|
|
369
|
+
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow')
|
|
370
|
+
];
|
|
371
|
+
if (hasIdentity) cells.push(model.identity || '-');
|
|
372
|
+
cells.push(model.description || '-');
|
|
373
|
+
return cells;
|
|
374
|
+
};
|
|
375
|
+
const leading = hasIdentity ? ['model', 'available', 'identity', 'description'] : ['model', 'available', 'description'];
|
|
354
376
|
|
|
355
377
|
if (!quotaSupported) {
|
|
356
|
-
return renderTable([
|
|
357
|
-
|
|
358
|
-
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
|
|
359
|
-
model.description || '-',
|
|
378
|
+
return renderTable([...leading, 'notes'], models.map((model) => [
|
|
379
|
+
...base(model),
|
|
360
380
|
cleanReason(model.reason || '-')
|
|
361
|
-
]), { ...options, wrapColumns: ['description', 'notes'] });
|
|
381
|
+
]), { ...options, wrapColumns: ['identity', 'description', 'notes'] });
|
|
362
382
|
}
|
|
363
383
|
|
|
364
|
-
return renderTable([
|
|
365
|
-
|
|
366
|
-
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
|
|
367
|
-
model.description || '-',
|
|
384
|
+
return renderTable([...leading, 'quota'], models.map((model) => [
|
|
385
|
+
...base(model),
|
|
368
386
|
cleanReason(model.quota || model.reason || '-')
|
|
369
|
-
]), { ...options, wrapColumns: ['description', 'quota'] });
|
|
387
|
+
]), { ...options, wrapColumns: ['identity', 'description', 'quota'] });
|
|
370
388
|
}
|
|
371
389
|
|
|
372
390
|
/**
|
|
@@ -390,6 +408,22 @@ export function renderModelsReport(models, options = {}) {
|
|
|
390
408
|
return lines.join('\n');
|
|
391
409
|
}
|
|
392
410
|
|
|
411
|
+
/**
|
|
412
|
+
* Run-to-run CV for a summary row. Reports written before `stabilityCv`
|
|
413
|
+
* existed only have the suite-wide `latency.cv`.
|
|
414
|
+
*/
|
|
415
|
+
export function stabilityCv(item) {
|
|
416
|
+
return Object.hasOwn(item, 'stabilityCv') ? item.stabilityCv : item.latency?.cv ?? null;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
/**
|
|
420
|
+
* Token-weighted decode rate for a summary row. Reports written before
|
|
421
|
+
* `decodeThroughput` existed only have the mean of per-run rates.
|
|
422
|
+
*/
|
|
423
|
+
export function decodeRate(item) {
|
|
424
|
+
return Object.hasOwn(item, 'decodeThroughput') ? item.decodeThroughput : item.decodeTokensPerSecond?.avg ?? null;
|
|
425
|
+
}
|
|
426
|
+
|
|
393
427
|
function formatRangeMs(low, high) {
|
|
394
428
|
if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
|
|
395
429
|
return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
|
|
@@ -739,7 +773,7 @@ function rankSummary(summary) {
|
|
|
739
773
|
totalTps: collectMetric(summary, (item) => item.totalTokenThroughput),
|
|
740
774
|
rps: collectMetric(summary, (item) => item.rps),
|
|
741
775
|
goodputRps: collectMetric(summary, (item) => item.goodputRps),
|
|
742
|
-
decodeTps: collectMetric(summary,
|
|
776
|
+
decodeTps: collectMetric(summary, decodeRate),
|
|
743
777
|
prefillTps: collectMetric(summary, (item) => item.prefillTokensPerSecond?.avg)
|
|
744
778
|
};
|
|
745
779
|
}
|
|
@@ -788,7 +822,7 @@ function metricTones(item, ranks, slo = {}) {
|
|
|
788
822
|
totalTps: rankTone(item.totalTokenThroughput, ranks.totalTps),
|
|
789
823
|
rps: rankTone(item.rps, ranks.rps),
|
|
790
824
|
goodputRps: goodputRpsTone(item, ranks.goodputRps),
|
|
791
|
-
decodeTps: rankTone(item
|
|
825
|
+
decodeTps: rankTone(decodeRate(item), ranks.decodeTps),
|
|
792
826
|
prefillTps: rankTone(item.prefillTokensPerSecond?.avg, ranks.prefillTps)
|
|
793
827
|
};
|
|
794
828
|
}
|