fm-bench 0.7.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -24
- package/docs/compatibility.md +10 -5
- package/docs/methodology.md +14 -9
- package/docs/releasing.md +33 -16
- package/docs/report-format.md +23 -14
- package/docs/supported-platforms.md +9 -4
- package/package.json +5 -5
- package/src/bench.js +73 -23
- package/src/capabilities.js +26 -9
- package/src/cli.js +80 -8
- package/src/compare.js +10 -1
- package/src/export.js +6 -4
- package/src/fm-help.js +62 -7
- package/src/fm.js +123 -2
- package/src/metrics.js +15 -7
- package/src/process.js +7 -1
- package/src/report.js +3 -2
- package/src/schema.js +25 -0
- package/src/stats.js +36 -0
- package/src/table.js +60 -26
package/src/bench.js
CHANGED
|
@@ -1,6 +1,15 @@
|
|
|
1
1
|
import crypto from 'node:crypto';
|
|
2
2
|
import { detectFmCapabilities } from './capabilities.js';
|
|
3
|
-
import {
|
|
3
|
+
import {
|
|
4
|
+
calibrateTokenCounter,
|
|
5
|
+
checkModelAvailability,
|
|
6
|
+
collectEnvironment,
|
|
7
|
+
countTokens,
|
|
8
|
+
fmBinaryFromOptions,
|
|
9
|
+
getQuotaUsage,
|
|
10
|
+
listModelStatus,
|
|
11
|
+
respond
|
|
12
|
+
} from './fm.js';
|
|
4
13
|
import { metricAvailability } from './metrics.js';
|
|
5
14
|
import { loadPrompts } from './prompts.js';
|
|
6
15
|
import { finalizeReportPayload } from './schema.js';
|
|
@@ -21,18 +30,22 @@ export async function inspectModels(options = {}) {
|
|
|
21
30
|
?? { name, description: 'Requested model not reported by this fm build' })
|
|
22
31
|
: discovered.models;
|
|
23
32
|
|
|
33
|
+
const modelList = models.length > 0 ? await listModelStatus(discovered.fmBin, { ...options, capabilities }) : null;
|
|
24
34
|
const inspected = [];
|
|
25
35
|
for (const model of models) {
|
|
26
36
|
const availability = await checkModelAvailability(discovered.fmBin, model.name, {
|
|
27
37
|
...options,
|
|
28
|
-
capabilities
|
|
29
|
-
|
|
30
|
-
const quota = await getQuotaUsage(discovered.fmBin, model.name, {
|
|
31
|
-
...options,
|
|
32
|
-
capabilities
|
|
38
|
+
capabilities,
|
|
39
|
+
modelList
|
|
33
40
|
});
|
|
41
|
+
// A model the build does not have has no quota; asking would only put fm's
|
|
42
|
+
// text for some other model where the "not supported" reason belongs.
|
|
43
|
+
const quota = availability.unsupported
|
|
44
|
+
? { supported: false, raw: '', reason: '' }
|
|
45
|
+
: await getQuotaUsage(discovered.fmBin, model.name, { ...options, capabilities });
|
|
34
46
|
inspected.push({
|
|
35
47
|
...model,
|
|
48
|
+
identity: availability.identity || '',
|
|
36
49
|
available: availability.available,
|
|
37
50
|
unsupported: Boolean(availability.unsupported),
|
|
38
51
|
reason: availability.available ? '' : (availability.reason || availability.raw || 'unavailable'),
|
|
@@ -73,7 +86,7 @@ export async function runBenchmark(options = {}) {
|
|
|
73
86
|
: inspection.models;
|
|
74
87
|
const runnableModels = modelStatuses.filter((model) => model.available);
|
|
75
88
|
if (runnableModels.length === 0) {
|
|
76
|
-
throw noRunnableModelsError(inspection.models, options);
|
|
89
|
+
throw noRunnableModelsError(inspection.models, capabilities.models, options);
|
|
77
90
|
}
|
|
78
91
|
const environment = await collectEnvironment(inspection.fmBin, { ...options, capabilities });
|
|
79
92
|
const metrics = metricAvailability(capabilities, {
|
|
@@ -84,6 +97,16 @@ export async function runBenchmark(options = {}) {
|
|
|
84
97
|
const concurrencies = normalizeConcurrencySweep(options);
|
|
85
98
|
const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
|
|
86
99
|
|
|
100
|
+
const tokenCounter = {
|
|
101
|
+
command: capabilities.features.tokenCountCommand,
|
|
102
|
+
overhead: 0,
|
|
103
|
+
calibrated: false
|
|
104
|
+
};
|
|
105
|
+
if (metrics.outputTokens.available) {
|
|
106
|
+
notify(options, { type: 'phase', phase: 'tokens', message: 'calibrating token counter' });
|
|
107
|
+
Object.assign(tokenCounter, await calibrateTokenCounter(inspection.fmBin, { ...options, capabilities }));
|
|
108
|
+
}
|
|
109
|
+
|
|
87
110
|
notify(options, {
|
|
88
111
|
type: 'tokens:start',
|
|
89
112
|
total: prompts.length,
|
|
@@ -123,6 +146,7 @@ export async function runBenchmark(options = {}) {
|
|
|
123
146
|
modelStatuses,
|
|
124
147
|
promptTokenCounts,
|
|
125
148
|
tokenCounting: metrics.outputTokens.available,
|
|
149
|
+
tokenOverhead: tokenCounter.overhead,
|
|
126
150
|
options,
|
|
127
151
|
concurrency,
|
|
128
152
|
scenarioIndex: scenarioIndex + 1,
|
|
@@ -171,6 +195,7 @@ export async function runBenchmark(options = {}) {
|
|
|
171
195
|
warnings: capabilities.warnings
|
|
172
196
|
},
|
|
173
197
|
metrics,
|
|
198
|
+
tokenCounter,
|
|
174
199
|
prompts: prompts.map((prompt) => ({
|
|
175
200
|
id: prompt.id,
|
|
176
201
|
prompt: prompt.prompt,
|
|
@@ -199,6 +224,7 @@ async function runScenario(context) {
|
|
|
199
224
|
modelStatuses,
|
|
200
225
|
promptTokenCounts,
|
|
201
226
|
tokenCounting,
|
|
227
|
+
tokenOverhead,
|
|
202
228
|
options,
|
|
203
229
|
concurrency,
|
|
204
230
|
scenarioIndex,
|
|
@@ -263,6 +289,7 @@ async function runScenario(context) {
|
|
|
263
289
|
job,
|
|
264
290
|
promptTokenCounts,
|
|
265
291
|
tokenCounting,
|
|
292
|
+
tokenOverhead,
|
|
266
293
|
options,
|
|
267
294
|
benchmarkStartedAt
|
|
268
295
|
});
|
|
@@ -289,7 +316,7 @@ async function runScenario(context) {
|
|
|
289
316
|
}
|
|
290
317
|
|
|
291
318
|
async function runSingleBenchmark(context) {
|
|
292
|
-
const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, options, benchmarkStartedAt } = context;
|
|
319
|
+
const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, tokenOverhead = 0, options, benchmarkStartedAt } = context;
|
|
293
320
|
const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
|
|
294
321
|
const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
295
322
|
let response;
|
|
@@ -309,24 +336,34 @@ async function runSingleBenchmark(context) {
|
|
|
309
336
|
|
|
310
337
|
const ok = response.ok;
|
|
311
338
|
const seconds = response.durationMs / 1000;
|
|
312
|
-
const chunks = response.stdoutChunks ?? 0;
|
|
313
339
|
|
|
314
|
-
//
|
|
315
|
-
//
|
|
316
|
-
//
|
|
340
|
+
// When one delivery carries the whole answer, the streamed portion is not
|
|
341
|
+
// separable: report generation time and TPOT as unavailable rather than as
|
|
342
|
+
// a near-zero decode phase. Otherwise the generation window runs from the
|
|
343
|
+
// first to the last delivery, which excludes process teardown.
|
|
317
344
|
const firstTokenMs = ok ? response.firstOutputMs : null;
|
|
318
|
-
const
|
|
319
|
-
|
|
345
|
+
const deliveryTimes = response.deliveryTimesMs ?? [];
|
|
346
|
+
const generationMs = ok && firstTokenMs != null && deliveryTimes.length > 1
|
|
347
|
+
? Math.max(0, deliveryTimes[deliveryTimes.length - 1] - firstTokenMs)
|
|
320
348
|
: null;
|
|
321
349
|
|
|
322
|
-
const
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
350
|
+
const countOptions = { ...options, capabilities };
|
|
351
|
+
const countedOutputTokens = ok && tokenCounting
|
|
352
|
+
? await countOutputTokens(fmBin, response.output, tokenOverhead, countOptions)
|
|
353
|
+
: null;
|
|
354
|
+
// fm streams coarse deltas: the first delivery carries about 20 tokens on
|
|
355
|
+
// macOS 27.2. Decode cadence is therefore measured over the tokens that
|
|
356
|
+
// arrived after it, not over "all tokens but one".
|
|
357
|
+
const firstChunkTokens = generationMs != null && countedOutputTokens != null
|
|
358
|
+
? await countOutputTokens(fmBin, response.firstChunkText ?? '', tokenOverhead, countOptions)
|
|
359
|
+
: null;
|
|
326
360
|
// Two decode tokens is the minimum for an inter-token interval that is not
|
|
327
361
|
// simply the inverse of a single chunk gap.
|
|
328
|
-
const
|
|
329
|
-
? countedOutputTokens -
|
|
362
|
+
const tokensAfterFirstChunk = firstChunkTokens != null
|
|
363
|
+
? countedOutputTokens - Math.min(firstChunkTokens, countedOutputTokens)
|
|
364
|
+
: null;
|
|
365
|
+
const decodeTokenCount = tokensAfterFirstChunk != null && tokensAfterFirstChunk >= 2
|
|
366
|
+
? tokensAfterFirstChunk
|
|
330
367
|
: null;
|
|
331
368
|
const hasDecodeCadence = generationMs != null && generationMs > 0 && decodeTokenCount != null;
|
|
332
369
|
const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
|
|
@@ -338,7 +375,7 @@ async function runSingleBenchmark(context) {
|
|
|
338
375
|
const prefillTokensPerSecond = promptTokens != null && firstTokenMs != null && firstTokenMs > 0
|
|
339
376
|
? promptTokens / (firstTokenMs / 1000)
|
|
340
377
|
: null;
|
|
341
|
-
const chunkGapsMs = ok ? chunkGaps(
|
|
378
|
+
const chunkGapsMs = ok ? chunkGaps(deliveryTimes) : [];
|
|
342
379
|
const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
|
|
343
380
|
|
|
344
381
|
return {
|
|
@@ -354,6 +391,8 @@ async function runSingleBenchmark(context) {
|
|
|
354
391
|
tpotMs,
|
|
355
392
|
promptTokens,
|
|
356
393
|
outputTokens: countedOutputTokens,
|
|
394
|
+
firstChunkTokens,
|
|
395
|
+
decodeTokens: hasDecodeCadence ? decodeTokenCount : null,
|
|
357
396
|
chars,
|
|
358
397
|
words,
|
|
359
398
|
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
@@ -379,6 +418,17 @@ async function runSingleBenchmark(context) {
|
|
|
379
418
|
};
|
|
380
419
|
}
|
|
381
420
|
|
|
421
|
+
/**
|
|
422
|
+
* Output token count with the token counter's framing overhead removed.
|
|
423
|
+
* Empty text is zero tokens; `fm count-tokens` rejects an empty prompt.
|
|
424
|
+
* @returns {Promise<number|null>}
|
|
425
|
+
*/
|
|
426
|
+
async function countOutputTokens(fmBin, text, overhead, options) {
|
|
427
|
+
if (!String(text).trim()) return 0;
|
|
428
|
+
const counted = await countTokens(fmBin, text, options);
|
|
429
|
+
return counted.ok ? Math.max(0, counted.count - overhead) : null;
|
|
430
|
+
}
|
|
431
|
+
|
|
382
432
|
async function runLimited(items, concurrency, worker, options = {}) {
|
|
383
433
|
let nextIndex = 0;
|
|
384
434
|
// fail-fast stops admitting new work; calls already in flight are allowed to
|
|
@@ -404,9 +454,9 @@ async function runLimited(items, concurrency, worker, options = {}) {
|
|
|
404
454
|
|
|
405
455
|
// A benchmark with nothing to run is a configuration error, not an empty
|
|
406
456
|
// report: say which models were asked for and which ones the build supports.
|
|
407
|
-
function noRunnableModelsError(models, options) {
|
|
457
|
+
function noRunnableModelsError(models, discoveredModels, options) {
|
|
408
458
|
const requested = normalizeModelSelection(options.models);
|
|
409
|
-
const supported =
|
|
459
|
+
const supported = discoveredModels.map((model) => model.name);
|
|
410
460
|
const lines = ['No benchmark was run: none of the requested models are usable right now.'];
|
|
411
461
|
if (requested.length > 0) lines.push(` requested: ${requested.join(', ')}`);
|
|
412
462
|
if (supported.length > 0) lines.push(` models reported by this fm build: ${supported.join(', ')}`);
|
package/src/capabilities.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
import crypto from 'node:crypto';
|
|
11
11
|
import { stripAnsi } from './ansi.js';
|
|
12
|
-
import { parseAvailabilityList, parseModelsFromHelp } from './fm-help.js';
|
|
12
|
+
import { parseAvailabilityList, parseModelList, parseModelsFromHelp } from './fm-help.js';
|
|
13
13
|
import { runProcess } from './process.js';
|
|
14
14
|
|
|
15
15
|
const SECTION_HEADER = /^\s*[A-Z][A-Z0-9 /-]+\s*$/;
|
|
@@ -72,12 +72,26 @@ export function resolveTokenCountCommand(commands = []) {
|
|
|
72
72
|
return null;
|
|
73
73
|
}
|
|
74
74
|
|
|
75
|
+
/**
|
|
76
|
+
* Pick the subcommand this `fm` build uses to report model availability.
|
|
77
|
+
* macOS 27.2 renamed `available` to `models`; the old name still works there
|
|
78
|
+
* but prints a deprecation warning, so prefer the new one.
|
|
79
|
+
* @returns {'models'|'available'|null}
|
|
80
|
+
*/
|
|
81
|
+
export function resolveModelListCommand(commands = []) {
|
|
82
|
+
if (commands.includes('models')) return 'models';
|
|
83
|
+
if (commands.includes('available')) return 'available';
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
|
|
75
87
|
function buildFeatures(helpText, respondHelpText, commands) {
|
|
76
88
|
const tokenCountCommand = resolveTokenCountCommand(commands);
|
|
77
89
|
const respondHelp = respondHelpText || '';
|
|
78
90
|
return {
|
|
79
91
|
tokenCounting: tokenCountCommand != null,
|
|
80
92
|
tokenCountCommand,
|
|
93
|
+
modelListCommand: resolveModelListCommand(commands),
|
|
94
|
+
license: commands.includes('license'),
|
|
81
95
|
quota: commands.includes('quota-usage'),
|
|
82
96
|
streaming: hasFlagInHelp(respondHelp, '--no-stream') || hasFlagInHelp(respondHelp, '--stream'),
|
|
83
97
|
modelSelection: hasFlagInHelp(respondHelp, '--model'),
|
|
@@ -153,8 +167,8 @@ export async function detectFmCapabilities(fmBin, options = {}) {
|
|
|
153
167
|
const features = buildFeatures(cleanHelp, respondHelpText, commands);
|
|
154
168
|
let models = parseModelsFromHelp(cleanHelp);
|
|
155
169
|
|
|
156
|
-
if (models.length === 0 &&
|
|
157
|
-
models = await
|
|
170
|
+
if (models.length === 0 && features.modelListCommand) {
|
|
171
|
+
models = await discoverModelsFromList(fmBin, features.modelListCommand, { ...options, env });
|
|
158
172
|
}
|
|
159
173
|
|
|
160
174
|
return {
|
|
@@ -170,17 +184,20 @@ export async function detectFmCapabilities(fmBin, options = {}) {
|
|
|
170
184
|
}
|
|
171
185
|
|
|
172
186
|
/**
|
|
173
|
-
* Fallback discovery: ask `fm
|
|
174
|
-
* model names it reports. Used when `fm --help` has
|
|
187
|
+
* Fallback discovery: ask `fm models` (or legacy `fm available`) without a
|
|
188
|
+
* model filter and read the model names it reports. Used when `fm --help` has
|
|
189
|
+
* no MODELS section.
|
|
175
190
|
*/
|
|
176
|
-
async function
|
|
177
|
-
const result = await runProcess(fmBin, [
|
|
191
|
+
async function discoverModelsFromList(fmBin, command, options = {}) {
|
|
192
|
+
const result = await runProcess(fmBin, [command], {
|
|
178
193
|
timeoutMs: options.timeoutMs ?? 15_000,
|
|
179
194
|
env: options.env ?? process.env
|
|
180
195
|
});
|
|
181
196
|
if (result.error) return [];
|
|
182
|
-
|
|
183
|
-
|
|
197
|
+
const output = `${result.stdout}${result.stderr}`;
|
|
198
|
+
const listed = parseModelList(output);
|
|
199
|
+
const models = listed.length > 0 ? listed : parseAvailabilityList(output);
|
|
200
|
+
return models.map((model) => ({ name: model.name, description: '' }));
|
|
184
201
|
}
|
|
185
202
|
|
|
186
203
|
function escapeRegExp(value) {
|
package/src/cli.js
CHANGED
|
@@ -3,6 +3,7 @@ import { createRequire } from 'node:module';
|
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { diffReports, renderCompareReport } from './compare.js';
|
|
5
5
|
import { renderHtmlReport } from './export.js';
|
|
6
|
+
import { getLicenseStatus } from './fm.js';
|
|
6
7
|
import { formatCapabilitySummary } from './metrics.js';
|
|
7
8
|
import { validateReport } from './schema.js';
|
|
8
9
|
import { loadHistory, renderHistoryReport } from './history.js';
|
|
@@ -30,6 +31,53 @@ function operationalError(message) {
|
|
|
30
31
|
return error;
|
|
31
32
|
}
|
|
32
33
|
|
|
34
|
+
// One unmeasured call per model absorbs the cold model load, which otherwise
|
|
35
|
+
// dominates the first measured run (several hundred ms on Apple silicon).
|
|
36
|
+
const DEFAULT_WARMUP = 1;
|
|
37
|
+
|
|
38
|
+
const COMMANDS = ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'validate', 'export', 'help'];
|
|
39
|
+
|
|
40
|
+
const KNOWN_OPTIONS = [
|
|
41
|
+
'--help', '--version', '--model', '--models', '--runs', '--warmup', '--concurrency', '--sweep-concurrency',
|
|
42
|
+
'--request-rate', '--ramp-up-ms', '--timeout', '--timeout-ms', '--slo-ttft-ms', '--slo-e2e-ms', '--slo-tpot-ms',
|
|
43
|
+
'--prompt', '--prompt-file', '--profile', '--instructions', '--fm-bin', '--use-case', '--guardrails',
|
|
44
|
+
'--greedy', '--no-greedy', '--stream', '--no-stream', '--json', '--csv', '--format', '--ascii', '--color',
|
|
45
|
+
'--no-color', '--progress', '--no-progress', '--compact', '--width', '--histogram', '--export-html', '--strict',
|
|
46
|
+
'--out', '--output-dir', '--capture-output', '--available-only', '--fail-fast', '--retry', '--ci', '--tag',
|
|
47
|
+
'--note', '--verbose'
|
|
48
|
+
];
|
|
49
|
+
|
|
50
|
+
/** Closest candidate within a small edit distance, or null. */
|
|
51
|
+
function closestMatch(input, candidates) {
|
|
52
|
+
let best = null;
|
|
53
|
+
let bestDistance = Infinity;
|
|
54
|
+
for (const candidate of candidates) {
|
|
55
|
+
const distance = editDistance(input, candidate);
|
|
56
|
+
if (distance < bestDistance) {
|
|
57
|
+
best = candidate;
|
|
58
|
+
bestDistance = distance;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
const limit = input.replace(/^-+/, '').length <= 6 ? 1 : 2;
|
|
62
|
+
return bestDistance > 0 && bestDistance <= limit ? best : null;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** Optimal string alignment distance: an adjacent swap counts as one edit. */
|
|
66
|
+
function editDistance(a, b) {
|
|
67
|
+
const rows = Array.from({ length: a.length + 1 }, (_, i) => [i, ...Array(b.length).fill(0)]);
|
|
68
|
+
for (let j = 1; j <= b.length; j += 1) rows[0][j] = j;
|
|
69
|
+
for (let i = 1; i <= a.length; i += 1) {
|
|
70
|
+
for (let j = 1; j <= b.length; j += 1) {
|
|
71
|
+
const cost = a[i - 1] === b[j - 1] ? 0 : 1;
|
|
72
|
+
rows[i][j] = Math.min(rows[i - 1][j] + 1, rows[i][j - 1] + 1, rows[i - 1][j - 1] + cost);
|
|
73
|
+
if (i > 1 && j > 1 && a[i - 1] === b[j - 2] && a[i - 2] === b[j - 1]) {
|
|
74
|
+
rows[i][j] = Math.min(rows[i][j], rows[i - 2][j - 2] + 1);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
return rows[a.length][b.length];
|
|
79
|
+
}
|
|
80
|
+
|
|
33
81
|
export async function runCli(argv = process.argv.slice(2), env = {}) {
|
|
34
82
|
const parsed = parseArgs(argv);
|
|
35
83
|
|
|
@@ -230,7 +278,7 @@ export function parseArgs(argv) {
|
|
|
230
278
|
models: [],
|
|
231
279
|
prompts: [],
|
|
232
280
|
runs: 1,
|
|
233
|
-
warmup:
|
|
281
|
+
warmup: DEFAULT_WARMUP,
|
|
234
282
|
concurrency: 1,
|
|
235
283
|
sweepConcurrency: [],
|
|
236
284
|
requestRate: null,
|
|
@@ -263,8 +311,15 @@ export function parseArgs(argv) {
|
|
|
263
311
|
};
|
|
264
312
|
|
|
265
313
|
const args = [...argv];
|
|
266
|
-
if (args[0] && !args[0].startsWith('-') &&
|
|
314
|
+
if (args[0] && !args[0].startsWith('-') && COMMANDS.includes(args[0])) {
|
|
267
315
|
options.command = args.shift();
|
|
316
|
+
} else if (args[0] && /^[a-z]{3,}$/.test(args[0])) {
|
|
317
|
+
// A bare word is a prompt, but one that is a near miss for a command is
|
|
318
|
+
// almost always a typo; benchmarking "modles" as a prompt helps nobody.
|
|
319
|
+
const suggestion = closestMatch(args[0], COMMANDS);
|
|
320
|
+
if (suggestion) {
|
|
321
|
+
throw usageError(`Unknown command "${args[0]}". Did you mean "${suggestion}"? To benchmark it as a prompt, use: fm-bench -- ${args[0]}`);
|
|
322
|
+
}
|
|
268
323
|
}
|
|
269
324
|
|
|
270
325
|
if (options.command === 'metrics') {
|
|
@@ -447,7 +502,8 @@ export function parseArgs(argv) {
|
|
|
447
502
|
break;
|
|
448
503
|
default:
|
|
449
504
|
if (arg.startsWith('-')) {
|
|
450
|
-
|
|
505
|
+
const suggestion = arg.startsWith('--') ? closestMatch(arg, KNOWN_OPTIONS) : null;
|
|
506
|
+
throw usageError(`Unknown option: ${arg}${suggestion ? `. Did you mean ${suggestion}?` : ''} (see fm-bench --help)`);
|
|
451
507
|
}
|
|
452
508
|
if (options.command === 'compare') {
|
|
453
509
|
options.compareFiles.push(arg);
|
|
@@ -665,8 +721,17 @@ async function runDoctor(options) {
|
|
|
665
721
|
checks.push(['fm token counting', capabilities.features.tokenCounting ? `yes (${capabilities.features.tokenCountCommand})` : 'no', capabilities.features.tokenCounting]);
|
|
666
722
|
checks.push(['fm streaming', capabilities.features.streaming ? 'yes' : 'no', capabilities.features.streaming]);
|
|
667
723
|
checks.push(['fm quota', capabilities.features.quota ? 'yes' : 'no (not exposed by this build)', true]);
|
|
724
|
+
if (capabilities.features.license) {
|
|
725
|
+
const license = await getLicenseStatus(inspection.fmBin, { capabilities });
|
|
726
|
+
checks.push(['fm license', license.agreed === false
|
|
727
|
+
? `${license.detail || 'not agreed'} — run "fm license" to review and accept the terms`
|
|
728
|
+
: license.detail, license.agreed !== false]);
|
|
729
|
+
}
|
|
668
730
|
for (const model of inspection.models) {
|
|
669
|
-
|
|
731
|
+
const detail = model.available
|
|
732
|
+
? `available${model.identity ? ` (${model.identity})` : ''}`
|
|
733
|
+
: model.reason || 'unavailable';
|
|
734
|
+
checks.push([`model:${model.name}`, detail, model.available]);
|
|
670
735
|
}
|
|
671
736
|
} catch (error) {
|
|
672
737
|
checks.push(['fm', error.message || String(error), false]);
|
|
@@ -681,7 +746,8 @@ async function runDoctor(options) {
|
|
|
681
746
|
if (json) {
|
|
682
747
|
console.log(JSON.stringify(payload, null, 2));
|
|
683
748
|
} else {
|
|
684
|
-
const
|
|
749
|
+
const nameWidth = Math.max(...checks.map(([name]) => name.length));
|
|
750
|
+
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(nameWidth)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
|
|
685
751
|
console.log(lines.join('\n'));
|
|
686
752
|
if (capabilities) {
|
|
687
753
|
console.log('');
|
|
@@ -795,7 +861,7 @@ Commands:
|
|
|
795
861
|
Run options:
|
|
796
862
|
-m, --models <list> Models to benchmark, comma-separated or repeated
|
|
797
863
|
-r, --runs <n> Runs per prompt/model (default: 1)
|
|
798
|
-
--warmup <n>
|
|
864
|
+
--warmup <n> Unmeasured warmup runs per model (default: 1; 0 measures cold start)
|
|
799
865
|
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
800
866
|
--sweep-concurrency <list>
|
|
801
867
|
Run separate operating points, e.g. 1,2,4
|
|
@@ -859,10 +925,16 @@ Exit codes:
|
|
|
859
925
|
2 usage or environment error (bad flags, missing arguments, unsupported macOS, fm not found)
|
|
860
926
|
|
|
861
927
|
Capability detection:
|
|
862
|
-
fm-bench probes "fm --help" and "fm respond --help" once per run
|
|
863
|
-
|
|
928
|
+
fm-bench probes "fm --help" and "fm respond --help" once per run and reads
|
|
929
|
+
model status from "fm models" (or "fm available" on older builds). Metrics
|
|
930
|
+
the installed fm cannot supply are reported as unavailable instead of being
|
|
864
931
|
guessed, and unsupported models are refused before any benchmark starts.
|
|
865
932
|
|
|
933
|
+
Private Cloud Compute:
|
|
934
|
+
fm reports the pcc model as "not available in this context" outside the
|
|
935
|
+
Terminal app (for example in editor terminals). Run fm-bench from Terminal
|
|
936
|
+
to benchmark pcc; elsewhere it is listed as skipped with that reason.
|
|
937
|
+
|
|
866
938
|
Examples:
|
|
867
939
|
fm-bench
|
|
868
940
|
fm-bench --models system --runs 3 --profile stress
|
package/src/compare.js
CHANGED
|
@@ -77,10 +77,19 @@ function buildDiffRow(key, before, after) {
|
|
|
77
77
|
goodputRate: diffPercent(before?.goodputRate, after?.goodputRate),
|
|
78
78
|
tokensPerSecond: diffNumber(before?.tokensPerSecond?.avg, after?.tokensPerSecond?.avg, false),
|
|
79
79
|
rps: diffNumber(before?.rps, after?.rps, false),
|
|
80
|
-
cv: diffPercent(before
|
|
80
|
+
cv: diffPercent(...comparableCv(before, after))
|
|
81
81
|
};
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
// Compare the run-to-run CV only when both reports carry it; otherwise fall
|
|
85
|
+
// back to the suite-wide latency CV on both sides so the definitions match.
|
|
86
|
+
function comparableCv(before, after) {
|
|
87
|
+
const bothStable = before && after && Object.hasOwn(before, 'stabilityCv') && Object.hasOwn(after, 'stabilityCv');
|
|
88
|
+
return bothStable
|
|
89
|
+
? [before.stabilityCv, after.stabilityCv]
|
|
90
|
+
: [before?.latency?.cv, after?.latency?.cv];
|
|
91
|
+
}
|
|
92
|
+
|
|
84
93
|
function diffMs(before, after) {
|
|
85
94
|
return {
|
|
86
95
|
before: before ?? null,
|
package/src/export.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { environmentFingerprint } from './schema.js';
|
|
2
|
+
import { stabilityCv } from './table.js';
|
|
2
3
|
|
|
3
4
|
/**
|
|
4
5
|
* Self-contained HTML report for sharing (paste, email, GitHub gist, static host).
|
|
@@ -33,12 +34,13 @@ export function renderHtmlReport(report) {
|
|
|
33
34
|
['macOS', fp.macOSProductVersion ?? '—'],
|
|
34
35
|
['Build', fp.macOSBuildVersion ?? '—'],
|
|
35
36
|
['Node', fp.node ?? '—'],
|
|
36
|
-
['fm CLI digest', fp.fmHelpDigest ?? '—']
|
|
37
|
+
['fm CLI digest', fp.fmHelpDigest ?? '—'],
|
|
38
|
+
['Models', Object.entries(fp.modelIdentities).map(([name, identity]) => `${name} = ${identity}`).join(', ') || '—']
|
|
37
39
|
];
|
|
38
40
|
|
|
39
41
|
const summaryRows = summary.map((row) => {
|
|
40
42
|
const ttft = /** @type {{ p50?: number, p95?: number }} */ (row.ttft ?? {});
|
|
41
|
-
const lat = /** @type {{ p50?: number, p95?: number
|
|
43
|
+
const lat = /** @type {{ p50?: number, p95?: number }} */ (row.latency ?? {});
|
|
42
44
|
const tps = /** @type {{ avg?: number }} */ (row.tokensPerSecond ?? {});
|
|
43
45
|
return `<tr>
|
|
44
46
|
<td>${esc(row.model)}</td>
|
|
@@ -51,7 +53,7 @@ export function renderHtmlReport(report) {
|
|
|
51
53
|
<td>${fmtMs(lat.p50)}</td>
|
|
52
54
|
<td>${fmtMs(lat.p95)}</td>
|
|
53
55
|
<td>${fmtNum(tps.avg)}</td>
|
|
54
|
-
<td>${fmtPct(
|
|
56
|
+
<td>${fmtPct(stabilityCv(row))}</td>
|
|
55
57
|
</tr>`;
|
|
56
58
|
}).join('\n');
|
|
57
59
|
|
|
@@ -114,7 +116,7 @@ export function renderHtmlReport(report) {
|
|
|
114
116
|
</section>
|
|
115
117
|
|
|
116
118
|
<footer>
|
|
117
|
-
Generated by <a href="https://github.com/
|
|
119
|
+
Generated by <a href="https://github.com/dvnold/fm-bench">fm-bench</a>.
|
|
118
120
|
Compare two reports: <code>fm-bench compare before.json after.json</code>.
|
|
119
121
|
Validate: <code>fm-bench validate report.json</code>.
|
|
120
122
|
</footer>
|
package/src/fm-help.js
CHANGED
|
@@ -60,6 +60,47 @@ export function parseModelsFromHelp(helpText = '') {
|
|
|
60
60
|
return [...models.values()];
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
+
/**
|
|
64
|
+
* Read `fm models` output (macOS 27.2+), one model per line:
|
|
65
|
+
*
|
|
66
|
+
* Apple Foundation Models
|
|
67
|
+
* ✓ system (AFM 3 Core Advanced)
|
|
68
|
+
* ✗ pcc (Private Cloud Compute is not available in this context. ...)
|
|
69
|
+
*
|
|
70
|
+
* The parenthesised text is the model identity for an available model and
|
|
71
|
+
* the reason for an unavailable one.
|
|
72
|
+
* @param {string} output
|
|
73
|
+
* @returns {{ name: string, available: boolean, identity: string, reason: string }[]}
|
|
74
|
+
*/
|
|
75
|
+
export function parseModelList(output = '') {
|
|
76
|
+
const models = new Map();
|
|
77
|
+
for (const line of withoutWarnings(output).split(/\r?\n/)) {
|
|
78
|
+
const match = line.match(/^\s*([✓✔✗✘×])\s+([A-Za-z0-9][A-Za-z0-9._:/-]*)\s*(?:\((.*)\))?\s*$/);
|
|
79
|
+
if (!match) continue;
|
|
80
|
+
const available = match[1] === '✓' || match[1] === '✔';
|
|
81
|
+
const detail = (match[3] ?? '').trim();
|
|
82
|
+
models.set(match[2], {
|
|
83
|
+
name: match[2],
|
|
84
|
+
available,
|
|
85
|
+
identity: available ? detail : '',
|
|
86
|
+
reason: available ? '' : detail
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
|
+
return [...models.values()];
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Drop `warning:` lines (for example the `fm available` rename notice) so
|
|
94
|
+
* they are never mistaken for a model status or an error cause.
|
|
95
|
+
* @param {string} text
|
|
96
|
+
*/
|
|
97
|
+
export function withoutWarnings(text = '') {
|
|
98
|
+
return stripAnsi(text)
|
|
99
|
+
.split(/\r?\n/)
|
|
100
|
+
.filter((line) => !/^\s*warning:/i.test(line))
|
|
101
|
+
.join('\n');
|
|
102
|
+
}
|
|
103
|
+
|
|
63
104
|
/**
|
|
64
105
|
* Read model names and availability out of `fm available` (no model filter).
|
|
65
106
|
* @param {string} output
|
|
@@ -67,7 +108,7 @@ export function parseModelsFromHelp(helpText = '') {
|
|
|
67
108
|
*/
|
|
68
109
|
export function parseAvailabilityList(output = '') {
|
|
69
110
|
const models = new Map();
|
|
70
|
-
for (const line of
|
|
111
|
+
for (const line of withoutWarnings(output).split(/\r?\n/)) {
|
|
71
112
|
const match = line.match(/^\s*([A-Za-z][A-Za-z0-9._-]*)(?:\s+model)?\s+(?:is\s+)?(available|unavailable|not available)\b/i);
|
|
72
113
|
if (!match) continue;
|
|
73
114
|
const name = match[1].toLowerCase();
|
|
@@ -88,7 +129,17 @@ export function parseAvailabilityList(output = '') {
|
|
|
88
129
|
* @param {number|null} code process exit code
|
|
89
130
|
*/
|
|
90
131
|
export function parseAvailabilityOutput(model, output = '', code = null) {
|
|
91
|
-
const clean =
|
|
132
|
+
const clean = withoutWarnings(output).trim();
|
|
133
|
+
const listed = parseModelList(clean).find((entry) => entry.name === model);
|
|
134
|
+
if (listed) {
|
|
135
|
+
return {
|
|
136
|
+
model,
|
|
137
|
+
available: listed.available,
|
|
138
|
+
identity: listed.identity,
|
|
139
|
+
raw: clean,
|
|
140
|
+
reason: listed.reason
|
|
141
|
+
};
|
|
142
|
+
}
|
|
92
143
|
const lower = clean.toLowerCase();
|
|
93
144
|
const modelLower = String(model).toLowerCase();
|
|
94
145
|
const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b|\bis invalid for\b/.test(lower);
|
|
@@ -106,14 +157,18 @@ export function parseAvailabilityOutput(model, output = '', code = null) {
|
|
|
106
157
|
|
|
107
158
|
/**
|
|
108
159
|
* Collapse `fm` diagnostics into one actionable line. `fm` writes multi-line
|
|
109
|
-
* usage blocks for argument errors; benchmark output
|
|
160
|
+
* usage blocks for argument errors and `warning:` notices; benchmark output
|
|
161
|
+
* only needs the line that states the cause (including any hint on it, such
|
|
162
|
+
* as "Please use the Terminal app.").
|
|
110
163
|
* @param {string} text
|
|
111
164
|
*/
|
|
112
165
|
export function firstLine(text = '') {
|
|
113
|
-
const
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
166
|
+
const lines = withoutWarnings(text)
|
|
167
|
+
.split(/\r?\n/)
|
|
168
|
+
.map((line) => line.replace(/\s+/g, ' ').trim())
|
|
169
|
+
.filter((line) => line && !/^(usage|help|see)\b/i.test(line));
|
|
170
|
+
if (lines.length === 0) return '';
|
|
171
|
+
const head = lines.find((line) => /error|invalid|unavailable|not available|not supported|failed/i.test(line)) || lines[0];
|
|
117
172
|
return head.replace(/^Error:\s*/i, '').trim();
|
|
118
173
|
}
|
|
119
174
|
|