fm-bench 0.6.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +88 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +34 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +43 -0
- package/package.json +11 -6
- package/src/bench.js +129 -37
- package/src/capabilities.js +196 -0
- package/src/cli.js +192 -76
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +110 -106
- package/src/history.js +2 -1
- package/src/macos.js +72 -0
- package/src/metrics.js +212 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
// Capability detection for the installed `fm` CLI.
|
|
2
|
+
//
|
|
3
|
+
// fm-bench talks to whatever `fm` build is on the machine, and Apple has
|
|
4
|
+
// changed subcommand names and flags between releases (for example
|
|
5
|
+
// `count-tokens` replaces the older `token-count`). Everything downstream
|
|
6
|
+
// reads this detection result instead of hardcoding subcommand names, so a
|
|
7
|
+
// compatible `fm` build is supported without a code change and an
|
|
8
|
+
// incompatible one is reported instead of producing wrong numbers.
|
|
9
|
+
|
|
10
|
+
import crypto from 'node:crypto';
|
|
11
|
+
import { stripAnsi } from './ansi.js';
|
|
12
|
+
import { parseModelsFromHelp } from './fm-help.js';
|
|
13
|
+
import { runProcess } from './process.js';
|
|
14
|
+
|
|
15
|
+
const SECTION_HEADER = /^\s*[A-Z][A-Z0-9 /-]+\s*$/;
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Parse the COMMANDS section of `fm --help`.
|
|
19
|
+
* @param {string} helpText
|
|
20
|
+
* @returns {string[]}
|
|
21
|
+
*/
|
|
22
|
+
export function parseCommandsFromHelp(helpText = '') {
|
|
23
|
+
const lines = stripAnsi(helpText).split(/\r?\n/);
|
|
24
|
+
const commands = [];
|
|
25
|
+
let inCommands = false;
|
|
26
|
+
|
|
27
|
+
for (const line of lines) {
|
|
28
|
+
if (/^\s*COMMANDS\s*$/.test(line)) {
|
|
29
|
+
inCommands = true;
|
|
30
|
+
continue;
|
|
31
|
+
}
|
|
32
|
+
if (inCommands && SECTION_HEADER.test(line)) {
|
|
33
|
+
inCommands = false;
|
|
34
|
+
}
|
|
35
|
+
if (!inCommands) continue;
|
|
36
|
+
|
|
37
|
+
const match = line.match(/^\s{2,}([a-z][a-z0-9-]*)\s{2,}\S/);
|
|
38
|
+
if (match) commands.push(match[1]);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
return commands;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* True when a long flag appears in `fm` help text. Apple prints boolean flags
|
|
46
|
+
* with a negatable form (`--[no-]stream`), so accept that shape too.
|
|
47
|
+
* @param {string} helpText
|
|
48
|
+
* @param {string} flag e.g. "--use-case"
|
|
49
|
+
*/
|
|
50
|
+
export function hasFlagInHelp(helpText, flag) {
|
|
51
|
+
const clean = stripAnsi(helpText);
|
|
52
|
+
if (new RegExp(`${escapeRegExp(flag)}\\b`).test(clean)) return true;
|
|
53
|
+
if (flag.startsWith('--no-')) {
|
|
54
|
+
return clean.includes(`[no-]${flag.slice('--no-'.length)}`);
|
|
55
|
+
}
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function helpDigest(helpText = '') {
|
|
60
|
+
const text = stripAnsi(helpText).trim();
|
|
61
|
+
if (!text) return null;
|
|
62
|
+
return crypto.createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Pick the subcommand this `fm` build exposes for token counting.
|
|
67
|
+
* @returns {string|null}
|
|
68
|
+
*/
|
|
69
|
+
export function resolveTokenCountCommand(commands = []) {
|
|
70
|
+
if (commands.includes('count-tokens')) return 'count-tokens';
|
|
71
|
+
if (commands.includes('token-count')) return 'token-count';
|
|
72
|
+
return null;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function buildFeatures(helpText, respondHelpText, commands) {
|
|
76
|
+
const tokenCountCommand = resolveTokenCountCommand(commands);
|
|
77
|
+
const respondHelp = respondHelpText || '';
|
|
78
|
+
return {
|
|
79
|
+
tokenCounting: tokenCountCommand != null,
|
|
80
|
+
tokenCountCommand,
|
|
81
|
+
quota: commands.includes('quota-usage'),
|
|
82
|
+
streaming: hasFlagInHelp(respondHelp, '--no-stream') || hasFlagInHelp(respondHelp, '--stream'),
|
|
83
|
+
modelSelection: hasFlagInHelp(respondHelp, '--model'),
|
|
84
|
+
instructions: hasFlagInHelp(respondHelp, '--instructions'),
|
|
85
|
+
greedy: hasFlagInHelp(respondHelp, '--greedy'),
|
|
86
|
+
useCase: hasFlagInHelp(respondHelp, '--use-case'),
|
|
87
|
+
guardrails: hasFlagInHelp(respondHelp, '--guardrails'),
|
|
88
|
+
images: hasFlagInHelp(respondHelp, '--image'),
|
|
89
|
+
tools: hasFlagInHelp(respondHelp, '--tool'),
|
|
90
|
+
structuredOutput: hasFlagInHelp(respondHelp, '--schema'),
|
|
91
|
+
server: commands.includes('serve')
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function capabilityWarnings(commands, features, models) {
|
|
96
|
+
const warnings = [];
|
|
97
|
+
if (!features.tokenCounting) {
|
|
98
|
+
warnings.push('this fm build exposes no token-counting command, so token counts and token throughput are unavailable');
|
|
99
|
+
}
|
|
100
|
+
if (!features.quota) {
|
|
101
|
+
warnings.push('this fm build exposes no quota command, so quota is not reported');
|
|
102
|
+
}
|
|
103
|
+
if (!features.streaming) {
|
|
104
|
+
warnings.push('this fm build does not document a streaming flag, so TTFT cannot be measured');
|
|
105
|
+
}
|
|
106
|
+
if (models.length === 0) {
|
|
107
|
+
warnings.push('could not discover any models from fm help output');
|
|
108
|
+
}
|
|
109
|
+
if (commands.length === 0) {
|
|
110
|
+
warnings.push('could not read the command list from fm --help');
|
|
111
|
+
}
|
|
112
|
+
return warnings;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Probe the installed `fm` binary once and describe what it can do.
|
|
117
|
+
*
|
|
118
|
+
* @param {string} fmBin
|
|
119
|
+
* @param {{ timeoutMs?: number, env?: NodeJS.ProcessEnv, help?: { text: string } }} [options]
|
|
120
|
+
*/
|
|
121
|
+
export async function detectFmCapabilities(fmBin, options = {}) {
|
|
122
|
+
const timeoutMs = options.timeoutMs ?? 10_000;
|
|
123
|
+
const env = options.env ?? process.env;
|
|
124
|
+
let helpText = options.help?.text ?? null;
|
|
125
|
+
|
|
126
|
+
if (helpText == null) {
|
|
127
|
+
const help = await runProcess(fmBin, ['--help'], { timeoutMs, env });
|
|
128
|
+
if (help.error) {
|
|
129
|
+
return {
|
|
130
|
+
ok: false,
|
|
131
|
+
bin: fmBin,
|
|
132
|
+
error: `Unable to execute ${fmBin}: ${help.stderr || help.error.message}`,
|
|
133
|
+
commands: [],
|
|
134
|
+
models: [],
|
|
135
|
+
features: buildFeatures('', '', []),
|
|
136
|
+
digest: null,
|
|
137
|
+
help: '',
|
|
138
|
+
warnings: [`cannot execute ${fmBin}`]
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
helpText = `${help.stdout}${help.stderr}`;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
const cleanHelp = stripAnsi(helpText);
|
|
145
|
+
const commands = parseCommandsFromHelp(cleanHelp);
|
|
146
|
+
|
|
147
|
+
let respondHelpText = '';
|
|
148
|
+
if (commands.includes('respond')) {
|
|
149
|
+
const respondHelp = await runProcess(fmBin, ['respond', '--help'], { timeoutMs, env });
|
|
150
|
+
if (!respondHelp.error) respondHelpText = `${respondHelp.stdout}${respondHelp.stderr}`;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const features = buildFeatures(cleanHelp, respondHelpText, commands);
|
|
154
|
+
let models = parseModelsFromHelp(cleanHelp);
|
|
155
|
+
|
|
156
|
+
if (models.length === 0 && commands.includes('available')) {
|
|
157
|
+
models = await discoverModelsFromAvailability(fmBin, { ...options, env });
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
ok: commands.length > 0,
|
|
162
|
+
bin: fmBin,
|
|
163
|
+
commands,
|
|
164
|
+
models,
|
|
165
|
+
features,
|
|
166
|
+
digest: helpDigest(cleanHelp),
|
|
167
|
+
help: cleanHelp,
|
|
168
|
+
warnings: capabilityWarnings(commands, features, models)
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Fallback discovery: ask `fm available` without a model filter and read the
|
|
174
|
+
* model names it reports. Used when `fm --help` has no MODELS section.
|
|
175
|
+
*/
|
|
176
|
+
async function discoverModelsFromAvailability(fmBin, options = {}) {
|
|
177
|
+
const result = await runProcess(fmBin, ['available'], {
|
|
178
|
+
timeoutMs: options.timeoutMs ?? 15_000,
|
|
179
|
+
env: options.env ?? process.env
|
|
180
|
+
});
|
|
181
|
+
if (result.error) return [];
|
|
182
|
+
const text = stripAnsi(`${result.stdout}${result.stderr}`);
|
|
183
|
+
const models = new Map();
|
|
184
|
+
for (const line of text.split(/\r?\n/)) {
|
|
185
|
+
const match = line.match(/^\s*([A-Za-z0-9._-]+)(?:\s+model)?\s+(?:is\s+)?(available|unavailable|not available)\b/i);
|
|
186
|
+
if (!match) continue;
|
|
187
|
+
const name = match[1].toLowerCase();
|
|
188
|
+
if (['the', 'no', 'model', 'system'].includes(name) && name !== 'system') continue;
|
|
189
|
+
models.set(name, { name, description: '' });
|
|
190
|
+
}
|
|
191
|
+
return [...models.values()];
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
function escapeRegExp(value) {
|
|
195
|
+
return String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
196
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -3,18 +3,34 @@ import { createRequire } from 'node:module';
|
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { diffReports, renderCompareReport } from './compare.js';
|
|
5
5
|
import { renderHtmlReport } from './export.js';
|
|
6
|
+
import { formatCapabilitySummary } from './metrics.js';
|
|
6
7
|
import { validateReport } from './schema.js';
|
|
7
8
|
import { loadHistory, renderHistoryReport } from './history.js';
|
|
9
|
+
import { detectMacosVersion, evaluateMacosSupport, formatMacosRequirementError, MIN_SUPPORTED_MACOS, parseMacosVersion } from './macos.js';
|
|
8
10
|
import { runProcess } from './process.js';
|
|
9
11
|
import { createProgress } from './progress.js';
|
|
10
12
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
11
13
|
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
12
|
-
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend,
|
|
14
|
+
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend, renderModelsReport } from './table.js';
|
|
13
15
|
|
|
14
16
|
const require = createRequire(import.meta.url);
|
|
15
17
|
const packageJson = require('../package.json');
|
|
16
18
|
|
|
17
|
-
|
|
19
|
+
/** Usage / environment error: bad flags, missing arguments, unsupported host. */
|
|
20
|
+
function usageError(message) {
|
|
21
|
+
const error = new Error(message);
|
|
22
|
+
error.exitCode = 2;
|
|
23
|
+
return error;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** Operational failure: invalid report data, failed benchmark gate. */
|
|
27
|
+
function operationalError(message) {
|
|
28
|
+
const error = new Error(message);
|
|
29
|
+
error.exitCode = 1;
|
|
30
|
+
return error;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export async function runCli(argv = process.argv.slice(2), env = {}) {
|
|
18
34
|
const parsed = parseArgs(argv);
|
|
19
35
|
|
|
20
36
|
if (parsed.help) {
|
|
@@ -27,6 +43,8 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
27
43
|
return;
|
|
28
44
|
}
|
|
29
45
|
|
|
46
|
+
await assertSupportedMacos(parsed, env);
|
|
47
|
+
|
|
30
48
|
if (parsed.command === 'legend') {
|
|
31
49
|
if (parsed.format === 'json') {
|
|
32
50
|
console.log(JSON.stringify(legendEntries(), null, 2));
|
|
@@ -66,9 +84,14 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
66
84
|
if (parsed.command === 'models') {
|
|
67
85
|
const inspection = await inspectModels(parsed);
|
|
68
86
|
if (parsed.format === 'json') {
|
|
87
|
+
// Shape stays a plain array so existing automation keeps working;
|
|
88
|
+
// full capability detail lives in `doctor --json` and in report payloads.
|
|
69
89
|
console.log(JSON.stringify(inspection.models, null, 2));
|
|
70
90
|
} else {
|
|
71
|
-
console.log(
|
|
91
|
+
console.log(renderModelsReport(inspection.models, {
|
|
92
|
+
...renderOptions(parsed),
|
|
93
|
+
capabilities: inspection.capabilities
|
|
94
|
+
}));
|
|
72
95
|
}
|
|
73
96
|
return;
|
|
74
97
|
}
|
|
@@ -145,14 +168,35 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
145
168
|
if (!ciResult.passed) {
|
|
146
169
|
const reasons = ciResult.reasons.join('; ');
|
|
147
170
|
console.error(`fm-bench ci: FAIL — ${reasons}`);
|
|
148
|
-
|
|
149
|
-
error.exitCode = 1;
|
|
150
|
-
throw error;
|
|
171
|
+
throw operationalError(`CI checks failed: ${reasons}`);
|
|
151
172
|
}
|
|
152
173
|
console.error(`fm-bench ci: PASS`);
|
|
153
174
|
}
|
|
154
175
|
}
|
|
155
176
|
|
|
177
|
+
// Offline report tools stay usable anywhere. `doctor` is also allowed through so
|
|
178
|
+
// it can print the detected version and the latest supported macOS when the host
|
|
179
|
+
// is too old. Only commands that launch `fm` benchmarks are hard-gated.
|
|
180
|
+
const SKIP_MACOS_GATE = new Set(['compare', 'history', 'validate', 'export', 'legend', 'doctor']);
|
|
181
|
+
|
|
182
|
+
async function assertSupportedMacos(parsed, env = {}) {
|
|
183
|
+
if (SKIP_MACOS_GATE.has(parsed.command)) return;
|
|
184
|
+
|
|
185
|
+
// The gate guards the default fm discovery path, because Apple only ships the
|
|
186
|
+
// CLI from macOS 27. An explicitly configured binary is honoured on any host;
|
|
187
|
+
// the capability probe below still fails with exit 2 if it is unusable.
|
|
188
|
+
if (parsed.fmBin || env.FM_BIN || process.env.FM_BIN) return;
|
|
189
|
+
|
|
190
|
+
const evaluation = evaluateMacosSupport(
|
|
191
|
+
env.platform ?? process.platform,
|
|
192
|
+
parseMacosVersion(await detectMacosVersion(env))
|
|
193
|
+
);
|
|
194
|
+
if (evaluation.supported) return;
|
|
195
|
+
|
|
196
|
+
const error = usageError(formatMacosRequirementError(evaluation));
|
|
197
|
+
throw error;
|
|
198
|
+
}
|
|
199
|
+
|
|
156
200
|
function evaluateCi(payload) {
|
|
157
201
|
const reasons = [];
|
|
158
202
|
const totalFailed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
|
|
@@ -293,7 +337,7 @@ export function parseArgs(argv) {
|
|
|
293
337
|
case '--profile':
|
|
294
338
|
options.profile = requireValue(arg, args);
|
|
295
339
|
if (!['quick', 'standard', 'interactive', 'throughput', 'client', 'stress', 'reasoning', 'coding', 'creative'].includes(options.profile)) {
|
|
296
|
-
throw
|
|
340
|
+
throw usageError('--profile must be one of: quick, standard, interactive, throughput, client, stress, reasoning, coding, creative');
|
|
297
341
|
}
|
|
298
342
|
break;
|
|
299
343
|
case '-i':
|
|
@@ -330,7 +374,7 @@ export function parseArgs(argv) {
|
|
|
330
374
|
case '--format':
|
|
331
375
|
options.format = requireValue(arg, args);
|
|
332
376
|
if (!['table', 'json', 'csv'].includes(options.format)) {
|
|
333
|
-
throw
|
|
377
|
+
throw usageError('--format must be one of: table, json, csv');
|
|
334
378
|
}
|
|
335
379
|
break;
|
|
336
380
|
case '--ascii':
|
|
@@ -403,7 +447,7 @@ export function parseArgs(argv) {
|
|
|
403
447
|
break;
|
|
404
448
|
default:
|
|
405
449
|
if (arg.startsWith('-')) {
|
|
406
|
-
throw
|
|
450
|
+
throw usageError(`Unknown option: ${arg}`);
|
|
407
451
|
}
|
|
408
452
|
if (options.command === 'compare') {
|
|
409
453
|
options.compareFiles.push(arg);
|
|
@@ -442,30 +486,27 @@ async function runHistory(options, renderOpts) {
|
|
|
442
486
|
async function runCompare(options, renderOpts) {
|
|
443
487
|
const files = options.compareFiles;
|
|
444
488
|
if (files.length < 2) {
|
|
445
|
-
throw
|
|
489
|
+
throw usageError('compare requires two JSON report files: fm-bench compare before.json after.json');
|
|
446
490
|
}
|
|
447
491
|
if (files.length > 2) {
|
|
448
|
-
throw
|
|
492
|
+
throw usageError('compare accepts exactly two JSON report files');
|
|
449
493
|
}
|
|
450
494
|
|
|
451
495
|
const [beforePath, afterPath] = files;
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
fs.readFile(afterPath, 'utf8')
|
|
455
|
-
]);
|
|
456
|
-
|
|
457
|
-
let before, after;
|
|
496
|
+
let beforeText;
|
|
497
|
+
let afterText;
|
|
458
498
|
try {
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
} catch {
|
|
466
|
-
throw new Error(`Cannot parse ${afterPath} as JSON`);
|
|
499
|
+
[beforeText, afterText] = await Promise.all([
|
|
500
|
+
fs.readFile(beforePath, 'utf8'),
|
|
501
|
+
fs.readFile(afterPath, 'utf8')
|
|
502
|
+
]);
|
|
503
|
+
} catch (error) {
|
|
504
|
+
throw operationalError(`Cannot read report: ${error.message}`);
|
|
467
505
|
}
|
|
468
506
|
|
|
507
|
+
const before = parseReportJson(beforeText, beforePath);
|
|
508
|
+
const after = parseReportJson(afterText, afterPath);
|
|
509
|
+
|
|
469
510
|
const diff = diffReports(before, after);
|
|
470
511
|
|
|
471
512
|
if (options.format === 'json') {
|
|
@@ -480,64 +521,77 @@ async function runCompare(options, renderOpts) {
|
|
|
480
521
|
}
|
|
481
522
|
|
|
482
523
|
if (options.strictCompare && diff.compatibility && !diff.compatibility.suiteMatch) {
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
524
|
+
throw usageError('compare: benchmark suites differ (--strict)');
|
|
525
|
+
}
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
function parseReportJson(text, filePath) {
|
|
529
|
+
try {
|
|
530
|
+
return JSON.parse(text);
|
|
531
|
+
} catch {
|
|
532
|
+
throw operationalError(`Cannot parse ${filePath} as JSON`);
|
|
486
533
|
}
|
|
487
534
|
}
|
|
488
535
|
|
|
489
536
|
async function runValidate(options) {
|
|
490
537
|
const files = options.validateFiles;
|
|
491
538
|
if (files.length === 0) {
|
|
492
|
-
throw
|
|
539
|
+
throw usageError('validate requires at least one JSON report: fm-bench validate report.json');
|
|
493
540
|
}
|
|
494
541
|
|
|
495
|
-
|
|
542
|
+
const results = [];
|
|
496
543
|
for (const filePath of files) {
|
|
497
544
|
let parsed;
|
|
498
545
|
try {
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
546
|
+
parsed = JSON.parse(await fs.readFile(filePath, 'utf8'));
|
|
547
|
+
} catch (error) {
|
|
548
|
+
results.push({ file: filePath, ok: false, errors: [error.code === 'ENOENT'
|
|
549
|
+
? 'file not found'
|
|
550
|
+
: 'cannot read or parse JSON'] });
|
|
504
551
|
continue;
|
|
505
552
|
}
|
|
506
553
|
const result = validateReport(parsed);
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
554
|
+
results.push(result.ok
|
|
555
|
+
? { file: filePath, ok: true, schema: result.report.schemaVersion ?? 'legacy', id: result.report.reportId ?? null }
|
|
556
|
+
: { file: filePath, ok: false, errors: result.errors });
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
if (options.format === 'json') {
|
|
560
|
+
console.log(JSON.stringify({ ok: results.every((item) => item.ok), files: results }, null, 2));
|
|
561
|
+
} else {
|
|
562
|
+
for (const item of results) {
|
|
563
|
+
if (item.ok) {
|
|
564
|
+
console.log(`ok ${item.file} schema=${item.schema} id=${item.id ?? '—'}`);
|
|
565
|
+
} else {
|
|
566
|
+
console.error(`invalid ${item.file} ${item.errors.join('; ')}`);
|
|
567
|
+
}
|
|
514
568
|
}
|
|
515
569
|
}
|
|
516
570
|
|
|
571
|
+
const failed = results.filter((item) => !item.ok).length;
|
|
517
572
|
if (failed > 0) {
|
|
518
|
-
|
|
519
|
-
error.exitCode = 1;
|
|
520
|
-
throw error;
|
|
573
|
+
throw operationalError(`${failed} report(s) failed validation`);
|
|
521
574
|
}
|
|
522
575
|
}
|
|
523
576
|
|
|
524
577
|
async function runExport(options) {
|
|
525
578
|
const files = options.validateFiles;
|
|
526
579
|
if (files.length === 0) {
|
|
527
|
-
throw
|
|
580
|
+
throw usageError('export requires a JSON report: fm-bench export report.json [-o out.html]');
|
|
528
581
|
}
|
|
529
582
|
|
|
530
583
|
const filePath = files[0];
|
|
531
|
-
const text = await fs.readFile(filePath, 'utf8');
|
|
532
584
|
let report;
|
|
533
585
|
try {
|
|
534
|
-
report = JSON.parse(
|
|
535
|
-
} catch {
|
|
536
|
-
throw
|
|
586
|
+
report = JSON.parse(await fs.readFile(filePath, 'utf8'));
|
|
587
|
+
} catch (error) {
|
|
588
|
+
throw error.code === 'ENOENT'
|
|
589
|
+
? operationalError(`Cannot read ${filePath}: file not found`)
|
|
590
|
+
: operationalError(`Cannot parse ${filePath} as JSON`);
|
|
537
591
|
}
|
|
538
592
|
const validation = validateReport(report);
|
|
539
593
|
if (!validation.ok) {
|
|
540
|
-
throw
|
|
594
|
+
throw operationalError(`Not a valid fm-bench report: ${validation.errors.join('; ')}`);
|
|
541
595
|
}
|
|
542
596
|
|
|
543
597
|
const html = renderHtmlReport(validation.report);
|
|
@@ -550,15 +604,19 @@ async function runExport(options) {
|
|
|
550
604
|
}
|
|
551
605
|
|
|
552
606
|
async function runDoctor(options) {
|
|
607
|
+
const json = options.format === 'json';
|
|
553
608
|
const checks = [];
|
|
554
609
|
checks.push(['node', process.version, true]);
|
|
555
610
|
checks.push(['platform', `${process.platform}/${process.arch}`, process.platform === 'darwin']);
|
|
556
611
|
|
|
557
|
-
const
|
|
558
|
-
const
|
|
559
|
-
const
|
|
560
|
-
|
|
561
|
-
|
|
612
|
+
const macOS = await detectMacosVersion();
|
|
613
|
+
const parsedVersion = parseMacosVersion(macOS);
|
|
614
|
+
const support = evaluateMacosSupport(process.platform, parsedVersion);
|
|
615
|
+
checks.push(['macOS', parsedVersion?.version || 'unknown', support.supported]);
|
|
616
|
+
if (!support.supported) {
|
|
617
|
+
checks.push(['macOS support', support.reason, false]);
|
|
618
|
+
checks.push(['latest supported', support.latestSupported, false]);
|
|
619
|
+
}
|
|
562
620
|
|
|
563
621
|
const hwModel = await runProcess('sysctl', ['-n', 'hw.model'], { timeoutMs: 3_000 });
|
|
564
622
|
const hwModelStr = (hwModel.stdout || '').trim();
|
|
@@ -595,41 +653,83 @@ async function runDoctor(options) {
|
|
|
595
653
|
checks.push(['battery', `${pct} (${source})`, ok]);
|
|
596
654
|
}
|
|
597
655
|
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
656
|
+
let models = [];
|
|
657
|
+
let capabilities = null;
|
|
658
|
+
try {
|
|
659
|
+
const inspection = await inspectModels(options);
|
|
660
|
+
models = inspection.models;
|
|
661
|
+
capabilities = inspection.capabilities;
|
|
662
|
+
checks.push(['fm', inspection.fmBin, capabilities.ok]);
|
|
663
|
+
if (capabilities.digest) checks.push(['fm help digest', capabilities.digest, true]);
|
|
664
|
+
checks.push(['fm commands', capabilities.commands.join(', ') || 'none found', capabilities.commands.length > 0]);
|
|
665
|
+
checks.push(['fm token counting', capabilities.features.tokenCounting ? `yes (${capabilities.features.tokenCountCommand})` : 'no', capabilities.features.tokenCounting]);
|
|
666
|
+
checks.push(['fm streaming', capabilities.features.streaming ? 'yes' : 'no', capabilities.features.streaming]);
|
|
667
|
+
checks.push(['fm quota', capabilities.features.quota ? 'yes' : 'no (not exposed by this build)', true]);
|
|
668
|
+
for (const model of inspection.models) {
|
|
669
|
+
checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
|
|
670
|
+
}
|
|
671
|
+
} catch (error) {
|
|
672
|
+
checks.push(['fm', error.message || String(error), false]);
|
|
602
673
|
}
|
|
603
674
|
|
|
604
|
-
const
|
|
605
|
-
|
|
675
|
+
const payload = {
|
|
676
|
+
checks: checks.map(([name, detail, ok]) => ({ name, detail: String(detail), ok })),
|
|
677
|
+
models,
|
|
678
|
+
capabilities: capabilities ? describeCapabilities(capabilities) : null
|
|
679
|
+
};
|
|
680
|
+
|
|
681
|
+
if (json) {
|
|
682
|
+
console.log(JSON.stringify(payload, null, 2));
|
|
683
|
+
} else {
|
|
684
|
+
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(16)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
|
|
685
|
+
console.log(lines.join('\n'));
|
|
686
|
+
if (capabilities) {
|
|
687
|
+
console.log('');
|
|
688
|
+
console.log(`fm capabilities: ${formatCapabilitySummary(capabilities)}`);
|
|
689
|
+
for (const warning of capabilities.warnings) {
|
|
690
|
+
console.log(`limit: ${warning}`);
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
}
|
|
606
694
|
|
|
607
695
|
if (options.out) {
|
|
608
|
-
await fs.writeFile(options.out, `${JSON.stringify(
|
|
696
|
+
await fs.writeFile(options.out, `${JSON.stringify(payload, null, 2)}\n`, 'utf8');
|
|
609
697
|
}
|
|
610
698
|
}
|
|
611
699
|
|
|
700
|
+
function describeCapabilities(capabilities) {
|
|
701
|
+
return {
|
|
702
|
+
bin: capabilities.bin,
|
|
703
|
+
ok: capabilities.ok,
|
|
704
|
+
digest: capabilities.digest,
|
|
705
|
+
commands: capabilities.commands,
|
|
706
|
+
models: capabilities.models,
|
|
707
|
+
features: capabilities.features,
|
|
708
|
+
warnings: capabilities.warnings
|
|
709
|
+
};
|
|
710
|
+
}
|
|
711
|
+
|
|
612
712
|
function requireValue(option, args) {
|
|
613
713
|
const value = args.shift();
|
|
614
|
-
if (value == null || value === '') throw
|
|
714
|
+
if (value == null || value === '') throw usageError(`${option} requires a value`);
|
|
615
715
|
return value;
|
|
616
716
|
}
|
|
617
717
|
|
|
618
718
|
function parsePositiveInt(value, option) {
|
|
619
719
|
const parsed = Number.parseInt(value, 10);
|
|
620
|
-
if (!Number.isInteger(parsed) || parsed < 1) throw
|
|
720
|
+
if (!Number.isInteger(parsed) || parsed < 1) throw usageError(`${option} must be a positive integer`);
|
|
621
721
|
return parsed;
|
|
622
722
|
}
|
|
623
723
|
|
|
624
724
|
function parsePositiveNumber(value, option) {
|
|
625
725
|
const parsed = Number.parseFloat(value);
|
|
626
|
-
if (!Number.isFinite(parsed) || parsed <= 0) throw
|
|
726
|
+
if (!Number.isFinite(parsed) || parsed <= 0) throw usageError(`${option} must be a positive number`);
|
|
627
727
|
return parsed;
|
|
628
728
|
}
|
|
629
729
|
|
|
630
730
|
function parseNonNegativeInt(value, option) {
|
|
631
731
|
const parsed = Number.parseInt(value, 10);
|
|
632
|
-
if (!Number.isInteger(parsed) || parsed < 0) throw
|
|
732
|
+
if (!Number.isInteger(parsed) || parsed < 0) throw usageError(`${option} must be a non-negative integer`);
|
|
633
733
|
return parsed;
|
|
634
734
|
}
|
|
635
735
|
|
|
@@ -639,7 +739,7 @@ function parsePositiveIntList(value, option) {
|
|
|
639
739
|
.map((item) => item.trim())
|
|
640
740
|
.filter(Boolean)
|
|
641
741
|
.map((item) => parsePositiveInt(item, option));
|
|
642
|
-
if (parsed.length === 0) throw
|
|
742
|
+
if (parsed.length === 0) throw usageError(`${option} requires at least one positive integer`);
|
|
643
743
|
return parsed;
|
|
644
744
|
}
|
|
645
745
|
|
|
@@ -669,27 +769,28 @@ function resolveProgress(parsed) {
|
|
|
669
769
|
function helpText() {
|
|
670
770
|
return `fm-bench ${packageJson.version}
|
|
671
771
|
|
|
672
|
-
Dynamic benchmark CLI for Apple's fm command on macOS
|
|
772
|
+
Dynamic benchmark CLI for Apple's fm command on macOS ${MIN_SUPPORTED_MACOS}+.
|
|
673
773
|
|
|
674
774
|
Usage:
|
|
675
775
|
fm-bench [run] [options]
|
|
676
776
|
fm-bench models [options]
|
|
677
777
|
fm-bench compare <before.json> <after.json> [options]
|
|
678
778
|
fm-bench history [dir] [options]
|
|
679
|
-
fm-bench validate <report.json> [more...]
|
|
779
|
+
fm-bench validate <report.json> [more...] [options]
|
|
680
780
|
fm-bench export <report.json> [-o report.html]
|
|
681
781
|
fm-bench legend [options]
|
|
682
782
|
fm-bench doctor [options]
|
|
683
783
|
|
|
684
784
|
Commands:
|
|
685
785
|
run Benchmark discovered or selected fm models
|
|
686
|
-
models List discovered models and
|
|
786
|
+
models List discovered models, availability, and fm capabilities
|
|
687
787
|
compare Compare two saved JSON reports and show metric deltas
|
|
688
788
|
history Show a trend table from all fm-bench JSON reports in a directory
|
|
689
789
|
validate Verify report JSON structure (schema v1)
|
|
690
790
|
export Render a shareable standalone HTML report from JSON
|
|
691
|
-
legend Explain every terminal table column and
|
|
692
|
-
doctor Check Node, macOS, fm, and model availability
|
|
791
|
+
legend Explain every terminal table column, color rule, and metric source
|
|
792
|
+
doctor Check Node, macOS, fm capabilities, and model availability
|
|
793
|
+
metrics Alias for legend
|
|
693
794
|
|
|
694
795
|
Run options:
|
|
695
796
|
-m, --models <list> Models to benchmark, comma-separated or repeated
|
|
@@ -744,15 +845,32 @@ Compare:
|
|
|
744
845
|
|
|
745
846
|
Environment:
|
|
746
847
|
--fm-bin <path> fm binary to execute (default: FM_BIN or fm)
|
|
848
|
+
-- Treat the rest of the line as the prompt
|
|
747
849
|
-h, --help Show this help
|
|
748
850
|
--version Print version
|
|
749
851
|
|
|
852
|
+
Machine-readable output:
|
|
853
|
+
--json and --csv write only data to stdout; progress and diagnostics go to stderr.
|
|
854
|
+
"validate --json" prints { ok, files }; "doctor --json" prints the full check list.
|
|
855
|
+
|
|
856
|
+
Exit codes:
|
|
857
|
+
0 success
|
|
858
|
+
1 operational failure (failed runs with --ci, invalid reports, fm errors)
|
|
859
|
+
2 usage or environment error (bad flags, missing arguments, unsupported macOS, fm not found)
|
|
860
|
+
|
|
861
|
+
Capability detection:
|
|
862
|
+
fm-bench probes "fm --help" and "fm respond --help" once per run. Metrics the
|
|
863
|
+
installed fm cannot supply are reported as unavailable instead of being
|
|
864
|
+
guessed, and unsupported models are refused before any benchmark starts.
|
|
865
|
+
|
|
750
866
|
Examples:
|
|
751
867
|
fm-bench
|
|
752
|
-
fm-bench --models system
|
|
868
|
+
fm-bench --models system --runs 3 --profile stress
|
|
753
869
|
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
754
870
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
755
871
|
fm-bench --profile reasoning --runs 5 --retry 2
|
|
872
|
+
fm-bench models
|
|
873
|
+
fm-bench doctor --json
|
|
756
874
|
fm-bench compare before.json after.json
|
|
757
875
|
fm-bench compare before.json after.json --json
|
|
758
876
|
fm-bench compare before.json after.json --strict
|
|
@@ -762,7 +880,5 @@ Examples:
|
|
|
762
880
|
fm-bench history ./reports
|
|
763
881
|
fm-bench history ./reports --json
|
|
764
882
|
fm-bench legend
|
|
765
|
-
fm-bench models
|
|
766
|
-
fm-bench doctor
|
|
767
883
|
`;
|
|
768
884
|
}
|