fm-bench 0.6.3 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +33 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +19 -2
- package/package.json +5 -4
- package/src/bench.js +129 -37
- package/src/capabilities.js +196 -0
- package/src/cli.js +153 -68
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +110 -106
- package/src/history.js +2 -1
- package/src/macos.js +2 -1
- package/src/metrics.js +212 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/src/cli.js
CHANGED
|
@@ -3,6 +3,7 @@ import { createRequire } from 'node:module';
|
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { diffReports, renderCompareReport } from './compare.js';
|
|
5
5
|
import { renderHtmlReport } from './export.js';
|
|
6
|
+
import { formatCapabilitySummary } from './metrics.js';
|
|
6
7
|
import { validateReport } from './schema.js';
|
|
7
8
|
import { loadHistory, renderHistoryReport } from './history.js';
|
|
8
9
|
import { detectMacosVersion, evaluateMacosSupport, formatMacosRequirementError, MIN_SUPPORTED_MACOS, parseMacosVersion } from './macos.js';
|
|
@@ -10,11 +11,25 @@ import { runProcess } from './process.js';
|
|
|
10
11
|
import { createProgress } from './progress.js';
|
|
11
12
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
12
13
|
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
13
|
-
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend,
|
|
14
|
+
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend, renderModelsReport } from './table.js';
|
|
14
15
|
|
|
15
16
|
const require = createRequire(import.meta.url);
|
|
16
17
|
const packageJson = require('../package.json');
|
|
17
18
|
|
|
19
|
+
/** Usage / environment error: bad flags, missing arguments, unsupported host. */
|
|
20
|
+
function usageError(message) {
|
|
21
|
+
const error = new Error(message);
|
|
22
|
+
error.exitCode = 2;
|
|
23
|
+
return error;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** Operational failure: invalid report data, failed benchmark gate. */
|
|
27
|
+
function operationalError(message) {
|
|
28
|
+
const error = new Error(message);
|
|
29
|
+
error.exitCode = 1;
|
|
30
|
+
return error;
|
|
31
|
+
}
|
|
32
|
+
|
|
18
33
|
export async function runCli(argv = process.argv.slice(2), env = {}) {
|
|
19
34
|
const parsed = parseArgs(argv);
|
|
20
35
|
|
|
@@ -69,9 +84,14 @@ export async function runCli(argv = process.argv.slice(2), env = {}) {
|
|
|
69
84
|
if (parsed.command === 'models') {
|
|
70
85
|
const inspection = await inspectModels(parsed);
|
|
71
86
|
if (parsed.format === 'json') {
|
|
87
|
+
// Shape stays a plain array so existing automation keeps working;
|
|
88
|
+
// full capability detail lives in `doctor --json` and in report payloads.
|
|
72
89
|
console.log(JSON.stringify(inspection.models, null, 2));
|
|
73
90
|
} else {
|
|
74
|
-
console.log(
|
|
91
|
+
console.log(renderModelsReport(inspection.models, {
|
|
92
|
+
...renderOptions(parsed),
|
|
93
|
+
capabilities: inspection.capabilities
|
|
94
|
+
}));
|
|
75
95
|
}
|
|
76
96
|
return;
|
|
77
97
|
}
|
|
@@ -148,9 +168,7 @@ export async function runCli(argv = process.argv.slice(2), env = {}) {
|
|
|
148
168
|
if (!ciResult.passed) {
|
|
149
169
|
const reasons = ciResult.reasons.join('; ');
|
|
150
170
|
console.error(`fm-bench ci: FAIL — ${reasons}`);
|
|
151
|
-
|
|
152
|
-
error.exitCode = 1;
|
|
153
|
-
throw error;
|
|
171
|
+
throw operationalError(`CI checks failed: ${reasons}`);
|
|
154
172
|
}
|
|
155
173
|
console.error(`fm-bench ci: PASS`);
|
|
156
174
|
}
|
|
@@ -164,14 +182,18 @@ const SKIP_MACOS_GATE = new Set(['compare', 'history', 'validate', 'export', 'le
|
|
|
164
182
|
async function assertSupportedMacos(parsed, env = {}) {
|
|
165
183
|
if (SKIP_MACOS_GATE.has(parsed.command)) return;
|
|
166
184
|
|
|
185
|
+
// The gate guards the default fm discovery path, because Apple only ships the
|
|
186
|
+
// CLI from macOS 27. An explicitly configured binary is honoured on any host;
|
|
187
|
+
// the capability probe below still fails with exit 2 if it is unusable.
|
|
188
|
+
if (parsed.fmBin || env.FM_BIN || process.env.FM_BIN) return;
|
|
189
|
+
|
|
167
190
|
const evaluation = evaluateMacosSupport(
|
|
168
191
|
env.platform ?? process.platform,
|
|
169
192
|
parseMacosVersion(await detectMacosVersion(env))
|
|
170
193
|
);
|
|
171
194
|
if (evaluation.supported) return;
|
|
172
195
|
|
|
173
|
-
const error =
|
|
174
|
-
error.exitCode = 2;
|
|
196
|
+
const error = usageError(formatMacosRequirementError(evaluation));
|
|
175
197
|
throw error;
|
|
176
198
|
}
|
|
177
199
|
|
|
@@ -315,7 +337,7 @@ export function parseArgs(argv) {
|
|
|
315
337
|
case '--profile':
|
|
316
338
|
options.profile = requireValue(arg, args);
|
|
317
339
|
if (!['quick', 'standard', 'interactive', 'throughput', 'client', 'stress', 'reasoning', 'coding', 'creative'].includes(options.profile)) {
|
|
318
|
-
throw
|
|
340
|
+
throw usageError('--profile must be one of: quick, standard, interactive, throughput, client, stress, reasoning, coding, creative');
|
|
319
341
|
}
|
|
320
342
|
break;
|
|
321
343
|
case '-i':
|
|
@@ -352,7 +374,7 @@ export function parseArgs(argv) {
|
|
|
352
374
|
case '--format':
|
|
353
375
|
options.format = requireValue(arg, args);
|
|
354
376
|
if (!['table', 'json', 'csv'].includes(options.format)) {
|
|
355
|
-
throw
|
|
377
|
+
throw usageError('--format must be one of: table, json, csv');
|
|
356
378
|
}
|
|
357
379
|
break;
|
|
358
380
|
case '--ascii':
|
|
@@ -425,7 +447,7 @@ export function parseArgs(argv) {
|
|
|
425
447
|
break;
|
|
426
448
|
default:
|
|
427
449
|
if (arg.startsWith('-')) {
|
|
428
|
-
throw
|
|
450
|
+
throw usageError(`Unknown option: ${arg}`);
|
|
429
451
|
}
|
|
430
452
|
if (options.command === 'compare') {
|
|
431
453
|
options.compareFiles.push(arg);
|
|
@@ -464,30 +486,27 @@ async function runHistory(options, renderOpts) {
|
|
|
464
486
|
async function runCompare(options, renderOpts) {
|
|
465
487
|
const files = options.compareFiles;
|
|
466
488
|
if (files.length < 2) {
|
|
467
|
-
throw
|
|
489
|
+
throw usageError('compare requires two JSON report files: fm-bench compare before.json after.json');
|
|
468
490
|
}
|
|
469
491
|
if (files.length > 2) {
|
|
470
|
-
throw
|
|
492
|
+
throw usageError('compare accepts exactly two JSON report files');
|
|
471
493
|
}
|
|
472
494
|
|
|
473
495
|
const [beforePath, afterPath] = files;
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
fs.readFile(afterPath, 'utf8')
|
|
477
|
-
]);
|
|
478
|
-
|
|
479
|
-
let before, after;
|
|
480
|
-
try {
|
|
481
|
-
before = JSON.parse(beforeText);
|
|
482
|
-
} catch {
|
|
483
|
-
throw new Error(`Cannot parse ${beforePath} as JSON`);
|
|
484
|
-
}
|
|
496
|
+
let beforeText;
|
|
497
|
+
let afterText;
|
|
485
498
|
try {
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
499
|
+
[beforeText, afterText] = await Promise.all([
|
|
500
|
+
fs.readFile(beforePath, 'utf8'),
|
|
501
|
+
fs.readFile(afterPath, 'utf8')
|
|
502
|
+
]);
|
|
503
|
+
} catch (error) {
|
|
504
|
+
throw operationalError(`Cannot read report: ${error.message}`);
|
|
489
505
|
}
|
|
490
506
|
|
|
507
|
+
const before = parseReportJson(beforeText, beforePath);
|
|
508
|
+
const after = parseReportJson(afterText, afterPath);
|
|
509
|
+
|
|
491
510
|
const diff = diffReports(before, after);
|
|
492
511
|
|
|
493
512
|
if (options.format === 'json') {
|
|
@@ -502,64 +521,77 @@ async function runCompare(options, renderOpts) {
|
|
|
502
521
|
}
|
|
503
522
|
|
|
504
523
|
if (options.strictCompare && diff.compatibility && !diff.compatibility.suiteMatch) {
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
524
|
+
throw usageError('compare: benchmark suites differ (--strict)');
|
|
525
|
+
}
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
function parseReportJson(text, filePath) {
|
|
529
|
+
try {
|
|
530
|
+
return JSON.parse(text);
|
|
531
|
+
} catch {
|
|
532
|
+
throw operationalError(`Cannot parse ${filePath} as JSON`);
|
|
508
533
|
}
|
|
509
534
|
}
|
|
510
535
|
|
|
511
536
|
async function runValidate(options) {
|
|
512
537
|
const files = options.validateFiles;
|
|
513
538
|
if (files.length === 0) {
|
|
514
|
-
throw
|
|
539
|
+
throw usageError('validate requires at least one JSON report: fm-bench validate report.json');
|
|
515
540
|
}
|
|
516
541
|
|
|
517
|
-
|
|
542
|
+
const results = [];
|
|
518
543
|
for (const filePath of files) {
|
|
519
544
|
let parsed;
|
|
520
545
|
try {
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
546
|
+
parsed = JSON.parse(await fs.readFile(filePath, 'utf8'));
|
|
547
|
+
} catch (error) {
|
|
548
|
+
results.push({ file: filePath, ok: false, errors: [error.code === 'ENOENT'
|
|
549
|
+
? 'file not found'
|
|
550
|
+
: 'cannot read or parse JSON'] });
|
|
526
551
|
continue;
|
|
527
552
|
}
|
|
528
553
|
const result = validateReport(parsed);
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
554
|
+
results.push(result.ok
|
|
555
|
+
? { file: filePath, ok: true, schema: result.report.schemaVersion ?? 'legacy', id: result.report.reportId ?? null }
|
|
556
|
+
: { file: filePath, ok: false, errors: result.errors });
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
if (options.format === 'json') {
|
|
560
|
+
console.log(JSON.stringify({ ok: results.every((item) => item.ok), files: results }, null, 2));
|
|
561
|
+
} else {
|
|
562
|
+
for (const item of results) {
|
|
563
|
+
if (item.ok) {
|
|
564
|
+
console.log(`ok ${item.file} schema=${item.schema} id=${item.id ?? '—'}`);
|
|
565
|
+
} else {
|
|
566
|
+
console.error(`invalid ${item.file} ${item.errors.join('; ')}`);
|
|
567
|
+
}
|
|
536
568
|
}
|
|
537
569
|
}
|
|
538
570
|
|
|
571
|
+
const failed = results.filter((item) => !item.ok).length;
|
|
539
572
|
if (failed > 0) {
|
|
540
|
-
|
|
541
|
-
error.exitCode = 1;
|
|
542
|
-
throw error;
|
|
573
|
+
throw operationalError(`${failed} report(s) failed validation`);
|
|
543
574
|
}
|
|
544
575
|
}
|
|
545
576
|
|
|
546
577
|
async function runExport(options) {
|
|
547
578
|
const files = options.validateFiles;
|
|
548
579
|
if (files.length === 0) {
|
|
549
|
-
throw
|
|
580
|
+
throw usageError('export requires a JSON report: fm-bench export report.json [-o out.html]');
|
|
550
581
|
}
|
|
551
582
|
|
|
552
583
|
const filePath = files[0];
|
|
553
|
-
const text = await fs.readFile(filePath, 'utf8');
|
|
554
584
|
let report;
|
|
555
585
|
try {
|
|
556
|
-
report = JSON.parse(
|
|
557
|
-
} catch {
|
|
558
|
-
throw
|
|
586
|
+
report = JSON.parse(await fs.readFile(filePath, 'utf8'));
|
|
587
|
+
} catch (error) {
|
|
588
|
+
throw error.code === 'ENOENT'
|
|
589
|
+
? operationalError(`Cannot read ${filePath}: file not found`)
|
|
590
|
+
: operationalError(`Cannot parse ${filePath} as JSON`);
|
|
559
591
|
}
|
|
560
592
|
const validation = validateReport(report);
|
|
561
593
|
if (!validation.ok) {
|
|
562
|
-
throw
|
|
594
|
+
throw operationalError(`Not a valid fm-bench report: ${validation.errors.join('; ')}`);
|
|
563
595
|
}
|
|
564
596
|
|
|
565
597
|
const html = renderHtmlReport(validation.report);
|
|
@@ -572,6 +604,7 @@ async function runExport(options) {
|
|
|
572
604
|
}
|
|
573
605
|
|
|
574
606
|
async function runDoctor(options) {
|
|
607
|
+
const json = options.format === 'json';
|
|
575
608
|
const checks = [];
|
|
576
609
|
checks.push(['node', process.version, true]);
|
|
577
610
|
checks.push(['platform', `${process.platform}/${process.arch}`, process.platform === 'darwin']);
|
|
@@ -621,10 +654,17 @@ async function runDoctor(options) {
|
|
|
621
654
|
}
|
|
622
655
|
|
|
623
656
|
let models = [];
|
|
657
|
+
let capabilities = null;
|
|
624
658
|
try {
|
|
625
659
|
const inspection = await inspectModels(options);
|
|
626
660
|
models = inspection.models;
|
|
627
|
-
|
|
661
|
+
capabilities = inspection.capabilities;
|
|
662
|
+
checks.push(['fm', inspection.fmBin, capabilities.ok]);
|
|
663
|
+
if (capabilities.digest) checks.push(['fm help digest', capabilities.digest, true]);
|
|
664
|
+
checks.push(['fm commands', capabilities.commands.join(', ') || 'none found', capabilities.commands.length > 0]);
|
|
665
|
+
checks.push(['fm token counting', capabilities.features.tokenCounting ? `yes (${capabilities.features.tokenCountCommand})` : 'no', capabilities.features.tokenCounting]);
|
|
666
|
+
checks.push(['fm streaming', capabilities.features.streaming ? 'yes' : 'no', capabilities.features.streaming]);
|
|
667
|
+
checks.push(['fm quota', capabilities.features.quota ? 'yes' : 'no (not exposed by this build)', true]);
|
|
628
668
|
for (const model of inspection.models) {
|
|
629
669
|
checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
|
|
630
670
|
}
|
|
@@ -632,35 +672,64 @@ async function runDoctor(options) {
|
|
|
632
672
|
checks.push(['fm', error.message || String(error), false]);
|
|
633
673
|
}
|
|
634
674
|
|
|
635
|
-
const
|
|
636
|
-
|
|
675
|
+
const payload = {
|
|
676
|
+
checks: checks.map(([name, detail, ok]) => ({ name, detail: String(detail), ok })),
|
|
677
|
+
models,
|
|
678
|
+
capabilities: capabilities ? describeCapabilities(capabilities) : null
|
|
679
|
+
};
|
|
680
|
+
|
|
681
|
+
if (json) {
|
|
682
|
+
console.log(JSON.stringify(payload, null, 2));
|
|
683
|
+
} else {
|
|
684
|
+
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(16)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
|
|
685
|
+
console.log(lines.join('\n'));
|
|
686
|
+
if (capabilities) {
|
|
687
|
+
console.log('');
|
|
688
|
+
console.log(`fm capabilities: ${formatCapabilitySummary(capabilities)}`);
|
|
689
|
+
for (const warning of capabilities.warnings) {
|
|
690
|
+
console.log(`limit: ${warning}`);
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
}
|
|
637
694
|
|
|
638
695
|
if (options.out) {
|
|
639
|
-
await fs.writeFile(options.out, `${JSON.stringify(
|
|
696
|
+
await fs.writeFile(options.out, `${JSON.stringify(payload, null, 2)}\n`, 'utf8');
|
|
640
697
|
}
|
|
641
698
|
}
|
|
642
699
|
|
|
700
|
+
function describeCapabilities(capabilities) {
|
|
701
|
+
return {
|
|
702
|
+
bin: capabilities.bin,
|
|
703
|
+
ok: capabilities.ok,
|
|
704
|
+
digest: capabilities.digest,
|
|
705
|
+
commands: capabilities.commands,
|
|
706
|
+
models: capabilities.models,
|
|
707
|
+
features: capabilities.features,
|
|
708
|
+
warnings: capabilities.warnings
|
|
709
|
+
};
|
|
710
|
+
}
|
|
711
|
+
|
|
643
712
|
function requireValue(option, args) {
|
|
644
713
|
const value = args.shift();
|
|
645
|
-
if (value == null || value === '') throw
|
|
714
|
+
if (value == null || value === '') throw usageError(`${option} requires a value`);
|
|
646
715
|
return value;
|
|
647
716
|
}
|
|
648
717
|
|
|
649
718
|
function parsePositiveInt(value, option) {
|
|
650
719
|
const parsed = Number.parseInt(value, 10);
|
|
651
|
-
if (!Number.isInteger(parsed) || parsed < 1) throw
|
|
720
|
+
if (!Number.isInteger(parsed) || parsed < 1) throw usageError(`${option} must be a positive integer`);
|
|
652
721
|
return parsed;
|
|
653
722
|
}
|
|
654
723
|
|
|
655
724
|
function parsePositiveNumber(value, option) {
|
|
656
725
|
const parsed = Number.parseFloat(value);
|
|
657
|
-
if (!Number.isFinite(parsed) || parsed <= 0) throw
|
|
726
|
+
if (!Number.isFinite(parsed) || parsed <= 0) throw usageError(`${option} must be a positive number`);
|
|
658
727
|
return parsed;
|
|
659
728
|
}
|
|
660
729
|
|
|
661
730
|
function parseNonNegativeInt(value, option) {
|
|
662
731
|
const parsed = Number.parseInt(value, 10);
|
|
663
|
-
if (!Number.isInteger(parsed) || parsed < 0) throw
|
|
732
|
+
if (!Number.isInteger(parsed) || parsed < 0) throw usageError(`${option} must be a non-negative integer`);
|
|
664
733
|
return parsed;
|
|
665
734
|
}
|
|
666
735
|
|
|
@@ -670,7 +739,7 @@ function parsePositiveIntList(value, option) {
|
|
|
670
739
|
.map((item) => item.trim())
|
|
671
740
|
.filter(Boolean)
|
|
672
741
|
.map((item) => parsePositiveInt(item, option));
|
|
673
|
-
if (parsed.length === 0) throw
|
|
742
|
+
if (parsed.length === 0) throw usageError(`${option} requires at least one positive integer`);
|
|
674
743
|
return parsed;
|
|
675
744
|
}
|
|
676
745
|
|
|
@@ -707,20 +776,21 @@ Usage:
|
|
|
707
776
|
fm-bench models [options]
|
|
708
777
|
fm-bench compare <before.json> <after.json> [options]
|
|
709
778
|
fm-bench history [dir] [options]
|
|
710
|
-
fm-bench validate <report.json> [more...]
|
|
779
|
+
fm-bench validate <report.json> [more...] [options]
|
|
711
780
|
fm-bench export <report.json> [-o report.html]
|
|
712
781
|
fm-bench legend [options]
|
|
713
782
|
fm-bench doctor [options]
|
|
714
783
|
|
|
715
784
|
Commands:
|
|
716
785
|
run Benchmark discovered or selected fm models
|
|
717
|
-
models List discovered models and
|
|
786
|
+
models List discovered models, availability, and fm capabilities
|
|
718
787
|
compare Compare two saved JSON reports and show metric deltas
|
|
719
788
|
history Show a trend table from all fm-bench JSON reports in a directory
|
|
720
789
|
validate Verify report JSON structure (schema v1)
|
|
721
790
|
export Render a shareable standalone HTML report from JSON
|
|
722
|
-
legend Explain every terminal table column and
|
|
723
|
-
doctor Check Node, macOS, fm, and model availability
|
|
791
|
+
legend Explain every terminal table column, color rule, and metric source
|
|
792
|
+
doctor Check Node, macOS, fm capabilities, and model availability
|
|
793
|
+
metrics Alias for legend
|
|
724
794
|
|
|
725
795
|
Run options:
|
|
726
796
|
-m, --models <list> Models to benchmark, comma-separated or repeated
|
|
@@ -775,15 +845,32 @@ Compare:
|
|
|
775
845
|
|
|
776
846
|
Environment:
|
|
777
847
|
--fm-bin <path> fm binary to execute (default: FM_BIN or fm)
|
|
848
|
+
-- Treat the rest of the line as the prompt
|
|
778
849
|
-h, --help Show this help
|
|
779
850
|
--version Print version
|
|
780
851
|
|
|
852
|
+
Machine-readable output:
|
|
853
|
+
--json and --csv write only data to stdout; progress and diagnostics go to stderr.
|
|
854
|
+
"validate --json" prints { ok, files }; "doctor --json" prints the full check list.
|
|
855
|
+
|
|
856
|
+
Exit codes:
|
|
857
|
+
0 success
|
|
858
|
+
1 operational failure (failed runs with --ci, invalid reports, fm errors)
|
|
859
|
+
2 usage or environment error (bad flags, missing arguments, unsupported macOS, fm not found)
|
|
860
|
+
|
|
861
|
+
Capability detection:
|
|
862
|
+
fm-bench probes "fm --help" and "fm respond --help" once per run. Metrics the
|
|
863
|
+
installed fm cannot supply are reported as unavailable instead of being
|
|
864
|
+
guessed, and unsupported models are refused before any benchmark starts.
|
|
865
|
+
|
|
781
866
|
Examples:
|
|
782
867
|
fm-bench
|
|
783
|
-
fm-bench --models system
|
|
868
|
+
fm-bench --models system --runs 3 --profile stress
|
|
784
869
|
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
785
870
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
786
871
|
fm-bench --profile reasoning --runs 5 --retry 2
|
|
872
|
+
fm-bench models
|
|
873
|
+
fm-bench doctor --json
|
|
787
874
|
fm-bench compare before.json after.json
|
|
788
875
|
fm-bench compare before.json after.json --json
|
|
789
876
|
fm-bench compare before.json after.json --strict
|
|
@@ -793,7 +880,5 @@ Examples:
|
|
|
793
880
|
fm-bench history ./reports
|
|
794
881
|
fm-bench history ./reports --json
|
|
795
882
|
fm-bench legend
|
|
796
|
-
fm-bench models
|
|
797
|
-
fm-bench doctor
|
|
798
883
|
`;
|
|
799
884
|
}
|
package/src/compare.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { stripAnsi } from './ansi.js';
|
|
1
2
|
import { compareCompatibility } from './schema.js';
|
|
2
3
|
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
3
4
|
|
|
@@ -266,12 +267,19 @@ function applyTone(text, tone) {
|
|
|
266
267
|
}
|
|
267
268
|
|
|
268
269
|
function pad(text, width) {
|
|
269
|
-
const
|
|
270
|
-
|
|
270
|
+
const str = safeCell(text);
|
|
271
|
+
const len = str.length;
|
|
272
|
+
return str + ' '.repeat(Math.max(0, width - len));
|
|
271
273
|
}
|
|
272
274
|
|
|
273
275
|
function fitCell(text, width) {
|
|
274
|
-
const str =
|
|
276
|
+
const str = safeCell(text);
|
|
275
277
|
if (str.length <= width) return str + ' '.repeat(width - str.length);
|
|
276
278
|
return `${str.slice(0, width - 1)}…`;
|
|
277
279
|
}
|
|
280
|
+
|
|
281
|
+
// Values can originate from report files that fm-bench did not write, so strip
|
|
282
|
+
// ANSI escapes and control characters before they reach the terminal.
|
|
283
|
+
function safeCell(text) {
|
|
284
|
+
return stripAnsi(String(text ?? '')).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
|
|
285
|
+
}
|
package/src/fm-help.js
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
// Pure parsers for `fm` help and status text.
|
|
2
|
+
//
|
|
3
|
+
// These functions take raw text and return plain data so they can be unit
|
|
4
|
+
// tested without spawning a process, and so the capability layer, the model
|
|
5
|
+
// discovery path, and the availability path agree on how `fm` output is read.
|
|
6
|
+
|
|
7
|
+
import { stripAnsi } from './ansi.js';
|
|
8
|
+
|
|
9
|
+
const SECTION_HEADER = /^\s*[A-Z][A-Z0-9 /-]+\s*$/;
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Extract models from the MODELS section of `fm --help`, falling back to a
|
|
13
|
+
* `--model` option list such as `Model to use (system, pcc)`.
|
|
14
|
+
* @param {string} helpText
|
|
15
|
+
* @returns {{ name: string, description: string }[]}
|
|
16
|
+
*/
|
|
17
|
+
export function parseModelsFromHelp(helpText = '') {
|
|
18
|
+
const lines = stripAnsi(helpText).split(/\r?\n/);
|
|
19
|
+
const models = new Map();
|
|
20
|
+
let inModels = false;
|
|
21
|
+
|
|
22
|
+
for (const line of lines) {
|
|
23
|
+
if (/^\s*MODELS\s*$/.test(line)) {
|
|
24
|
+
inModels = true;
|
|
25
|
+
continue;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
if (inModels && SECTION_HEADER.test(line)) {
|
|
29
|
+
inModels = false;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
if (inModels) {
|
|
33
|
+
const match = line.match(/^\s*([A-Za-z0-9._:-]+)\s{2,}(.+?)\s*$/);
|
|
34
|
+
if (match) {
|
|
35
|
+
const name = match[1];
|
|
36
|
+
const description = match[2].replace(/\s*\(default\)\s*$/, '').trim();
|
|
37
|
+
const existing = models.get(name);
|
|
38
|
+
// The MODELS section is authoritative over a bare option list.
|
|
39
|
+
if (existing) {
|
|
40
|
+
if (!existing.description) existing.description = description;
|
|
41
|
+
} else {
|
|
42
|
+
models.set(name, { name, description });
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
if (/(^|\s)--model\b/.test(line)) {
|
|
48
|
+
const optionMatch = line.match(/\(([^)]*)\)/);
|
|
49
|
+
if (optionMatch) {
|
|
50
|
+
for (const raw of optionMatch[1].split(',')) {
|
|
51
|
+
const name = raw.trim();
|
|
52
|
+
if (/^[A-Za-z0-9._:-]+$/.test(name) && !models.has(name)) {
|
|
53
|
+
models.set(name, { name, description: '' });
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
return [...models.values()];
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Read model names and availability out of `fm available` (no model filter).
|
|
65
|
+
* @param {string} output
|
|
66
|
+
* @returns {{ name: string, available: boolean }[]}
|
|
67
|
+
*/
|
|
68
|
+
export function parseAvailabilityList(output = '') {
|
|
69
|
+
const models = new Map();
|
|
70
|
+
for (const line of stripAnsi(output).split(/\r?\n/)) {
|
|
71
|
+
const match = line.match(/^\s*([A-Za-z][A-Za-z0-9._-]*)(?:\s+model)?\s+(?:is\s+)?(available|unavailable|not available)\b/i);
|
|
72
|
+
if (!match) continue;
|
|
73
|
+
const name = match[1].toLowerCase();
|
|
74
|
+
models.set(name, { name, available: /^available$/i.test(match[2]) });
|
|
75
|
+
}
|
|
76
|
+
return [...models.values()];
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Interpret `fm available --model <model>` output.
|
|
81
|
+
*
|
|
82
|
+
* `fm` reports the model plus the word "available" on success. Anything with
|
|
83
|
+
* an explicit error/unavailable marker, or a non-zero exit code, counts as
|
|
84
|
+
* unavailable so a partially supported host is never reported as working.
|
|
85
|
+
*
|
|
86
|
+
* @param {string} model
|
|
87
|
+
* @param {string} output combined stdout+stderr
|
|
88
|
+
* @param {number|null} code process exit code
|
|
89
|
+
*/
|
|
90
|
+
export function parseAvailabilityOutput(model, output = '', code = null) {
|
|
91
|
+
const clean = stripAnsi(output).trim();
|
|
92
|
+
const lower = clean.toLowerCase();
|
|
93
|
+
const modelLower = String(model).toLowerCase();
|
|
94
|
+
const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b|\bis invalid for\b/.test(lower);
|
|
95
|
+
const modelPattern = escapeRegExp(modelLower);
|
|
96
|
+
const hasAvailable = new RegExp(`\\b${modelPattern}\\b[\\s\\S]{0,80}\\bavailable\\b`).test(lower)
|
|
97
|
+
|| new RegExp(`\\bavailable\\b[\\s\\S]{0,80}\\b${modelPattern}\\b`).test(lower);
|
|
98
|
+
|
|
99
|
+
return {
|
|
100
|
+
model,
|
|
101
|
+
available: code === 0 && hasAvailable && !hasError,
|
|
102
|
+
raw: clean,
|
|
103
|
+
reason: hasError ? firstLine(clean) : ''
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Collapse `fm` diagnostics into one actionable line. `fm` writes multi-line
|
|
109
|
+
* usage blocks for argument errors; benchmark output only needs the cause.
|
|
110
|
+
* @param {string} text
|
|
111
|
+
*/
|
|
112
|
+
export function firstLine(text = '') {
|
|
113
|
+
const clean = stripAnsi(text).replace(/\s+/g, ' ').trim();
|
|
114
|
+
if (!clean) return '';
|
|
115
|
+
const sentences = clean.split(/(?<=\.)\s+(?=[A-Z])/);
|
|
116
|
+
const head = sentences.find((part) => /error|invalid|unavailable|not supported|failed/i.test(part)) || sentences[0];
|
|
117
|
+
return head.replace(/^Error:\s*/i, '').trim();
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* True when an `fm` diagnostic means "this build does not have that model"
|
|
122
|
+
* rather than "the model exists but is currently unusable".
|
|
123
|
+
* @param {string} text
|
|
124
|
+
*/
|
|
125
|
+
export function isUnsupportedModelError(text = '') {
|
|
126
|
+
return /is invalid for '--model|unknown model|no such model|unrecognized model/i.test(stripAnsi(text));
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function escapeRegExp(value) {
|
|
130
|
+
return String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
131
|
+
}
|