fm-bench 0.5.2 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -10
- package/docs/methodology.md +4 -0
- package/docs/releasing.md +35 -0
- package/docs/report-format.md +52 -0
- package/package.json +2 -2
- package/src/bench.js +3 -2
- package/src/cli.js +121 -7
- package/src/compare.js +27 -3
- package/src/export.js +145 -0
- package/src/fm.js +46 -1
- package/src/history.js +14 -3
- package/src/report.js +9 -3
- package/src/schema.js +158 -0
package/README.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# fm-bench
|
|
2
2
|
|
|
3
|
+
[](https://github.com/devinoldenburg/fm-bench/actions/workflows/ci.yml)
|
|
4
|
+
|
|
3
5
|
Benchmark Apple's `fm` command on macOS 27+.
|
|
4
6
|
|
|
5
7
|
Measure latency, throughput, streaming smoothness, stability, and goodput across Apple Foundation Models — with repeatable prompt suites and JSON/CSV reports for automation.
|
|
@@ -28,17 +30,18 @@ fm-bench
|
|
|
28
30
|
One command discovers your models, runs the standard prompt suite, and prints a full benchmark report:
|
|
29
31
|
|
|
30
32
|
```text
|
|
31
|
-
fm-bench 0.
|
|
32
|
-
prompts
|
|
33
|
+
fm-bench 0.6.0 | darwin/arm64 | fm
|
|
34
|
+
prompts 3 | runs 5 | concurrency 1 | stream on | measured 15 | failed 0 | skipped 0 | elapsed 38.20s
|
|
33
35
|
|
|
34
36
|
┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
|
|
35
37
|
│ C │ MODEL │ STATUS │ OK/RUNS │ SUCC │ GOOD │ GOOD RPS │ TTFT │ E2E P95 │ SYS │ CV │
|
|
36
38
|
├───┼────────┼────────┼─────────┼──────┼──────┼──────────┼──────┼──────────┼─────┼─────┤
|
|
37
39
|
│ 1 │ system │ ok │ 15/15 │ 100% │ 93% │ 0.4 │ 318ms│ 3.20s │ 42 │ 12% │
|
|
38
|
-
│ 2 │ system │ ok │ 15/15 │ 100% │ 80% │ 0.7 │ 501ms│ 4.40s │ 68 │ 21% │
|
|
39
40
|
└───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
|
|
40
41
|
```
|
|
41
42
|
|
|
43
|
+
Default profile is `standard` (3 prompts). Use `--runs 5`, `--sweep-concurrency 1,2`, or `--profile client` for heavier suites.
|
|
44
|
+
|
|
42
45
|
Wide terminals add TTFT P95, TPOT, decode/prefill throughput, chunk-gap smoothness, and 95% CI columns. Narrow terminals switch to compact model cards automatically.
|
|
43
46
|
|
|
44
47
|
## Install
|
|
@@ -68,8 +71,10 @@ fm-bench doctor # verify your setup
|
|
|
68
71
|
|---------|-------------|
|
|
69
72
|
| `fm-bench` | Run the full benchmark (default) |
|
|
70
73
|
| `fm-bench models` | List discovered models, availability, and quota |
|
|
71
|
-
| `fm-bench compare <a.json> <b.json>` | Regression diff
|
|
72
|
-
| `fm-bench history [dir]` |
|
|
74
|
+
| `fm-bench compare <a.json> <b.json>` | Regression diff with suite/hardware warnings; `--strict` for CI |
|
|
75
|
+
| `fm-bench history [dir]` | Trend table from saved reports (sorted by time, tags visible) |
|
|
76
|
+
| `fm-bench validate <report.json>` | Verify report JSON (schema v1) before sharing |
|
|
77
|
+
| `fm-bench export <report.json>` | Standalone HTML report with embedded JSON |
|
|
73
78
|
| `fm-bench legend` | Definitions for every table column and color rule |
|
|
74
79
|
| `fm-bench doctor` | Environment check: Node, macOS, `fm`, CPU, memory, thermals, battery |
|
|
75
80
|
|
|
@@ -93,9 +98,10 @@ fm-bench --profile reasoning --runs 5
|
|
|
93
98
|
fm-bench --profile coding --runs 3 --histogram
|
|
94
99
|
|
|
95
100
|
# Archive runs and compare before/after a macOS update
|
|
96
|
-
fm-bench --output-dir reports/ --tag before-update
|
|
97
|
-
fm-bench --output-dir reports/ --tag after-update
|
|
98
|
-
fm-bench
|
|
101
|
+
fm-bench --output-dir reports/ --tag before-update --export-html
|
|
102
|
+
fm-bench --output-dir reports/ --tag after-update --export-html
|
|
103
|
+
fm-bench validate reports/*.json
|
|
104
|
+
fm-bench compare reports/fm-bench_*before*.json reports/fm-bench_*after*.json --strict
|
|
99
105
|
|
|
100
106
|
# Fail CI when SLOs regress
|
|
101
107
|
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
|
|
@@ -140,8 +146,9 @@ fm-bench --format csv --out bench.csv
|
|
|
140
146
|
| Flag | Description |
|
|
141
147
|
|------|-------------|
|
|
142
148
|
| `--json` / `--csv` | Output format (also `--format table\|json\|csv`) |
|
|
143
|
-
| `-o, --out <file>` | Save
|
|
144
|
-
| `--output-dir <dir>` | Auto-save
|
|
149
|
+
| `-o, --out <file>` | Save JSON (`.json`), per-run CSV (`.csv`), or shareable HTML (`.html`) |
|
|
150
|
+
| `--output-dir <dir>` | Auto-save timestamped JSON (and optional HTML with `--export-html`) |
|
|
151
|
+
| `--export-html` | With `--output-dir`, also write a matching `.html` report |
|
|
145
152
|
| `--tag <name>` | Label this run; repeatable; appears in payload and header |
|
|
146
153
|
| `--note <text>` | Freeform annotation in payload and header |
|
|
147
154
|
| `--histogram` | Print ASCII latency distribution chart after the report |
|
|
@@ -174,6 +181,14 @@ Nine built-in suites, choose the one that matches your use case:
|
|
|
174
181
|
| `coding` | 5 | Code review, refactoring, algorithms, system design |
|
|
175
182
|
| `creative` | 5 | Product copy, analogies, commit messages, docs |
|
|
176
183
|
|
|
184
|
+
## Sharing and comparing results
|
|
185
|
+
|
|
186
|
+
Reports from 0.6.0+ include **schema v1**: `reportId`, hardware fingerprint, and a **suite key** so you can tell if two JSON files used the same prompts and run settings. See [docs/report-format.md](docs/report-format.md).
|
|
187
|
+
|
|
188
|
+
- Share **HTML** with teammates who do not use the CLI: `fm-bench export bench.json -o bench.html`
|
|
189
|
+
- Gate uploads in CI: `fm-bench validate artifact.json`
|
|
190
|
+
- Apples-to-apples regressions: same `--profile` and `--runs`, then `fm-bench compare a.json b.json`
|
|
191
|
+
|
|
177
192
|
## Regression Tracking
|
|
178
193
|
|
|
179
194
|
Track performance across macOS updates, model changes, or hardware swaps:
|
package/docs/methodology.md
CHANGED
|
@@ -57,3 +57,7 @@ Client-side measurements include process startup, local queueing, model prefill,
|
|
|
57
57
|
Stream smoothness metrics use stdout chunk arrival times. A chunk can contain more than one token, and terminal or pipe buffering can affect chunk boundaries. Treat `second_chunk_ms` and `chunk_gap` as user-visible streaming diagnostics, not raw decoder telemetry.
|
|
58
58
|
|
|
59
59
|
For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput or client profiles, compare models at the same concurrency operating points, set SLOs that match your real UX budget, and save JSON reports for later analysis.
|
|
60
|
+
|
|
61
|
+
## Report artifacts
|
|
62
|
+
|
|
63
|
+
Saved JSON includes client-side environment metadata (hardware model, macOS build, `fm` help digest, power/thermal snapshot) so shared results remain interpretable on other machines. Use `fm-bench validate` before publishing and `fm-bench compare --strict` when you require identical prompt suites. Format details: [report-format.md](./report-format.md).
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# Releasing
|
|
2
|
+
|
|
3
|
+
`fm-bench` uses semver tags (`v*.*.*`). Pushing a tag runs the **Release** workflow: test, lint, npm publish (with provenance), and GitHub release notes.
|
|
4
|
+
|
|
5
|
+
## Prerequisites
|
|
6
|
+
|
|
7
|
+
- Repository secret **`NPM_TOKEN`**: npm automation token with publish access to `fm-bench`.
|
|
8
|
+
- **`main`** is green on CI.
|
|
9
|
+
|
|
10
|
+
## Option A — GitHub Actions (recommended)
|
|
11
|
+
|
|
12
|
+
1. Open **Actions → Version → Run workflow**.
|
|
13
|
+
2. Choose `patch`, `minor`, `major`, or an exact semver.
|
|
14
|
+
3. The job runs `npm version`, pushes the commit and tag to `main`.
|
|
15
|
+
4. The tag push triggers **Release** automatically.
|
|
16
|
+
|
|
17
|
+
## Option B — Local
|
|
18
|
+
|
|
19
|
+
```sh
|
|
20
|
+
npm ci && npm test && npm run lint
|
|
21
|
+
npm version patch # or minor / major
|
|
22
|
+
git push --follow-tags
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Re-run Release without republishing
|
|
26
|
+
|
|
27
|
+
If npm already has the version but GitHub release failed (or vice versa), use **Actions → Release → Run workflow** and enter the existing tag (for example `v0.5.3`). The workflow skips npm publish when that version is already on the registry and skips creating a duplicate GitHub release.
|
|
28
|
+
|
|
29
|
+
## Dry run
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
npm run publish:dry-run
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
CI runs the same dry-run on every push to `main` and on pull requests.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Report format (schema v1)
|
|
2
|
+
|
|
3
|
+
Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** include a versioned schema so you can validate, share, and compare results across machines.
|
|
4
|
+
|
|
5
|
+
## Top-level fields
|
|
6
|
+
|
|
7
|
+
| Field | Description |
|
|
8
|
+
|-------|-------------|
|
|
9
|
+
| `tool` | Always `"fm-bench"` |
|
|
10
|
+
| `version` | fm-bench package version that produced the report |
|
|
11
|
+
| `schemaVersion` | `"1"` for reports from 0.6.0+ (older reports omit this) |
|
|
12
|
+
| `reportId` | Random hex id for citing a single run |
|
|
13
|
+
| `startedAt` / `finishedAt` | ISO-8601 timestamps |
|
|
14
|
+
| `options` | Public run configuration (profile, runs, concurrency, SLOs, tags, note) |
|
|
15
|
+
| `environment` | Host fingerprint: platform, arch, Node, hardware model, CPU, memory, macOS version, `fm` help digest, thermal/power snapshot |
|
|
16
|
+
| `suite` | Derived suite key + fingerprint for apples-to-apples comparison |
|
|
17
|
+
| `prompts` | Prompt ids, text, and token counts |
|
|
18
|
+
| `models` | Discovered models and availability |
|
|
19
|
+
| `summary` | Per-model / per-concurrency roll-up statistics |
|
|
20
|
+
| `results` | Per-run measurements (optional `output` when `--capture-output`) |
|
|
21
|
+
|
|
22
|
+
## Sharing results
|
|
23
|
+
|
|
24
|
+
1. **JSON** — best for automation and `fm-bench compare`. Save with `--out bench.json` or `--output-dir reports/`.
|
|
25
|
+
2. **HTML** — self-contained page for humans: `--out bench.html`, `fm-bench export bench.json -o bench.html`, or `--output-dir reports/ --export-html`.
|
|
26
|
+
3. **CSV** — per-run rows only: `--format csv` or `--out runs.csv`.
|
|
27
|
+
|
|
28
|
+
Validate before publishing:
|
|
29
|
+
|
|
30
|
+
```sh
|
|
31
|
+
fm-bench validate my-report.json
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Comparable benchmarks
|
|
35
|
+
|
|
36
|
+
For fair comparison, match:
|
|
37
|
+
|
|
38
|
+
- Same `--profile` (or same `--prompt-file`)
|
|
39
|
+
- Same `--runs` and `--warmup`
|
|
40
|
+
- Same concurrency operating points (`--concurrency` or `--sweep-concurrency`)
|
|
41
|
+
- Similar power/thermal state (see `environment.power` and `environment.thermal`)
|
|
42
|
+
|
|
43
|
+
```sh
|
|
44
|
+
fm-bench compare before.json after.json
|
|
45
|
+
fm-bench compare before.json after.json --strict # exit 2 if suites differ
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`compare` warns when hardware, macOS, or suite configuration differ. Use `--tag` and `--note` so `fm-bench history` stays readable.
|
|
49
|
+
|
|
50
|
+
## Legacy reports
|
|
51
|
+
|
|
52
|
+
Reports from fm-bench before 0.6.0 remain valid JSON. They lack `schemaVersion`, `reportId`, `suite`, and enriched `environment`. `validate` still checks required fields; `compare` works on `summary` as before.
|
package/package.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "fm-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
|
-
"fm-bench": "
|
|
7
|
+
"fm-bench": "bin/fm-bench.js"
|
|
8
8
|
},
|
|
9
9
|
"files": [
|
|
10
10
|
"bin",
|
package/src/bench.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import crypto from 'node:crypto';
|
|
2
2
|
import { checkModelAvailability, collectEnvironment, countTokens, discoverModels, getQuotaUsage, respond } from './fm.js';
|
|
3
3
|
import { loadPrompts } from './prompts.js';
|
|
4
|
+
import { finalizeReportPayload } from './schema.js';
|
|
4
5
|
import { summarizeByModel } from './stats.js';
|
|
5
6
|
|
|
6
7
|
export async function inspectModels(options = {}) {
|
|
@@ -115,7 +116,7 @@ export async function runBenchmark(options = {}) {
|
|
|
115
116
|
|| a.run - b.run);
|
|
116
117
|
|
|
117
118
|
const summary = summarizeByModel(results, modelStatuses, { concurrencies });
|
|
118
|
-
const payload = {
|
|
119
|
+
const payload = finalizeReportPayload({
|
|
119
120
|
tool: 'fm-bench',
|
|
120
121
|
version: options.version,
|
|
121
122
|
startedAt,
|
|
@@ -131,7 +132,7 @@ export async function runBenchmark(options = {}) {
|
|
|
131
132
|
scenarios,
|
|
132
133
|
summary,
|
|
133
134
|
results
|
|
134
|
-
};
|
|
135
|
+
});
|
|
135
136
|
notify(options, {
|
|
136
137
|
type: 'benchmark:complete',
|
|
137
138
|
completed: completedRuns,
|
package/src/cli.js
CHANGED
|
@@ -2,6 +2,8 @@ import fs from 'node:fs/promises';
|
|
|
2
2
|
import { createRequire } from 'node:module';
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { diffReports, renderCompareReport } from './compare.js';
|
|
5
|
+
import { renderHtmlReport } from './export.js';
|
|
6
|
+
import { validateReport } from './schema.js';
|
|
5
7
|
import { loadHistory, renderHistoryReport } from './history.js';
|
|
6
8
|
import { runProcess } from './process.js';
|
|
7
9
|
import { createProgress } from './progress.js';
|
|
@@ -46,6 +48,16 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
46
48
|
return;
|
|
47
49
|
}
|
|
48
50
|
|
|
51
|
+
if (parsed.command === 'validate') {
|
|
52
|
+
await runValidate(parsed);
|
|
53
|
+
return;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
if (parsed.command === 'export') {
|
|
57
|
+
await runExport(parsed);
|
|
58
|
+
return;
|
|
59
|
+
}
|
|
60
|
+
|
|
49
61
|
if (parsed.command === 'history') {
|
|
50
62
|
await runHistory(parsed, renderOptions(parsed));
|
|
51
63
|
return;
|
|
@@ -99,7 +111,9 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
99
111
|
}
|
|
100
112
|
|
|
101
113
|
if (parsed.out) {
|
|
102
|
-
const reportFormat = parsed.out.endsWith('.csv') ? 'csv'
|
|
114
|
+
const reportFormat = parsed.out.endsWith('.csv') ? 'csv'
|
|
115
|
+
: parsed.out.endsWith('.html') ? 'html'
|
|
116
|
+
: 'json';
|
|
103
117
|
const written = await writeReport(parsed.out, payload, reportFormat);
|
|
104
118
|
if (parsed.format !== 'json') {
|
|
105
119
|
console.error(`Saved ${reportFormat.toUpperCase()} report to ${written}`);
|
|
@@ -110,9 +124,17 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
110
124
|
const stamp = payload.startedAt.replace(/[:.]/g, '-').replace('T', '_').slice(0, 19);
|
|
111
125
|
const firstModel = parsed.models?.flatMap((m) => String(m).split(',')).map((m) => m.trim()).filter(Boolean)[0] || 'all';
|
|
112
126
|
const modelSlug = firstModel.replace(/[^a-z0-9]/gi, '_');
|
|
113
|
-
const
|
|
114
|
-
const
|
|
115
|
-
const
|
|
127
|
+
const tagSlug = parsed.tags?.length ? `_${parsed.tags.join('-').replace(/[^a-z0-9-]/gi, '_')}` : '';
|
|
128
|
+
const base = `fm-bench_${stamp}_${modelSlug}${tagSlug}`;
|
|
129
|
+
const jsonPath = `${parsed.outputDir}/${base}.json`;
|
|
130
|
+
const written = await writeReport(jsonPath, payload, 'json');
|
|
131
|
+
if (parsed.exportHtml) {
|
|
132
|
+
const htmlPath = `${parsed.outputDir}/${base}.html`;
|
|
133
|
+
await writeReport(htmlPath, payload, 'html');
|
|
134
|
+
if (parsed.format !== 'json') {
|
|
135
|
+
console.error(`Saved HTML report to ${htmlPath}`);
|
|
136
|
+
}
|
|
137
|
+
}
|
|
116
138
|
if (parsed.format !== 'json') {
|
|
117
139
|
console.error(`Saved JSON report to ${written}`);
|
|
118
140
|
}
|
|
@@ -190,11 +212,14 @@ export function parseArgs(argv) {
|
|
|
190
212
|
progress: 'auto',
|
|
191
213
|
compact: false,
|
|
192
214
|
width: null,
|
|
193
|
-
histogram: false
|
|
215
|
+
histogram: false,
|
|
216
|
+
exportHtml: false,
|
|
217
|
+
strictCompare: false,
|
|
218
|
+
validateFiles: []
|
|
194
219
|
};
|
|
195
220
|
|
|
196
221
|
const args = [...argv];
|
|
197
|
-
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'help'].includes(args[0])) {
|
|
222
|
+
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'validate', 'export', 'help'].includes(args[0])) {
|
|
198
223
|
options.command = args.shift();
|
|
199
224
|
}
|
|
200
225
|
|
|
@@ -332,6 +357,12 @@ export function parseArgs(argv) {
|
|
|
332
357
|
case '--histogram':
|
|
333
358
|
options.histogram = true;
|
|
334
359
|
break;
|
|
360
|
+
case '--export-html':
|
|
361
|
+
options.exportHtml = true;
|
|
362
|
+
break;
|
|
363
|
+
case '--strict':
|
|
364
|
+
options.strictCompare = true;
|
|
365
|
+
break;
|
|
335
366
|
case '-o':
|
|
336
367
|
case '--out':
|
|
337
368
|
options.out = requireValue(arg, args);
|
|
@@ -378,6 +409,8 @@ export function parseArgs(argv) {
|
|
|
378
409
|
options.compareFiles.push(arg);
|
|
379
410
|
} else if (options.command === 'history') {
|
|
380
411
|
options.historyDir = arg;
|
|
412
|
+
} else if (options.command === 'validate' || options.command === 'export') {
|
|
413
|
+
options.validateFiles.push(arg);
|
|
381
414
|
} else {
|
|
382
415
|
options.prompts.push([arg, ...args].join(' '));
|
|
383
416
|
args.length = 0;
|
|
@@ -445,6 +478,75 @@ async function runCompare(options, renderOpts) {
|
|
|
445
478
|
await fs.writeFile(options.out, `${JSON.stringify(diff, null, 2)}\n`, 'utf8');
|
|
446
479
|
console.error(`Saved compare report to ${options.out}`);
|
|
447
480
|
}
|
|
481
|
+
|
|
482
|
+
if (options.strictCompare && diff.compatibility && !diff.compatibility.suiteMatch) {
|
|
483
|
+
const error = new Error('compare: benchmark suites differ (--strict)');
|
|
484
|
+
error.exitCode = 2;
|
|
485
|
+
throw error;
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
async function runValidate(options) {
|
|
490
|
+
const files = options.validateFiles;
|
|
491
|
+
if (files.length === 0) {
|
|
492
|
+
throw new Error('validate requires at least one JSON report: fm-bench validate report.json');
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
let failed = 0;
|
|
496
|
+
for (const filePath of files) {
|
|
497
|
+
let parsed;
|
|
498
|
+
try {
|
|
499
|
+
const text = await fs.readFile(filePath, 'utf8');
|
|
500
|
+
parsed = JSON.parse(text);
|
|
501
|
+
} catch {
|
|
502
|
+
console.error(`invalid ${filePath} (cannot read or parse JSON)`);
|
|
503
|
+
failed += 1;
|
|
504
|
+
continue;
|
|
505
|
+
}
|
|
506
|
+
const result = validateReport(parsed);
|
|
507
|
+
if (result.ok) {
|
|
508
|
+
const id = result.report.reportId ?? '—';
|
|
509
|
+
const schema = result.report.schemaVersion ?? 'legacy';
|
|
510
|
+
console.log(`ok ${filePath} schema=${schema} id=${id}`);
|
|
511
|
+
} else {
|
|
512
|
+
console.error(`invalid ${filePath} ${result.errors.join('; ')}`);
|
|
513
|
+
failed += 1;
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
if (failed > 0) {
|
|
518
|
+
const error = new Error(`${failed} report(s) failed validation`);
|
|
519
|
+
error.exitCode = 1;
|
|
520
|
+
throw error;
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
async function runExport(options) {
|
|
525
|
+
const files = options.validateFiles;
|
|
526
|
+
if (files.length === 0) {
|
|
527
|
+
throw new Error('export requires a JSON report: fm-bench export report.json [-o out.html]');
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
const filePath = files[0];
|
|
531
|
+
const text = await fs.readFile(filePath, 'utf8');
|
|
532
|
+
let report;
|
|
533
|
+
try {
|
|
534
|
+
report = JSON.parse(text);
|
|
535
|
+
} catch {
|
|
536
|
+
throw new Error(`Cannot parse ${filePath} as JSON`);
|
|
537
|
+
}
|
|
538
|
+
const validation = validateReport(report);
|
|
539
|
+
if (!validation.ok) {
|
|
540
|
+
throw new Error(`Not a valid fm-bench report: ${validation.errors.join('; ')}`);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
const html = renderHtmlReport(validation.report);
|
|
544
|
+
if (options.out) {
|
|
545
|
+
await fs.writeFile(options.out, html, 'utf8');
|
|
546
|
+
console.error(`Wrote HTML to ${options.out}`);
|
|
547
|
+
} else {
|
|
548
|
+
console.log(html);
|
|
549
|
+
}
|
|
448
550
|
}
|
|
449
551
|
|
|
450
552
|
async function runDoctor(options) {
|
|
@@ -574,6 +676,8 @@ Usage:
|
|
|
574
676
|
fm-bench models [options]
|
|
575
677
|
fm-bench compare <before.json> <after.json> [options]
|
|
576
678
|
fm-bench history [dir] [options]
|
|
679
|
+
fm-bench validate <report.json> [more...]
|
|
680
|
+
fm-bench export <report.json> [-o report.html]
|
|
577
681
|
fm-bench legend [options]
|
|
578
682
|
fm-bench doctor [options]
|
|
579
683
|
|
|
@@ -582,6 +686,8 @@ Commands:
|
|
|
582
686
|
models List discovered models and availability
|
|
583
687
|
compare Compare two saved JSON reports and show metric deltas
|
|
584
688
|
history Show a trend table from all fm-bench JSON reports in a directory
|
|
689
|
+
validate Verify report JSON structure (schema v1)
|
|
690
|
+
export Render a shareable standalone HTML report from JSON
|
|
585
691
|
legend Explain every terminal table column and color rule
|
|
586
692
|
doctor Check Node, macOS, fm, and model availability
|
|
587
693
|
|
|
@@ -628,10 +734,14 @@ Output:
|
|
|
628
734
|
--compact Force compact terminal layout
|
|
629
735
|
--width <n> Render for a specific terminal width
|
|
630
736
|
--histogram Print an ASCII latency distribution histogram after the report
|
|
631
|
-
-o, --out <file> Save JSON or
|
|
737
|
+
-o, --out <file> Save JSON, CSV, or HTML report based on file extension
|
|
632
738
|
--output-dir <dir> Save a timestamped JSON report to a directory automatically
|
|
739
|
+
--export-html With --output-dir, also write a matching .html report
|
|
633
740
|
-v, --verbose Include per-run CSV after the summary table
|
|
634
741
|
|
|
742
|
+
Compare:
|
|
743
|
+
--strict Exit 2 when before/after benchmark suites differ
|
|
744
|
+
|
|
635
745
|
Environment:
|
|
636
746
|
--fm-bin <path> fm binary to execute (default: FM_BIN or fm)
|
|
637
747
|
-h, --help Show this help
|
|
@@ -645,6 +755,10 @@ Examples:
|
|
|
645
755
|
fm-bench --profile reasoning --runs 5 --retry 2
|
|
646
756
|
fm-bench compare before.json after.json
|
|
647
757
|
fm-bench compare before.json after.json --json
|
|
758
|
+
fm-bench compare before.json after.json --strict
|
|
759
|
+
fm-bench validate reports/*.json
|
|
760
|
+
fm-bench export bench.json -o bench.html
|
|
761
|
+
fm-bench --output-dir reports/ --export-html --tag nightly
|
|
648
762
|
fm-bench history ./reports
|
|
649
763
|
fm-bench history ./reports --json
|
|
650
764
|
fm-bench legend
|
package/src/compare.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import { compareCompatibility } from './schema.js';
|
|
1
2
|
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
2
3
|
|
|
3
4
|
export function diffReports(before, after) {
|
|
5
|
+
const compatibility = compareCompatibility(before, after);
|
|
4
6
|
const beforeByKey = indexSummary(before.summary ?? []);
|
|
5
7
|
const afterByKey = indexSummary(after.summary ?? []);
|
|
6
8
|
|
|
@@ -16,18 +18,28 @@ export function diffReports(before, after) {
|
|
|
16
18
|
return {
|
|
17
19
|
before: reportMeta(before),
|
|
18
20
|
after: reportMeta(after),
|
|
21
|
+
compatibility,
|
|
19
22
|
rows
|
|
20
23
|
};
|
|
21
24
|
}
|
|
22
25
|
|
|
23
26
|
function reportMeta(report) {
|
|
27
|
+
const tags = report.options?.tags;
|
|
24
28
|
return {
|
|
25
29
|
version: report.version ?? '?',
|
|
30
|
+
schemaVersion: report.schemaVersion ?? null,
|
|
31
|
+
reportId: report.reportId ?? null,
|
|
26
32
|
startedAt: report.startedAt ?? '',
|
|
27
33
|
finishedAt: report.finishedAt ?? '',
|
|
28
34
|
runs: report.options?.runs ?? null,
|
|
29
35
|
profile: report.options?.profile ?? null,
|
|
30
|
-
concurrency: report.options?.concurrency ?? null
|
|
36
|
+
concurrency: report.options?.concurrency ?? null,
|
|
37
|
+
tags: Array.isArray(tags) ? tags : [],
|
|
38
|
+
note: report.options?.note ?? null,
|
|
39
|
+
hwModel: report.environment?.hwModel ?? report.suite?.fingerprint?.hwModel ?? null,
|
|
40
|
+
macOS: report.suite?.fingerprint?.macOSProductVersion
|
|
41
|
+
?? report.environment?.macOS?.match(/ProductVersion:\s*([^\n]+)/)?.[1]?.trim()
|
|
42
|
+
?? null
|
|
31
43
|
};
|
|
32
44
|
}
|
|
33
45
|
|
|
@@ -122,8 +134,20 @@ export function renderCompareReport(diff, options = {}) {
|
|
|
122
134
|
const afterLabel = `${diff.after.version} ${diff.after.startedAt ? diff.after.startedAt.slice(0, 19).replace('T', ' ') : '?'}`;
|
|
123
135
|
|
|
124
136
|
lines.push(`fm-bench compare`);
|
|
125
|
-
lines.push(` before: ${beforeLabel}`);
|
|
126
|
-
lines.push(` after: ${afterLabel}`);
|
|
137
|
+
lines.push(` before: ${beforeLabel}${diff.before.hwModel ? ` | ${diff.before.hwModel}` : ''}`);
|
|
138
|
+
lines.push(` after: ${afterLabel}${diff.after.hwModel ? ` | ${diff.after.hwModel}` : ''}`);
|
|
139
|
+
if (diff.before.profile || diff.after.profile) {
|
|
140
|
+
lines.push(` suite: profile=${diff.before.profile ?? '?'} runs=${diff.before.runs ?? '?'} (before) → profile=${diff.after.profile ?? '?'} runs=${diff.after.runs ?? '?'}`);
|
|
141
|
+
}
|
|
142
|
+
const compat = diff.compatibility;
|
|
143
|
+
if (compat?.warnings?.length) {
|
|
144
|
+
for (const w of compat.warnings) {
|
|
145
|
+
lines.push(` warn: ${w}`);
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
if (compat && !compat.suiteMatch) {
|
|
149
|
+
lines.push(' note: use identical --profile, --runs, and prompts for apples-to-apples comparison');
|
|
150
|
+
}
|
|
127
151
|
lines.push('');
|
|
128
152
|
|
|
129
153
|
const colWidths = [8, 3, 10, 10, 10, 10, 10, 10, 8, 8, 9, 7, 7];
|
package/src/export.js
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
import { environmentFingerprint } from './schema.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Self-contained HTML report for sharing (paste, email, GitHub gist, static host).
|
|
5
|
+
* @param {Record<string, unknown>} report
|
|
6
|
+
*/
|
|
7
|
+
export function renderHtmlReport(report) {
|
|
8
|
+
const fp = environmentFingerprint(report);
|
|
9
|
+
const options = /** @type {Record<string, unknown>} */ (report.options ?? {});
|
|
10
|
+
const tags = Array.isArray(options.tags) ? options.tags : [];
|
|
11
|
+
const summary = /** @type {Array<Record<string, unknown>>} */ (report.summary ?? []);
|
|
12
|
+
|
|
13
|
+
const esc = (s) => String(s ?? '')
|
|
14
|
+
.replace(/&/g, '&')
|
|
15
|
+
.replace(/</g, '<')
|
|
16
|
+
.replace(/>/g, '>')
|
|
17
|
+
.replace(/"/g, '"');
|
|
18
|
+
|
|
19
|
+
const metaRows = [
|
|
20
|
+
['Report ID', report.reportId ?? '—'],
|
|
21
|
+
['fm-bench', report.version ?? '—'],
|
|
22
|
+
['Schema', report.schemaVersion ?? 'legacy'],
|
|
23
|
+
['Started', report.startedAt ?? '—'],
|
|
24
|
+
['Finished', report.finishedAt ?? '—'],
|
|
25
|
+
['Profile', options.profile ?? '—'],
|
|
26
|
+
['Runs / prompt', options.runs ?? '—'],
|
|
27
|
+
['Concurrency', formatConcurrency(options)],
|
|
28
|
+
['Tags', tags.length ? tags.join(', ') : '—'],
|
|
29
|
+
['Note', options.note ?? '—'],
|
|
30
|
+
['Hardware', fp.hwModel ?? '—'],
|
|
31
|
+
['CPU', fp.cpuBrand ?? '—'],
|
|
32
|
+
['Memory', fp.memoryGb != null ? `${fp.memoryGb} GB` : '—'],
|
|
33
|
+
['macOS', fp.macOSProductVersion ?? '—'],
|
|
34
|
+
['Build', fp.macOSBuildVersion ?? '—'],
|
|
35
|
+
['Node', fp.node ?? '—'],
|
|
36
|
+
['fm CLI digest', fp.fmHelpDigest ?? '—']
|
|
37
|
+
];
|
|
38
|
+
|
|
39
|
+
const summaryRows = summary.map((row) => {
|
|
40
|
+
const ttft = /** @type {{ p50?: number, p95?: number }} */ (row.ttft ?? {});
|
|
41
|
+
const lat = /** @type {{ p50?: number, p95?: number, cv?: number }} */ (row.latency ?? {});
|
|
42
|
+
const tps = /** @type {{ avg?: number }} */ (row.tokensPerSecond ?? {});
|
|
43
|
+
return `<tr>
|
|
44
|
+
<td>${esc(row.model)}</td>
|
|
45
|
+
<td>${esc(row.concurrency ?? 1)}</td>
|
|
46
|
+
<td>${row.available ? 'yes' : 'no'}</td>
|
|
47
|
+
<td>${fmtPct(row.successRate)}</td>
|
|
48
|
+
<td>${fmtPct(row.goodputRate)}</td>
|
|
49
|
+
<td>${fmtMs(ttft.p50)}</td>
|
|
50
|
+
<td>${fmtMs(ttft.p95)}</td>
|
|
51
|
+
<td>${fmtMs(lat.p50)}</td>
|
|
52
|
+
<td>${fmtMs(lat.p95)}</td>
|
|
53
|
+
<td>${fmtNum(tps.avg)}</td>
|
|
54
|
+
<td>${fmtPct(lat.cv)}</td>
|
|
55
|
+
</tr>`;
|
|
56
|
+
}).join('\n');
|
|
57
|
+
|
|
58
|
+
const jsonEmbed = esc(JSON.stringify(report, null, 2));
|
|
59
|
+
|
|
60
|
+
return `<!DOCTYPE html>
|
|
61
|
+
<html lang="en">
|
|
62
|
+
<head>
|
|
63
|
+
<meta charset="utf-8" />
|
|
64
|
+
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
|
65
|
+
<title>fm-bench report ${esc(report.reportId ?? '')}</title>
|
|
66
|
+
<style>
|
|
67
|
+
:root { --bg: #0f1419; --card: #1a2332; --text: #e7ecf3; --muted: #8b9cb3; --accent: #3d8bfd; --border: #2a3548; }
|
|
68
|
+
* { box-sizing: border-box; }
|
|
69
|
+
body { font-family: ui-sans-serif, system-ui, -apple-system, sans-serif; background: var(--bg); color: var(--text); margin: 0; padding: 2rem 1.25rem; line-height: 1.5; }
|
|
70
|
+
h1 { font-size: 1.35rem; margin: 0 0 0.25rem; }
|
|
71
|
+
.sub { color: var(--muted); font-size: 0.9rem; margin-bottom: 1.5rem; }
|
|
72
|
+
section { background: var(--card); border: 1px solid var(--border); border-radius: 10px; padding: 1.25rem; margin-bottom: 1.25rem; }
|
|
73
|
+
h2 { font-size: 1rem; margin: 0 0 1rem; color: var(--accent); }
|
|
74
|
+
dl { display: grid; grid-template-columns: 10rem 1fr; gap: 0.35rem 1rem; margin: 0; font-size: 0.88rem; }
|
|
75
|
+
dt { color: var(--muted); }
|
|
76
|
+
dd { margin: 0; }
|
|
77
|
+
table { width: 100%; border-collapse: collapse; font-size: 0.82rem; }
|
|
78
|
+
th, td { text-align: left; padding: 0.45rem 0.5rem; border-bottom: 1px solid var(--border); }
|
|
79
|
+
th { color: var(--muted); font-weight: 600; }
|
|
80
|
+
pre { background: #0b0f14; border: 1px solid var(--border); border-radius: 8px; padding: 1rem; overflow: auto; font-size: 0.72rem; max-height: 320px; }
|
|
81
|
+
footer { color: var(--muted); font-size: 0.8rem; margin-top: 2rem; }
|
|
82
|
+
a { color: var(--accent); }
|
|
83
|
+
</style>
|
|
84
|
+
</head>
|
|
85
|
+
<body>
|
|
86
|
+
<h1>fm-bench benchmark report</h1>
|
|
87
|
+
<p class="sub">Share this file to compare runs on identical prompt suites. Attach JSON for automation.</p>
|
|
88
|
+
|
|
89
|
+
<section>
|
|
90
|
+
<h2>Run metadata</h2>
|
|
91
|
+
<dl>
|
|
92
|
+
${metaRows.map(([k, v]) => `<dt>${esc(k)}</dt><dd>${esc(v)}</dd>`).join('\n ')}
|
|
93
|
+
</dl>
|
|
94
|
+
</section>
|
|
95
|
+
|
|
96
|
+
<section>
|
|
97
|
+
<h2>Summary by model</h2>
|
|
98
|
+
<table>
|
|
99
|
+
<thead>
|
|
100
|
+
<tr>
|
|
101
|
+
<th>Model</th><th>C</th><th>Avail</th><th>Success</th><th>Goodput</th>
|
|
102
|
+
<th>TTFT P50</th><th>TTFT P95</th><th>E2E P50</th><th>E2E P95</th><th>User tok/s</th><th>CV</th>
|
|
103
|
+
</tr>
|
|
104
|
+
</thead>
|
|
105
|
+
<tbody>
|
|
106
|
+
${summaryRows || '<tr><td colspan="11">No summary rows</td></tr>'}
|
|
107
|
+
</tbody>
|
|
108
|
+
</table>
|
|
109
|
+
</section>
|
|
110
|
+
|
|
111
|
+
<section>
|
|
112
|
+
<h2>Embedded JSON (machine-readable)</h2>
|
|
113
|
+
<pre>${jsonEmbed}</pre>
|
|
114
|
+
</section>
|
|
115
|
+
|
|
116
|
+
<footer>
|
|
117
|
+
Generated by <a href="https://github.com/devinoldenburg/fm-bench">fm-bench</a>.
|
|
118
|
+
Compare two reports: <code>fm-bench compare before.json after.json</code>.
|
|
119
|
+
Validate: <code>fm-bench validate report.json</code>.
|
|
120
|
+
</footer>
|
|
121
|
+
</body>
|
|
122
|
+
</html>
|
|
123
|
+
`;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function formatConcurrency(options) {
|
|
127
|
+
const sweep = options.sweepConcurrency;
|
|
128
|
+
if (Array.isArray(sweep) && sweep.length > 0) return sweep.join(', ');
|
|
129
|
+
return String(options.concurrency ?? 1);
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function fmtMs(v) {
|
|
133
|
+
if (!Number.isFinite(v)) return '—';
|
|
134
|
+
return v >= 1000 ? `${(v / 1000).toFixed(2)}s` : `${Math.round(v)}ms`;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function fmtPct(v) {
|
|
138
|
+
if (!Number.isFinite(v)) return '—';
|
|
139
|
+
return `${Math.round(v * 100)}%`;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function fmtNum(v) {
|
|
143
|
+
if (!Number.isFinite(v)) return '—';
|
|
144
|
+
return String(Math.round(v * 10) / 10);
|
|
145
|
+
}
|
package/src/fm.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import crypto from 'node:crypto';
|
|
1
2
|
import os from 'node:os';
|
|
2
3
|
import { stripAnsi } from './ansi.js';
|
|
3
4
|
import { runProcess } from './process.js';
|
|
5
|
+
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
4
6
|
|
|
5
7
|
const DEFAULT_MODELS = [
|
|
6
8
|
{ name: 'system', description: 'On-device Apple Foundation Model' },
|
|
@@ -183,13 +185,56 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
183
185
|
|
|
184
186
|
export async function collectEnvironment(fmBin) {
|
|
185
187
|
const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
|
|
188
|
+
const macOS = stripAnsi(swVers.stdout).trim() || null;
|
|
189
|
+
|
|
190
|
+
const hwModel = await runProcess('sysctl', ['-n', 'hw.model'], { timeoutMs: 3_000 });
|
|
191
|
+
const cpuBrand = await runProcess('sysctl', ['-n', 'machdep.cpu.brand_string'], { timeoutMs: 3_000 });
|
|
192
|
+
const memBytes = await runProcess('sysctl', ['-n', 'hw.memsize'], { timeoutMs: 3_000 });
|
|
193
|
+
|
|
194
|
+
const thermalResult = await runProcess('pmset', ['-g', 'therm'], { timeoutMs: 5_000 });
|
|
195
|
+
const thermal = parseThermalOutput(`${thermalResult.stdout || ''}${thermalResult.stderr || ''}`);
|
|
196
|
+
const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
|
|
197
|
+
const battery = parseBatteryOutput(`${batteryResult.stdout || ''}${batteryResult.stderr || ''}`);
|
|
198
|
+
|
|
199
|
+
let fmHelpDigest = null;
|
|
200
|
+
try {
|
|
201
|
+
const help = await getFmHelp(fmBin, 10_000);
|
|
202
|
+
const text = stripAnsi(help.text).trim();
|
|
203
|
+
if (text) {
|
|
204
|
+
fmHelpDigest = crypto.createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
205
|
+
}
|
|
206
|
+
} catch {
|
|
207
|
+
fmHelpDigest = null;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
const memRaw = (memBytes.stdout || '').trim();
|
|
211
|
+
const memoryGb = memRaw && Number.isFinite(Number(memRaw))
|
|
212
|
+
? Math.round(Number(memRaw) / (1024 ** 3))
|
|
213
|
+
: null;
|
|
214
|
+
|
|
186
215
|
return {
|
|
187
216
|
platform: process.platform,
|
|
188
217
|
arch: process.arch,
|
|
189
218
|
node: process.version,
|
|
190
219
|
host: os.hostname(),
|
|
191
220
|
fmBin,
|
|
192
|
-
macOS
|
|
221
|
+
macOS,
|
|
222
|
+
hwModel: (hwModel.stdout || '').trim() || null,
|
|
223
|
+
cpuBrand: (cpuBrand.stdout || '').trim() || null,
|
|
224
|
+
memoryGb,
|
|
225
|
+
fmHelpDigest,
|
|
226
|
+
thermal: thermal.available
|
|
227
|
+
? {
|
|
228
|
+
schedulerLimit: thermal.schedulerLimit,
|
|
229
|
+
healthyIdle: Boolean(thermal.healthyIdle)
|
|
230
|
+
}
|
|
231
|
+
: null,
|
|
232
|
+
power: battery.present
|
|
233
|
+
? {
|
|
234
|
+
pct: battery.pct,
|
|
235
|
+
onAC: battery.onAC
|
|
236
|
+
}
|
|
237
|
+
: null
|
|
193
238
|
};
|
|
194
239
|
}
|
|
195
240
|
|
package/src/history.js
CHANGED
|
@@ -29,6 +29,13 @@ export async function loadHistory(dir) {
|
|
|
29
29
|
}
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
reports.sort((a, b) => {
|
|
33
|
+
const ta = Date.parse(a.report.startedAt ?? '') || 0;
|
|
34
|
+
const tb = Date.parse(b.report.startedAt ?? '') || 0;
|
|
35
|
+
if (tb !== ta) return tb - ta;
|
|
36
|
+
return a.filePath.localeCompare(b.filePath);
|
|
37
|
+
});
|
|
38
|
+
|
|
32
39
|
return reports;
|
|
33
40
|
}
|
|
34
41
|
|
|
@@ -50,8 +57,8 @@ export function renderHistoryReport(reports, options = {}) {
|
|
|
50
57
|
const MJ = ascii ? '+' : '┼';
|
|
51
58
|
const BJ = ascii ? '+' : '┴';
|
|
52
59
|
|
|
53
|
-
const colWidths = [19, 8, 3, 6, 9, 9, 9, 9, 5];
|
|
54
|
-
const headers = ['STARTED AT', 'MODEL', 'C', 'RUNS', 'TTFT P50', 'E2E P50', 'E2E P95', 'USER T/S', 'SUCC'];
|
|
60
|
+
const colWidths = [19, 8, 10, 3, 6, 9, 9, 9, 9, 5];
|
|
61
|
+
const headers = ['STARTED AT', 'MODEL', 'TAG / NOTE', 'C', 'RUNS', 'TTFT P50', 'E2E P50', 'E2E P95', 'USER T/S', 'SUCC'];
|
|
55
62
|
|
|
56
63
|
const renderRule = (l, j, r) => `${l}${colWidths.map((w) => H.repeat(w + 2)).join(j)}${r}`;
|
|
57
64
|
const renderRowLine = (cells) => `${V}${cells.map((text, i) => ` ${fit(text, colWidths[i])} `).join(V)}${V}`;
|
|
@@ -63,15 +70,19 @@ export function renderHistoryReport(reports, options = {}) {
|
|
|
63
70
|
lines.push(renderRowLine(headers.map((h, i) => h)));
|
|
64
71
|
lines.push(renderRule(ML, MJ, MR));
|
|
65
72
|
|
|
66
|
-
for (const {
|
|
73
|
+
for (const { report } of reports) {
|
|
67
74
|
const started = report.startedAt ? report.startedAt.slice(0, 19).replace('T', ' ') : '?';
|
|
68
75
|
const modelRows = report.summary ?? [];
|
|
76
|
+
const tags = Array.isArray(report.options?.tags) ? report.options.tags.join(',') : '';
|
|
77
|
+
const note = report.options?.note ? String(report.options.note) : '';
|
|
78
|
+
const label = tags || note || report.options?.profile || '—';
|
|
69
79
|
|
|
70
80
|
for (const item of modelRows) {
|
|
71
81
|
if (!item.available) continue;
|
|
72
82
|
const row = [
|
|
73
83
|
started,
|
|
74
84
|
item.model ?? '-',
|
|
85
|
+
label,
|
|
75
86
|
String(item.concurrency ?? 1),
|
|
76
87
|
String(item.successes ?? '-'),
|
|
77
88
|
item.ttft?.p50 != null ? formatMs(item.ttft.p50) : '-',
|
package/src/report.js
CHANGED
|
@@ -43,9 +43,15 @@ export function flattenResults(results) {
|
|
|
43
43
|
export async function writeReport(filePath, payload, format) {
|
|
44
44
|
const target = path.resolve(filePath);
|
|
45
45
|
await fs.mkdir(path.dirname(target), { recursive: true });
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
46
|
+
let content;
|
|
47
|
+
if (format === 'csv') {
|
|
48
|
+
content = toCsv(flattenResults(payload.results));
|
|
49
|
+
} else if (format === 'html') {
|
|
50
|
+
const { renderHtmlReport } = await import('./export.js');
|
|
51
|
+
content = renderHtmlReport(payload);
|
|
52
|
+
} else {
|
|
53
|
+
content = `${JSON.stringify(payload, null, 2)}\n`;
|
|
54
|
+
}
|
|
49
55
|
await fs.writeFile(target, content, 'utf8');
|
|
50
56
|
return target;
|
|
51
57
|
}
|
package/src/schema.js
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/** @typedef {'1'} ReportSchemaVersion */
|
|
2
|
+
|
|
3
|
+
export const REPORT_SCHEMA_VERSION = '1';
|
|
4
|
+
|
|
5
|
+
const REQUIRED_TOP_LEVEL = ['tool', 'version', 'startedAt', 'summary', 'options', 'environment'];
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* @param {unknown} value
|
|
9
|
+
* @returns {value is Record<string, unknown>}
|
|
10
|
+
*/
|
|
11
|
+
export function isFmBenchReport(value) {
|
|
12
|
+
if (!value || typeof value !== 'object') return false;
|
|
13
|
+
const report = /** @type {Record<string, unknown>} */ (value);
|
|
14
|
+
if (report.tool !== 'fm-bench') return false;
|
|
15
|
+
for (const key of REQUIRED_TOP_LEVEL) {
|
|
16
|
+
if (!(key in report)) return false;
|
|
17
|
+
}
|
|
18
|
+
if (!Array.isArray(report.summary)) return false;
|
|
19
|
+
return true;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Stable key for comparing two runs (profile, prompts, runs, concurrency sweep).
|
|
24
|
+
* @param {Record<string, unknown>} report
|
|
25
|
+
*/
|
|
26
|
+
export function suiteKey(report) {
|
|
27
|
+
const options = /** @type {Record<string, unknown>} */ (report.options ?? {});
|
|
28
|
+
const prompts = Array.isArray(report.prompts)
|
|
29
|
+
? report.prompts.map((p) => /** @type {{ id?: string }} */ (p).id).filter(Boolean).sort()
|
|
30
|
+
: [];
|
|
31
|
+
const sweep = Array.isArray(options.sweepConcurrency) ? [...options.sweepConcurrency].sort((a, b) => a - b) : [];
|
|
32
|
+
return JSON.stringify({
|
|
33
|
+
profile: options.profile ?? null,
|
|
34
|
+
runs: options.runs ?? null,
|
|
35
|
+
warmup: options.warmup ?? 0,
|
|
36
|
+
concurrency: options.concurrency ?? 1,
|
|
37
|
+
sweepConcurrency: sweep,
|
|
38
|
+
promptIds: prompts,
|
|
39
|
+
stream: options.stream ?? true,
|
|
40
|
+
greedy: options.greedy ?? true
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Human-readable hardware + OS fingerprint for sharing and leaderboards.
|
|
46
|
+
* @param {Record<string, unknown>} report
|
|
47
|
+
*/
|
|
48
|
+
export function environmentFingerprint(report) {
|
|
49
|
+
const env = /** @type {Record<string, unknown>} */ (report.environment ?? {});
|
|
50
|
+
const mac = typeof env.macOS === 'string' ? env.macOS : '';
|
|
51
|
+
const product = mac.match(/ProductVersion:\s*([^\n]+)/)?.[1]?.trim() ?? null;
|
|
52
|
+
const build = mac.match(/BuildVersion:\s*([^\n]+)/)?.[1]?.trim() ?? null;
|
|
53
|
+
return {
|
|
54
|
+
platform: env.platform ?? null,
|
|
55
|
+
arch: env.arch ?? null,
|
|
56
|
+
node: env.node ?? null,
|
|
57
|
+
host: env.host ?? null,
|
|
58
|
+
hwModel: env.hwModel ?? null,
|
|
59
|
+
cpuBrand: env.cpuBrand ?? null,
|
|
60
|
+
memoryGb: env.memoryGb ?? null,
|
|
61
|
+
macOSProductVersion: product,
|
|
62
|
+
macOSBuildVersion: build,
|
|
63
|
+
fmBin: env.fmBin ?? null,
|
|
64
|
+
fmHelpDigest: env.fmHelpDigest ?? null,
|
|
65
|
+
thermal: env.thermal ?? null,
|
|
66
|
+
power: env.power ?? null
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* @param {Record<string, unknown>} before
|
|
72
|
+
* @param {Record<string, unknown>} after
|
|
73
|
+
*/
|
|
74
|
+
export function compareCompatibility(before, after) {
|
|
75
|
+
const warnings = [];
|
|
76
|
+
const errors = [];
|
|
77
|
+
|
|
78
|
+
if (!isFmBenchReport(before)) errors.push('before file is not a valid fm-bench report');
|
|
79
|
+
if (!isFmBenchReport(after)) errors.push('after file is not a valid fm-bench report');
|
|
80
|
+
if (errors.length > 0) {
|
|
81
|
+
return { compatible: false, warnings, errors, suiteMatch: false };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const beforeKey = suiteKey(before);
|
|
85
|
+
const afterKey = suiteKey(after);
|
|
86
|
+
const suiteMatch = beforeKey === afterKey;
|
|
87
|
+
if (!suiteMatch) {
|
|
88
|
+
warnings.push('benchmark suites differ (profile, runs, prompts, or concurrency). Summary deltas may be misleading.');
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const bFp = environmentFingerprint(before);
|
|
92
|
+
const aFp = environmentFingerprint(after);
|
|
93
|
+
if (bFp.hwModel && aFp.hwModel && bFp.hwModel !== aFp.hwModel) {
|
|
94
|
+
warnings.push(`hardware model differs (${bFp.hwModel} vs ${aFp.hwModel})`);
|
|
95
|
+
}
|
|
96
|
+
if (bFp.macOSProductVersion && aFp.macOSProductVersion && bFp.macOSProductVersion !== aFp.macOSProductVersion) {
|
|
97
|
+
warnings.push(`macOS version differs (${bFp.macOSProductVersion} vs ${aFp.macOSProductVersion})`);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
return {
|
|
101
|
+
compatible: errors.length === 0,
|
|
102
|
+
warnings,
|
|
103
|
+
errors,
|
|
104
|
+
suiteMatch,
|
|
105
|
+
beforeSuite: beforeKey,
|
|
106
|
+
afterSuite: afterKey,
|
|
107
|
+
beforeFingerprint: bFp,
|
|
108
|
+
afterFingerprint: aFp
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* @param {unknown} value
|
|
114
|
+
* @returns {{ ok: true, report: Record<string, unknown> } | { ok: false, errors: string[] }}
|
|
115
|
+
*/
|
|
116
|
+
export function validateReport(value) {
|
|
117
|
+
const errors = [];
|
|
118
|
+
if (!value || typeof value !== 'object') {
|
|
119
|
+
return { ok: false, errors: ['root must be a JSON object'] };
|
|
120
|
+
}
|
|
121
|
+
const report = /** @type {Record<string, unknown>} */ (value);
|
|
122
|
+
if (report.tool !== 'fm-bench') errors.push('tool must be "fm-bench"');
|
|
123
|
+
for (const key of REQUIRED_TOP_LEVEL) {
|
|
124
|
+
if (!(key in report)) errors.push(`missing required field: ${key}`);
|
|
125
|
+
}
|
|
126
|
+
if (!Array.isArray(report.summary)) errors.push('summary must be an array');
|
|
127
|
+
if (report.schemaVersion != null && report.schemaVersion !== REPORT_SCHEMA_VERSION) {
|
|
128
|
+
errors.push(`unsupported schemaVersion: ${report.schemaVersion} (expected ${REPORT_SCHEMA_VERSION})`);
|
|
129
|
+
}
|
|
130
|
+
if (errors.length > 0) return { ok: false, errors };
|
|
131
|
+
return { ok: true, report };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Attach schema version and suite metadata without mutating caller's object deeply.
|
|
136
|
+
* @param {Record<string, unknown>} payload
|
|
137
|
+
* @param {{ reportId?: string }} [meta]
|
|
138
|
+
*/
|
|
139
|
+
export function finalizeReportPayload(payload, meta = {}) {
|
|
140
|
+
const reportId = meta.reportId ?? cryptoRandomId();
|
|
141
|
+
return {
|
|
142
|
+
...payload,
|
|
143
|
+
schemaVersion: REPORT_SCHEMA_VERSION,
|
|
144
|
+
reportId,
|
|
145
|
+
suite: {
|
|
146
|
+
key: suiteKey(payload),
|
|
147
|
+
profile: payload.options?.profile ?? null,
|
|
148
|
+
promptCount: Array.isArray(payload.prompts) ? payload.prompts.length : 0,
|
|
149
|
+
fingerprint: environmentFingerprint(payload)
|
|
150
|
+
}
|
|
151
|
+
};
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function cryptoRandomId() {
|
|
155
|
+
const bytes = new Uint8Array(8);
|
|
156
|
+
globalThis.crypto.getRandomValues(bytes);
|
|
157
|
+
return [...bytes].map((b) => b.toString(16).padStart(2, '0')).join('');
|
|
158
|
+
}
|