fm-bench 0.6.3 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +33 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +19 -2
- package/package.json +6 -5
- package/src/bench.js +129 -37
- package/src/capabilities.js +188 -0
- package/src/cli.js +153 -68
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +100 -112
- package/src/history.js +2 -1
- package/src/macos.js +2 -1
- package/src/metrics.js +203 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/docs/report-format.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Report format (schema v1)
|
|
2
2
|
|
|
3
|
-
Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** include a versioned schema so you can validate, share, and compare results across machines.
|
|
3
|
+
Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** include a versioned schema so you can validate, share, and compare results across machines. The schema version stayed at `1` in 0.7.0: the new `capabilities`, `metrics`, and per-result `attempts` fields are additive, so older readers and older reports both keep working.
|
|
4
4
|
|
|
5
5
|
## Top-level fields
|
|
6
6
|
|
|
@@ -13,12 +13,88 @@ Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** includ
|
|
|
13
13
|
| `startedAt` / `finishedAt` | ISO-8601 timestamps |
|
|
14
14
|
| `options` | Public run configuration (profile, runs, concurrency, SLOs, tags, note) |
|
|
15
15
|
| `environment` | Host fingerprint: platform, arch, Node, hardware model, CPU, memory, macOS version/build, `fm` help digest, thermal/power snapshot |
|
|
16
|
+
| `capabilities` | What the installed `fm` build exposes (see below) |
|
|
17
|
+
| `metrics` | Per-metric availability and provenance for this run (see below) |
|
|
16
18
|
| `suite` | Derived suite key + fingerprint for apples-to-apples comparison |
|
|
17
19
|
| `prompts` | Prompt ids, text, and token counts |
|
|
18
20
|
| `models` | Discovered models and availability |
|
|
19
21
|
| `summary` | Per-model / per-concurrency roll-up statistics |
|
|
20
22
|
| `results` | Per-run measurements (optional `output` when `--capture-output`) |
|
|
21
23
|
|
|
24
|
+
## `capabilities`
|
|
25
|
+
|
|
26
|
+
Detected once per run from `fm --help` and `fm respond --help`.
|
|
27
|
+
|
|
28
|
+
```json
|
|
29
|
+
{
|
|
30
|
+
"capabilities": {
|
|
31
|
+
"bin": "fm",
|
|
32
|
+
"digest": "921a7839714e3707",
|
|
33
|
+
"commands": ["available", "chat", "count-tokens", "license", "respond", "schema", "serve"],
|
|
34
|
+
"models": [{ "name": "system", "description": "On-device Apple Foundation Model" }],
|
|
35
|
+
"features": {
|
|
36
|
+
"tokenCounting": true,
|
|
37
|
+
"tokenCountCommand": "count-tokens",
|
|
38
|
+
"quota": false,
|
|
39
|
+
"streaming": true,
|
|
40
|
+
"modelSelection": true,
|
|
41
|
+
"instructions": true,
|
|
42
|
+
"greedy": true,
|
|
43
|
+
"useCase": true,
|
|
44
|
+
"guardrails": true,
|
|
45
|
+
"images": true,
|
|
46
|
+
"tools": true,
|
|
47
|
+
"structuredOutput": true,
|
|
48
|
+
"server": true
|
|
49
|
+
},
|
|
50
|
+
"warnings": ["this fm build exposes no quota command, so quota is not reported"]
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
`digest` is a short hash of the normalized `fm --help` output, so a report records which CLI surface produced it. Reports from two different `fm` builds are not directly comparable even on the same machine.
|
|
56
|
+
|
|
57
|
+
## `metrics`
|
|
58
|
+
|
|
59
|
+
Each entry states how a metric was obtained for this run and whether it was available. `kind` is one of `measured`, `proxy`, `derived`, or `controlled`.
|
|
60
|
+
|
|
61
|
+
```json
|
|
62
|
+
{
|
|
63
|
+
"metrics": {
|
|
64
|
+
"ttft": {
|
|
65
|
+
"label": "TTFT",
|
|
66
|
+
"kind": "proxy",
|
|
67
|
+
"source": "arrival time of the first streamed stdout chunk",
|
|
68
|
+
"available": true,
|
|
69
|
+
"unavailableReason": ""
|
|
70
|
+
},
|
|
71
|
+
"quota": {
|
|
72
|
+
"label": "quota",
|
|
73
|
+
"kind": "measured",
|
|
74
|
+
"source": "fm quota-usage",
|
|
75
|
+
"available": false,
|
|
76
|
+
"unavailableReason": "this fm build exposes no quota command"
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Consumers should read `available` before trusting a metric. When a metric is unavailable, the corresponding fields are `null` (JSON) or blank (CSV) — never `0`.
|
|
83
|
+
|
|
84
|
+
Per-run rows in `results` carry:
|
|
85
|
+
|
|
86
|
+
| Field | Notes |
|
|
87
|
+
|-------|-------|
|
|
88
|
+
| `attempts` | Total `fm` invocations for this measured run, including retries. `1` when no retry was needed. |
|
|
89
|
+
| `ok` | Whether the call produced a response and exited cleanly. |
|
|
90
|
+
| `firstTokenMs`, `generationMs`, `tpotMs` | `null` when the run cannot supply them (no streaming, single-chunk answer, failed run). |
|
|
91
|
+
| `promptTokens`, `outputTokens`, `tokensPerSecond`, `decodeTokensPerSecond`, `prefillTokensPerSecond` | `null` when the `fm` build cannot count tokens. |
|
|
92
|
+
| `error` | Actionable failure text (`timed out after 30000ms`, `fm exited with code 3`, the first actionable line of `fm` stderr). |
|
|
93
|
+
|
|
94
|
+
## CSV
|
|
95
|
+
|
|
96
|
+
`--format csv` and `--out runs.csv` write per-run rows. Columns are stable; `attempts` was added in 0.7.0. Text cells that begin with `=`, `+`, `@`, or a non-numeric `-` are prefixed with a single quote so spreadsheets do not execute prompt or model output as a formula.
|
|
97
|
+
|
|
22
98
|
## Sharing results
|
|
23
99
|
|
|
24
100
|
1. **JSON** — best for automation and `fm-bench compare`. Save with `--out bench.json` or `--output-dir reports/`.
|
|
@@ -38,6 +114,7 @@ For fair comparison, match:
|
|
|
38
114
|
- Same `--profile` (or same `--prompt-file`)
|
|
39
115
|
- Same `--runs` and `--warmup`
|
|
40
116
|
- Same concurrency operating points (`--concurrency` or `--sweep-concurrency`)
|
|
117
|
+
- Same `fm` build (compare `capabilities.digest` / `environment.fmHelpDigest`)
|
|
41
118
|
- Same or intentionally changed macOS build (especially for beta-to-beta comparisons)
|
|
42
119
|
- Similar power/thermal state (see `environment.power` and `environment.thermal`)
|
|
43
120
|
|
|
@@ -50,4 +127,4 @@ fm-bench compare before.json after.json --strict # exit 2 if suites differ
|
|
|
50
127
|
|
|
51
128
|
## Legacy reports
|
|
52
129
|
|
|
53
|
-
Reports from fm-bench before 0.6.0 remain valid JSON. They lack `schemaVersion`, `reportId`, `suite`, and enriched `environment`. `validate` still checks required fields
|
|
130
|
+
Reports from fm-bench before 0.6.0 remain valid JSON. They lack `schemaVersion`, `reportId`, `suite`, `capabilities`, and the enriched `environment`. `validate` still checks required fields, and `compare` works on `summary` as before.
|
|
@@ -8,8 +8,6 @@
|
|
|
8
8
|
- **Node.js 20 or newer**.
|
|
9
9
|
- **Apple Intelligence enabled** on the device.
|
|
10
10
|
|
|
11
|
-
`pcc` (Private Cloud Compute) availability additionally depends on Apple's current eligibility. `fm-bench` reports it as skipped when `fm available --model pcc` reports unavailable.
|
|
12
|
-
|
|
13
11
|
## Version enforcement
|
|
14
12
|
|
|
15
13
|
Commands that launch `fm` benchmarks — the default `run` command and `models` — check the macOS version first and refuse to start on anything older than macOS 27:
|
|
@@ -21,6 +19,25 @@ Latest supported: macOS 27.0 or newer (fm is not available on older macOS releas
|
|
|
21
19
|
|
|
22
20
|
The process exits with code `2`. This is deliberate: running on an unsupported macOS could never produce a valid benchmark, so the CLI fails fast with the exact version it found and the latest supported macOS version.
|
|
23
21
|
|
|
22
|
+
The gate guards the **default** `fm` discovery path, because Apple only ships the CLI from macOS 27. If you explicitly provide a binary with `--fm-bin <path>` or `FM_BIN`, the host version is not a constraint and fm-bench proceeds on any platform — the capability probe still exits `2` if that binary is unusable.
|
|
23
|
+
|
|
24
24
|
Commands that only read local report files — `compare`, `history`, `validate`, `export`, and `legend` — are not gated and work anywhere Node.js runs.
|
|
25
25
|
|
|
26
26
|
`fm-bench doctor` still runs on unsupported hosts so you can diagnose the environment: it prints the detected macOS version, an explicit `macOS support` line, and the latest supported version.
|
|
27
|
+
|
|
28
|
+
## Which `fm` builds work
|
|
29
|
+
|
|
30
|
+
Support is capability-based rather than version-pinned. The CLI probes the installed `fm` and reports what it can measure; a build that exposes different subcommand names or fewer flags still works, with the unsupported metrics reported as unavailable. See [compatibility.md](./compatibility.md) for the detection policy and the verified build table.
|
|
31
|
+
|
|
32
|
+
## Models
|
|
33
|
+
|
|
34
|
+
`fm-bench` benchmarks exactly the models the installed `fm` reports — it does not assume that any particular cloud or adapter model exists. On the verified macOS 27.0 build that is the on-device `system` model only. If a build adds models (for example a Private Cloud Compute model), they are discovered automatically and reported with their own availability.
|
|
35
|
+
|
|
36
|
+
Requesting only models the build cannot run exits with code `2` before any benchmark starts, and `fm-bench models` shows the same reasons without failing:
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
fm-bench: No benchmark was run: none of the requested models are usable right now.
|
|
40
|
+
requested: pcc
|
|
41
|
+
pcc: not supported by this fm build (supported: system)
|
|
42
|
+
run "fm-bench models" to see availability and reasons
|
|
43
|
+
```
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "fm-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.7.1",
|
|
4
4
|
"description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -14,13 +14,14 @@
|
|
|
14
14
|
"LICENSE"
|
|
15
15
|
],
|
|
16
16
|
"scripts": {
|
|
17
|
-
"test": "node --test",
|
|
18
|
-
"lint": "node --check bin/fm-bench.js && find src test -name '*.js' -print0 | xargs -0 -n1 node --check",
|
|
19
|
-
"
|
|
17
|
+
"test": "node --test test/*.test.js",
|
|
18
|
+
"lint": "node --check bin/fm-bench.js && find src test scripts \\( -name '*.js' -o -name '*.mjs' \\) -print0 | xargs -0 -n1 node --check",
|
|
19
|
+
"check": "npm run lint && npm test && npm run check:pack",
|
|
20
|
+
"prepack": "npm run check >&2",
|
|
20
21
|
"release:patch": "npm version patch && git push --follow-tags",
|
|
21
22
|
"release:minor": "npm version minor && git push --follow-tags",
|
|
22
23
|
"release:major": "npm version major && git push --follow-tags",
|
|
23
|
-
"publish:dry-run": "
|
|
24
|
+
"publish:dry-run": "node scripts/publish-dry-run.mjs",
|
|
24
25
|
"check:pack": "node scripts/check-package.mjs"
|
|
25
26
|
},
|
|
26
27
|
"keywords": [
|
package/src/bench.js
CHANGED
|
@@ -1,51 +1,85 @@
|
|
|
1
1
|
import crypto from 'node:crypto';
|
|
2
|
-
import {
|
|
2
|
+
import { detectFmCapabilities } from './capabilities.js';
|
|
3
|
+
import { checkModelAvailability, collectEnvironment, countTokens, fmBinaryFromOptions, getQuotaUsage, respond } from './fm.js';
|
|
4
|
+
import { metricAvailability } from './metrics.js';
|
|
3
5
|
import { loadPrompts } from './prompts.js';
|
|
4
6
|
import { finalizeReportPayload } from './schema.js';
|
|
5
7
|
import { summarizeByModel } from './stats.js';
|
|
6
8
|
|
|
7
9
|
export async function inspectModels(options = {}) {
|
|
8
|
-
const
|
|
10
|
+
const fmBin = fmBinaryFromOptions(options);
|
|
11
|
+
const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
|
|
12
|
+
const discovered = {
|
|
13
|
+
fmBin,
|
|
14
|
+
models: capabilities.models,
|
|
15
|
+
help: capabilities.help,
|
|
16
|
+
capabilities
|
|
17
|
+
};
|
|
9
18
|
const requested = normalizeModelSelection(options.models);
|
|
10
19
|
const models = requested.length > 0
|
|
11
|
-
? discovered.models.
|
|
20
|
+
? requested.map((name) => discovered.models.find((model) => model.name === name)
|
|
21
|
+
?? { name, description: 'Requested model not reported by this fm build' })
|
|
12
22
|
: discovered.models;
|
|
13
23
|
|
|
14
|
-
const missing = requested.filter((name) => !models.some((model) => model.name === name));
|
|
15
|
-
for (const name of missing) {
|
|
16
|
-
models.push({ name, description: 'Requested model not reported by fm --help' });
|
|
17
|
-
}
|
|
18
|
-
|
|
19
24
|
const inspected = [];
|
|
20
25
|
for (const model of models) {
|
|
21
|
-
const availability = await checkModelAvailability(discovered.fmBin, model.name,
|
|
22
|
-
|
|
26
|
+
const availability = await checkModelAvailability(discovered.fmBin, model.name, {
|
|
27
|
+
...options,
|
|
28
|
+
capabilities
|
|
29
|
+
});
|
|
30
|
+
const quota = await getQuotaUsage(discovered.fmBin, model.name, {
|
|
31
|
+
...options,
|
|
32
|
+
capabilities
|
|
33
|
+
});
|
|
23
34
|
inspected.push({
|
|
24
35
|
...model,
|
|
25
36
|
available: availability.available,
|
|
26
|
-
|
|
27
|
-
|
|
37
|
+
unsupported: Boolean(availability.unsupported),
|
|
38
|
+
reason: availability.available ? '' : (availability.reason || availability.raw || 'unavailable'),
|
|
39
|
+
quota: quota.supported ? (quota.raw || quota.reason) : '',
|
|
40
|
+
quotaSupported: quota.supported,
|
|
41
|
+
quotaReason: quota.reason
|
|
28
42
|
});
|
|
29
43
|
}
|
|
30
44
|
|
|
31
45
|
return {
|
|
32
46
|
fmBin: discovered.fmBin,
|
|
33
47
|
models: inspected,
|
|
34
|
-
help: discovered.help
|
|
48
|
+
help: discovered.help,
|
|
49
|
+
capabilities
|
|
35
50
|
};
|
|
36
51
|
}
|
|
37
52
|
|
|
38
53
|
export async function runBenchmark(options = {}) {
|
|
39
54
|
const startedAt = new Date().toISOString();
|
|
55
|
+
const fmBin = fmBinaryFromOptions(options);
|
|
56
|
+
|
|
57
|
+
notify(options, { type: 'phase', phase: 'capabilities', message: 'probing fm capabilities' });
|
|
58
|
+
const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
|
|
59
|
+
if (!capabilities.ok) {
|
|
60
|
+
const error = new Error(capabilities.error
|
|
61
|
+
? `${capabilities.error}\nInstall Apple's fm CLI (macOS 27+) or point --fm-bin / FM_BIN at a compatible binary.`
|
|
62
|
+
: `No usable fm commands were found in ${fmBin} --help.`);
|
|
63
|
+
error.exitCode = 2;
|
|
64
|
+
throw error;
|
|
65
|
+
}
|
|
66
|
+
|
|
40
67
|
notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
|
|
41
68
|
const prompts = await loadPrompts(options);
|
|
42
69
|
notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
|
|
43
|
-
const inspection = await inspectModels(options);
|
|
70
|
+
const inspection = await inspectModels({ ...options, capabilities });
|
|
44
71
|
const modelStatuses = options.availableOnly
|
|
45
72
|
? inspection.models.filter((model) => model.available)
|
|
46
73
|
: inspection.models;
|
|
47
74
|
const runnableModels = modelStatuses.filter((model) => model.available);
|
|
48
|
-
|
|
75
|
+
if (runnableModels.length === 0) {
|
|
76
|
+
throw noRunnableModelsError(inspection.models, options);
|
|
77
|
+
}
|
|
78
|
+
const environment = await collectEnvironment(inspection.fmBin, { ...options, capabilities });
|
|
79
|
+
const metrics = metricAvailability(capabilities, {
|
|
80
|
+
stream: options.stream,
|
|
81
|
+
slo: Boolean(options.sloTtftMs || options.sloE2eMs || options.sloTpotMs)
|
|
82
|
+
});
|
|
49
83
|
const promptTokenCounts = new Map();
|
|
50
84
|
const concurrencies = normalizeConcurrencySweep(options);
|
|
51
85
|
const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
|
|
@@ -53,10 +87,13 @@ export async function runBenchmark(options = {}) {
|
|
|
53
87
|
notify(options, {
|
|
54
88
|
type: 'tokens:start',
|
|
55
89
|
total: prompts.length,
|
|
56
|
-
|
|
90
|
+
supported: metrics.promptTokens.available,
|
|
91
|
+
message: metrics.promptTokens.available ? 'counting prompt tokens' : 'token counting unavailable'
|
|
57
92
|
});
|
|
58
93
|
for (const prompt of prompts) {
|
|
59
|
-
const counted =
|
|
94
|
+
const counted = metrics.promptTokens.available
|
|
95
|
+
? await countTokens(inspection.fmBin, prompt.prompt, { ...options, capabilities })
|
|
96
|
+
: { ok: false, count: null };
|
|
60
97
|
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
61
98
|
notify(options, {
|
|
62
99
|
type: 'tokens:progress',
|
|
@@ -80,10 +117,12 @@ export async function runBenchmark(options = {}) {
|
|
|
80
117
|
for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
|
|
81
118
|
const scenario = await runScenario({
|
|
82
119
|
fmBin: inspection.fmBin,
|
|
120
|
+
capabilities,
|
|
83
121
|
prompts,
|
|
84
122
|
runnableModels,
|
|
85
123
|
modelStatuses,
|
|
86
124
|
promptTokenCounts,
|
|
125
|
+
tokenCounting: metrics.outputTokens.available,
|
|
87
126
|
options,
|
|
88
127
|
concurrency,
|
|
89
128
|
scenarioIndex: scenarioIndex + 1,
|
|
@@ -123,6 +162,15 @@ export async function runBenchmark(options = {}) {
|
|
|
123
162
|
finishedAt: new Date().toISOString(),
|
|
124
163
|
options: publicOptions(options),
|
|
125
164
|
environment,
|
|
165
|
+
capabilities: {
|
|
166
|
+
bin: capabilities.bin,
|
|
167
|
+
digest: capabilities.digest,
|
|
168
|
+
commands: capabilities.commands,
|
|
169
|
+
models: capabilities.models,
|
|
170
|
+
features: capabilities.features,
|
|
171
|
+
warnings: capabilities.warnings
|
|
172
|
+
},
|
|
173
|
+
metrics,
|
|
126
174
|
prompts: prompts.map((prompt) => ({
|
|
127
175
|
id: prompt.id,
|
|
128
176
|
prompt: prompt.prompt,
|
|
@@ -145,10 +193,12 @@ export async function runBenchmark(options = {}) {
|
|
|
145
193
|
async function runScenario(context) {
|
|
146
194
|
const {
|
|
147
195
|
fmBin,
|
|
196
|
+
capabilities,
|
|
148
197
|
prompts,
|
|
149
198
|
runnableModels,
|
|
150
199
|
modelStatuses,
|
|
151
200
|
promptTokenCounts,
|
|
201
|
+
tokenCounting,
|
|
152
202
|
options,
|
|
153
203
|
concurrency,
|
|
154
204
|
scenarioIndex,
|
|
@@ -172,6 +222,7 @@ async function runScenario(context) {
|
|
|
172
222
|
for (const model of runnableModels) {
|
|
173
223
|
await respond(fmBin, model.name, prompts[0].prompt, {
|
|
174
224
|
...options,
|
|
225
|
+
capabilities,
|
|
175
226
|
stream: false
|
|
176
227
|
});
|
|
177
228
|
warmupCompleted += 1;
|
|
@@ -206,7 +257,15 @@ async function runScenario(context) {
|
|
|
206
257
|
total: jobs.length
|
|
207
258
|
});
|
|
208
259
|
await runLimited(jobs, concurrency, async (job) => {
|
|
209
|
-
const result = await runSingleBenchmark(
|
|
260
|
+
const result = await runSingleBenchmark({
|
|
261
|
+
fmBin,
|
|
262
|
+
capabilities,
|
|
263
|
+
job,
|
|
264
|
+
promptTokenCounts,
|
|
265
|
+
tokenCounting,
|
|
266
|
+
options,
|
|
267
|
+
benchmarkStartedAt
|
|
268
|
+
});
|
|
210
269
|
results.push(result);
|
|
211
270
|
if (onMeasuredResult) onMeasuredResult(result);
|
|
212
271
|
if (!result.ok && options.failFast) {
|
|
@@ -229,13 +288,17 @@ async function runScenario(context) {
|
|
|
229
288
|
};
|
|
230
289
|
}
|
|
231
290
|
|
|
232
|
-
async function runSingleBenchmark(
|
|
291
|
+
async function runSingleBenchmark(context) {
|
|
292
|
+
const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, options, benchmarkStartedAt } = context;
|
|
233
293
|
const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
|
|
234
294
|
const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
235
295
|
let response;
|
|
296
|
+
let attempts = 0;
|
|
236
297
|
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
298
|
+
attempts = attempt;
|
|
237
299
|
response = await respond(fmBin, job.model.name, job.prompt.prompt, {
|
|
238
300
|
...options,
|
|
301
|
+
capabilities,
|
|
239
302
|
stream: options.stream
|
|
240
303
|
});
|
|
241
304
|
if (response.ok || attempt >= maxAttempts) break;
|
|
@@ -243,28 +306,39 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
243
306
|
await new Promise((resolve) => setTimeout(resolve, backoffMs));
|
|
244
307
|
}
|
|
245
308
|
const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
: { ok: false, count: null };
|
|
309
|
+
|
|
310
|
+
const ok = response.ok;
|
|
249
311
|
const seconds = response.durationMs / 1000;
|
|
250
|
-
const
|
|
251
|
-
|
|
312
|
+
const chunks = response.stdoutChunks ?? 0;
|
|
313
|
+
|
|
314
|
+
// A single stdout chunk carries the whole answer, so the streamed portion is
|
|
315
|
+
// not separable: report generation time and TPOT as unavailable rather than
|
|
316
|
+
// as a near-zero decode phase.
|
|
317
|
+
const firstTokenMs = ok ? response.firstOutputMs : null;
|
|
318
|
+
const generationMs = ok && firstTokenMs != null && chunks > 1
|
|
252
319
|
? Math.max(0, response.durationMs - firstTokenMs)
|
|
253
320
|
: null;
|
|
321
|
+
|
|
322
|
+
const outputTokens = ok && tokenCounting
|
|
323
|
+
? await countTokens(fmBin, response.output, { ...options, capabilities })
|
|
324
|
+
: { ok: false, count: null };
|
|
254
325
|
const countedOutputTokens = outputTokens.ok ? outputTokens.count : null;
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
const
|
|
258
|
-
|
|
259
|
-
? decodeTokenCount / (generationMs / 1000)
|
|
326
|
+
// Two decode tokens is the minimum for an inter-token interval that is not
|
|
327
|
+
// simply the inverse of a single chunk gap.
|
|
328
|
+
const decodeTokenCount = countedOutputTokens != null && countedOutputTokens > 2
|
|
329
|
+
? countedOutputTokens - 1
|
|
260
330
|
: null;
|
|
331
|
+
const hasDecodeCadence = generationMs != null && generationMs > 0 && decodeTokenCount != null;
|
|
332
|
+
const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
|
|
333
|
+
const decodeTokensPerSecond = hasDecodeCadence ? decodeTokenCount / (generationMs / 1000) : null;
|
|
334
|
+
|
|
261
335
|
const chars = response.output.length;
|
|
262
336
|
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
263
|
-
const promptTokens = promptTokenCounts.get(job.prompt.id);
|
|
264
|
-
const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
|
|
337
|
+
const promptTokens = promptTokenCounts.get(job.prompt.id) ?? null;
|
|
338
|
+
const prefillTokensPerSecond = promptTokens != null && firstTokenMs != null && firstTokenMs > 0
|
|
265
339
|
? promptTokens / (firstTokenMs / 1000)
|
|
266
340
|
: null;
|
|
267
|
-
const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
|
|
341
|
+
const chunkGapsMs = ok ? chunkGaps(response.stdoutChunkTimesMs) : [];
|
|
268
342
|
const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
|
|
269
343
|
|
|
270
344
|
return {
|
|
@@ -272,7 +346,8 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
272
346
|
concurrency: job.concurrency,
|
|
273
347
|
promptId: job.prompt.id,
|
|
274
348
|
run: job.run,
|
|
275
|
-
|
|
349
|
+
attempts,
|
|
350
|
+
ok,
|
|
276
351
|
durationMs: response.durationMs,
|
|
277
352
|
firstTokenMs,
|
|
278
353
|
generationMs,
|
|
@@ -284,7 +359,7 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
284
359
|
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
285
360
|
decodeTokensPerSecond,
|
|
286
361
|
prefillTokensPerSecond,
|
|
287
|
-
charsPerSecond: seconds > 0 ? chars / seconds :
|
|
362
|
+
charsPerSecond: seconds > 0 ? chars / seconds : null,
|
|
288
363
|
startOffsetMs,
|
|
289
364
|
endOffsetMs,
|
|
290
365
|
streamed: response.streamed,
|
|
@@ -293,14 +368,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
293
368
|
chunkGapsMs,
|
|
294
369
|
chunkGapAvgMs: average(chunkGapsMs),
|
|
295
370
|
chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
|
|
296
|
-
outputHash:
|
|
297
|
-
good:
|
|
371
|
+
outputHash: ok ? hashOutput(response.output) : null,
|
|
372
|
+
good: ok ? evaluateSlo({
|
|
298
373
|
firstTokenMs,
|
|
299
374
|
durationMs: response.durationMs,
|
|
300
375
|
tpotMs
|
|
301
376
|
}, options) : false,
|
|
302
377
|
output: options.captureOutput ? response.output : undefined,
|
|
303
|
-
error:
|
|
378
|
+
error: ok ? '' : (response.error || `fm exited with code ${response.code ?? response.signal}`)
|
|
304
379
|
};
|
|
305
380
|
}
|
|
306
381
|
|
|
@@ -318,6 +393,23 @@ async function runLimited(items, concurrency, worker, options = {}) {
|
|
|
318
393
|
await Promise.all(workers);
|
|
319
394
|
}
|
|
320
395
|
|
|
396
|
+
// A benchmark with nothing to run is a configuration error, not an empty
|
|
397
|
+
// report: say which models were asked for and which ones the build supports.
|
|
398
|
+
function noRunnableModelsError(models, options) {
|
|
399
|
+
const requested = normalizeModelSelection(options.models);
|
|
400
|
+
const supported = models.filter((model) => !model.unsupported).map((model) => model.name);
|
|
401
|
+
const lines = ['No benchmark was run: none of the requested models are usable right now.'];
|
|
402
|
+
if (requested.length > 0) lines.push(` requested: ${requested.join(', ')}`);
|
|
403
|
+
if (supported.length > 0) lines.push(` models reported by this fm build: ${supported.join(', ')}`);
|
|
404
|
+
for (const model of models) {
|
|
405
|
+
if (model.reason) lines.push(` ${model.name}: ${model.reason}`);
|
|
406
|
+
}
|
|
407
|
+
lines.push(' run "fm-bench models" to see availability and reasons');
|
|
408
|
+
const error = new Error(lines.join('\n'));
|
|
409
|
+
error.exitCode = 2;
|
|
410
|
+
return error;
|
|
411
|
+
}
|
|
412
|
+
|
|
321
413
|
function normalizeModelSelection(models) {
|
|
322
414
|
if (!models) return [];
|
|
323
415
|
const values = Array.isArray(models) ? models : [models];
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
// Capability detection for the installed `fm` CLI.
|
|
2
|
+
//
|
|
3
|
+
// fm-bench talks to whatever `fm` build is on the machine, and Apple has
|
|
4
|
+
// changed subcommand names and flags between releases (for example
|
|
5
|
+
// `count-tokens` replaces the older `token-count`). Everything downstream
|
|
6
|
+
// reads this detection result instead of hardcoding subcommand names, so a
|
|
7
|
+
// compatible `fm` build is supported without a code change and an
|
|
8
|
+
// incompatible one is reported instead of producing wrong numbers.
|
|
9
|
+
|
|
10
|
+
import crypto from 'node:crypto';
|
|
11
|
+
import { stripAnsi } from './ansi.js';
|
|
12
|
+
import { parseAvailabilityList, parseModelsFromHelp } from './fm-help.js';
|
|
13
|
+
import { runProcess } from './process.js';
|
|
14
|
+
|
|
15
|
+
const SECTION_HEADER = /^\s*[A-Z][A-Z0-9 /-]+\s*$/;
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Parse the COMMANDS section of `fm --help`.
|
|
19
|
+
* @param {string} helpText
|
|
20
|
+
* @returns {string[]}
|
|
21
|
+
*/
|
|
22
|
+
export function parseCommandsFromHelp(helpText = '') {
|
|
23
|
+
const lines = stripAnsi(helpText).split(/\r?\n/);
|
|
24
|
+
const commands = [];
|
|
25
|
+
let inCommands = false;
|
|
26
|
+
|
|
27
|
+
for (const line of lines) {
|
|
28
|
+
if (/^\s*COMMANDS\s*$/.test(line)) {
|
|
29
|
+
inCommands = true;
|
|
30
|
+
continue;
|
|
31
|
+
}
|
|
32
|
+
if (inCommands && SECTION_HEADER.test(line)) {
|
|
33
|
+
inCommands = false;
|
|
34
|
+
}
|
|
35
|
+
if (!inCommands) continue;
|
|
36
|
+
|
|
37
|
+
const match = line.match(/^\s{2,}([a-z][a-z0-9-]*)\s{2,}\S/);
|
|
38
|
+
if (match) commands.push(match[1]);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
return commands;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* True when a long flag appears in `fm` help text. Apple prints boolean flags
|
|
46
|
+
* with a negatable form (`--[no-]stream`), so accept that shape too.
|
|
47
|
+
* @param {string} helpText
|
|
48
|
+
* @param {string} flag e.g. "--use-case"
|
|
49
|
+
*/
|
|
50
|
+
export function hasFlagInHelp(helpText, flag) {
|
|
51
|
+
const clean = stripAnsi(helpText);
|
|
52
|
+
if (new RegExp(`${escapeRegExp(flag)}\\b`).test(clean)) return true;
|
|
53
|
+
if (flag.startsWith('--no-')) {
|
|
54
|
+
return clean.includes(`[no-]${flag.slice('--no-'.length)}`);
|
|
55
|
+
}
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function helpDigest(helpText = '') {
|
|
60
|
+
const text = stripAnsi(helpText).trim();
|
|
61
|
+
if (!text) return null;
|
|
62
|
+
return crypto.createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Pick the subcommand this `fm` build exposes for token counting.
|
|
67
|
+
* @returns {string|null}
|
|
68
|
+
*/
|
|
69
|
+
export function resolveTokenCountCommand(commands = []) {
|
|
70
|
+
if (commands.includes('count-tokens')) return 'count-tokens';
|
|
71
|
+
if (commands.includes('token-count')) return 'token-count';
|
|
72
|
+
return null;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function buildFeatures(helpText, respondHelpText, commands) {
|
|
76
|
+
const tokenCountCommand = resolveTokenCountCommand(commands);
|
|
77
|
+
const respondHelp = respondHelpText || '';
|
|
78
|
+
return {
|
|
79
|
+
tokenCounting: tokenCountCommand != null,
|
|
80
|
+
tokenCountCommand,
|
|
81
|
+
quota: commands.includes('quota-usage'),
|
|
82
|
+
streaming: hasFlagInHelp(respondHelp, '--no-stream') || hasFlagInHelp(respondHelp, '--stream'),
|
|
83
|
+
modelSelection: hasFlagInHelp(respondHelp, '--model'),
|
|
84
|
+
instructions: hasFlagInHelp(respondHelp, '--instructions'),
|
|
85
|
+
greedy: hasFlagInHelp(respondHelp, '--greedy'),
|
|
86
|
+
useCase: hasFlagInHelp(respondHelp, '--use-case'),
|
|
87
|
+
guardrails: hasFlagInHelp(respondHelp, '--guardrails'),
|
|
88
|
+
images: hasFlagInHelp(respondHelp, '--image'),
|
|
89
|
+
tools: hasFlagInHelp(respondHelp, '--tool'),
|
|
90
|
+
structuredOutput: hasFlagInHelp(respondHelp, '--schema'),
|
|
91
|
+
server: commands.includes('serve')
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function capabilityWarnings(commands, features, models) {
|
|
96
|
+
const warnings = [];
|
|
97
|
+
if (!features.tokenCounting) {
|
|
98
|
+
warnings.push('this fm build exposes no token-counting command, so token counts and token throughput are unavailable');
|
|
99
|
+
}
|
|
100
|
+
if (!features.quota) {
|
|
101
|
+
warnings.push('this fm build exposes no quota command, so quota is not reported');
|
|
102
|
+
}
|
|
103
|
+
if (!features.streaming) {
|
|
104
|
+
warnings.push('this fm build does not document a streaming flag, so TTFT cannot be measured');
|
|
105
|
+
}
|
|
106
|
+
if (models.length === 0) {
|
|
107
|
+
warnings.push('could not discover any models from fm help output');
|
|
108
|
+
}
|
|
109
|
+
if (commands.length === 0) {
|
|
110
|
+
warnings.push('could not read the command list from fm --help');
|
|
111
|
+
}
|
|
112
|
+
return warnings;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Probe the installed `fm` binary once and describe what it can do.
|
|
117
|
+
*
|
|
118
|
+
* @param {string} fmBin
|
|
119
|
+
* @param {{ timeoutMs?: number, env?: NodeJS.ProcessEnv, help?: { text: string } }} [options]
|
|
120
|
+
*/
|
|
121
|
+
export async function detectFmCapabilities(fmBin, options = {}) {
|
|
122
|
+
const timeoutMs = options.timeoutMs ?? 10_000;
|
|
123
|
+
const env = options.env ?? process.env;
|
|
124
|
+
let helpText = options.help?.text ?? null;
|
|
125
|
+
|
|
126
|
+
if (helpText == null) {
|
|
127
|
+
const help = await runProcess(fmBin, ['--help'], { timeoutMs, env });
|
|
128
|
+
if (help.error) {
|
|
129
|
+
return {
|
|
130
|
+
ok: false,
|
|
131
|
+
bin: fmBin,
|
|
132
|
+
error: `Unable to execute ${fmBin}: ${help.stderr || help.error.message}`,
|
|
133
|
+
commands: [],
|
|
134
|
+
models: [],
|
|
135
|
+
features: buildFeatures('', '', []),
|
|
136
|
+
digest: null,
|
|
137
|
+
help: '',
|
|
138
|
+
warnings: [`cannot execute ${fmBin}`]
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
helpText = `${help.stdout}${help.stderr}`;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
const cleanHelp = stripAnsi(helpText);
|
|
145
|
+
const commands = parseCommandsFromHelp(cleanHelp);
|
|
146
|
+
|
|
147
|
+
let respondHelpText = '';
|
|
148
|
+
if (commands.includes('respond')) {
|
|
149
|
+
const respondHelp = await runProcess(fmBin, ['respond', '--help'], { timeoutMs, env });
|
|
150
|
+
if (!respondHelp.error) respondHelpText = `${respondHelp.stdout}${respondHelp.stderr}`;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const features = buildFeatures(cleanHelp, respondHelpText, commands);
|
|
154
|
+
let models = parseModelsFromHelp(cleanHelp);
|
|
155
|
+
|
|
156
|
+
if (models.length === 0 && commands.includes('available')) {
|
|
157
|
+
models = await discoverModelsFromAvailability(fmBin, { ...options, env });
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
ok: commands.length > 0,
|
|
162
|
+
bin: fmBin,
|
|
163
|
+
commands,
|
|
164
|
+
models,
|
|
165
|
+
features,
|
|
166
|
+
digest: helpDigest(cleanHelp),
|
|
167
|
+
help: cleanHelp,
|
|
168
|
+
warnings: capabilityWarnings(commands, features, models)
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Fallback discovery: ask `fm available` without a model filter and read the
|
|
174
|
+
* model names it reports. Used when `fm --help` has no MODELS section.
|
|
175
|
+
*/
|
|
176
|
+
async function discoverModelsFromAvailability(fmBin, options = {}) {
|
|
177
|
+
const result = await runProcess(fmBin, ['available'], {
|
|
178
|
+
timeoutMs: options.timeoutMs ?? 15_000,
|
|
179
|
+
env: options.env ?? process.env
|
|
180
|
+
});
|
|
181
|
+
if (result.error) return [];
|
|
182
|
+
return parseAvailabilityList(`${result.stdout}${result.stderr}`)
|
|
183
|
+
.map((model) => ({ name: model.name, description: '' }));
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
function escapeRegExp(value) {
|
|
187
|
+
return String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
188
|
+
}
|