fm-bench 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  # Report format (schema v1)
2
2
 
3
- Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** include a versioned schema so you can validate, share, and compare results across machines.
3
+ Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** include a versioned schema so you can validate, share, and compare results across machines. The schema version stayed at `1` in 0.7.0: the new `capabilities`, `metrics`, and per-result `attempts` fields are additive, so older readers and older reports both keep working.
4
4
 
5
5
  ## Top-level fields
6
6
 
@@ -13,12 +13,88 @@ Every measured run can be saved as JSON. Reports from fm-bench **0.6.0+** includ
13
13
  | `startedAt` / `finishedAt` | ISO-8601 timestamps |
14
14
  | `options` | Public run configuration (profile, runs, concurrency, SLOs, tags, note) |
15
15
  | `environment` | Host fingerprint: platform, arch, Node, hardware model, CPU, memory, macOS version/build, `fm` help digest, thermal/power snapshot |
16
+ | `capabilities` | What the installed `fm` build exposes (see below) |
17
+ | `metrics` | Per-metric availability and provenance for this run (see below) |
16
18
  | `suite` | Derived suite key + fingerprint for apples-to-apples comparison |
17
19
  | `prompts` | Prompt ids, text, and token counts |
18
20
  | `models` | Discovered models and availability |
19
21
  | `summary` | Per-model / per-concurrency roll-up statistics |
20
22
  | `results` | Per-run measurements (optional `output` when `--capture-output`) |
21
23
 
24
+ ## `capabilities`
25
+
26
+ Detected once per run from `fm --help` and `fm respond --help`.
27
+
28
+ ```json
29
+ {
30
+ "capabilities": {
31
+ "bin": "fm",
32
+ "digest": "921a7839714e3707",
33
+ "commands": ["available", "chat", "count-tokens", "license", "respond", "schema", "serve"],
34
+ "models": [{ "name": "system", "description": "On-device Apple Foundation Model" }],
35
+ "features": {
36
+ "tokenCounting": true,
37
+ "tokenCountCommand": "count-tokens",
38
+ "quota": false,
39
+ "streaming": true,
40
+ "modelSelection": true,
41
+ "instructions": true,
42
+ "greedy": true,
43
+ "useCase": true,
44
+ "guardrails": true,
45
+ "images": true,
46
+ "tools": true,
47
+ "structuredOutput": true,
48
+ "server": true
49
+ },
50
+ "warnings": ["this fm build exposes no quota command, so quota is not reported"]
51
+ }
52
+ }
53
+ ```
54
+
55
+ `digest` is a short hash of the normalized `fm --help` output, so a report records which CLI surface produced it. Reports from two different `fm` builds are not directly comparable even on the same machine.
56
+
57
+ ## `metrics`
58
+
59
+ Each entry states how a metric was obtained for this run and whether it was available. `kind` is one of `measured`, `proxy`, `derived`, or `controlled`.
60
+
61
+ ```json
62
+ {
63
+ "metrics": {
64
+ "ttft": {
65
+ "label": "TTFT",
66
+ "kind": "proxy",
67
+ "source": "arrival time of the first streamed stdout chunk",
68
+ "available": true,
69
+ "unavailableReason": ""
70
+ },
71
+ "quota": {
72
+ "label": "quota",
73
+ "kind": "measured",
74
+ "source": "fm quota-usage",
75
+ "available": false,
76
+ "unavailableReason": "this fm build exposes no quota command"
77
+ }
78
+ }
79
+ }
80
+ ```
81
+
82
+ Consumers should read `available` before trusting a metric. When a metric is unavailable, the corresponding fields are `null` (JSON) or blank (CSV) — never `0`.
83
+
84
+ Per-run rows in `results` carry:
85
+
86
+ | Field | Notes |
87
+ |-------|-------|
88
+ | `attempts` | Total `fm` invocations for this measured run, including retries. `1` when no retry was needed. |
89
+ | `ok` | Whether the call produced a response and exited cleanly. |
90
+ | `firstTokenMs`, `generationMs`, `tpotMs` | `null` when the run cannot supply them (no streaming, single-chunk answer, failed run). |
91
+ | `promptTokens`, `outputTokens`, `tokensPerSecond`, `decodeTokensPerSecond`, `prefillTokensPerSecond` | `null` when the `fm` build cannot count tokens. |
92
+ | `error` | Actionable failure text (`timed out after 30000ms`, `fm exited with code 3`, the first actionable line of `fm` stderr). |
93
+
94
+ ## CSV
95
+
96
+ `--format csv` and `--out runs.csv` write per-run rows. Columns are stable; `attempts` was added in 0.7.0. Text cells that begin with `=`, `+`, `@`, or a non-numeric `-` are prefixed with a single quote so spreadsheets do not execute prompt or model output as a formula.
97
+
22
98
  ## Sharing results
23
99
 
24
100
  1. **JSON** — best for automation and `fm-bench compare`. Save with `--out bench.json` or `--output-dir reports/`.
@@ -38,6 +114,7 @@ For fair comparison, match:
38
114
  - Same `--profile` (or same `--prompt-file`)
39
115
  - Same `--runs` and `--warmup`
40
116
  - Same concurrency operating points (`--concurrency` or `--sweep-concurrency`)
117
+ - Same `fm` build (compare `capabilities.digest` / `environment.fmHelpDigest`)
41
118
  - Same or intentionally changed macOS build (especially for beta-to-beta comparisons)
42
119
  - Similar power/thermal state (see `environment.power` and `environment.thermal`)
43
120
 
@@ -50,4 +127,4 @@ fm-bench compare before.json after.json --strict # exit 2 if suites differ
50
127
 
51
128
  ## Legacy reports
52
129
 
53
- Reports from fm-bench before 0.6.0 remain valid JSON. They lack `schemaVersion`, `reportId`, `suite`, and enriched `environment`. `validate` still checks required fields; `compare` works on `summary` as before.
130
+ Reports from fm-bench before 0.6.0 remain valid JSON. They lack `schemaVersion`, `reportId`, `suite`, `capabilities`, and the enriched `environment`. `validate` still checks required fields, and `compare` works on `summary` as before.
@@ -0,0 +1,43 @@
1
+ # Supported platforms
2
+
3
+ `fm-bench` benchmarks Apple's `fm` command, so support follows the platforms where Apple ships that CLI.
4
+
5
+ ## Requirements
6
+
7
+ - **macOS 27.0 or newer** — Apple's `fm` CLI is preinstalled starting with macOS 27. Older macOS releases do not ship `fm`, so `fm-bench` cannot run there.
8
+ - **Node.js 20 or newer**.
9
+ - **Apple Intelligence enabled** on the device.
10
+
11
+ ## Version enforcement
12
+
13
+ Commands that launch `fm` benchmarks — the default `run` command and `models` — check the macOS version first and refuse to start on anything older than macOS 27:
14
+
15
+ ```text
16
+ fm-bench: unsupported macOS: detected macOS 26.1, but fm-bench requires macOS 27 or newer (Apple's fm CLI is preinstalled there).
17
+ Latest supported: macOS 27.0 or newer (fm is not available on older macOS releases).
18
+ ```
19
+
20
+ The process exits with code `2`. This is deliberate: running on an unsupported macOS could never produce a valid benchmark, so the CLI fails fast with the exact version it found and the latest supported macOS version.
21
+
22
+ The gate guards the **default** `fm` discovery path, because Apple only ships the CLI from macOS 27. If you explicitly provide a binary with `--fm-bin <path>` or `FM_BIN`, the host version is not a constraint and fm-bench proceeds on any platform — the capability probe still exits `2` if that binary is unusable.
23
+
24
+ Commands that only read local report files — `compare`, `history`, `validate`, `export`, and `legend` — are not gated and work anywhere Node.js runs.
25
+
26
+ `fm-bench doctor` still runs on unsupported hosts so you can diagnose the environment: it prints the detected macOS version, an explicit `macOS support` line, and the latest supported version.
27
+
28
+ ## Which `fm` builds work
29
+
30
+ Support is capability-based rather than version-pinned. The CLI probes the installed `fm` and reports what it can measure; a build that exposes different subcommand names or fewer flags still works, with the unsupported metrics reported as unavailable. See [compatibility.md](./compatibility.md) for the detection policy and the verified build table.
31
+
32
+ ## Models
33
+
34
+ `fm-bench` benchmarks exactly the models the installed `fm` reports — it does not assume that any particular cloud or adapter model exists. On the verified macOS 27.0 build that is the on-device `system` model only. If a build adds models (for example a Private Cloud Compute model), they are discovered automatically and reported with their own availability.
35
+
36
+ Requesting only models the build cannot run exits with code `2` before any benchmark starts, and `fm-bench models` shows the same reasons without failing:
37
+
38
+ ```text
39
+ fm-bench: No benchmark was run: none of the requested models are usable right now.
40
+ requested: pcc
41
+ pcc: not supported by this fm build (supported: system)
42
+ run "fm-bench models" to see availability and reasons
43
+ ```
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.6.2",
3
+ "version": "0.7.0",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -14,13 +14,15 @@
14
14
  "LICENSE"
15
15
  ],
16
16
  "scripts": {
17
- "test": "node --test",
18
- "lint": "node --check bin/fm-bench.js && find src test -name '*.js' -print0 | xargs -0 -n1 node --check",
19
- "prepack": "npm test && npm run lint",
17
+ "test": "node --test test/*.test.js",
18
+ "lint": "node --check bin/fm-bench.js && find src test scripts \\( -name '*.js' -o -name '*.mjs' \\) -print0 | xargs -0 -n1 node --check",
19
+ "check": "npm run lint && npm test && npm run check:pack",
20
+ "prepack": "npm run check >&2",
20
21
  "release:patch": "npm version patch && git push --follow-tags",
21
22
  "release:minor": "npm version minor && git push --follow-tags",
22
23
  "release:major": "npm version major && git push --follow-tags",
23
- "publish:dry-run": "npm publish --dry-run --access public"
24
+ "publish:dry-run": "npm publish --dry-run --access public",
25
+ "check:pack": "node scripts/check-package.mjs"
24
26
  },
25
27
  "keywords": [
26
28
  "apple",
@@ -46,5 +48,8 @@
46
48
  "publishConfig": {
47
49
  "access": "public",
48
50
  "registry": "https://registry.npmjs.org/"
49
- }
51
+ },
52
+ "os": [
53
+ "darwin"
54
+ ]
50
55
  }
package/src/bench.js CHANGED
@@ -1,51 +1,85 @@
1
1
  import crypto from 'node:crypto';
2
- import { checkModelAvailability, collectEnvironment, countTokens, discoverModels, getQuotaUsage, respond } from './fm.js';
2
+ import { detectFmCapabilities } from './capabilities.js';
3
+ import { checkModelAvailability, collectEnvironment, countTokens, getQuotaUsage, respond } from './fm.js';
4
+ import { metricAvailability } from './metrics.js';
3
5
  import { loadPrompts } from './prompts.js';
4
6
  import { finalizeReportPayload } from './schema.js';
5
7
  import { summarizeByModel } from './stats.js';
6
8
 
7
9
  export async function inspectModels(options = {}) {
8
- const discovered = await discoverModels(options);
10
+ const fmBin = options.fmBin || process.env.FM_BIN || 'fm';
11
+ const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
12
+ const discovered = {
13
+ fmBin,
14
+ models: capabilities.models,
15
+ help: capabilities.help,
16
+ capabilities
17
+ };
9
18
  const requested = normalizeModelSelection(options.models);
10
19
  const models = requested.length > 0
11
- ? discovered.models.filter((model) => requested.includes(model.name))
20
+ ? requested.map((name) => discovered.models.find((model) => model.name === name)
21
+ ?? { name, description: 'Requested model not reported by this fm build' })
12
22
  : discovered.models;
13
23
 
14
- const missing = requested.filter((name) => !models.some((model) => model.name === name));
15
- for (const name of missing) {
16
- models.push({ name, description: 'Requested model not reported by fm --help' });
17
- }
18
-
19
24
  const inspected = [];
20
25
  for (const model of models) {
21
- const availability = await checkModelAvailability(discovered.fmBin, model.name, options);
22
- const quota = await getQuotaUsage(discovered.fmBin, model.name, options);
26
+ const availability = await checkModelAvailability(discovered.fmBin, model.name, {
27
+ ...options,
28
+ capabilities
29
+ });
30
+ const quota = await getQuotaUsage(discovered.fmBin, model.name, {
31
+ ...options,
32
+ capabilities
33
+ });
23
34
  inspected.push({
24
35
  ...model,
25
36
  available: availability.available,
26
- reason: availability.reason || availability.raw,
27
- quota: quota.raw
37
+ unsupported: Boolean(availability.unsupported),
38
+ reason: availability.available ? '' : (availability.reason || availability.raw || 'unavailable'),
39
+ quota: quota.supported ? (quota.raw || quota.reason) : '',
40
+ quotaSupported: quota.supported,
41
+ quotaReason: quota.reason
28
42
  });
29
43
  }
30
44
 
31
45
  return {
32
46
  fmBin: discovered.fmBin,
33
47
  models: inspected,
34
- help: discovered.help
48
+ help: discovered.help,
49
+ capabilities
35
50
  };
36
51
  }
37
52
 
38
53
  export async function runBenchmark(options = {}) {
39
54
  const startedAt = new Date().toISOString();
55
+ const fmBin = options.fmBin || process.env.FM_BIN || 'fm';
56
+
57
+ notify(options, { type: 'phase', phase: 'capabilities', message: 'probing fm capabilities' });
58
+ const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
59
+ if (!capabilities.ok) {
60
+ const error = new Error(capabilities.error
61
+ ? `${capabilities.error}\nInstall Apple's fm CLI (macOS 27+) or point --fm-bin / FM_BIN at a compatible binary.`
62
+ : `No usable fm commands were found in ${fmBin} --help.`);
63
+ error.exitCode = 2;
64
+ throw error;
65
+ }
66
+
40
67
  notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
41
68
  const prompts = await loadPrompts(options);
42
69
  notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
43
- const inspection = await inspectModels(options);
70
+ const inspection = await inspectModels({ ...options, capabilities });
44
71
  const modelStatuses = options.availableOnly
45
72
  ? inspection.models.filter((model) => model.available)
46
73
  : inspection.models;
47
74
  const runnableModels = modelStatuses.filter((model) => model.available);
48
- const environment = await collectEnvironment(inspection.fmBin);
75
+ if (runnableModels.length === 0) {
76
+ throw noRunnableModelsError(inspection.models, options);
77
+ }
78
+ const environment = await collectEnvironment(inspection.fmBin, { ...options, capabilities });
79
+ const metrics = metricAvailability(capabilities, {
80
+ stream: options.stream,
81
+ slo: Boolean(options.sloTtftMs || options.sloE2eMs || options.sloTpotMs)
82
+ });
49
83
  const promptTokenCounts = new Map();
50
84
  const concurrencies = normalizeConcurrencySweep(options);
51
85
  const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
@@ -53,10 +87,13 @@ export async function runBenchmark(options = {}) {
53
87
  notify(options, {
54
88
  type: 'tokens:start',
55
89
  total: prompts.length,
56
- message: 'counting prompt tokens'
90
+ supported: metrics.promptTokens.available,
91
+ message: metrics.promptTokens.available ? 'counting prompt tokens' : 'token counting unavailable'
57
92
  });
58
93
  for (const prompt of prompts) {
59
- const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
94
+ const counted = metrics.promptTokens.available
95
+ ? await countTokens(inspection.fmBin, prompt.prompt, { ...options, capabilities })
96
+ : { ok: false, count: null };
60
97
  promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
61
98
  notify(options, {
62
99
  type: 'tokens:progress',
@@ -80,10 +117,12 @@ export async function runBenchmark(options = {}) {
80
117
  for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
81
118
  const scenario = await runScenario({
82
119
  fmBin: inspection.fmBin,
120
+ capabilities,
83
121
  prompts,
84
122
  runnableModels,
85
123
  modelStatuses,
86
124
  promptTokenCounts,
125
+ tokenCounting: metrics.outputTokens.available,
87
126
  options,
88
127
  concurrency,
89
128
  scenarioIndex: scenarioIndex + 1,
@@ -123,6 +162,15 @@ export async function runBenchmark(options = {}) {
123
162
  finishedAt: new Date().toISOString(),
124
163
  options: publicOptions(options),
125
164
  environment,
165
+ capabilities: {
166
+ bin: capabilities.bin,
167
+ digest: capabilities.digest,
168
+ commands: capabilities.commands,
169
+ models: capabilities.models,
170
+ features: capabilities.features,
171
+ warnings: capabilities.warnings
172
+ },
173
+ metrics,
126
174
  prompts: prompts.map((prompt) => ({
127
175
  id: prompt.id,
128
176
  prompt: prompt.prompt,
@@ -145,10 +193,12 @@ export async function runBenchmark(options = {}) {
145
193
  async function runScenario(context) {
146
194
  const {
147
195
  fmBin,
196
+ capabilities,
148
197
  prompts,
149
198
  runnableModels,
150
199
  modelStatuses,
151
200
  promptTokenCounts,
201
+ tokenCounting,
152
202
  options,
153
203
  concurrency,
154
204
  scenarioIndex,
@@ -172,6 +222,7 @@ async function runScenario(context) {
172
222
  for (const model of runnableModels) {
173
223
  await respond(fmBin, model.name, prompts[0].prompt, {
174
224
  ...options,
225
+ capabilities,
175
226
  stream: false
176
227
  });
177
228
  warmupCompleted += 1;
@@ -206,7 +257,15 @@ async function runScenario(context) {
206
257
  total: jobs.length
207
258
  });
208
259
  await runLimited(jobs, concurrency, async (job) => {
209
- const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
260
+ const result = await runSingleBenchmark({
261
+ fmBin,
262
+ capabilities,
263
+ job,
264
+ promptTokenCounts,
265
+ tokenCounting,
266
+ options,
267
+ benchmarkStartedAt
268
+ });
210
269
  results.push(result);
211
270
  if (onMeasuredResult) onMeasuredResult(result);
212
271
  if (!result.ok && options.failFast) {
@@ -229,13 +288,17 @@ async function runScenario(context) {
229
288
  };
230
289
  }
231
290
 
232
- async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt) {
291
+ async function runSingleBenchmark(context) {
292
+ const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, options, benchmarkStartedAt } = context;
233
293
  const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
234
294
  const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
235
295
  let response;
296
+ let attempts = 0;
236
297
  for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
298
+ attempts = attempt;
237
299
  response = await respond(fmBin, job.model.name, job.prompt.prompt, {
238
300
  ...options,
301
+ capabilities,
239
302
  stream: options.stream
240
303
  });
241
304
  if (response.ok || attempt >= maxAttempts) break;
@@ -243,28 +306,39 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
243
306
  await new Promise((resolve) => setTimeout(resolve, backoffMs));
244
307
  }
245
308
  const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
246
- const outputTokens = response.ok
247
- ? await countTokens(fmBin, response.output, options)
248
- : { ok: false, count: null };
309
+
310
+ const ok = response.ok;
249
311
  const seconds = response.durationMs / 1000;
250
- const firstTokenMs = response.firstOutputMs;
251
- const generationMs = response.ok && firstTokenMs != null
312
+ const chunks = response.stdoutChunks ?? 0;
313
+
314
+ // A single stdout chunk carries the whole answer, so the streamed portion is
315
+ // not separable: report generation time and TPOT as unavailable rather than
316
+ // as a near-zero decode phase.
317
+ const firstTokenMs = ok ? response.firstOutputMs : null;
318
+ const generationMs = ok && firstTokenMs != null && chunks > 1
252
319
  ? Math.max(0, response.durationMs - firstTokenMs)
253
320
  : null;
321
+
322
+ const outputTokens = ok && tokenCounting
323
+ ? await countTokens(fmBin, response.output, { ...options, capabilities })
324
+ : { ok: false, count: null };
254
325
  const countedOutputTokens = outputTokens.ok ? outputTokens.count : null;
255
- const decodeTokenCount = countedOutputTokens != null ? Math.max(0, countedOutputTokens - 1) : null;
256
- const hasDecodeCadence = response.stdoutChunks > 2 && generationMs != null && generationMs > 0 && decodeTokenCount > 0;
257
- const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
258
- const decodeTokensPerSecond = hasDecodeCadence
259
- ? decodeTokenCount / (generationMs / 1000)
326
+ // Two decode tokens is the minimum for an inter-token interval that is not
327
+ // simply the inverse of a single chunk gap.
328
+ const decodeTokenCount = countedOutputTokens != null && countedOutputTokens > 2
329
+ ? countedOutputTokens - 1
260
330
  : null;
331
+ const hasDecodeCadence = generationMs != null && generationMs > 0 && decodeTokenCount != null;
332
+ const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
333
+ const decodeTokensPerSecond = hasDecodeCadence ? decodeTokenCount / (generationMs / 1000) : null;
334
+
261
335
  const chars = response.output.length;
262
336
  const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
263
- const promptTokens = promptTokenCounts.get(job.prompt.id);
264
- const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
337
+ const promptTokens = promptTokenCounts.get(job.prompt.id) ?? null;
338
+ const prefillTokensPerSecond = promptTokens != null && firstTokenMs != null && firstTokenMs > 0
265
339
  ? promptTokens / (firstTokenMs / 1000)
266
340
  : null;
267
- const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
341
+ const chunkGapsMs = ok ? chunkGaps(response.stdoutChunkTimesMs) : [];
268
342
  const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
269
343
 
270
344
  return {
@@ -272,7 +346,8 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
272
346
  concurrency: job.concurrency,
273
347
  promptId: job.prompt.id,
274
348
  run: job.run,
275
- ok: response.ok,
349
+ attempts,
350
+ ok,
276
351
  durationMs: response.durationMs,
277
352
  firstTokenMs,
278
353
  generationMs,
@@ -284,7 +359,7 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
284
359
  tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
285
360
  decodeTokensPerSecond,
286
361
  prefillTokensPerSecond,
287
- charsPerSecond: seconds > 0 ? chars / seconds : 0,
362
+ charsPerSecond: seconds > 0 ? chars / seconds : null,
288
363
  startOffsetMs,
289
364
  endOffsetMs,
290
365
  streamed: response.streamed,
@@ -293,14 +368,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
293
368
  chunkGapsMs,
294
369
  chunkGapAvgMs: average(chunkGapsMs),
295
370
  chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
296
- outputHash: response.ok ? hashOutput(response.output) : null,
297
- good: response.ok ? evaluateSlo({
371
+ outputHash: ok ? hashOutput(response.output) : null,
372
+ good: ok ? evaluateSlo({
298
373
  firstTokenMs,
299
374
  durationMs: response.durationMs,
300
375
  tpotMs
301
376
  }, options) : false,
302
377
  output: options.captureOutput ? response.output : undefined,
303
- error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
378
+ error: ok ? '' : (response.error || `fm exited with code ${response.code ?? response.signal}`)
304
379
  };
305
380
  }
306
381
 
@@ -318,6 +393,23 @@ async function runLimited(items, concurrency, worker, options = {}) {
318
393
  await Promise.all(workers);
319
394
  }
320
395
 
396
+ // A benchmark with nothing to run is a configuration error, not an empty
397
+ // report: say which models were asked for and which ones the build supports.
398
+ function noRunnableModelsError(models, options) {
399
+ const requested = normalizeModelSelection(options.models);
400
+ const supported = models.filter((model) => !model.unsupported).map((model) => model.name);
401
+ const lines = ['No benchmark was run: none of the requested models are usable right now.'];
402
+ if (requested.length > 0) lines.push(` requested: ${requested.join(', ')}`);
403
+ if (supported.length > 0) lines.push(` models reported by this fm build: ${supported.join(', ')}`);
404
+ for (const model of models) {
405
+ if (model.reason) lines.push(` ${model.name}: ${model.reason}`);
406
+ }
407
+ lines.push(' run "fm-bench models" to see availability and reasons');
408
+ const error = new Error(lines.join('\n'));
409
+ error.exitCode = 2;
410
+ return error;
411
+ }
412
+
321
413
  function normalizeModelSelection(models) {
322
414
  if (!models) return [];
323
415
  const values = Array.isArray(models) ? models : [models];