fm-bench 0.7.2 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/bench.js CHANGED
@@ -1,6 +1,15 @@
1
1
  import crypto from 'node:crypto';
2
2
  import { detectFmCapabilities } from './capabilities.js';
3
- import { checkModelAvailability, collectEnvironment, countTokens, fmBinaryFromOptions, getQuotaUsage, respond } from './fm.js';
3
+ import {
4
+ calibrateTokenCounter,
5
+ checkModelAvailability,
6
+ collectEnvironment,
7
+ countTokens,
8
+ fmBinaryFromOptions,
9
+ getQuotaUsage,
10
+ listModelStatus,
11
+ respond
12
+ } from './fm.js';
4
13
  import { metricAvailability } from './metrics.js';
5
14
  import { loadPrompts } from './prompts.js';
6
15
  import { finalizeReportPayload } from './schema.js';
@@ -21,18 +30,22 @@ export async function inspectModels(options = {}) {
21
30
  ?? { name, description: 'Requested model not reported by this fm build' })
22
31
  : discovered.models;
23
32
 
33
+ const modelList = models.length > 0 ? await listModelStatus(discovered.fmBin, { ...options, capabilities }) : null;
24
34
  const inspected = [];
25
35
  for (const model of models) {
26
36
  const availability = await checkModelAvailability(discovered.fmBin, model.name, {
27
37
  ...options,
28
- capabilities
29
- });
30
- const quota = await getQuotaUsage(discovered.fmBin, model.name, {
31
- ...options,
32
- capabilities
38
+ capabilities,
39
+ modelList
33
40
  });
41
+ // A model the build does not have has no quota; asking would only put fm's
42
+ // text for some other model where the "not supported" reason belongs.
43
+ const quota = availability.unsupported
44
+ ? { supported: false, raw: '', reason: '' }
45
+ : await getQuotaUsage(discovered.fmBin, model.name, { ...options, capabilities });
34
46
  inspected.push({
35
47
  ...model,
48
+ identity: availability.identity || '',
36
49
  available: availability.available,
37
50
  unsupported: Boolean(availability.unsupported),
38
51
  reason: availability.available ? '' : (availability.reason || availability.raw || 'unavailable'),
@@ -73,7 +86,7 @@ export async function runBenchmark(options = {}) {
73
86
  : inspection.models;
74
87
  const runnableModels = modelStatuses.filter((model) => model.available);
75
88
  if (runnableModels.length === 0) {
76
- throw noRunnableModelsError(inspection.models, options);
89
+ throw noRunnableModelsError(inspection.models, capabilities.models, options);
77
90
  }
78
91
  const environment = await collectEnvironment(inspection.fmBin, { ...options, capabilities });
79
92
  const metrics = metricAvailability(capabilities, {
@@ -84,6 +97,16 @@ export async function runBenchmark(options = {}) {
84
97
  const concurrencies = normalizeConcurrencySweep(options);
85
98
  const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
86
99
 
100
+ const tokenCounter = {
101
+ command: capabilities.features.tokenCountCommand,
102
+ overhead: 0,
103
+ calibrated: false
104
+ };
105
+ if (metrics.outputTokens.available) {
106
+ notify(options, { type: 'phase', phase: 'tokens', message: 'calibrating token counter' });
107
+ Object.assign(tokenCounter, await calibrateTokenCounter(inspection.fmBin, { ...options, capabilities }));
108
+ }
109
+
87
110
  notify(options, {
88
111
  type: 'tokens:start',
89
112
  total: prompts.length,
@@ -123,6 +146,7 @@ export async function runBenchmark(options = {}) {
123
146
  modelStatuses,
124
147
  promptTokenCounts,
125
148
  tokenCounting: metrics.outputTokens.available,
149
+ tokenOverhead: tokenCounter.overhead,
126
150
  options,
127
151
  concurrency,
128
152
  scenarioIndex: scenarioIndex + 1,
@@ -171,6 +195,7 @@ export async function runBenchmark(options = {}) {
171
195
  warnings: capabilities.warnings
172
196
  },
173
197
  metrics,
198
+ tokenCounter,
174
199
  prompts: prompts.map((prompt) => ({
175
200
  id: prompt.id,
176
201
  prompt: prompt.prompt,
@@ -199,6 +224,7 @@ async function runScenario(context) {
199
224
  modelStatuses,
200
225
  promptTokenCounts,
201
226
  tokenCounting,
227
+ tokenOverhead,
202
228
  options,
203
229
  concurrency,
204
230
  scenarioIndex,
@@ -263,6 +289,7 @@ async function runScenario(context) {
263
289
  job,
264
290
  promptTokenCounts,
265
291
  tokenCounting,
292
+ tokenOverhead,
266
293
  options,
267
294
  benchmarkStartedAt
268
295
  });
@@ -289,7 +316,7 @@ async function runScenario(context) {
289
316
  }
290
317
 
291
318
  async function runSingleBenchmark(context) {
292
- const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, options, benchmarkStartedAt } = context;
319
+ const { fmBin, capabilities, job, promptTokenCounts, tokenCounting, tokenOverhead = 0, options, benchmarkStartedAt } = context;
293
320
  const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
294
321
  const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
295
322
  let response;
@@ -309,24 +336,34 @@ async function runSingleBenchmark(context) {
309
336
 
310
337
  const ok = response.ok;
311
338
  const seconds = response.durationMs / 1000;
312
- const chunks = response.stdoutChunks ?? 0;
313
339
 
314
- // A single stdout chunk carries the whole answer, so the streamed portion is
315
- // not separable: report generation time and TPOT as unavailable rather than
316
- // as a near-zero decode phase.
340
+ // When one delivery carries the whole answer, the streamed portion is not
341
+ // separable: report generation time and TPOT as unavailable rather than as
342
+ // a near-zero decode phase. Otherwise the generation window runs from the
343
+ // first to the last delivery, which excludes process teardown.
317
344
  const firstTokenMs = ok ? response.firstOutputMs : null;
318
- const generationMs = ok && firstTokenMs != null && chunks > 1
319
- ? Math.max(0, response.durationMs - firstTokenMs)
345
+ const deliveryTimes = response.deliveryTimesMs ?? [];
346
+ const generationMs = ok && firstTokenMs != null && deliveryTimes.length > 1
347
+ ? Math.max(0, deliveryTimes[deliveryTimes.length - 1] - firstTokenMs)
320
348
  : null;
321
349
 
322
- const outputTokens = ok && tokenCounting
323
- ? await countTokens(fmBin, response.output, { ...options, capabilities })
324
- : { ok: false, count: null };
325
- const countedOutputTokens = outputTokens.ok ? outputTokens.count : null;
350
+ const countOptions = { ...options, capabilities };
351
+ const countedOutputTokens = ok && tokenCounting
352
+ ? await countOutputTokens(fmBin, response.output, tokenOverhead, countOptions)
353
+ : null;
354
+ // fm streams coarse deltas: the first delivery carries about 20 tokens on
355
+ // macOS 27.2. Decode cadence is therefore measured over the tokens that
356
+ // arrived after it, not over "all tokens but one".
357
+ const firstChunkTokens = generationMs != null && countedOutputTokens != null
358
+ ? await countOutputTokens(fmBin, response.firstChunkText ?? '', tokenOverhead, countOptions)
359
+ : null;
326
360
  // Two decode tokens is the minimum for an inter-token interval that is not
327
361
  // simply the inverse of a single chunk gap.
328
- const decodeTokenCount = countedOutputTokens != null && countedOutputTokens > 2
329
- ? countedOutputTokens - 1
362
+ const tokensAfterFirstChunk = firstChunkTokens != null
363
+ ? countedOutputTokens - Math.min(firstChunkTokens, countedOutputTokens)
364
+ : null;
365
+ const decodeTokenCount = tokensAfterFirstChunk != null && tokensAfterFirstChunk >= 2
366
+ ? tokensAfterFirstChunk
330
367
  : null;
331
368
  const hasDecodeCadence = generationMs != null && generationMs > 0 && decodeTokenCount != null;
332
369
  const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
@@ -338,7 +375,7 @@ async function runSingleBenchmark(context) {
338
375
  const prefillTokensPerSecond = promptTokens != null && firstTokenMs != null && firstTokenMs > 0
339
376
  ? promptTokens / (firstTokenMs / 1000)
340
377
  : null;
341
- const chunkGapsMs = ok ? chunkGaps(response.stdoutChunkTimesMs) : [];
378
+ const chunkGapsMs = ok ? chunkGaps(deliveryTimes) : [];
342
379
  const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
343
380
 
344
381
  return {
@@ -354,6 +391,8 @@ async function runSingleBenchmark(context) {
354
391
  tpotMs,
355
392
  promptTokens,
356
393
  outputTokens: countedOutputTokens,
394
+ firstChunkTokens,
395
+ decodeTokens: hasDecodeCadence ? decodeTokenCount : null,
357
396
  chars,
358
397
  words,
359
398
  tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
@@ -379,6 +418,17 @@ async function runSingleBenchmark(context) {
379
418
  };
380
419
  }
381
420
 
421
+ /**
422
+ * Output token count with the token counter's framing overhead removed.
423
+ * Empty text is zero tokens; `fm count-tokens` rejects an empty prompt.
424
+ * @returns {Promise<number|null>}
425
+ */
426
+ async function countOutputTokens(fmBin, text, overhead, options) {
427
+ if (!String(text).trim()) return 0;
428
+ const counted = await countTokens(fmBin, text, options);
429
+ return counted.ok ? Math.max(0, counted.count - overhead) : null;
430
+ }
431
+
382
432
  async function runLimited(items, concurrency, worker, options = {}) {
383
433
  let nextIndex = 0;
384
434
  // fail-fast stops admitting new work; calls already in flight are allowed to
@@ -404,9 +454,9 @@ async function runLimited(items, concurrency, worker, options = {}) {
404
454
 
405
455
  // A benchmark with nothing to run is a configuration error, not an empty
406
456
  // report: say which models were asked for and which ones the build supports.
407
- function noRunnableModelsError(models, options) {
457
+ function noRunnableModelsError(models, discoveredModels, options) {
408
458
  const requested = normalizeModelSelection(options.models);
409
- const supported = models.filter((model) => !model.unsupported).map((model) => model.name);
459
+ const supported = discoveredModels.map((model) => model.name);
410
460
  const lines = ['No benchmark was run: none of the requested models are usable right now.'];
411
461
  if (requested.length > 0) lines.push(` requested: ${requested.join(', ')}`);
412
462
  if (supported.length > 0) lines.push(` models reported by this fm build: ${supported.join(', ')}`);
@@ -9,7 +9,7 @@
9
9
 
10
10
  import crypto from 'node:crypto';
11
11
  import { stripAnsi } from './ansi.js';
12
- import { parseAvailabilityList, parseModelsFromHelp } from './fm-help.js';
12
+ import { parseAvailabilityList, parseModelList, parseModelsFromHelp } from './fm-help.js';
13
13
  import { runProcess } from './process.js';
14
14
 
15
15
  const SECTION_HEADER = /^\s*[A-Z][A-Z0-9 /-]+\s*$/;
@@ -72,12 +72,26 @@ export function resolveTokenCountCommand(commands = []) {
72
72
  return null;
73
73
  }
74
74
 
75
+ /**
76
+ * Pick the subcommand this `fm` build uses to report model availability.
77
+ * macOS 27.2 renamed `available` to `models`; the old name still works there
78
+ * but prints a deprecation warning, so prefer the new one.
79
+ * @returns {'models'|'available'|null}
80
+ */
81
+ export function resolveModelListCommand(commands = []) {
82
+ if (commands.includes('models')) return 'models';
83
+ if (commands.includes('available')) return 'available';
84
+ return null;
85
+ }
86
+
75
87
  function buildFeatures(helpText, respondHelpText, commands) {
76
88
  const tokenCountCommand = resolveTokenCountCommand(commands);
77
89
  const respondHelp = respondHelpText || '';
78
90
  return {
79
91
  tokenCounting: tokenCountCommand != null,
80
92
  tokenCountCommand,
93
+ modelListCommand: resolveModelListCommand(commands),
94
+ license: commands.includes('license'),
81
95
  quota: commands.includes('quota-usage'),
82
96
  streaming: hasFlagInHelp(respondHelp, '--no-stream') || hasFlagInHelp(respondHelp, '--stream'),
83
97
  modelSelection: hasFlagInHelp(respondHelp, '--model'),
@@ -153,8 +167,8 @@ export async function detectFmCapabilities(fmBin, options = {}) {
153
167
  const features = buildFeatures(cleanHelp, respondHelpText, commands);
154
168
  let models = parseModelsFromHelp(cleanHelp);
155
169
 
156
- if (models.length === 0 && commands.includes('available')) {
157
- models = await discoverModelsFromAvailability(fmBin, { ...options, env });
170
+ if (models.length === 0 && features.modelListCommand) {
171
+ models = await discoverModelsFromList(fmBin, features.modelListCommand, { ...options, env });
158
172
  }
159
173
 
160
174
  return {
@@ -170,17 +184,20 @@ export async function detectFmCapabilities(fmBin, options = {}) {
170
184
  }
171
185
 
172
186
  /**
173
- * Fallback discovery: ask `fm available` without a model filter and read the
174
- * model names it reports. Used when `fm --help` has no MODELS section.
187
+ * Fallback discovery: ask `fm models` (or legacy `fm available`) without a
188
+ * model filter and read the model names it reports. Used when `fm --help` has
189
+ * no MODELS section.
175
190
  */
176
- async function discoverModelsFromAvailability(fmBin, options = {}) {
177
- const result = await runProcess(fmBin, ['available'], {
191
+ async function discoverModelsFromList(fmBin, command, options = {}) {
192
+ const result = await runProcess(fmBin, [command], {
178
193
  timeoutMs: options.timeoutMs ?? 15_000,
179
194
  env: options.env ?? process.env
180
195
  });
181
196
  if (result.error) return [];
182
- return parseAvailabilityList(`${result.stdout}${result.stderr}`)
183
- .map((model) => ({ name: model.name, description: '' }));
197
+ const output = `${result.stdout}${result.stderr}`;
198
+ const listed = parseModelList(output);
199
+ const models = listed.length > 0 ? listed : parseAvailabilityList(output);
200
+ return models.map((model) => ({ name: model.name, description: '' }));
184
201
  }
185
202
 
186
203
  function escapeRegExp(value) {
package/src/cli.js CHANGED
@@ -3,6 +3,7 @@ import { createRequire } from 'node:module';
3
3
  import { inspectModels, runBenchmark } from './bench.js';
4
4
  import { diffReports, renderCompareReport } from './compare.js';
5
5
  import { renderHtmlReport } from './export.js';
6
+ import { getLicenseStatus } from './fm.js';
6
7
  import { formatCapabilitySummary } from './metrics.js';
7
8
  import { validateReport } from './schema.js';
8
9
  import { loadHistory, renderHistoryReport } from './history.js';
@@ -30,6 +31,53 @@ function operationalError(message) {
30
31
  return error;
31
32
  }
32
33
 
34
+ // One unmeasured call per model absorbs the cold model load, which otherwise
35
+ // dominates the first measured run (several hundred ms on Apple silicon).
36
+ const DEFAULT_WARMUP = 1;
37
+
38
+ const COMMANDS = ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'validate', 'export', 'help'];
39
+
40
+ const KNOWN_OPTIONS = [
41
+ '--help', '--version', '--model', '--models', '--runs', '--warmup', '--concurrency', '--sweep-concurrency',
42
+ '--request-rate', '--ramp-up-ms', '--timeout', '--timeout-ms', '--slo-ttft-ms', '--slo-e2e-ms', '--slo-tpot-ms',
43
+ '--prompt', '--prompt-file', '--profile', '--instructions', '--fm-bin', '--use-case', '--guardrails',
44
+ '--greedy', '--no-greedy', '--stream', '--no-stream', '--json', '--csv', '--format', '--ascii', '--color',
45
+ '--no-color', '--progress', '--no-progress', '--compact', '--width', '--histogram', '--export-html', '--strict',
46
+ '--out', '--output-dir', '--capture-output', '--available-only', '--fail-fast', '--retry', '--ci', '--tag',
47
+ '--note', '--verbose'
48
+ ];
49
+
50
+ /** Closest candidate within a small edit distance, or null. */
51
+ function closestMatch(input, candidates) {
52
+ let best = null;
53
+ let bestDistance = Infinity;
54
+ for (const candidate of candidates) {
55
+ const distance = editDistance(input, candidate);
56
+ if (distance < bestDistance) {
57
+ best = candidate;
58
+ bestDistance = distance;
59
+ }
60
+ }
61
+ const limit = input.replace(/^-+/, '').length <= 6 ? 1 : 2;
62
+ return bestDistance > 0 && bestDistance <= limit ? best : null;
63
+ }
64
+
65
+ /** Optimal string alignment distance: an adjacent swap counts as one edit. */
66
+ function editDistance(a, b) {
67
+ const rows = Array.from({ length: a.length + 1 }, (_, i) => [i, ...Array(b.length).fill(0)]);
68
+ for (let j = 1; j <= b.length; j += 1) rows[0][j] = j;
69
+ for (let i = 1; i <= a.length; i += 1) {
70
+ for (let j = 1; j <= b.length; j += 1) {
71
+ const cost = a[i - 1] === b[j - 1] ? 0 : 1;
72
+ rows[i][j] = Math.min(rows[i - 1][j] + 1, rows[i][j - 1] + 1, rows[i - 1][j - 1] + cost);
73
+ if (i > 1 && j > 1 && a[i - 1] === b[j - 2] && a[i - 2] === b[j - 1]) {
74
+ rows[i][j] = Math.min(rows[i][j], rows[i - 2][j - 2] + 1);
75
+ }
76
+ }
77
+ }
78
+ return rows[a.length][b.length];
79
+ }
80
+
33
81
  export async function runCli(argv = process.argv.slice(2), env = {}) {
34
82
  const parsed = parseArgs(argv);
35
83
 
@@ -230,7 +278,7 @@ export function parseArgs(argv) {
230
278
  models: [],
231
279
  prompts: [],
232
280
  runs: 1,
233
- warmup: 0,
281
+ warmup: DEFAULT_WARMUP,
234
282
  concurrency: 1,
235
283
  sweepConcurrency: [],
236
284
  requestRate: null,
@@ -263,8 +311,15 @@ export function parseArgs(argv) {
263
311
  };
264
312
 
265
313
  const args = [...argv];
266
- if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'validate', 'export', 'help'].includes(args[0])) {
314
+ if (args[0] && !args[0].startsWith('-') && COMMANDS.includes(args[0])) {
267
315
  options.command = args.shift();
316
+ } else if (args[0] && /^[a-z]{3,}$/.test(args[0])) {
317
+ // A bare word is a prompt, but one that is a near miss for a command is
318
+ // almost always a typo; benchmarking "modles" as a prompt helps nobody.
319
+ const suggestion = closestMatch(args[0], COMMANDS);
320
+ if (suggestion) {
321
+ throw usageError(`Unknown command "${args[0]}". Did you mean "${suggestion}"? To benchmark it as a prompt, use: fm-bench -- ${args[0]}`);
322
+ }
268
323
  }
269
324
 
270
325
  if (options.command === 'metrics') {
@@ -447,7 +502,8 @@ export function parseArgs(argv) {
447
502
  break;
448
503
  default:
449
504
  if (arg.startsWith('-')) {
450
- throw usageError(`Unknown option: ${arg}`);
505
+ const suggestion = arg.startsWith('--') ? closestMatch(arg, KNOWN_OPTIONS) : null;
506
+ throw usageError(`Unknown option: ${arg}${suggestion ? `. Did you mean ${suggestion}?` : ''} (see fm-bench --help)`);
451
507
  }
452
508
  if (options.command === 'compare') {
453
509
  options.compareFiles.push(arg);
@@ -665,8 +721,17 @@ async function runDoctor(options) {
665
721
  checks.push(['fm token counting', capabilities.features.tokenCounting ? `yes (${capabilities.features.tokenCountCommand})` : 'no', capabilities.features.tokenCounting]);
666
722
  checks.push(['fm streaming', capabilities.features.streaming ? 'yes' : 'no', capabilities.features.streaming]);
667
723
  checks.push(['fm quota', capabilities.features.quota ? 'yes' : 'no (not exposed by this build)', true]);
724
+ if (capabilities.features.license) {
725
+ const license = await getLicenseStatus(inspection.fmBin, { capabilities });
726
+ checks.push(['fm license', license.agreed === false
727
+ ? `${license.detail || 'not agreed'} — run "fm license" to review and accept the terms`
728
+ : license.detail, license.agreed !== false]);
729
+ }
668
730
  for (const model of inspection.models) {
669
- checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
731
+ const detail = model.available
732
+ ? `available${model.identity ? ` (${model.identity})` : ''}`
733
+ : model.reason || 'unavailable';
734
+ checks.push([`model:${model.name}`, detail, model.available]);
670
735
  }
671
736
  } catch (error) {
672
737
  checks.push(['fm', error.message || String(error), false]);
@@ -681,7 +746,8 @@ async function runDoctor(options) {
681
746
  if (json) {
682
747
  console.log(JSON.stringify(payload, null, 2));
683
748
  } else {
684
- const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(16)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
749
+ const nameWidth = Math.max(...checks.map(([name]) => name.length));
750
+ const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(nameWidth)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
685
751
  console.log(lines.join('\n'));
686
752
  if (capabilities) {
687
753
  console.log('');
@@ -795,7 +861,7 @@ Commands:
795
861
  Run options:
796
862
  -m, --models <list> Models to benchmark, comma-separated or repeated
797
863
  -r, --runs <n> Runs per prompt/model (default: 1)
798
- --warmup <n> Warmup runs per model before measurement
864
+ --warmup <n> Unmeasured warmup runs per model (default: 1; 0 measures cold start)
799
865
  -c, --concurrency <n> Parallel fm processes (default: 1)
800
866
  --sweep-concurrency <list>
801
867
  Run separate operating points, e.g. 1,2,4
@@ -859,10 +925,16 @@ Exit codes:
859
925
  2 usage or environment error (bad flags, missing arguments, unsupported macOS, fm not found)
860
926
 
861
927
  Capability detection:
862
- fm-bench probes "fm --help" and "fm respond --help" once per run. Metrics the
863
- installed fm cannot supply are reported as unavailable instead of being
928
+ fm-bench probes "fm --help" and "fm respond --help" once per run and reads
929
+ model status from "fm models" (or "fm available" on older builds). Metrics
930
+ the installed fm cannot supply are reported as unavailable instead of being
864
931
  guessed, and unsupported models are refused before any benchmark starts.
865
932
 
933
+ Private Cloud Compute:
934
+ fm reports the pcc model as "not available in this context" outside the
935
+ Terminal app (for example in editor terminals). Run fm-bench from Terminal
936
+ to benchmark pcc; elsewhere it is listed as skipped with that reason.
937
+
866
938
  Examples:
867
939
  fm-bench
868
940
  fm-bench --models system --runs 3 --profile stress
package/src/compare.js CHANGED
@@ -77,10 +77,19 @@ function buildDiffRow(key, before, after) {
77
77
  goodputRate: diffPercent(before?.goodputRate, after?.goodputRate),
78
78
  tokensPerSecond: diffNumber(before?.tokensPerSecond?.avg, after?.tokensPerSecond?.avg, false),
79
79
  rps: diffNumber(before?.rps, after?.rps, false),
80
- cv: diffPercent(before?.latency?.cv, after?.latency?.cv)
80
+ cv: diffPercent(...comparableCv(before, after))
81
81
  };
82
82
  }
83
83
 
84
+ // Compare the run-to-run CV only when both reports carry it; otherwise fall
85
+ // back to the suite-wide latency CV on both sides so the definitions match.
86
+ function comparableCv(before, after) {
87
+ const bothStable = before && after && Object.hasOwn(before, 'stabilityCv') && Object.hasOwn(after, 'stabilityCv');
88
+ return bothStable
89
+ ? [before.stabilityCv, after.stabilityCv]
90
+ : [before?.latency?.cv, after?.latency?.cv];
91
+ }
92
+
84
93
  function diffMs(before, after) {
85
94
  return {
86
95
  before: before ?? null,
package/src/export.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { environmentFingerprint } from './schema.js';
2
+ import { stabilityCv } from './table.js';
2
3
 
3
4
  /**
4
5
  * Self-contained HTML report for sharing (paste, email, GitHub gist, static host).
@@ -33,12 +34,13 @@ export function renderHtmlReport(report) {
33
34
  ['macOS', fp.macOSProductVersion ?? '—'],
34
35
  ['Build', fp.macOSBuildVersion ?? '—'],
35
36
  ['Node', fp.node ?? '—'],
36
- ['fm CLI digest', fp.fmHelpDigest ?? '—']
37
+ ['fm CLI digest', fp.fmHelpDigest ?? '—'],
38
+ ['Models', Object.entries(fp.modelIdentities).map(([name, identity]) => `${name} = ${identity}`).join(', ') || '—']
37
39
  ];
38
40
 
39
41
  const summaryRows = summary.map((row) => {
40
42
  const ttft = /** @type {{ p50?: number, p95?: number }} */ (row.ttft ?? {});
41
- const lat = /** @type {{ p50?: number, p95?: number, cv?: number }} */ (row.latency ?? {});
43
+ const lat = /** @type {{ p50?: number, p95?: number }} */ (row.latency ?? {});
42
44
  const tps = /** @type {{ avg?: number }} */ (row.tokensPerSecond ?? {});
43
45
  return `<tr>
44
46
  <td>${esc(row.model)}</td>
@@ -51,7 +53,7 @@ export function renderHtmlReport(report) {
51
53
  <td>${fmtMs(lat.p50)}</td>
52
54
  <td>${fmtMs(lat.p95)}</td>
53
55
  <td>${fmtNum(tps.avg)}</td>
54
- <td>${fmtPct(lat.cv)}</td>
56
+ <td>${fmtPct(stabilityCv(row))}</td>
55
57
  </tr>`;
56
58
  }).join('\n');
57
59
 
@@ -114,7 +116,7 @@ export function renderHtmlReport(report) {
114
116
  </section>
115
117
 
116
118
  <footer>
117
- Generated by <a href="https://github.com/devinoldenburg/fm-bench">fm-bench</a>.
119
+ Generated by <a href="https://github.com/dvnold/fm-bench">fm-bench</a>.
118
120
  Compare two reports: <code>fm-bench compare before.json after.json</code>.
119
121
  Validate: <code>fm-bench validate report.json</code>.
120
122
  </footer>
package/src/fm-help.js CHANGED
@@ -60,6 +60,47 @@ export function parseModelsFromHelp(helpText = '') {
60
60
  return [...models.values()];
61
61
  }
62
62
 
63
+ /**
64
+ * Read `fm models` output (macOS 27.2+), one model per line:
65
+ *
66
+ * Apple Foundation Models
67
+ * ✓ system (AFM 3 Core Advanced)
68
+ * ✗ pcc (Private Cloud Compute is not available in this context. ...)
69
+ *
70
+ * The parenthesised text is the model identity for an available model and
71
+ * the reason for an unavailable one.
72
+ * @param {string} output
73
+ * @returns {{ name: string, available: boolean, identity: string, reason: string }[]}
74
+ */
75
+ export function parseModelList(output = '') {
76
+ const models = new Map();
77
+ for (const line of withoutWarnings(output).split(/\r?\n/)) {
78
+ const match = line.match(/^\s*([✓✔✗✘×])\s+([A-Za-z0-9][A-Za-z0-9._:/-]*)\s*(?:\((.*)\))?\s*$/);
79
+ if (!match) continue;
80
+ const available = match[1] === '✓' || match[1] === '✔';
81
+ const detail = (match[3] ?? '').trim();
82
+ models.set(match[2], {
83
+ name: match[2],
84
+ available,
85
+ identity: available ? detail : '',
86
+ reason: available ? '' : detail
87
+ });
88
+ }
89
+ return [...models.values()];
90
+ }
91
+
92
+ /**
93
+ * Drop `warning:` lines (for example the `fm available` rename notice) so
94
+ * they are never mistaken for a model status or an error cause.
95
+ * @param {string} text
96
+ */
97
+ export function withoutWarnings(text = '') {
98
+ return stripAnsi(text)
99
+ .split(/\r?\n/)
100
+ .filter((line) => !/^\s*warning:/i.test(line))
101
+ .join('\n');
102
+ }
103
+
63
104
  /**
64
105
  * Read model names and availability out of `fm available` (no model filter).
65
106
  * @param {string} output
@@ -67,7 +108,7 @@ export function parseModelsFromHelp(helpText = '') {
67
108
  */
68
109
  export function parseAvailabilityList(output = '') {
69
110
  const models = new Map();
70
- for (const line of stripAnsi(output).split(/\r?\n/)) {
111
+ for (const line of withoutWarnings(output).split(/\r?\n/)) {
71
112
  const match = line.match(/^\s*([A-Za-z][A-Za-z0-9._-]*)(?:\s+model)?\s+(?:is\s+)?(available|unavailable|not available)\b/i);
72
113
  if (!match) continue;
73
114
  const name = match[1].toLowerCase();
@@ -88,7 +129,17 @@ export function parseAvailabilityList(output = '') {
88
129
  * @param {number|null} code process exit code
89
130
  */
90
131
  export function parseAvailabilityOutput(model, output = '', code = null) {
91
- const clean = stripAnsi(output).trim();
132
+ const clean = withoutWarnings(output).trim();
133
+ const listed = parseModelList(clean).find((entry) => entry.name === model);
134
+ if (listed) {
135
+ return {
136
+ model,
137
+ available: listed.available,
138
+ identity: listed.identity,
139
+ raw: clean,
140
+ reason: listed.reason
141
+ };
142
+ }
92
143
  const lower = clean.toLowerCase();
93
144
  const modelLower = String(model).toLowerCase();
94
145
  const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b|\bis invalid for\b/.test(lower);
@@ -106,14 +157,18 @@ export function parseAvailabilityOutput(model, output = '', code = null) {
106
157
 
107
158
  /**
108
159
  * Collapse `fm` diagnostics into one actionable line. `fm` writes multi-line
109
- * usage blocks for argument errors; benchmark output only needs the cause.
160
+ * usage blocks for argument errors and `warning:` notices; benchmark output
161
+ * only needs the line that states the cause (including any hint on it, such
162
+ * as "Please use the Terminal app.").
110
163
  * @param {string} text
111
164
  */
112
165
  export function firstLine(text = '') {
113
- const clean = stripAnsi(text).replace(/\s+/g, ' ').trim();
114
- if (!clean) return '';
115
- const sentences = clean.split(/(?<=\.)\s+(?=[A-Z])/);
116
- const head = sentences.find((part) => /error|invalid|unavailable|not supported|failed/i.test(part)) || sentences[0];
166
+ const lines = withoutWarnings(text)
167
+ .split(/\r?\n/)
168
+ .map((line) => line.replace(/\s+/g, ' ').trim())
169
+ .filter((line) => line && !/^(usage|help|see)\b/i.test(line));
170
+ if (lines.length === 0) return '';
171
+ const head = lines.find((line) => /error|invalid|unavailable|not available|not supported|failed/i.test(line)) || lines[0];
117
172
  return head.replace(/^Error:\s*/i, '').trim();
118
173
  }
119
174