fm-bench 0.4.3 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -51,6 +51,8 @@ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skip
51
51
  ```sh
52
52
  fm-bench [run] [options]
53
53
  fm-bench models [options]
54
+ fm-bench compare <before.json> <after.json> [options]
55
+ fm-bench history [dir] [options]
54
56
  fm-bench legend [options]
55
57
  fm-bench doctor [options]
56
58
  ```
@@ -59,9 +61,13 @@ fm-bench doctor [options]
59
61
 
60
62
  `models` lists discovered models, availability, descriptions, and quota output.
61
63
 
64
+ `compare` reads two saved JSON reports and prints a side-by-side regression table showing absolute and percent change for every latency, throughput, and reliability metric. Green/yellow/red coloring applies lower-is-better logic for latency and CV and higher-is-better for throughput. Use `--json` to get the diff as structured data.
65
+
66
+ `history` scans a directory for fm-bench JSON report files and prints a chronological trend table. Pairs with `--output-dir` to build a persistent benchmark archive.
67
+
62
68
  `legend` explains every terminal table column, compact-card field, model-list column, and color rule. It does not run `fm`.
63
69
 
64
- `doctor` checks Node, macOS, `fm`, and model availability.
70
+ `doctor` checks Node, macOS, `fm`, model availability, CPU, memory, thermal throttle state, and battery status.
65
71
 
66
72
  ## Benchmark Options
67
73
 
@@ -72,9 +78,15 @@ fm-bench --models system --runs 3 --profile throughput --warmup 1
72
78
  fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
73
79
  fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5 --ramp-up-ms 2000
74
80
  fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
81
+ fm-bench --profile reasoning --runs 5 --retry 2
82
+ fm-bench --profile coding --runs 3 --tag pre-update --output-dir reports/
75
83
  fm-bench --prompt "Reply with exactly: ok" --runs 5
76
84
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
77
85
  fm-bench --format csv --out reports/bench.csv
86
+ fm-bench --histogram
87
+ fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000
88
+ fm-bench compare reports/before.json reports/after.json
89
+ fm-bench history reports/
78
90
  fm-bench legend
79
91
  fm-bench legend --json
80
92
  ```
@@ -89,11 +101,15 @@ Useful flags:
89
101
  - `--request-rate <rps>`: pace request starts at a target requests-per-second rate.
90
102
  - `--ramp-up-ms <n>`: gradually ramp request pacing over `n` milliseconds.
91
103
  - `--timeout-ms <n>`: timeout per `fm` call.
104
+ - `--retry <n>`: retry failed `fm` calls up to `n` times with exponential backoff (500ms–4s). Useful for handling transient model busy errors.
92
105
  - `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
93
- - `--profile quick|standard|interactive|throughput|client|stress`: built-in prompt suite.
106
+ - `--ci`: exit with code 1 if any run fails or any SLO budget is violated. Disables color and progress. Designed for GitHub Actions and CI pipelines.
107
+ - `--profile quick|standard|interactive|throughput|client|stress|reasoning|coding|creative`: built-in prompt suite.
94
108
  - `--prompt <text>`: custom prompt, repeatable.
95
109
  - `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
96
110
  - `--instructions <text>`: passed to `fm respond`.
111
+ - `--tag <name>`: tag this run (repeatable). Tags appear in the JSON payload and report header.
112
+ - `--note <text>`: freeform note attached to the JSON payload and report header.
97
113
  - `--available-only`: hide unavailable discovered models.
98
114
  - `--capture-output`: include raw model output in JSON reports.
99
115
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
@@ -102,7 +118,60 @@ Useful flags:
102
118
  - `--progress`, `--no-progress`: force or disable the live progress status line on stderr.
103
119
  - `--compact`: force the narrow terminal layout.
104
120
  - `--width <n>`: render as if the terminal has `n` columns.
121
+ - `--histogram`: print an ASCII latency distribution bar chart after the report.
105
122
  - `--out <file>`: save a report.
123
+ - `--output-dir <dir>`: auto-save a timestamped JSON report to a directory on every run.
124
+
125
+ ## Prompt Profiles
126
+
127
+ Nine built-in profiles cover a range of workloads:
128
+
129
+ | Profile | Prompts | Focus |
130
+ |---------|---------|-------|
131
+ | `quick` | 1 | Single-prompt smoke test |
132
+ | `standard` | 3 | Short chat, structured JSON, medium generation |
133
+ | `interactive` | 3 | Short conversational turns |
134
+ | `throughput` | 3 | Longer generation and transformation tasks |
135
+ | `client` | 5 | Broad real-world mix: chat, content, extraction, summarization, code review |
136
+ | `stress` | 5 | Diverse stress mix with math and reasoning |
137
+ | `reasoning` | 5 | Multi-step math, logic, causal chains, estimation, debugging |
138
+ | `coding` | 5 | Code review, refactoring, algorithms, code explanation, system design |
139
+ | `creative` | 5 | Product copy, error messages, analogies, commit messages, doc writing |
140
+
141
+ Use `--profile reasoning` or `--profile coding` for richer signal when evaluating a model's capability alongside raw speed.
142
+
143
+ ## Compare Reports
144
+
145
+ Save two runs to JSON and diff them:
146
+
147
+ ```sh
148
+ fm-bench --runs 5 --json --out before.json
149
+ # ... update your system, wait, or run again ...
150
+ fm-bench --runs 5 --json --out after.json
151
+ fm-bench compare before.json after.json
152
+ ```
153
+
154
+ Output shows each model/concurrency row with before value, delta percentage (color-coded green/yellow/red), and after value for TTFT, E2E, TPOT, tokens/s, RPS, success rate, and CV.
155
+
156
+ ## Benchmark History
157
+
158
+ Use `--output-dir` to build an archive, then view trends with `history`:
159
+
160
+ ```sh
161
+ fm-bench --output-dir reports/ --runs 3
162
+ fm-bench --output-dir reports/ --runs 3
163
+ fm-bench history reports/
164
+ ```
165
+
166
+ ## CI Integration
167
+
168
+ Use `--ci` to fail the pipeline when quality regresses:
169
+
170
+ ```sh
171
+ fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
172
+ ```
173
+
174
+ Exit code is 0 on pass, 1 on any failure or SLO violation. Prints `fm-bench ci: PASS` or `fm-bench ci: FAIL — <reasons>` to stderr.
106
175
 
107
176
  ## Prompt Files
108
177
 
@@ -149,7 +218,7 @@ Token counts come from `fm token-count --quiet`. If `fm` cannot count a response
149
218
 
150
219
  Measured runs stream by default so `fm-bench` can capture TTFT and streaming smoothness. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT, TPOT, second-chunk, and chunk-gap fields that depend on streaming will be blank.
151
220
 
152
- Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
221
+ Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Long `NOTE`, `DESCRIPTION`, and model quota cells wrap so error context stays visible instead of being hidden behind ellipses. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
153
222
 
154
223
  ## Table Legend
155
224
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.4.3",
3
+ "version": "0.5.0",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/bench.js CHANGED
@@ -229,11 +229,18 @@ async function runScenario(context) {
229
229
  }
230
230
 
231
231
  async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt) {
232
+ const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
232
233
  const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
233
- const response = await respond(fmBin, job.model.name, job.prompt.prompt, {
234
- ...options,
235
- stream: options.stream
236
- });
234
+ let response;
235
+ for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
236
+ response = await respond(fmBin, job.model.name, job.prompt.prompt, {
237
+ ...options,
238
+ stream: options.stream
239
+ });
240
+ if (response.ok || attempt >= maxAttempts) break;
241
+ const backoffMs = Math.min(500 * 2 ** (attempt - 1), 4000);
242
+ await new Promise((resolve) => setTimeout(resolve, backoffMs));
243
+ }
237
244
  const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
238
245
  const outputTokens = response.ok
239
246
  ? await countTokens(fmBin, response.output, options)
@@ -338,7 +345,10 @@ function publicOptions(options) {
338
345
  e2eMs: options.sloE2eMs || null,
339
346
  tpotMs: options.sloTpotMs || null
340
347
  },
341
- instructions: options.instructions ? '[set]' : ''
348
+ instructions: options.instructions ? '[set]' : '',
349
+ retry: options.retry ?? 0,
350
+ tags: options.tags?.length ? options.tags : [],
351
+ note: options.note ?? null
342
352
  };
343
353
  }
344
354
 
package/src/cli.js CHANGED
@@ -1,10 +1,12 @@
1
1
  import fs from 'node:fs/promises';
2
2
  import { createRequire } from 'node:module';
3
3
  import { inspectModels, runBenchmark } from './bench.js';
4
+ import { diffReports, renderCompareReport } from './compare.js';
5
+ import { loadHistory, renderHistoryReport } from './history.js';
4
6
  import { runProcess } from './process.js';
5
7
  import { createProgress } from './progress.js';
6
8
  import { flattenResults, toCsv, writeReport } from './report.js';
7
- import { legendEntries, renderBenchmarkReport, renderLegend, renderModelsTable } from './table.js';
9
+ import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend, renderModelsTable } from './table.js';
8
10
 
9
11
  const require = createRequire(import.meta.url);
10
12
  const packageJson = require('../package.json');
@@ -38,6 +40,16 @@ export async function runCli(argv = process.argv.slice(2)) {
38
40
  return;
39
41
  }
40
42
 
43
+ if (parsed.command === 'compare') {
44
+ await runCompare(parsed, renderOptions(parsed));
45
+ return;
46
+ }
47
+
48
+ if (parsed.command === 'history') {
49
+ await runHistory(parsed, renderOptions(parsed));
50
+ return;
51
+ }
52
+
41
53
  if (parsed.command === 'models') {
42
54
  const inspection = await inspectModels(parsed);
43
55
  if (parsed.format === 'json') {
@@ -48,6 +60,11 @@ export async function runCli(argv = process.argv.slice(2)) {
48
60
  return;
49
61
  }
50
62
 
63
+ if (parsed.ci) {
64
+ if (parsed.color === 'auto') parsed.color = 'never';
65
+ if (parsed.progress === 'auto') parsed.progress = 'never';
66
+ }
67
+
51
68
  const progress = createProgress({
52
69
  ...renderOptions(parsed),
53
70
  enabled: resolveProgress(parsed),
@@ -70,6 +87,10 @@ export async function runCli(argv = process.argv.slice(2)) {
70
87
  console.log(toCsv(flattenResults(payload.results)));
71
88
  } else {
72
89
  console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
90
+ if (parsed.histogram) {
91
+ console.log();
92
+ console.log(renderLatencyHistogram(payload.results, renderOptions(parsed)));
93
+ }
73
94
  if (parsed.verbose) {
74
95
  console.log();
75
96
  console.log(toCsv(flattenResults(payload.results)));
@@ -83,11 +104,62 @@ export async function runCli(argv = process.argv.slice(2)) {
83
104
  console.error(`Saved ${reportFormat.toUpperCase()} report to ${written}`);
84
105
  }
85
106
  }
107
+
108
+ if (parsed.outputDir) {
109
+ const stamp = payload.startedAt.replace(/[:.]/g, '-').replace('T', '_').slice(0, 19);
110
+ const firstModel = parsed.models?.flatMap((m) => String(m).split(',')).map((m) => m.trim()).filter(Boolean)[0] || 'all';
111
+ const modelSlug = firstModel.replace(/[^a-z0-9]/gi, '_');
112
+ const filename = `fm-bench_${stamp}_${modelSlug}.json`;
113
+ const filePath = `${parsed.outputDir}/${filename}`;
114
+ const written = await writeReport(filePath, payload, 'json');
115
+ if (parsed.format !== 'json') {
116
+ console.error(`Saved JSON report to ${written}`);
117
+ }
118
+ }
119
+
120
+ if (parsed.ci) {
121
+ const ciResult = evaluateCi(payload);
122
+ if (!ciResult.passed) {
123
+ const reasons = ciResult.reasons.join('; ');
124
+ console.error(`fm-bench ci: FAIL — ${reasons}`);
125
+ const error = new Error(`CI checks failed: ${reasons}`);
126
+ error.exitCode = 1;
127
+ throw error;
128
+ }
129
+ console.error(`fm-bench ci: PASS`);
130
+ }
131
+ }
132
+
133
+ function evaluateCi(payload) {
134
+ const reasons = [];
135
+ const totalFailed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
136
+ if (totalFailed > 0) {
137
+ reasons.push(`${totalFailed} run(s) failed`);
138
+ }
139
+
140
+ const hasSlo = payload.options?.slo && (
141
+ payload.options.slo.ttftMs || payload.options.slo.e2eMs || payload.options.slo.tpotMs
142
+ );
143
+ if (hasSlo) {
144
+ for (const item of payload.summary) {
145
+ if (!item.available) continue;
146
+ if (item.goodputRate != null && item.goodputRate < 1) {
147
+ const pct = Math.round((1 - item.goodputRate) * 100);
148
+ reasons.push(`${item.model} c${item.concurrency ?? 1}: ${pct}% of runs violated SLO`);
149
+ }
150
+ }
151
+ }
152
+
153
+ return {
154
+ passed: reasons.length === 0,
155
+ reasons
156
+ };
86
157
  }
87
158
 
88
159
  export function parseArgs(argv) {
89
160
  const options = {
90
161
  command: 'run',
162
+ compareFiles: [],
91
163
  models: [],
92
164
  prompts: [],
93
165
  runs: 1,
@@ -107,16 +179,21 @@ export function parseArgs(argv) {
107
179
  captureOutput: false,
108
180
  availableOnly: false,
109
181
  failFast: false,
182
+ retry: 0,
183
+ ci: false,
184
+ tags: [],
185
+ note: null,
110
186
  verbose: false,
111
187
  ascii: false,
112
188
  color: 'auto',
113
189
  progress: 'auto',
114
190
  compact: false,
115
- width: null
191
+ width: null,
192
+ histogram: false
116
193
  };
117
194
 
118
195
  const args = [...argv];
119
- if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'help'].includes(args[0])) {
196
+ if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'help'].includes(args[0])) {
120
197
  options.command = args.shift();
121
198
  }
122
199
 
@@ -189,6 +266,9 @@ export function parseArgs(argv) {
189
266
  break;
190
267
  case '--profile':
191
268
  options.profile = requireValue(arg, args);
269
+ if (!['quick', 'standard', 'interactive', 'throughput', 'client', 'stress', 'reasoning', 'coding', 'creative'].includes(options.profile)) {
270
+ throw new Error(`--profile must be one of: quick, standard, interactive, throughput, client, stress, reasoning, coding, creative`);
271
+ }
192
272
  break;
193
273
  case '-i':
194
274
  case '--instructions':
@@ -248,10 +328,16 @@ export function parseArgs(argv) {
248
328
  case '--width':
249
329
  options.width = parsePositiveInt(requireValue(arg, args), arg);
250
330
  break;
331
+ case '--histogram':
332
+ options.histogram = true;
333
+ break;
251
334
  case '-o':
252
335
  case '--out':
253
336
  options.out = requireValue(arg, args);
254
337
  break;
338
+ case '--output-dir':
339
+ options.outputDir = requireValue(arg, args);
340
+ break;
255
341
  case '--capture-output':
256
342
  options.captureOutput = true;
257
343
  break;
@@ -261,6 +347,18 @@ export function parseArgs(argv) {
261
347
  case '--fail-fast':
262
348
  options.failFast = true;
263
349
  break;
350
+ case '--retry':
351
+ options.retry = parseNonNegativeInt(requireValue(arg, args), arg);
352
+ break;
353
+ case '--ci':
354
+ options.ci = true;
355
+ break;
356
+ case '--tag':
357
+ options.tags.push(requireValue(arg, args));
358
+ break;
359
+ case '--note':
360
+ options.note = requireValue(arg, args);
361
+ break;
264
362
  case '-v':
265
363
  case '--verbose':
266
364
  options.verbose = true;
@@ -275,8 +373,14 @@ export function parseArgs(argv) {
275
373
  if (arg.startsWith('-')) {
276
374
  throw new Error(`Unknown option: ${arg}`);
277
375
  }
278
- options.prompts.push([arg, ...args].join(' '));
279
- args.length = 0;
376
+ if (options.command === 'compare') {
377
+ options.compareFiles.push(arg);
378
+ } else if (options.command === 'history') {
379
+ options.historyDir = arg;
380
+ } else {
381
+ options.prompts.push([arg, ...args].join(' '));
382
+ args.length = 0;
383
+ }
280
384
  break;
281
385
  }
282
386
  }
@@ -284,6 +388,64 @@ export function parseArgs(argv) {
284
388
  return options;
285
389
  }
286
390
 
391
+ async function runHistory(options, renderOpts) {
392
+ const dir = options.historyDir || options.outputDir || '.';
393
+ const reports = await loadHistory(dir);
394
+
395
+ if (options.format === 'json') {
396
+ const data = reports.map(({ filePath, report }) => ({
397
+ file: filePath,
398
+ startedAt: report.startedAt,
399
+ version: report.version,
400
+ summary: report.summary
401
+ }));
402
+ console.log(JSON.stringify(data, null, 2));
403
+ } else {
404
+ console.log(renderHistoryReport(reports, renderOpts));
405
+ }
406
+ }
407
+
408
+ async function runCompare(options, renderOpts) {
409
+ const files = options.compareFiles;
410
+ if (files.length < 2) {
411
+ throw new Error('compare requires two JSON report files: fm-bench compare before.json after.json');
412
+ }
413
+ if (files.length > 2) {
414
+ throw new Error('compare accepts exactly two JSON report files');
415
+ }
416
+
417
+ const [beforePath, afterPath] = files;
418
+ const [beforeText, afterText] = await Promise.all([
419
+ fs.readFile(beforePath, 'utf8'),
420
+ fs.readFile(afterPath, 'utf8')
421
+ ]);
422
+
423
+ let before, after;
424
+ try {
425
+ before = JSON.parse(beforeText);
426
+ } catch {
427
+ throw new Error(`Cannot parse ${beforePath} as JSON`);
428
+ }
429
+ try {
430
+ after = JSON.parse(afterText);
431
+ } catch {
432
+ throw new Error(`Cannot parse ${afterPath} as JSON`);
433
+ }
434
+
435
+ const diff = diffReports(before, after);
436
+
437
+ if (options.format === 'json') {
438
+ console.log(JSON.stringify(diff, null, 2));
439
+ } else {
440
+ console.log(renderCompareReport(diff, renderOpts));
441
+ }
442
+
443
+ if (options.out) {
444
+ await fs.writeFile(options.out, `${JSON.stringify(diff, null, 2)}\n`, 'utf8');
445
+ console.error(`Saved compare report to ${options.out}`);
446
+ }
447
+ }
448
+
287
449
  async function runDoctor(options) {
288
450
  const checks = [];
289
451
  checks.push(['node', process.version, true]);
@@ -295,13 +457,46 @@ async function runDoctor(options) {
295
457
  const major = versionMatch ? Number.parseInt(versionMatch[1].split('.')[0], 10) : null;
296
458
  checks.push(['macOS', versionMatch?.[1] || 'unknown', major == null || major >= 27]);
297
459
 
460
+ const hwModel = await runProcess('sysctl', ['-n', 'hw.model'], { timeoutMs: 3_000 });
461
+ const hwModelStr = (hwModel.stdout || '').trim();
462
+ if (hwModelStr) checks.push(['hw.model', hwModelStr, true]);
463
+
464
+ const cpuBrand = await runProcess('sysctl', ['-n', 'machdep.cpu.brand_string'], { timeoutMs: 3_000 });
465
+ const cpuBrandStr = (cpuBrand.stdout || '').trim();
466
+ if (cpuBrandStr) checks.push(['cpu', cpuBrandStr, true]);
467
+
468
+ const memBytes = await runProcess('sysctl', ['-n', 'hw.memsize'], { timeoutMs: 3_000 });
469
+ const memRaw = (memBytes.stdout || '').trim();
470
+ if (memRaw) {
471
+ const gb = (Number(memRaw) / (1024 ** 3)).toFixed(0);
472
+ checks.push(['memory', `${gb} GB`, true]);
473
+ }
474
+
475
+ const thermalResult = await runProcess('pmset', ['-g', 'therm'], { timeoutMs: 5_000 });
476
+ const thermalOut = (thermalResult.stdout || thermalResult.stderr || '').trim();
477
+ const thermalMatch = thermalOut.match(/CPU_Scheduler_Limit\s*=\s*(\d+)/);
478
+ if (thermalMatch) {
479
+ const limit = Number.parseInt(thermalMatch[1], 10);
480
+ checks.push(['thermal limit', `${limit}%`, limit >= 100]);
481
+ }
482
+
483
+ const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
484
+ const batteryOut = (batteryResult.stdout || '').trim();
485
+ const batteryLine = batteryOut.split('\n').find((line) => line.includes('%'));
486
+ if (batteryLine) {
487
+ const pctMatch = batteryLine.match(/(\d+)%/);
488
+ const charging = /charging|AC Power|charged/i.test(batteryLine);
489
+ const pct = pctMatch ? `${pctMatch[1]}%` : '?%';
490
+ checks.push(['battery', `${pct}${charging ? ' (charging/AC)' : ' (battery)'}`, charging || Number.parseInt(pctMatch?.[1], 10) >= 20]);
491
+ }
492
+
298
493
  const inspection = await inspectModels(options);
299
494
  checks.push(['fm', inspection.fmBin, inspection.models.length > 0]);
300
495
  for (const model of inspection.models) {
301
496
  checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
302
497
  }
303
498
 
304
- const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(14)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
499
+ const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(16)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
305
500
  console.log(lines.join('\n'));
306
501
 
307
502
  if (options.out) {
@@ -374,12 +569,16 @@ Dynamic benchmark CLI for Apple's fm command on macOS 27+.
374
569
  Usage:
375
570
  fm-bench [run] [options]
376
571
  fm-bench models [options]
572
+ fm-bench compare <before.json> <after.json> [options]
573
+ fm-bench history [dir] [options]
377
574
  fm-bench legend [options]
378
575
  fm-bench doctor [options]
379
576
 
380
577
  Commands:
381
578
  run Benchmark discovered or selected fm models
382
579
  models List discovered models and availability
580
+ compare Compare two saved JSON reports and show metric deltas
581
+ history Show a trend table from all fm-bench JSON reports in a directory
383
582
  legend Explain every terminal table column and color rule
384
583
  doctor Check Node, macOS, fm, and model availability
385
584
 
@@ -398,7 +597,7 @@ Run options:
398
597
  --slo-tpot-ms <n> Count request as good only if TPOT is <= n
399
598
  -p, --prompt <text> Prompt to benchmark; repeatable
400
599
  --prompt-file <file> .json, .jsonl, or blank-line separated text prompts
401
- --profile <name> quick, standard, interactive, throughput, client, or stress
600
+ --profile <name> quick, standard, interactive, throughput, client, stress, reasoning, coding, or creative
402
601
  -i, --instructions <text> Instructions passed to fm respond
403
602
  --use-case <case> Pass a system model use case through to fm
404
603
  --guardrails <level> Pass a system model guardrail level through to fm
@@ -409,6 +608,10 @@ Run options:
409
608
  --available-only Hide unavailable discovered models
410
609
  --capture-output Include raw model output in JSON reports
411
610
  --fail-fast Stop after the first failed measured run
611
+ --retry <n> Retry failed fm calls up to n times with exponential backoff
612
+ --ci Exit 1 if any run fails or any SLO is violated; disables color and progress
613
+ --tag <name> Tag this run; repeatable; included in JSON payload and report header
614
+ --note <text> Freeform note included in JSON payload and report header
412
615
 
413
616
  Output:
414
617
  --format <type> table, json, or csv (default: table)
@@ -421,7 +624,9 @@ Output:
421
624
  --no-progress Disable live progress on stderr
422
625
  --compact Force compact terminal layout
423
626
  --width <n> Render for a specific terminal width
627
+ --histogram Print an ASCII latency distribution histogram after the report
424
628
  -o, --out <file> Save JSON or CSV report based on file extension
629
+ --output-dir <dir> Save a timestamped JSON report to a directory automatically
425
630
  -v, --verbose Include per-run CSV after the summary table
426
631
 
427
632
  Environment:
@@ -434,6 +639,11 @@ Examples:
434
639
  fm-bench --models system,pcc --runs 3 --profile stress
435
640
  fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
436
641
  fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
642
+ fm-bench --profile reasoning --runs 5 --retry 2
643
+ fm-bench compare before.json after.json
644
+ fm-bench compare before.json after.json --json
645
+ fm-bench history ./reports
646
+ fm-bench history ./reports --json
437
647
  fm-bench legend
438
648
  fm-bench models
439
649
  fm-bench doctor
package/src/compare.js ADDED
@@ -0,0 +1,241 @@
1
+ import { formatMs, formatNumber, formatPercent } from './table.js';
2
+
3
+ export function diffReports(before, after) {
4
+ const beforeByKey = indexSummary(before.summary ?? []);
5
+ const afterByKey = indexSummary(after.summary ?? []);
6
+
7
+ const keys = new Set([...beforeByKey.keys(), ...afterByKey.keys()]);
8
+ const rows = [];
9
+
10
+ for (const key of [...keys].sort()) {
11
+ const b = beforeByKey.get(key);
12
+ const a = afterByKey.get(key);
13
+ rows.push(buildDiffRow(key, b ?? null, a ?? null));
14
+ }
15
+
16
+ return {
17
+ before: reportMeta(before),
18
+ after: reportMeta(after),
19
+ rows
20
+ };
21
+ }
22
+
23
+ function reportMeta(report) {
24
+ return {
25
+ version: report.version ?? '?',
26
+ startedAt: report.startedAt ?? '',
27
+ finishedAt: report.finishedAt ?? '',
28
+ runs: report.options?.runs ?? null,
29
+ profile: report.options?.profile ?? null,
30
+ concurrency: report.options?.concurrency ?? null
31
+ };
32
+ }
33
+
34
+ function indexSummary(summary) {
35
+ const map = new Map();
36
+ for (const item of summary) {
37
+ const key = `${item.model}::${item.concurrency ?? 1}`;
38
+ map.set(key, item);
39
+ }
40
+ return map;
41
+ }
42
+
43
+ function buildDiffRow(key, before, after) {
44
+ const [model, concurrencyStr] = key.split('::');
45
+ return {
46
+ model,
47
+ concurrency: Number.parseInt(concurrencyStr, 10) || 1,
48
+ available: {
49
+ before: before?.available ?? null,
50
+ after: after?.available ?? null
51
+ },
52
+ ttftP50: diffMs(before?.ttft?.p50, after?.ttft?.p50),
53
+ ttftP95: diffMs(before?.ttft?.p95, after?.ttft?.p95),
54
+ e2eP50: diffMs(before?.latency?.p50, after?.latency?.p50),
55
+ e2eP95: diffMs(before?.latency?.p95, after?.latency?.p95),
56
+ e2eP99: diffMs(before?.latency?.p99, after?.latency?.p99),
57
+ tpotP50: diffMs(before?.tpot?.p50, after?.tpot?.p50),
58
+ successRate: diffPercent(before?.successRate, after?.successRate),
59
+ goodputRate: diffPercent(before?.goodputRate, after?.goodputRate),
60
+ tokensPerSecond: diffNumber(before?.tokensPerSecond?.avg, after?.tokensPerSecond?.avg, false),
61
+ rps: diffNumber(before?.rps, after?.rps, false),
62
+ cv: diffPercent(before?.latency?.cv, after?.latency?.cv)
63
+ };
64
+ }
65
+
66
+ function diffMs(before, after) {
67
+ return {
68
+ before: before ?? null,
69
+ after: after ?? null,
70
+ delta: numDelta(before, after),
71
+ deltaPercent: percentDelta(before, after)
72
+ };
73
+ }
74
+
75
+ function diffNumber(before, after, lowerIsBetter = true) {
76
+ return {
77
+ before: before ?? null,
78
+ after: after ?? null,
79
+ delta: numDelta(before, after),
80
+ deltaPercent: percentDelta(before, after),
81
+ lowerIsBetter
82
+ };
83
+ }
84
+
85
+ function diffPercent(before, after) {
86
+ return {
87
+ before: before ?? null,
88
+ after: after ?? null,
89
+ delta: numDelta(before, after),
90
+ deltaPercent: percentDelta(before, after)
91
+ };
92
+ }
93
+
94
+ function numDelta(before, after) {
95
+ if (!Number.isFinite(before) || !Number.isFinite(after)) return null;
96
+ return after - before;
97
+ }
98
+
99
+ function percentDelta(before, after) {
100
+ if (!Number.isFinite(before) || !Number.isFinite(after) || before === 0) return null;
101
+ return (after - before) / Math.abs(before);
102
+ }
103
+
104
+ export function renderCompareReport(diff, options = {}) {
105
+ const { color = false, ascii = false } = options;
106
+ const width = options.width || process.stdout.columns || 120;
107
+ const V = ascii ? '|' : '│';
108
+ const H = ascii ? '-' : '─';
109
+ const TL = ascii ? '+' : '┌';
110
+ const TR = ascii ? '+' : '┐';
111
+ const BL = ascii ? '+' : '└';
112
+ const BR = ascii ? '+' : '┘';
113
+ const ML = ascii ? '+' : '├';
114
+ const MR = ascii ? '+' : '┤';
115
+ const TJ = ascii ? '+' : '┬';
116
+ const MJ = ascii ? '+' : '┼';
117
+ const BJ = ascii ? '+' : '┴';
118
+
119
+ const lines = [];
120
+
121
+ const beforeLabel = `${diff.before.version} ${diff.before.startedAt ? diff.before.startedAt.slice(0, 19).replace('T', ' ') : '?'}`;
122
+ const afterLabel = `${diff.after.version} ${diff.after.startedAt ? diff.after.startedAt.slice(0, 19).replace('T', ' ') : '?'}`;
123
+
124
+ lines.push(`fm-bench compare`);
125
+ lines.push(` before: ${beforeLabel}`);
126
+ lines.push(` after: ${afterLabel}`);
127
+ lines.push('');
128
+
129
+ const colWidths = [8, 3, 10, 10, 10, 10, 10, 10, 8, 8, 9, 7, 7];
130
+ const headers = ['MODEL', 'C', 'TTFT P50', 'TTFT P95', 'E2E P50', 'E2E P95', 'E2E P99', 'TPOT P50', 'SUCC', 'GOOD', 'USER T/S', 'RPS', 'CV'];
131
+
132
+ const renderRule = (l, j, r) => `${l}${colWidths.map((w) => H.repeat(w + 2)).join(j)}${r}`;
133
+ const renderRow = (cells) => `${V}${cells.map((cell, i) => ` ${fitCell(cell, colWidths[i])} `).join(V)}${V}`;
134
+
135
+ lines.push(renderRule(TL, TJ, TR));
136
+ lines.push(renderRow(headers.map((h, i) => pad(h, colWidths[i]))));
137
+ lines.push(renderRule(ML, MJ, MR));
138
+
139
+ for (const row of diff.rows) {
140
+ const bLine = renderRow([
141
+ fitCell(row.model, colWidths[0]),
142
+ fitCell(String(row.concurrency), colWidths[1]),
143
+ fmtDiffMs(row.ttftP50, 'before', color),
144
+ fmtDiffMs(row.ttftP95, 'before', color),
145
+ fmtDiffMs(row.e2eP50, 'before', color),
146
+ fmtDiffMs(row.e2eP95, 'before', color),
147
+ fmtDiffMs(row.e2eP99, 'before', color),
148
+ fmtDiffMs(row.tpotP50, 'before', color),
149
+ fitCell(row.successRate.before != null ? formatPercent(row.successRate.before) : '-', colWidths[8]),
150
+ fitCell(row.goodputRate.before != null ? formatPercent(row.goodputRate.before) : '-', colWidths[9]),
151
+ fitCell(row.tokensPerSecond.before != null ? formatNumber(row.tokensPerSecond.before) : '-', colWidths[10]),
152
+ fitCell(row.rps.before != null ? formatNumber(row.rps.before) : '-', colWidths[11]),
153
+ fitCell(row.cv.before != null ? formatPercent(row.cv.before) : '-', colWidths[12])
154
+ ]);
155
+
156
+ const deltaLine = renderRow([
157
+ fitCell('', colWidths[0]),
158
+ fitCell('', colWidths[1]),
159
+ fmtDelta(row.ttftP50, true, color, colWidths[2]),
160
+ fmtDelta(row.ttftP95, true, color, colWidths[3]),
161
+ fmtDelta(row.e2eP50, true, color, colWidths[4]),
162
+ fmtDelta(row.e2eP95, true, color, colWidths[5]),
163
+ fmtDelta(row.e2eP99, true, color, colWidths[6]),
164
+ fmtDelta(row.tpotP50, true, color, colWidths[7]),
165
+ fmtDelta(row.successRate, false, color, colWidths[8]),
166
+ fmtDelta(row.goodputRate, false, color, colWidths[9]),
167
+ fmtDelta(row.tokensPerSecond, false, color, colWidths[10]),
168
+ fmtDelta(row.rps, false, color, colWidths[11]),
169
+ fmtDelta(row.cv, true, color, colWidths[12])
170
+ ]);
171
+
172
+ const aLine = renderRow([
173
+ fitCell('', colWidths[0]),
174
+ fitCell('', colWidths[1]),
175
+ fmtDiffMs(row.ttftP50, 'after', color),
176
+ fmtDiffMs(row.ttftP95, 'after', color),
177
+ fmtDiffMs(row.e2eP50, 'after', color),
178
+ fmtDiffMs(row.e2eP95, 'after', color),
179
+ fmtDiffMs(row.e2eP99, 'after', color),
180
+ fmtDiffMs(row.tpotP50, 'after', color),
181
+ fitCell(row.successRate.after != null ? formatPercent(row.successRate.after) : '-', colWidths[8]),
182
+ fitCell(row.goodputRate.after != null ? formatPercent(row.goodputRate.after) : '-', colWidths[9]),
183
+ fitCell(row.tokensPerSecond.after != null ? formatNumber(row.tokensPerSecond.after) : '-', colWidths[10]),
184
+ fitCell(row.rps.after != null ? formatNumber(row.rps.after) : '-', colWidths[11]),
185
+ fitCell(row.cv.after != null ? formatPercent(row.cv.after) : '-', colWidths[12])
186
+ ]);
187
+
188
+ lines.push(bLine);
189
+ lines.push(deltaLine);
190
+ lines.push(aLine);
191
+ }
192
+
193
+ lines.push(renderRule(BL, BJ, BR));
194
+ lines.push('');
195
+ lines.push('Rows: before → delta (% change) → after | Lower is better for latency and CV; higher for throughput and success.');
196
+
197
+ return lines.join('\n');
198
+ }
199
+
200
+ function fmtDiffMs(diff, which, color) {
201
+ const value = diff[which];
202
+ if (value == null) return '-';
203
+ return formatMs(value);
204
+ }
205
+
206
+ function fmtDelta(diff, lowerIsBetter, color, width) {
207
+ const { delta, deltaPercent } = diff;
208
+ if (delta == null || deltaPercent == null) return fitCell('n/a', width ?? 10);
209
+
210
+ const pct = Math.round(deltaPercent * 100);
211
+ const sign = delta >= 0 ? '+' : '';
212
+ const label = `${sign}${pct}%`;
213
+
214
+ let tone = null;
215
+ if (lowerIsBetter) {
216
+ tone = pct < -5 ? 'green' : pct > 5 ? 'red' : 'yellow';
217
+ } else {
218
+ tone = pct > 5 ? 'green' : pct < -5 ? 'red' : 'yellow';
219
+ }
220
+
221
+ const text = fitCell(label, width ?? 10);
222
+ if (color) return applyTone(text, tone);
223
+ return text;
224
+ }
225
+
226
+ function applyTone(text, tone) {
227
+ const TONES = { green: '\x1b[32m', yellow: '\x1b[33m', red: '\x1b[31m' };
228
+ const reset = '\x1b[0m';
229
+ return tone && TONES[tone] ? `${TONES[tone]}${text}${reset}` : text;
230
+ }
231
+
232
+ function pad(text, width) {
233
+ const len = String(text ?? '').length;
234
+ return String(text ?? '') + ' '.repeat(Math.max(0, width - len));
235
+ }
236
+
237
+ function fitCell(text, width) {
238
+ const str = String(text ?? '');
239
+ if (str.length <= width) return str + ' '.repeat(width - str.length);
240
+ return `${str.slice(0, width - 1)}…`;
241
+ }
package/src/history.js ADDED
@@ -0,0 +1,95 @@
1
+ import fs from 'node:fs/promises';
2
+ import path from 'node:path';
3
+ import { formatMs, formatNumber, formatPercent } from './table.js';
4
+
5
+ export async function loadHistory(dir) {
6
+ const absDir = path.resolve(dir);
7
+ let entries;
8
+ try {
9
+ entries = await fs.readdir(absDir);
10
+ } catch {
11
+ throw new Error(`Cannot read directory: ${absDir}`);
12
+ }
13
+
14
+ const jsonFiles = entries
15
+ .filter((name) => name.endsWith('.json'))
16
+ .map((name) => path.join(absDir, name))
17
+ .sort();
18
+
19
+ const reports = [];
20
+ for (const filePath of jsonFiles) {
21
+ try {
22
+ const text = await fs.readFile(filePath, 'utf8');
23
+ const parsed = JSON.parse(text);
24
+ if (parsed.tool === 'fm-bench' && parsed.summary) {
25
+ reports.push({ filePath, report: parsed });
26
+ }
27
+ } catch {
28
+ // skip unparseable files
29
+ }
30
+ }
31
+
32
+ return reports;
33
+ }
34
+
35
+ export function renderHistoryReport(reports, options = {}) {
36
+ if (reports.length === 0) {
37
+ return 'No fm-bench JSON reports found in the given directory.';
38
+ }
39
+
40
+ const { color = false, ascii = false } = options;
41
+ const V = ascii ? '|' : '│';
42
+ const H = ascii ? '-' : '─';
43
+ const TL = ascii ? '+' : '┌';
44
+ const TR = ascii ? '+' : '┐';
45
+ const BL = ascii ? '+' : '└';
46
+ const BR = ascii ? '+' : '┘';
47
+ const ML = ascii ? '+' : '├';
48
+ const MR = ascii ? '+' : '┤';
49
+ const TJ = ascii ? '+' : '┬';
50
+ const MJ = ascii ? '+' : '┼';
51
+ const BJ = ascii ? '+' : '┴';
52
+
53
+ const colWidths = [19, 8, 3, 6, 9, 9, 9, 9, 5];
54
+ const headers = ['STARTED AT', 'MODEL', 'C', 'RUNS', 'TTFT P50', 'E2E P50', 'E2E P95', 'USER T/S', 'SUCC'];
55
+
56
+ const renderRule = (l, j, r) => `${l}${colWidths.map((w) => H.repeat(w + 2)).join(j)}${r}`;
57
+ const renderRowLine = (cells) => `${V}${cells.map((text, i) => ` ${fit(text, colWidths[i])} `).join(V)}${V}`;
58
+
59
+ const lines = [];
60
+ lines.push(`fm-bench history (${reports.length} report${reports.length === 1 ? '' : 's'})`);
61
+ lines.push('');
62
+ lines.push(renderRule(TL, TJ, TR));
63
+ lines.push(renderRowLine(headers.map((h, i) => h)));
64
+ lines.push(renderRule(ML, MJ, MR));
65
+
66
+ for (const { filePath, report } of reports) {
67
+ const started = report.startedAt ? report.startedAt.slice(0, 19).replace('T', ' ') : '?';
68
+ const modelRows = report.summary ?? [];
69
+
70
+ for (const item of modelRows) {
71
+ if (!item.available) continue;
72
+ const row = [
73
+ started,
74
+ item.model ?? '-',
75
+ String(item.concurrency ?? 1),
76
+ String(item.successes ?? '-'),
77
+ item.ttft?.p50 != null ? formatMs(item.ttft.p50) : '-',
78
+ item.latency?.p50 != null ? formatMs(item.latency.p50) : '-',
79
+ item.latency?.p95 != null ? formatMs(item.latency.p95) : '-',
80
+ item.tokensPerSecond?.avg != null ? formatNumber(item.tokensPerSecond.avg) : '-',
81
+ item.successRate != null ? formatPercent(item.successRate) : '-'
82
+ ];
83
+ lines.push(renderRowLine(row));
84
+ }
85
+ }
86
+
87
+ lines.push(renderRule(BL, BJ, BR));
88
+ return lines.join('\n');
89
+ }
90
+
91
+ function fit(text, width) {
92
+ const str = String(text ?? '');
93
+ if (str.length <= width) return str + ' '.repeat(width - str.length);
94
+ return `${str.slice(0, width - 1)}…`;
95
+ }
package/src/prompts.js CHANGED
@@ -93,6 +93,72 @@ const PROFILES = {
93
93
  id: 'summarize',
94
94
  prompt: 'Summarize this in two bullets: local model benchmarks should measure first-token latency, total latency, throughput, failures, and the exact prompt suite so results can be compared later.'
95
95
  }
96
+ ],
97
+ reasoning: [
98
+ {
99
+ id: 'math-word',
100
+ prompt: 'A server processes 1,200 requests per minute at peak. Each request uses 0.8 ms of CPU time on average. How many CPU cores are needed to handle peak load at 70% utilization? Show your reasoning step-by-step, then give a single final answer.'
101
+ },
102
+ {
103
+ id: 'logic-sequence',
104
+ prompt: 'Five engineers — Alice, Bob, Carol, Dave, Eve — each deploy one service. Alice deploys before Bob. Carol deploys after Dave but before Eve. Bob deploys before Dave. List the deployment order from first to last.'
105
+ },
106
+ {
107
+ id: 'causal-chain',
108
+ prompt: 'A CI pipeline has: lint (2 min), unit tests (5 min, parallel), integration tests (8 min, depends on unit), build (3 min, depends on integration), deploy (1 min, depends on build). What is the minimum wall-clock time from start to deployed? Explain each step.'
109
+ },
110
+ {
111
+ id: 'estimation',
112
+ prompt: 'Estimate how many tokens per day a popular AI coding assistant might process if it has 500,000 daily active users, each averaging 30 completions of 200 output tokens. Show your calculation and state any assumptions.'
113
+ },
114
+ {
115
+ id: 'debug-logic',
116
+ prompt: 'This function should return the median of a list: def median(lst): lst.sort(); n=len(lst); return lst[n//2] if n%2 else (lst[n//2-1]+lst[n//2])/2. Find all bugs and explain why each is a bug.'
117
+ }
118
+ ],
119
+ coding: [
120
+ {
121
+ id: 'code-review',
122
+ prompt: 'Review this TypeScript for correctness, performance, and readability issues: async function fetchAll(urls: string[]) { const results = []; for (const url of urls) { const r = await fetch(url); results.push(await r.json()); } return results; } Give three specific improvements with brief explanations.'
123
+ },
124
+ {
125
+ id: 'refactor',
126
+ prompt: 'Refactor this JavaScript to be cleaner and handle edge cases: function getUser(id, cb) { db.query("SELECT * FROM users WHERE id=" + id, function(err, rows) { if (err) { cb(null, err); } else { cb(rows[0]); } }); } Return only the improved code and a two-sentence explanation.'
127
+ },
128
+ {
129
+ id: 'algorithm',
130
+ prompt: 'Write a JavaScript function findDuplicates(arr) that returns all duplicate values in O(n) time and O(n) space. Include a brief complexity explanation and two edge-case examples.'
131
+ },
132
+ {
133
+ id: 'explain-code',
134
+ prompt: 'Explain what this code does, why it might be used, and one potential problem: const cache = new WeakMap(); function memoize(fn) { return function(...args) { if (!cache.has(this)) cache.set(this, new Map()); const key = JSON.stringify(args); if (!cache.get(this).has(key)) cache.get(this).set(key, fn.apply(this, args)); return cache.get(this).get(key); }; }'
135
+ },
136
+ {
137
+ id: 'system-design',
138
+ prompt: 'Design a rate limiter in 120 words or fewer: specify the data structure, the algorithm (token bucket, sliding window, or fixed window), and how you handle distributed deployments. Be precise and practical.'
139
+ }
140
+ ],
141
+ creative: [
142
+ {
143
+ id: 'product-announcement',
144
+ prompt: 'Write a 100-word product announcement for a developer tool called "PulseDB" that shows real-time query performance heatmaps in the terminal. Tone: enthusiastic but not hypey. Include one concrete example of what a user would see.'
145
+ },
146
+ {
147
+ id: 'error-message',
148
+ prompt: 'Rewrite this cryptic error into a helpful, actionable message a junior developer could act on: "Error: ECONNREFUSED 127.0.0.1:5432 errno: -111 syscall: connect code: ECONNREFUSED". Include what likely caused it and the first two things to check.'
149
+ },
150
+ {
151
+ id: 'technical-analogy',
152
+ prompt: 'Explain CPU context switching to someone who has never programmed, using a single concrete analogy from everyday life. Keep it under 80 words and make sure the analogy captures the performance cost.'
153
+ },
154
+ {
155
+ id: 'commit-message',
156
+ prompt: 'Write a clear and conventional git commit message for a change that adds retry logic with exponential backoff and jitter to the HTTP client. Include a subject line and a 3-bullet body.'
157
+ },
158
+ {
159
+ id: 'doc-summary',
160
+ prompt: 'Write a one-paragraph README introduction for an open-source CLI tool called "logslice" that extracts time-bounded log windows from large log files without loading them fully into memory. Target audience: backend engineers.'
161
+ }
96
162
  ]
97
163
  };
98
164
 
package/src/table.js CHANGED
@@ -3,24 +3,30 @@ import { stripAnsi } from './ansi.js';
3
3
  export function renderTable(headers, rows, options = {}) {
4
4
  const ascii = Boolean(options.ascii);
5
5
  const maxCellWidth = options.maxCellWidth || 60;
6
- const normalizedRows = rows.map((row) => row.map((cell) => {
6
+ const wrapColumns = normalizeWrapColumns(headers, options.wrapColumns);
7
+ const normalizedRows = rows.map((row) => row.map((cell, index) => {
7
8
  const normalized = normalizeCell(cell);
8
9
  return {
9
10
  ...normalized,
10
- text: truncate(normalized.text, maxCellWidth)
11
+ text: wrapColumns.has(index) ? normalized.text : truncate(normalized.text, maxCellWidth)
11
12
  };
12
13
  }));
13
- const widths = headers.map((header, index) => {
14
+ const widths = fitTableWidths(headers, headers.map((header, index) => {
14
15
  const values = [truncate(header, maxCellWidth), ...normalizedRows.map((row) => row[index]?.text ?? '')];
16
+ if (wrapColumns.has(index)) {
17
+ return Math.max(...values.map((value) => Math.min(maxCellWidth, visibleLength(value))));
18
+ }
15
19
  return Math.max(...values.map(visibleLength));
16
- });
20
+ }), wrapColumns, terminalWidth(options));
17
21
  const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
18
22
 
19
23
  const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
20
24
  const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
21
25
  const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
22
26
  const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true, options);
23
- const bodyLines = normalizedRows.map((row) => rowLine(row, widths, style, false, options));
27
+ const bodyLines = wrapColumns.size > 0
28
+ ? normalizedRows.flatMap((row) => wrappedRowLines(row, widths, style, wrapColumns, options))
29
+ : normalizedRows.map((row) => rowLine(row, widths, style, false, options));
24
30
 
25
31
  return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
26
32
  }
@@ -44,8 +50,12 @@ export function renderBenchmarkReport(payload, options = {}) {
44
50
 
45
51
  const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
46
52
  const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
53
+ const tags = payload.options?.tags?.length ? payload.options.tags : [];
54
+ const note = payload.options?.note ?? null;
47
55
  lines.push(truncate(title, width));
48
56
  lines.push(truncate(meta, width));
57
+ if (tags.length > 0) lines.push(truncate(`tags: ${tags.join(', ')}`, width));
58
+ if (note) lines.push(truncate(`note: ${note}`, width));
49
59
  lines.push('');
50
60
 
51
61
  if (mode === 'compact') {
@@ -115,6 +125,49 @@ export function renderLegend(options = {}) {
115
125
  return renderWrappedTable(['table', 'column', 'definition', 'rule'], rows, legendColumnWidths(width), options);
116
126
  }
117
127
 
128
+ export function renderLatencyHistogram(results, options = {}) {
129
+ const { ascii = false, color = false, width: termWidth = 80 } = options;
130
+ const successes = results.filter((r) => r.ok && Number.isFinite(r.durationMs));
131
+ if (successes.length === 0) return 'No successful results to histogram.';
132
+
133
+ const values = successes.map((r) => r.durationMs).sort((a, b) => a - b);
134
+ const min = values[0];
135
+ const max = values[values.length - 1];
136
+
137
+ if (min === max) {
138
+ return `E2E latency histogram: all ${values.length} values = ${formatMs(min)}`;
139
+ }
140
+
141
+ const numBuckets = Math.min(20, Math.max(5, Math.floor((termWidth - 20) / 4)));
142
+ const bucketSize = (max - min) / numBuckets;
143
+ const counts = new Array(numBuckets).fill(0);
144
+
145
+ for (const v of values) {
146
+ const idx = Math.min(numBuckets - 1, Math.floor((v - min) / bucketSize));
147
+ counts[idx] += 1;
148
+ }
149
+
150
+ const maxCount = Math.max(...counts);
151
+ const barWidth = Math.max(1, termWidth - 24);
152
+ const lines = [];
153
+ const bar = ascii ? '#' : '█';
154
+ const halfBar = ascii ? ':' : '▌';
155
+
156
+ lines.push(`E2E latency distribution (n=${values.length}, ${formatMs(min)}…${formatMs(max)}):`);
157
+
158
+ for (let i = 0; i < numBuckets; i += 1) {
159
+ const bucketMin = min + i * bucketSize;
160
+ const bucketMax = bucketMin + bucketSize;
161
+ const label = `${formatMs(bucketMin)}`.padStart(7);
162
+ const fillLength = Math.round((counts[i] / maxCount) * barWidth);
163
+ const filled = bar.repeat(fillLength);
164
+ const countStr = counts[i] > 0 ? String(counts[i]) : '';
165
+ lines.push(`${label} ${filled}${countStr ? ` ${countStr}` : ''}`);
166
+ }
167
+
168
+ return lines.join('\n');
169
+ }
170
+
118
171
  export function formatMs(value) {
119
172
  if (value == null || !Number.isFinite(value)) return '-';
120
173
  if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
@@ -155,7 +208,7 @@ export function renderSummaryTable(summary, options = {}) {
155
208
  cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
156
209
  cell(formatNumber(item.rps), tones.rps),
157
210
  cell(formatPercent(item.latency.cv), cvTone(item.latency.cv)),
158
- cell(item.available ? '' : compactReason(item.skippedReason), item.available ? null : 'yellow')
211
+ cell(item.available ? '' : cleanReason(item.skippedReason), item.available ? null : 'yellow')
159
212
  ];
160
213
 
161
214
  if (mode === 'medium') {
@@ -194,7 +247,7 @@ export function renderSummaryTable(summary, options = {}) {
194
247
  ? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good' && header !== 'good rps'))
195
248
  : (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good' && header !== 'good rps'));
196
249
 
197
- return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
250
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52, wrapColumns: ['note'] });
198
251
  }
199
252
 
200
253
  export function renderDetailTable(summary, options = {}) {
@@ -241,7 +294,7 @@ export function renderDetailTable(summary, options = {}) {
241
294
  'description'
242
295
  ];
243
296
 
244
- return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52 });
297
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52, wrapColumns: ['description'] });
245
298
  }
246
299
 
247
300
  export function renderCompactSummary(summary, options = {}) {
@@ -286,8 +339,8 @@ export function renderModelsTable(models, options = {}) {
286
339
  cell(model.name),
287
340
  cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
288
341
  model.description || '-',
289
- compactReason(model.quota || model.reason || '-')
290
- ]), options);
342
+ cleanReason(model.quota || model.reason || '-')
343
+ ]), { ...options, wrapColumns: ['description', 'quota'] });
291
344
  }
292
345
 
293
346
  function formatRangeMs(low, high) {
@@ -372,6 +425,69 @@ function renderWrappedTable(headers, rows, widths, options = {}) {
372
425
  return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
373
426
  }
374
427
 
428
+ function normalizeWrapColumns(headers, wrapColumns = []) {
429
+ const normalized = new Set();
430
+ for (const column of wrapColumns || []) {
431
+ if (Number.isInteger(column)) {
432
+ if (column >= 0 && column < headers.length) normalized.add(column);
433
+ continue;
434
+ }
435
+
436
+ const index = headers.findIndex((header) => String(header).toLowerCase() === String(column).toLowerCase());
437
+ if (index >= 0) normalized.add(index);
438
+ }
439
+ return normalized;
440
+ }
441
+
442
+ function fitTableWidths(headers, widths, wrapColumns, targetWidth) {
443
+ if (!targetWidth || wrapColumns.size === 0 || renderedTableWidth(widths) <= targetWidth) return widths;
444
+
445
+ const fitted = [...widths];
446
+ const minimums = fitted.map((width, index) => (
447
+ wrapColumns.has(index)
448
+ ? Math.max(visibleLength(String(headers[index]).toUpperCase()), Math.min(12, width))
449
+ : width
450
+ ));
451
+ let excess = renderedTableWidth(fitted) - targetWidth;
452
+
453
+ while (excess > 0) {
454
+ const candidates = [...wrapColumns]
455
+ .filter((index) => fitted[index] > minimums[index])
456
+ .sort((a, b) => fitted[b] - fitted[a]);
457
+ if (candidates.length === 0) break;
458
+
459
+ for (const index of candidates) {
460
+ if (excess <= 0) break;
461
+ fitted[index] -= 1;
462
+ excess -= 1;
463
+ }
464
+ }
465
+
466
+ return fitted;
467
+ }
468
+
469
+ function renderedTableWidth(widths) {
470
+ return widths.reduce((sum, width) => sum + width, 0) + (widths.length * 3) + 1;
471
+ }
472
+
473
+ function wrappedRowLines(row, widths, style, wrapColumns, options = {}) {
474
+ const wrappedCells = row.map((value, index) => {
475
+ const normalized = normalizeCell(value);
476
+ return {
477
+ tone: normalized.tone,
478
+ lines: wrapColumns.has(index) ? wrapText(normalized.text, widths[index]) : [normalized.text]
479
+ };
480
+ });
481
+ const height = Math.max(...wrappedCells.map((item) => item.lines.length));
482
+ const lines = [];
483
+
484
+ for (let lineIndex = 0; lineIndex < height; lineIndex += 1) {
485
+ lines.push(rowLine(wrappedCells.map((item) => cell(item.lines[lineIndex] ?? '', item.tone)), widths, style, false, options));
486
+ }
487
+
488
+ return lines;
489
+ }
490
+
375
491
  function rowLine(row, widths, style, header = false, options = {}) {
376
492
  return `${style.vertical}${row.map((cell, index) => {
377
493
  const normalized = header ? { text: String(cell).toUpperCase(), tone: 'header' } : normalizeCell(cell);
@@ -417,8 +533,12 @@ function visibleLength(value) {
417
533
  return stripAnsi(value).length;
418
534
  }
419
535
 
536
+ function cleanReason(value) {
537
+ return String(value || '').replace(/\s+/g, ' ').trim();
538
+ }
539
+
420
540
  function compactReason(value) {
421
- const clean = String(value || '').replace(/\s+/g, ' ').trim();
541
+ const clean = cleanReason(value);
422
542
  if (clean.length <= 58) return clean;
423
543
  return `${clean.slice(0, 55)}...`;
424
544
  }