fm-bench 0.3.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -95,6 +95,7 @@ Useful flags:
95
95
  - `--capture-output`: include raw model output in JSON reports.
96
96
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
97
97
  - `--ascii`: use plain ASCII table borders.
98
+ - `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
98
99
  - `--compact`: force the narrow terminal layout.
99
100
  - `--width <n>`: render as if the terminal has `n` columns.
100
101
  - `--out <file>`: save a report.
@@ -143,6 +144,18 @@ Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream
143
144
 
144
145
  Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
145
146
 
147
+ ## Terminal Colors
148
+
149
+ Table output uses semantic ANSI color on interactive terminals:
150
+
151
+ - green: passing, steadier, or better than the current comparison set.
152
+ - yellow: marginal, partial, or near a budget.
153
+ - red: failing a budget, unstable, or slower/lower than peers.
154
+
155
+ Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
156
+
157
+ Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
158
+
146
159
  See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
147
160
 
148
161
  ## Requirements
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.3.0",
3
+ "version": "0.3.1",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/cli.js CHANGED
@@ -84,6 +84,7 @@ export function parseArgs(argv) {
84
84
  failFast: false,
85
85
  verbose: false,
86
86
  ascii: false,
87
+ color: 'auto',
87
88
  compact: false,
88
89
  width: null
89
90
  };
@@ -193,6 +194,12 @@ export function parseArgs(argv) {
193
194
  case '--ascii':
194
195
  options.ascii = true;
195
196
  break;
197
+ case '--color':
198
+ options.color = 'always';
199
+ break;
200
+ case '--no-color':
201
+ options.color = 'never';
202
+ break;
196
203
  case '--compact':
197
204
  options.compact = true;
198
205
  break;
@@ -291,11 +298,20 @@ function parsePositiveIntList(value, option) {
291
298
  function renderOptions(parsed) {
292
299
  return {
293
300
  ascii: parsed.ascii,
301
+ color: resolveColor(parsed.color),
294
302
  compact: parsed.compact,
295
303
  width: parsed.width
296
304
  };
297
305
  }
298
306
 
307
+ function resolveColor(value) {
308
+ if (value === 'always') return true;
309
+ if (value === 'never') return false;
310
+ if (process.env.NO_COLOR) return false;
311
+ if (process.env.FORCE_COLOR && process.env.FORCE_COLOR !== '0') return true;
312
+ return Boolean(process.stdout.isTTY);
313
+ }
314
+
299
315
  function helpText() {
300
316
  return `fm-bench ${packageJson.version}
301
317
 
@@ -336,6 +352,8 @@ Output:
336
352
  --json Alias for --format json
337
353
  --csv Alias for --format csv
338
354
  --ascii Use plain ASCII tables instead of Unicode
355
+ --color Force ANSI colors in table output
356
+ --no-color Disable ANSI colors in table output
339
357
  --compact Force compact terminal layout
340
358
  --width <n> Render for a specific terminal width
341
359
  -o, --out <file> Save JSON or CSV report based on file extension
package/src/table.js CHANGED
@@ -1,9 +1,17 @@
1
+ import { stripAnsi } from './ansi.js';
2
+
1
3
  export function renderTable(headers, rows, options = {}) {
2
4
  const ascii = Boolean(options.ascii);
3
5
  const maxCellWidth = options.maxCellWidth || 60;
4
- const stringRows = rows.map((row) => row.map((cell) => truncate(formatCell(cell), maxCellWidth)));
6
+ const normalizedRows = rows.map((row) => row.map((cell) => {
7
+ const normalized = normalizeCell(cell);
8
+ return {
9
+ ...normalized,
10
+ text: truncate(normalized.text, maxCellWidth)
11
+ };
12
+ }));
5
13
  const widths = headers.map((header, index) => {
6
- const values = [truncate(header, maxCellWidth), ...stringRows.map((row) => row[index] ?? '')];
14
+ const values = [truncate(header, maxCellWidth), ...normalizedRows.map((row) => row[index]?.text ?? '')];
7
15
  return Math.max(...values.map(visibleLength));
8
16
  });
9
17
  const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
@@ -11,8 +19,8 @@ export function renderTable(headers, rows, options = {}) {
11
19
  const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
12
20
  const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
13
21
  const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
14
- const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true);
15
- const bodyLines = stringRows.map((row) => rowLine(row, widths, style));
22
+ const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true, options);
23
+ const bodyLines = normalizedRows.map((row) => rowLine(row, widths, style, false, options));
16
24
 
17
25
  return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
18
26
  }
@@ -41,11 +49,11 @@ export function renderBenchmarkReport(payload, options = {}) {
41
49
  lines.push('');
42
50
 
43
51
  if (mode === 'compact') {
44
- lines.push(renderCompactSummary(payload.summary, { ...options, width }));
52
+ lines.push(renderCompactSummary(payload.summary, { ...options, width, slo: payload.options.slo }));
45
53
  } else {
46
- lines.push(renderSummaryTable(payload.summary, { ...options, mode, width }));
54
+ lines.push(renderSummaryTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
47
55
  lines.push('');
48
- lines.push(renderDetailTable(payload.summary, { ...options, mode, width }));
56
+ lines.push(renderDetailTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
49
57
  }
50
58
 
51
59
  lines.push('');
@@ -74,24 +82,26 @@ export function formatPercent(value, digits = 0) {
74
82
  export function renderSummaryTable(summary, options = {}) {
75
83
  const mode = options.mode || 'wide';
76
84
  const hasGoodput = summary.some((item) => item.goodputRate != null);
85
+ const ranks = rankSummary(summary);
77
86
  const rows = summary.map((item) => {
78
87
  const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
88
+ const tones = metricTones(item, ranks, options.slo);
79
89
  const base = [
80
- formatConcurrency(item.concurrency),
81
- item.model,
82
- status,
83
- item.attempted ? `${item.successes}/${item.attempted}` : '-',
84
- formatPercent(item.successRate),
85
- formatPercent(item.goodputRate),
86
- formatMs(item.ttft.p50),
87
- formatMs(item.ttft.p95),
88
- formatMs(item.latency.p50),
89
- formatMs(item.latency.p95),
90
- formatNumber(item.tokensPerSecond.avg),
91
- formatNumber(item.outputTokenThroughput),
92
- formatNumber(item.rps),
93
- formatPercent(item.latency.cv),
94
- item.available ? '' : compactReason(item.skippedReason)
90
+ cell(formatConcurrency(item.concurrency)),
91
+ cell(item.model, item.available ? null : 'muted'),
92
+ cell(status, statusTone(status)),
93
+ cell(item.attempted ? `${item.successes}/${item.attempted}` : '-', item.failures > 0 ? 'yellow' : item.available ? 'green' : 'muted'),
94
+ cell(formatPercent(item.successRate), percentTone(item.successRate, 0.95, 1)),
95
+ cell(formatPercent(item.goodputRate), percentTone(item.goodputRate, 0.8, 1)),
96
+ cell(formatMs(item.ttft.p50), tones.ttft),
97
+ cell(formatMs(item.ttft.p95), tones.ttftP95),
98
+ cell(formatMs(item.latency.p50), tones.e2e),
99
+ cell(formatMs(item.latency.p95), tones.e2eP95),
100
+ cell(formatNumber(item.tokensPerSecond.avg), tones.userTps),
101
+ cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
102
+ cell(formatNumber(item.rps), tones.rps),
103
+ cell(formatPercent(item.latency.cv), cvTone(item.latency.cv)),
104
+ cell(item.available ? '' : compactReason(item.skippedReason), item.available ? null : 'yellow')
95
105
  ];
96
106
 
97
107
  if (mode === 'medium') {
@@ -113,8 +123,8 @@ export function renderSummaryTable(summary, options = {}) {
113
123
  base[7],
114
124
  base[8],
115
125
  base[9],
116
- formatMs(item.tpot.p50),
117
- formatMs(item.tpot.p95),
126
+ cell(formatMs(item.tpot.p50), tones.tpot),
127
+ cell(formatMs(item.tpot.p95), tones.tpotP95),
118
128
  base[10],
119
129
  base[11],
120
130
  base[12],
@@ -135,17 +145,19 @@ export function renderSummaryTable(summary, options = {}) {
135
145
 
136
146
  export function renderDetailTable(summary, options = {}) {
137
147
  const mode = options.mode || 'wide';
148
+ const ranks = rankSummary(summary);
138
149
  const rows = summary.map((item) => {
150
+ const tones = metricTones(item, ranks, options.slo);
139
151
  const base = [
140
- formatConcurrency(item.concurrency),
141
- item.model,
142
- formatNumber(item.promptTokens.avg, 0),
143
- formatNumber(item.outputTokens.avg, 0),
144
- formatNumber(item.decodeTokensPerSecond.avg),
145
- formatMs(item.latency.p99),
146
- formatRangeMs(item.latency.ci95Low, item.latency.ci95High),
147
- formatPercent(item.repeatability),
148
- item.description || '-'
152
+ cell(formatConcurrency(item.concurrency)),
153
+ cell(item.model, item.available ? null : 'muted'),
154
+ cell(formatNumber(item.promptTokens.avg, 0)),
155
+ cell(formatNumber(item.outputTokens.avg, 0)),
156
+ cell(formatNumber(item.decodeTokensPerSecond.avg), tones.decodeTps),
157
+ cell(formatMs(item.latency.p99), tones.e2eP99),
158
+ cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
159
+ cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
160
+ cell(item.description || '-', 'muted')
149
161
  ];
150
162
 
151
163
  if (mode === 'medium') {
@@ -176,19 +188,21 @@ export function renderCompactSummary(summary, options = {}) {
176
188
  const width = options.width || 80;
177
189
  const separator = options.ascii ? '-' : '─';
178
190
  const lines = [];
191
+ const ranks = rankSummary(summary);
179
192
 
180
193
  for (const item of summary) {
181
194
  const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
182
195
  const title = `${item.model} c${formatConcurrency(item.concurrency)} ${status} ${item.attempted ? `${item.successes}/${item.attempted}` : '-'}`;
183
- lines.push(truncate(title, width));
196
+ lines.push(toneText(truncate(title, width), statusTone(status), options));
184
197
 
185
198
  if (item.available) {
186
- lines.push(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width));
187
199
  const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
188
- lines.push(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width));
189
- lines.push(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width));
200
+ const tones = metricTones(item, ranks, options.slo);
201
+ lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
202
+ lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
203
+ lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width), worstTone(tones.tpot, percentTone(item.repeatability, 0.5, 0.9)), options));
190
204
  } else {
191
- lines.push(truncate(` ${compactReason(item.skippedReason)}`, width));
205
+ lines.push(toneText(truncate(` ${compactReason(item.skippedReason)}`, width), 'yellow', options));
192
206
  }
193
207
  lines.push(separator.repeat(Math.min(width, 72)));
194
208
  }
@@ -201,12 +215,15 @@ export function renderModelsTable(models, options = {}) {
201
215
  const width = terminalWidth(options);
202
216
  const compact = options.compact || width < 88;
203
217
  if (compact) {
204
- return models.map((model) => `${model.name} ${model.available ? 'yes' : 'no'} ${compactReason(model.reason || model.description || '-')}`).join('\n');
218
+ return models.map((model) => {
219
+ const status = model.available ? 'yes' : 'no';
220
+ return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(model.reason || model.description || '-')}`;
221
+ }).join('\n');
205
222
  }
206
223
 
207
224
  return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
208
- model.name,
209
- model.available ? 'yes' : 'no',
225
+ cell(model.name),
226
+ cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
210
227
  model.description || '-',
211
228
  compactReason(model.quota || model.reason || '-')
212
229
  ]), options);
@@ -270,13 +287,32 @@ function rule(left, join, right, horizontal, widths) {
270
287
  return `${left}${widths.map((width) => horizontal.repeat(width + 2)).join(join)}${right}`;
271
288
  }
272
289
 
273
- function rowLine(row, widths, style, header = false) {
290
+ function rowLine(row, widths, style, header = false, options = {}) {
274
291
  return `${style.vertical}${row.map((cell, index) => {
275
- const value = header ? String(cell).toUpperCase() : formatCell(cell);
276
- return ` ${pad(value, widths[index], !header && isNumericCell(value))} `;
292
+ const normalized = header ? { text: String(cell).toUpperCase(), tone: 'header' } : normalizeCell(cell);
293
+ const value = normalized.text;
294
+ const padded = pad(value, widths[index], !header && isNumericCell(value));
295
+ return ` ${toneText(padded, normalized.tone, options)} `;
277
296
  }).join(style.vertical)}${style.vertical}`;
278
297
  }
279
298
 
299
+ function cell(text, tone = null) {
300
+ return { text, tone };
301
+ }
302
+
303
+ function normalizeCell(value) {
304
+ if (value && typeof value === 'object' && Object.hasOwn(value, 'text')) {
305
+ return {
306
+ text: formatCell(value.text),
307
+ tone: value.tone || null
308
+ };
309
+ }
310
+ return {
311
+ text: formatCell(value),
312
+ tone: null
313
+ };
314
+ }
315
+
280
316
  function formatCell(value) {
281
317
  if (value == null) return '';
282
318
  return String(value);
@@ -293,7 +329,7 @@ function isNumericCell(value) {
293
329
  }
294
330
 
295
331
  function visibleLength(value) {
296
- return String(value).length;
332
+ return stripAnsi(value).length;
297
333
  }
298
334
 
299
335
  function compactReason(value) {
@@ -308,3 +344,112 @@ function truncate(value, width) {
308
344
  if (width <= 1) return '…';
309
345
  return `${text.slice(0, width - 1)}…`;
310
346
  }
347
+
348
+ const TONES = {
349
+ header: ['\x1b[1m', '\x1b[0m'],
350
+ green: ['\x1b[32m', '\x1b[0m'],
351
+ yellow: ['\x1b[33m', '\x1b[0m'],
352
+ red: ['\x1b[31m', '\x1b[0m'],
353
+ muted: ['\x1b[2m', '\x1b[0m']
354
+ };
355
+
356
+ function toneText(text, tone, options = {}) {
357
+ if (!options.color || !tone || !TONES[tone]) return text;
358
+ const [open, close] = TONES[tone];
359
+ return `${open}${text}${close}`;
360
+ }
361
+
362
+ function statusTone(status) {
363
+ if (status === 'ok') return 'green';
364
+ if (status === 'partial') return 'yellow';
365
+ if (status === 'skipped') return 'yellow';
366
+ return 'red';
367
+ }
368
+
369
+ function percentTone(value, yellowAt, greenAt) {
370
+ if (value == null || !Number.isFinite(value)) return null;
371
+ if (value >= greenAt) return 'green';
372
+ if (value >= yellowAt) return 'yellow';
373
+ return 'red';
374
+ }
375
+
376
+ function cvTone(value) {
377
+ if (value == null || !Number.isFinite(value)) return null;
378
+ if (value <= 0.1) return 'green';
379
+ if (value <= 0.25) return 'yellow';
380
+ return 'red';
381
+ }
382
+
383
+ function thresholdTone(value, threshold) {
384
+ if (value == null || !Number.isFinite(value) || !Number.isFinite(threshold)) return null;
385
+ if (value <= threshold * 0.8) return 'green';
386
+ if (value <= threshold) return 'yellow';
387
+ return 'red';
388
+ }
389
+
390
+ function rankSummary(summary) {
391
+ return {
392
+ ttft: collectMetric(summary, (item) => item.ttft?.p50),
393
+ ttftP95: collectMetric(summary, (item) => item.ttft?.p95),
394
+ e2e: collectMetric(summary, (item) => item.latency?.p50),
395
+ e2eP95: collectMetric(summary, (item) => item.latency?.p95),
396
+ e2eP99: collectMetric(summary, (item) => item.latency?.p99),
397
+ tpot: collectMetric(summary, (item) => item.tpot?.p50),
398
+ tpotP95: collectMetric(summary, (item) => item.tpot?.p95),
399
+ userTps: collectMetric(summary, (item) => item.tokensPerSecond?.avg),
400
+ systemTps: collectMetric(summary, (item) => item.outputTokenThroughput),
401
+ rps: collectMetric(summary, (item) => item.rps),
402
+ decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg)
403
+ };
404
+ }
405
+
406
+ function collectMetric(summary, accessor) {
407
+ return summary.map(accessor).filter((value) => Number.isFinite(value));
408
+ }
409
+
410
+ function rankTone(value, values) {
411
+ if (value == null || !Number.isFinite(value) || values.length === 0) return null;
412
+ if (values.length === 1) return 'green';
413
+ const min = Math.min(...values);
414
+ const max = Math.max(...values);
415
+ if (max === min) return 'green';
416
+ const position = (value - min) / (max - min);
417
+ if (position >= 0.67) return 'green';
418
+ if (position >= 0.34) return 'yellow';
419
+ return 'red';
420
+ }
421
+
422
+ function rankToneLower(value, values) {
423
+ if (value == null || !Number.isFinite(value) || values.length === 0) return null;
424
+ if (values.length === 1) return 'green';
425
+ const min = Math.min(...values);
426
+ const max = Math.max(...values);
427
+ if (max === min) return 'green';
428
+ const position = (max - value) / (max - min);
429
+ if (position >= 0.67) return 'green';
430
+ if (position >= 0.34) return 'yellow';
431
+ return 'red';
432
+ }
433
+
434
+ function metricTones(item, ranks, slo = {}) {
435
+ return {
436
+ ttft: thresholdTone(item.ttft?.p50, slo.ttftMs) || rankToneLower(item.ttft?.p50, ranks.ttft),
437
+ ttftP95: thresholdTone(item.ttft?.p95, slo.ttftMs) || rankToneLower(item.ttft?.p95, ranks.ttftP95),
438
+ e2e: thresholdTone(item.latency?.p50, slo.e2eMs) || rankToneLower(item.latency?.p50, ranks.e2e),
439
+ e2eP95: thresholdTone(item.latency?.p95, slo.e2eMs) || rankToneLower(item.latency?.p95, ranks.e2eP95),
440
+ e2eP99: thresholdTone(item.latency?.p99, slo.e2eMs) || rankToneLower(item.latency?.p99, ranks.e2eP99),
441
+ tpot: thresholdTone(item.tpot?.p50, slo.tpotMs) || rankToneLower(item.tpot?.p50, ranks.tpot),
442
+ tpotP95: thresholdTone(item.tpot?.p95, slo.tpotMs) || rankToneLower(item.tpot?.p95, ranks.tpotP95),
443
+ userTps: rankTone(item.tokensPerSecond?.avg, ranks.userTps),
444
+ systemTps: rankTone(item.outputTokenThroughput, ranks.systemTps),
445
+ rps: rankTone(item.rps, ranks.rps),
446
+ decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps)
447
+ };
448
+ }
449
+
450
+ function worstTone(...tones) {
451
+ if (tones.includes('red')) return 'red';
452
+ if (tones.includes('yellow')) return 'yellow';
453
+ if (tones.includes('green')) return 'green';
454
+ return null;
455
+ }