fm-bench 0.6.3 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/report.js CHANGED
@@ -16,6 +16,7 @@ export function flattenResults(results) {
16
16
  concurrency: result.concurrency ?? '',
17
17
  prompt_id: result.promptId,
18
18
  run: result.run,
19
+ attempts: result.attempts ?? 1,
19
20
  ok: result.ok,
20
21
  duration_ms: round(result.durationMs),
21
22
  ttft_ms: round(result.firstTokenMs),
@@ -58,9 +59,19 @@ export async function writeReport(filePath, payload, format) {
58
59
 
59
60
  function csvEscape(value) {
60
61
  const text = String(value ?? '');
61
- if (/[",\n\r]/.test(text)) {
62
- return `"${text.replaceAll('"', '""')}"`;
62
+ const safe = guardFormula(text);
63
+ if (/[",\n\r]/.test(safe)) {
64
+ return `"${safe.replaceAll('"', '""')}"`;
63
65
  }
66
+ return safe;
67
+ }
68
+
69
+ // Spreadsheet programs execute cells that start with =, +, @, or a non-numeric
70
+ // leading -. Prompt text and captured output are untrusted input, so prefix
71
+ // those cells with a single quote to keep them literal text.
72
+ function guardFormula(text) {
73
+ if (/^[=+@\t\r]/.test(text)) return `'${text}`;
74
+ if (text.startsWith('-') && !/^-?\d+(\.\d+)?$/.test(text)) return `'${text}`;
64
75
  return text;
65
76
  }
66
77
 
package/src/schema.js CHANGED
@@ -126,10 +126,26 @@ export function validateReport(value) {
126
126
  for (const key of REQUIRED_TOP_LEVEL) {
127
127
  if (!(key in report)) errors.push(`missing required field: ${key}`);
128
128
  }
129
- if (!Array.isArray(report.summary)) errors.push('summary must be an array');
129
+ if (!Array.isArray(report.summary)) {
130
+ errors.push('summary must be an array');
131
+ } else {
132
+ for (const [index, item] of report.summary.entries()) {
133
+ if (!item || typeof item !== 'object') {
134
+ errors.push(`summary[${index}] must be an object`);
135
+ continue;
136
+ }
137
+ const row = /** @type {Record<string, unknown>} */ (item);
138
+ if (typeof row.model !== 'string' || row.model === '') {
139
+ errors.push(`summary[${index}].model must be a non-empty string`);
140
+ }
141
+ }
142
+ }
130
143
  if (report.schemaVersion != null && report.schemaVersion !== REPORT_SCHEMA_VERSION) {
131
144
  errors.push(`unsupported schemaVersion: ${report.schemaVersion} (expected ${REPORT_SCHEMA_VERSION})`);
132
145
  }
146
+ if (report.metrics != null && (typeof report.metrics !== 'object' || Array.isArray(report.metrics))) {
147
+ errors.push('metrics must be an object when present');
148
+ }
133
149
  if (errors.length > 0) return { ok: false, errors };
134
150
  return { ok: true, report };
135
151
  }
package/src/stats.js CHANGED
@@ -1,3 +1,12 @@
1
+ /**
2
+ * Sample statistics for one metric.
3
+ *
4
+ * Spread metrics (`stddev`, `cv`, `ci95*`) need at least two samples. With a
5
+ * single sample they are `null` rather than `0`, because a "0% variation" or a
6
+ * zero-width confidence interval is invented precision, not a measurement.
7
+ *
8
+ * @param {number[]} values
9
+ */
1
10
  export function summarizeNumbers(values) {
2
11
  const clean = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
3
12
  if (clean.length === 0) {
@@ -20,11 +29,12 @@ export function summarizeNumbers(values) {
20
29
 
21
30
  const total = clean.reduce((sum, value) => sum + value, 0);
22
31
  const avg = total / clean.length;
23
- const variance = clean.length > 1
32
+ const hasSpread = clean.length > 1;
33
+ const variance = hasSpread
24
34
  ? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
25
- : 0;
26
- const stddev = Math.sqrt(variance);
27
- const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
35
+ : null;
36
+ const stddev = variance == null ? null : Math.sqrt(variance);
37
+ const margin = hasSpread ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : null;
28
38
  return {
29
39
  count: clean.length,
30
40
  min: clean[0],
@@ -32,9 +42,9 @@ export function summarizeNumbers(values) {
32
42
  avg,
33
43
  sum: total,
34
44
  stddev,
35
- cv: avg !== 0 ? stddev / Math.abs(avg) : null,
36
- ci95Low: avg - margin,
37
- ci95High: avg + margin,
45
+ cv: hasSpread && avg !== 0 ? stddev / Math.abs(avg) : null,
46
+ ci95Low: hasSpread ? avg - margin : null,
47
+ ci95High: hasSpread ? avg + margin : null,
38
48
  p50: percentile(clean, 50),
39
49
  p90: percentile(clean, 90),
40
50
  p95: percentile(clean, 95),
@@ -42,16 +52,25 @@ export function summarizeNumbers(values) {
42
52
  };
43
53
  }
44
54
 
45
- export function percentile(sortedValues, percentileValue) {
46
- if (sortedValues.length === 0) return null;
47
- if (sortedValues.length === 1) return sortedValues[0];
55
+ /**
56
+ * Percentile with linear interpolation between closest ranks (the same method
57
+ * as Excel's PERCENTILE.INC / NumPy's default). Accepts unsorted input so
58
+ * callers cannot silently get a wrong answer from an unsorted array.
59
+ *
60
+ * @param {number[]} values
61
+ * @param {number} percentileValue 0-100
62
+ */
63
+ export function percentile(values, percentileValue) {
64
+ const sorted = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
65
+ if (sorted.length === 0) return null;
66
+ if (sorted.length === 1) return sorted[0];
48
67
 
49
- const rank = (percentileValue / 100) * (sortedValues.length - 1);
68
+ const rank = (percentileValue / 100) * (sorted.length - 1);
50
69
  const low = Math.floor(rank);
51
70
  const high = Math.ceil(rank);
52
- if (low === high) return sortedValues[low];
71
+ if (low === high) return sorted[low];
53
72
  const weight = rank - low;
54
- return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
73
+ return sorted[low] * (1 - weight) + sorted[high] * weight;
55
74
  }
56
75
 
57
76
  export function summarizeByModel(results, modelStatuses = [], options = {}) {
@@ -66,6 +85,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
66
85
  concurrency,
67
86
  description: status.description,
68
87
  available: status.available,
88
+ unsupported: Boolean(status.unsupported),
69
89
  skippedReason: status.available ? '' : status.reason || 'Unavailable',
70
90
  results: []
71
91
  });
@@ -80,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
80
100
  concurrency: result.concurrency,
81
101
  description: '',
82
102
  available: true,
103
+ unsupported: false,
83
104
  skippedReason: '',
84
105
  results: []
85
106
  });
@@ -98,7 +119,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
98
119
  const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
99
120
  const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
100
121
  const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
101
- const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
122
+ const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
102
123
  const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
103
124
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
104
125
  const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
@@ -110,14 +131,18 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
110
131
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
111
132
  const totalTokens = promptTokens.sum + outputTokens.sum;
112
133
  const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
134
+ const attempts = entry.results.reduce((sum, result) => sum + (result.attempts ?? 1), 0);
113
135
 
114
136
  return {
115
137
  model: entry.model,
116
138
  concurrency: entry.concurrency,
117
139
  description: entry.description,
118
140
  available: entry.available,
141
+ unsupported: entry.unsupported,
119
142
  skippedReason: entry.skippedReason,
120
143
  attempted: entry.results.length,
144
+ attempts,
145
+ retried: entry.results.length > 0 ? Math.max(0, attempts - entry.results.length) : 0,
121
146
  successes: successes.length,
122
147
  failures: failures.length,
123
148
  successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
@@ -147,6 +172,10 @@ function summaryKey(model, concurrency) {
147
172
  return `${model}::${concurrency ?? 'default'}`;
148
173
  }
149
174
 
175
+ // Two-sided 95% t critical values. Exact table entries up to 30 degrees of
176
+ // freedom, then the standard 2.0 / 1.96 approximations for larger samples.
177
+ // fm-bench uses this for a mean confidence interval, which is context for
178
+ // small samples rather than a hypothesis test.
150
179
  function tCritical95(n) {
151
180
  const df = Math.max(1, n - 1);
152
181
  const table = {
package/src/table.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { stripAnsi } from './ansi.js';
2
+ import { formatCapabilitySummary } from './metrics.js';
2
3
 
3
4
  export function renderTable(headers, rows, options = {}) {
4
5
  const ascii = Boolean(options.ascii);
@@ -52,10 +53,19 @@ export function renderBenchmarkReport(payload, options = {}) {
52
53
  const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
53
54
  const tags = payload.options?.tags?.length ? payload.options.tags : [];
54
55
  const note = payload.options?.note ?? null;
55
- lines.push(truncate(title, width));
56
- lines.push(truncate(meta, width));
57
- if (tags.length > 0) lines.push(truncate(`tags: ${tags.join(', ')}`, width));
58
- if (note) lines.push(truncate(`note: ${note}`, width));
56
+ lines.push(...wrapText(title, width));
57
+ lines.push(...wrapText(meta, width));
58
+ if (tags.length > 0) {
59
+ for (const line of wrapText(`tags: ${tags.join(', ')}`, width)) lines.push(line);
60
+ }
61
+ if (note) {
62
+ for (const line of wrapText(`note: ${note}`, width)) lines.push(line);
63
+ }
64
+ for (const metricNote of unavailableMetricNotes(payload.metrics)) {
65
+ for (const line of wrapText(`unavailable: ${metricNote}`, width)) {
66
+ lines.push(line);
67
+ }
68
+ }
59
69
  lines.push('');
60
70
 
61
71
  if (mode === 'compact') {
@@ -71,41 +81,43 @@ export function renderBenchmarkReport(payload, options = {}) {
71
81
 
72
82
  export function legendEntries() {
73
83
  return [
74
- entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.'),
75
- entry('summary', 'MODEL', 'fm model name, such as system or pcc.', ''),
76
- entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable.'),
77
- entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.'),
78
- entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.'),
79
- entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Green 100%, yellow >=80%, red <80%.'),
80
- entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.'),
81
- entry('summary', 'TTFT', 'Time to first streamed output chunk, p50.', 'Lower is better. Uses SLO threshold when set; otherwise relative ranking.'),
82
- entry('summary', 'TTFT P95', '95th percentile time to first streamed output chunk.', 'Lower is better.'),
83
- entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better.'),
84
- entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.'),
85
- entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Lower is better. Requires streaming and token counts.'),
86
- entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.'),
87
- entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Higher is better; relative ranking.'),
88
- entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better; relative ranking.'),
89
- entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Higher is better; relative ranking.'),
90
- entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.'),
91
- entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', ''),
92
- entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm token-count.', ''),
93
- entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm token-count.', ''),
94
- entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Higher is better; estimates prompt-processing speed for streaming runs.'),
95
- entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Higher is better; requires streaming and token counts.'),
96
- entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.'),
97
- entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Lower is smoother streaming.'),
98
- entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.'),
99
- entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', 'Narrower usually means a steadier estimate. Treat small samples carefully.'),
100
- entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.'),
101
- entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', ''),
102
- entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', ''),
103
- entry('models', 'QUOTA', 'Raw fm quota-usage output or unavailable reason.', 'Mostly relevant for Private Cloud Compute.'),
104
- entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.'),
105
- entry('colors', 'GREEN', 'Passing, steadier, faster, or better within this benchmark context.', ''),
106
- entry('colors', 'YELLOW', 'Marginal, partial, near a budget, or middle-ranked within this benchmark context.', ''),
107
- entry('colors', 'RED', 'Failed budget, unstable, slower, lower, or worse within this benchmark context.', ''),
108
- entry('colors', 'MUTED', 'Unavailable model, skipped value, or descriptive text.', '')
84
+ entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.', 'controlled'),
85
+ entry('summary', 'MODEL', 'fm model name reported by the detected fm build.', '', 'measured'),
86
+ entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable or unsupported.', 'measured'),
87
+ entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.', 'measured'),
88
+ entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.', 'measured'),
89
+ entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Runs whose SLO metric is unmeasurable count as not good.', 'derived'),
90
+ entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.', 'derived'),
91
+ entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity, not per token. Lower is better.', 'proxy'),
92
+ entry('summary', 'TTFT P95', '95th percentile time to the first streamed stdout chunk.', 'Lower is better.', 'proxy'),
93
+ entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better. This is a direct wall-clock measurement.', 'measured'),
94
+ entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.', 'measured'),
95
+ entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Derived from fm token counts and stream timings. Lower is better.', 'derived'),
96
+ entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.', 'derived'),
97
+ entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Derived from fm token counts. Higher is better.', 'derived'),
98
+ entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better.', 'derived'),
99
+ entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Measured from process timings. Higher is better.', 'measured'),
100
+ entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.', 'derived'),
101
+ entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', '', 'measured'),
102
+ entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
103
+ entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
104
+ entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Proxy: prefill is inferred from time to first chunk, not observed directly. Higher is better.', 'proxy'),
105
+ entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Derived; requires streaming and token counts. Higher is better.', 'derived'),
106
+ entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.', 'proxy'),
107
+ entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Proxy for decode smoothness at chunk granularity. Lower is smoother.', 'proxy'),
108
+ entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.', 'measured'),
109
+ entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', 'Blank with fewer than two successful samples; treat small samples carefully.', 'derived'),
110
+ entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.', 'derived'),
111
+ entry('detail', 'ATTEMPTS', 'Total fm invocations including retries.', 'Shown in reports; a retried run is still one measured result.', 'measured'),
112
+ entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', '', 'measured'),
113
+ entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', '', 'measured'),
114
+ entry('models', 'QUOTA', 'Quota information when the fm build exposes a quota command.', 'Column is omitted entirely when the installed fm has no quota command.', 'measured'),
115
+ entry('metrics', 'SOURCE', 'How a metric is obtained: measured, proxy, derived, or controlled.', 'measured = observed directly; proxy = observed at coarser granularity; derived = computed from measured values.', 'measured'),
116
+ entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.', 'derived'),
117
+ entry('colors', 'GREEN', 'Passing, steadier, faster, or better within this benchmark context.', '', 'controlled'),
118
+ entry('colors', 'YELLOW', 'Marginal, partial, near a budget, or middle-ranked within this benchmark context.', '', 'controlled'),
119
+ entry('colors', 'RED', 'Failed budget, unstable, slower, lower, or worse within this benchmark context.', '', 'controlled'),
120
+ entry('colors', 'MUTED', 'Unavailable model, skipped value, or descriptive text.', '', 'controlled')
109
121
  ];
110
122
  }
111
123
 
@@ -114,6 +126,7 @@ export function renderLegend(options = {}) {
114
126
  const rows = legendEntries().map((item) => [
115
127
  item.table,
116
128
  item.column,
129
+ item.kind || 'measured',
117
130
  item.definition,
118
131
  item.rule || '-'
119
132
  ]);
@@ -122,7 +135,7 @@ export function renderLegend(options = {}) {
122
135
  return renderCompactLegend(width);
123
136
  }
124
137
 
125
- return renderWrappedTable(['table', 'column', 'definition', 'rule'], rows, legendColumnWidths(width), options);
138
+ return renderWrappedTable(['table', 'column', 'source', 'definition', 'rule'], rows, legendColumnWidths(width), options);
126
139
  }
127
140
 
128
141
  export function renderLatencyHistogram(results, options = {}) {
@@ -335,6 +348,19 @@ export function renderModelsTable(models, options = {}) {
335
348
  }).join('\n');
336
349
  }
337
350
 
351
+ const quotaSupported = options.capabilities
352
+ ? Boolean(options.capabilities.features?.quota)
353
+ : models.some((model) => model.quotaSupported === true || (model.quota != null && model.quota !== ''));
354
+
355
+ if (!quotaSupported) {
356
+ return renderTable(['model', 'available', 'description', 'notes'], models.map((model) => [
357
+ cell(model.name),
358
+ cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
359
+ model.description || '-',
360
+ cleanReason(model.reason || '-')
361
+ ]), { ...options, wrapColumns: ['description', 'notes'] });
362
+ }
363
+
338
364
  return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
339
365
  cell(model.name),
340
366
  cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
@@ -343,6 +369,27 @@ export function renderModelsTable(models, options = {}) {
343
369
  ]), { ...options, wrapColumns: ['description', 'quota'] });
344
370
  }
345
371
 
372
+ /**
373
+ * `fm-bench models` body: fm capability summary, then the model table.
374
+ * Unsupported capabilities are stated up front so a missing quota column or
375
+ * blank metric never looks like a bug.
376
+ */
377
+ export function renderModelsReport(models, options = {}) {
378
+ const capabilities = options.capabilities;
379
+ const lines = [];
380
+ if (capabilities) {
381
+ lines.push(`fm ${capabilities.bin}${capabilities.digest ? ` help ${capabilities.digest}` : ''}${capabilities.ok ? '' : ' (no commands detected)'}`);
382
+ lines.push(`commands ${capabilities.commands.join(', ') || 'none detected'}`);
383
+ lines.push(`features ${formatCapabilitySummary(capabilities)}`);
384
+ for (const warning of capabilities.warnings) {
385
+ lines.push(`warn ${warning}`);
386
+ }
387
+ lines.push('');
388
+ }
389
+ lines.push(renderModelsTable(models, options));
390
+ return lines.join('\n');
391
+ }
392
+
346
393
  function formatRangeMs(low, high) {
347
394
  if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
348
395
  return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
@@ -364,15 +411,38 @@ function formatSlo(slo = {}) {
364
411
  return parts.length ? `SLO ${parts.join(',')}` : '';
365
412
  }
366
413
 
367
- function entry(table, column, definition, rule) {
414
+ function entry(table, column, definition, rule, kind = 'measured') {
368
415
  return {
369
416
  table,
370
417
  column,
371
418
  definition,
372
- rule
419
+ rule,
420
+ kind
373
421
  };
374
422
  }
375
423
 
424
+ /**
425
+ * Reasons why metric columns are blank. Keeps "unavailable" visible in table
426
+ * output instead of leaving readers to guess why a column is empty.
427
+ */
428
+ function unavailableMetricNotes(metrics) {
429
+ if (!metrics) return [];
430
+ const notes = [];
431
+ const tokenMetric = metrics.promptTokens;
432
+ if (tokenMetric && !tokenMetric.available) {
433
+ notes.push(`${tokenMetric.label} unavailable — ${tokenMetric.unavailableReason}`);
434
+ }
435
+ const ttft = metrics.ttft;
436
+ if (ttft && !ttft.available) {
437
+ notes.push(`${ttft.label} unavailable — ${ttft.unavailableReason}`);
438
+ }
439
+ const quota = metrics.quota;
440
+ if (quota && !quota.available) {
441
+ notes.push(`${quota.label} unavailable — ${quota.unavailableReason}`);
442
+ }
443
+ return notes;
444
+ }
445
+
376
446
  const ASCII_TABLE = {
377
447
  topLeft: '+',
378
448
  topJoin: '+',
@@ -516,7 +586,10 @@ function normalizeCell(value) {
516
586
 
517
587
  function formatCell(value) {
518
588
  if (value == null) return '';
519
- return String(value);
589
+ // Terminal output carries prompt text, captured model output, and fm
590
+ // diagnostics. Strip ANSI escapes and control characters so they cannot
591
+ // corrupt the table layout or emit terminal control sequences.
592
+ return stripAnsi(String(value)).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
520
593
  }
521
594
 
522
595
  function pad(value, width, left = false) {
@@ -546,7 +619,7 @@ function compactReason(value) {
546
619
  function renderCompactLegend(width) {
547
620
  const lines = [];
548
621
  for (const item of legendEntries()) {
549
- lines.push(...wrapText(`${item.table.toUpperCase()} ${item.column}`, width));
622
+ lines.push(...wrapText(`${item.table.toUpperCase()} ${item.column} (${item.kind || 'measured'})`, width));
550
623
  lines.push(...wrapText(` Definition: ${item.definition}`, width));
551
624
  if (item.rule) {
552
625
  lines.push(...wrapText(` Rule: ${item.rule}`, width));
@@ -558,13 +631,14 @@ function renderCompactLegend(width) {
558
631
  }
559
632
 
560
633
  function legendColumnWidths(width) {
561
- const available = Math.max(40, width - 13);
634
+ const available = Math.max(40, width - 16);
562
635
  const tableWidth = 7;
563
- const columnWidth = Math.min(25, Math.max(18, Math.floor(available * 0.28)));
564
- const remaining = Math.max(36, available - tableWidth - columnWidth);
565
- const definitionWidth = Math.max(18, Math.floor(remaining * 0.52));
566
- const ruleWidth = Math.max(18, remaining - definitionWidth);
567
- return [tableWidth, columnWidth, definitionWidth, ruleWidth];
636
+ const columnWidth = Math.min(25, Math.max(18, Math.floor(available * 0.22)));
637
+ const sourceWidth = 11;
638
+ const remaining = Math.max(32, available - tableWidth - columnWidth - sourceWidth);
639
+ const definitionWidth = Math.max(18, Math.floor(remaining * 0.58));
640
+ const ruleWidth = Math.max(14, remaining - definitionWidth);
641
+ return [tableWidth, columnWidth, sourceWidth, definitionWidth, ruleWidth];
568
642
  }
569
643
 
570
644
  function wrapText(value, width) {