fm-bench 0.4.1 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -51,6 +51,7 @@ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skip
51
51
  ```sh
52
52
  fm-bench [run] [options]
53
53
  fm-bench models [options]
54
+ fm-bench legend [options]
54
55
  fm-bench doctor [options]
55
56
  ```
56
57
 
@@ -58,6 +59,8 @@ fm-bench doctor [options]
58
59
 
59
60
  `models` lists discovered models, availability, descriptions, and quota output.
60
61
 
62
+ `legend` explains every terminal table column, compact-card field, model-list column, and color rule. It does not run `fm`.
63
+
61
64
  `doctor` checks Node, macOS, `fm`, and model availability.
62
65
 
63
66
  ## Benchmark Options
@@ -72,6 +75,8 @@ fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
72
75
  fm-bench --prompt "Reply with exactly: ok" --runs 5
73
76
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
74
77
  fm-bench --format csv --out reports/bench.csv
78
+ fm-bench legend
79
+ fm-bench legend --json
75
80
  ```
76
81
 
77
82
  Useful flags:
@@ -146,6 +151,16 @@ Measured runs stream by default so `fm-bench` can capture TTFT and streaming smo
146
151
 
147
152
  Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
148
153
 
154
+ ## Table Legend
155
+
156
+ Benchmark output stays focused on results and does not print the metric legend footer. Use `fm-bench legend` when you want definitions for every table column and color rule:
157
+
158
+ ```sh
159
+ fm-bench legend
160
+ fm-bench legend --json
161
+ fm-bench legend --csv
162
+ ```
163
+
149
164
  ## Live Progress
150
165
 
151
166
  Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
@@ -16,6 +16,8 @@ The metric set follows common LLM inference benchmark practice:
16
16
 
17
17
  ## Metrics
18
18
 
19
+ For a column-by-column terminal reference, run `fm-bench legend`.
20
+
19
21
  - `TTFT`: time from starting `fm respond` to the first streamed stdout chunk. This is a practical terminal-side proxy for time to first token.
20
22
  - `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
21
23
  - `generation_ms`: `E2E - TTFT`.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.4.1",
3
+ "version": "0.4.2",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/cli.js CHANGED
@@ -4,7 +4,7 @@ import { inspectModels, runBenchmark } from './bench.js';
4
4
  import { runProcess } from './process.js';
5
5
  import { createProgress } from './progress.js';
6
6
  import { flattenResults, toCsv, writeReport } from './report.js';
7
- import { renderBenchmarkReport, renderModelsTable } from './table.js';
7
+ import { legendEntries, renderBenchmarkReport, renderLegend, renderModelsTable } from './table.js';
8
8
 
9
9
  const require = createRequire(import.meta.url);
10
10
  const packageJson = require('../package.json');
@@ -22,6 +22,17 @@ export async function runCli(argv = process.argv.slice(2)) {
22
22
  return;
23
23
  }
24
24
 
25
+ if (parsed.command === 'legend') {
26
+ if (parsed.format === 'json') {
27
+ console.log(JSON.stringify(legendEntries(), null, 2));
28
+ } else if (parsed.format === 'csv') {
29
+ console.log(toCsv(legendEntries()));
30
+ } else {
31
+ console.log(renderLegend(renderOptions(parsed)));
32
+ }
33
+ return;
34
+ }
35
+
25
36
  if (parsed.command === 'doctor') {
26
37
  await runDoctor(parsed);
27
38
  return;
@@ -105,10 +116,14 @@ export function parseArgs(argv) {
105
116
  };
106
117
 
107
118
  const args = [...argv];
108
- if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'help'].includes(args[0])) {
119
+ if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'help'].includes(args[0])) {
109
120
  options.command = args.shift();
110
121
  }
111
122
 
123
+ if (options.command === 'metrics') {
124
+ options.command = 'legend';
125
+ }
126
+
112
127
  if (options.command === 'help') {
113
128
  options.help = true;
114
129
  return options;
@@ -359,8 +374,15 @@ Dynamic benchmark CLI for Apple's fm command on macOS 27+.
359
374
  Usage:
360
375
  fm-bench [run] [options]
361
376
  fm-bench models [options]
377
+ fm-bench legend [options]
362
378
  fm-bench doctor [options]
363
379
 
380
+ Commands:
381
+ run Benchmark discovered or selected fm models
382
+ models List discovered models and availability
383
+ legend Explain every terminal table column and color rule
384
+ doctor Check Node, macOS, fm, and model availability
385
+
364
386
  Run options:
365
387
  -m, --models <list> Models to benchmark, comma-separated or repeated
366
388
  -r, --runs <n> Runs per prompt/model (default: 1)
@@ -412,6 +434,7 @@ Examples:
412
434
  fm-bench --models system,pcc --runs 3 --profile stress
413
435
  fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
414
436
  fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
437
+ fm-bench legend
415
438
  fm-bench models
416
439
  fm-bench doctor
417
440
  `;
package/src/table.js CHANGED
@@ -56,12 +56,71 @@ export function renderBenchmarkReport(payload, options = {}) {
56
56
  lines.push(renderDetailTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
57
57
  }
58
58
 
59
- lines.push('');
60
- lines.push(compactLegend(width));
61
-
62
59
  return lines.join('\n');
63
60
  }
64
61
 
62
+ export function legendEntries() {
63
+ return [
64
+ entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.'),
65
+ entry('summary', 'MODEL', 'fm model name, such as system or pcc.', ''),
66
+ entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable.'),
67
+ entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.'),
68
+ entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.'),
69
+ entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Green 100%, yellow >=80%, red <80%.'),
70
+ entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.'),
71
+ entry('summary', 'TTFT', 'Time to first streamed output chunk, p50.', 'Lower is better. Uses SLO threshold when set; otherwise relative ranking.'),
72
+ entry('summary', 'TTFT P95', '95th percentile time to first streamed output chunk.', 'Lower is better.'),
73
+ entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better.'),
74
+ entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.'),
75
+ entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Lower is better. Requires streaming and token counts.'),
76
+ entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.'),
77
+ entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Higher is better; relative ranking.'),
78
+ entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better; relative ranking.'),
79
+ entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Higher is better; relative ranking.'),
80
+ entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.'),
81
+ entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', ''),
82
+ entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm token-count.', ''),
83
+ entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm token-count.', ''),
84
+ entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Higher is better; estimates prompt-processing speed for streaming runs.'),
85
+ entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Higher is better; requires streaming and token counts.'),
86
+ entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.'),
87
+ entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Lower is smoother streaming.'),
88
+ entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.'),
89
+ entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', 'Narrower usually means a steadier estimate. Treat small samples carefully.'),
90
+ entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.'),
91
+ entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', ''),
92
+ entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', ''),
93
+ entry('models', 'QUOTA', 'Raw fm quota-usage output or unavailable reason.', 'Mostly relevant for Private Cloud Compute.'),
94
+ entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.'),
95
+ entry('colors', 'GREEN', 'Passing, steadier, faster, or better within this benchmark context.', ''),
96
+ entry('colors', 'YELLOW', 'Marginal, partial, near a budget, or middle-ranked within this benchmark context.', ''),
97
+ entry('colors', 'RED', 'Failed budget, unstable, slower, lower, or worse within this benchmark context.', ''),
98
+ entry('colors', 'MUTED', 'Unavailable model, skipped value, or descriptive text.', '')
99
+ ];
100
+ }
101
+
102
+ export function renderLegend(options = {}) {
103
+ const width = terminalWidth(options);
104
+ const rows = legendEntries().map((item) => [
105
+ item.table,
106
+ item.column,
107
+ item.definition,
108
+ item.rule || '-'
109
+ ]);
110
+
111
+ if (options.compact || width < 88) {
112
+ return legendEntries().map((item) => {
113
+ const suffix = item.rule ? ` Rule: ${item.rule}` : '';
114
+ return truncate(`${item.table.toUpperCase()} ${item.column}: ${item.definition}${suffix}`, width);
115
+ }).join('\n');
116
+ }
117
+
118
+ return renderTable(['table', 'column', 'definition', 'rule'], rows, {
119
+ ...options,
120
+ maxCellWidth: Math.max(20, Math.floor((width - 48) / 2))
121
+ });
122
+ }
123
+
65
124
  export function formatMs(value) {
66
125
  if (value == null || !Number.isFinite(value)) return '-';
67
126
  if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
@@ -250,13 +309,6 @@ function terminalWidth(options = {}) {
250
309
  return options.width || process.stdout.columns || 120;
251
310
  }
252
311
 
253
- function compactLegend(width) {
254
- return [
255
- 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence.',
256
- 'CV = latency variation; lower is steadier: green <=10%, yellow <=25%, red >25%.'
257
- ].map((line) => truncate(line, width)).join('\n');
258
- }
259
-
260
312
  function formatSlo(slo = {}) {
261
313
  const parts = [];
262
314
  if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
@@ -265,6 +317,15 @@ function formatSlo(slo = {}) {
265
317
  return parts.length ? `SLO ${parts.join(',')}` : '';
266
318
  }
267
319
 
320
+ function entry(table, column, definition, rule) {
321
+ return {
322
+ table,
323
+ column,
324
+ definition,
325
+ rule
326
+ };
327
+ }
328
+
268
329
  const ASCII_TABLE = {
269
330
  topLeft: '+',
270
331
  topJoin: '+',