fm-bench 0.4.0 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -1
- package/docs/methodology.md +2 -0
- package/package.json +1 -1
- package/src/cli.js +25 -2
- package/src/table.js +71 -8
package/README.md
CHANGED
|
@@ -51,6 +51,7 @@ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skip
|
|
|
51
51
|
```sh
|
|
52
52
|
fm-bench [run] [options]
|
|
53
53
|
fm-bench models [options]
|
|
54
|
+
fm-bench legend [options]
|
|
54
55
|
fm-bench doctor [options]
|
|
55
56
|
```
|
|
56
57
|
|
|
@@ -58,6 +59,8 @@ fm-bench doctor [options]
|
|
|
58
59
|
|
|
59
60
|
`models` lists discovered models, availability, descriptions, and quota output.
|
|
60
61
|
|
|
62
|
+
`legend` explains every terminal table column, compact-card field, model-list column, and color rule. It does not run `fm`.
|
|
63
|
+
|
|
61
64
|
`doctor` checks Node, macOS, `fm`, and model availability.
|
|
62
65
|
|
|
63
66
|
## Benchmark Options
|
|
@@ -72,6 +75,8 @@ fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
|
72
75
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
73
76
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
74
77
|
fm-bench --format csv --out reports/bench.csv
|
|
78
|
+
fm-bench legend
|
|
79
|
+
fm-bench legend --json
|
|
75
80
|
```
|
|
76
81
|
|
|
77
82
|
Useful flags:
|
|
@@ -146,6 +151,16 @@ Measured runs stream by default so `fm-bench` can capture TTFT and streaming smo
|
|
|
146
151
|
|
|
147
152
|
Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
|
|
148
153
|
|
|
154
|
+
## Table Legend
|
|
155
|
+
|
|
156
|
+
Benchmark output stays focused on results and does not print the metric legend footer. Use `fm-bench legend` when you want definitions for every table column and color rule:
|
|
157
|
+
|
|
158
|
+
```sh
|
|
159
|
+
fm-bench legend
|
|
160
|
+
fm-bench legend --json
|
|
161
|
+
fm-bench legend --csv
|
|
162
|
+
```
|
|
163
|
+
|
|
149
164
|
## Live Progress
|
|
150
165
|
|
|
151
166
|
Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
|
|
@@ -160,7 +175,7 @@ Table output uses semantic ANSI color on interactive terminals:
|
|
|
160
175
|
- yellow: marginal, partial, or near a budget.
|
|
161
176
|
- red: failing a budget, unstable, or slower/lower than peers.
|
|
162
177
|
|
|
163
|
-
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
178
|
+
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. CV is green at `<=10%`, yellow at `<=25%`, and red above `25%` because higher CV means less steady latency. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
164
179
|
|
|
165
180
|
Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
|
|
166
181
|
|
package/docs/methodology.md
CHANGED
|
@@ -16,6 +16,8 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
16
16
|
|
|
17
17
|
## Metrics
|
|
18
18
|
|
|
19
|
+
For a column-by-column terminal reference, run `fm-bench legend`.
|
|
20
|
+
|
|
19
21
|
- `TTFT`: time from starting `fm respond` to the first streamed stdout chunk. This is a practical terminal-side proxy for time to first token.
|
|
20
22
|
- `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
|
|
21
23
|
- `generation_ms`: `E2E - TTFT`.
|
package/package.json
CHANGED
package/src/cli.js
CHANGED
|
@@ -4,7 +4,7 @@ import { inspectModels, runBenchmark } from './bench.js';
|
|
|
4
4
|
import { runProcess } from './process.js';
|
|
5
5
|
import { createProgress } from './progress.js';
|
|
6
6
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
7
|
-
import { renderBenchmarkReport, renderModelsTable } from './table.js';
|
|
7
|
+
import { legendEntries, renderBenchmarkReport, renderLegend, renderModelsTable } from './table.js';
|
|
8
8
|
|
|
9
9
|
const require = createRequire(import.meta.url);
|
|
10
10
|
const packageJson = require('../package.json');
|
|
@@ -22,6 +22,17 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
22
22
|
return;
|
|
23
23
|
}
|
|
24
24
|
|
|
25
|
+
if (parsed.command === 'legend') {
|
|
26
|
+
if (parsed.format === 'json') {
|
|
27
|
+
console.log(JSON.stringify(legendEntries(), null, 2));
|
|
28
|
+
} else if (parsed.format === 'csv') {
|
|
29
|
+
console.log(toCsv(legendEntries()));
|
|
30
|
+
} else {
|
|
31
|
+
console.log(renderLegend(renderOptions(parsed)));
|
|
32
|
+
}
|
|
33
|
+
return;
|
|
34
|
+
}
|
|
35
|
+
|
|
25
36
|
if (parsed.command === 'doctor') {
|
|
26
37
|
await runDoctor(parsed);
|
|
27
38
|
return;
|
|
@@ -105,10 +116,14 @@ export function parseArgs(argv) {
|
|
|
105
116
|
};
|
|
106
117
|
|
|
107
118
|
const args = [...argv];
|
|
108
|
-
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'help'].includes(args[0])) {
|
|
119
|
+
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'help'].includes(args[0])) {
|
|
109
120
|
options.command = args.shift();
|
|
110
121
|
}
|
|
111
122
|
|
|
123
|
+
if (options.command === 'metrics') {
|
|
124
|
+
options.command = 'legend';
|
|
125
|
+
}
|
|
126
|
+
|
|
112
127
|
if (options.command === 'help') {
|
|
113
128
|
options.help = true;
|
|
114
129
|
return options;
|
|
@@ -359,8 +374,15 @@ Dynamic benchmark CLI for Apple's fm command on macOS 27+.
|
|
|
359
374
|
Usage:
|
|
360
375
|
fm-bench [run] [options]
|
|
361
376
|
fm-bench models [options]
|
|
377
|
+
fm-bench legend [options]
|
|
362
378
|
fm-bench doctor [options]
|
|
363
379
|
|
|
380
|
+
Commands:
|
|
381
|
+
run Benchmark discovered or selected fm models
|
|
382
|
+
models List discovered models and availability
|
|
383
|
+
legend Explain every terminal table column and color rule
|
|
384
|
+
doctor Check Node, macOS, fm, and model availability
|
|
385
|
+
|
|
364
386
|
Run options:
|
|
365
387
|
-m, --models <list> Models to benchmark, comma-separated or repeated
|
|
366
388
|
-r, --runs <n> Runs per prompt/model (default: 1)
|
|
@@ -412,6 +434,7 @@ Examples:
|
|
|
412
434
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
413
435
|
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
414
436
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
437
|
+
fm-bench legend
|
|
415
438
|
fm-bench models
|
|
416
439
|
fm-bench doctor
|
|
417
440
|
`;
|
package/src/table.js
CHANGED
|
@@ -56,12 +56,71 @@ export function renderBenchmarkReport(payload, options = {}) {
|
|
|
56
56
|
lines.push(renderDetailTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
-
lines.push('');
|
|
60
|
-
lines.push(compactLegend(width));
|
|
61
|
-
|
|
62
59
|
return lines.join('\n');
|
|
63
60
|
}
|
|
64
61
|
|
|
62
|
+
export function legendEntries() {
|
|
63
|
+
return [
|
|
64
|
+
entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.'),
|
|
65
|
+
entry('summary', 'MODEL', 'fm model name, such as system or pcc.', ''),
|
|
66
|
+
entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable.'),
|
|
67
|
+
entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.'),
|
|
68
|
+
entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.'),
|
|
69
|
+
entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Green 100%, yellow >=80%, red <80%.'),
|
|
70
|
+
entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.'),
|
|
71
|
+
entry('summary', 'TTFT', 'Time to first streamed output chunk, p50.', 'Lower is better. Uses SLO threshold when set; otherwise relative ranking.'),
|
|
72
|
+
entry('summary', 'TTFT P95', '95th percentile time to first streamed output chunk.', 'Lower is better.'),
|
|
73
|
+
entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better.'),
|
|
74
|
+
entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.'),
|
|
75
|
+
entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Lower is better. Requires streaming and token counts.'),
|
|
76
|
+
entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.'),
|
|
77
|
+
entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Higher is better; relative ranking.'),
|
|
78
|
+
entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better; relative ranking.'),
|
|
79
|
+
entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Higher is better; relative ranking.'),
|
|
80
|
+
entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.'),
|
|
81
|
+
entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', ''),
|
|
82
|
+
entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm token-count.', ''),
|
|
83
|
+
entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm token-count.', ''),
|
|
84
|
+
entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Higher is better; estimates prompt-processing speed for streaming runs.'),
|
|
85
|
+
entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Higher is better; requires streaming and token counts.'),
|
|
86
|
+
entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.'),
|
|
87
|
+
entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Lower is smoother streaming.'),
|
|
88
|
+
entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.'),
|
|
89
|
+
entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', 'Narrower usually means a steadier estimate. Treat small samples carefully.'),
|
|
90
|
+
entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.'),
|
|
91
|
+
entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', ''),
|
|
92
|
+
entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', ''),
|
|
93
|
+
entry('models', 'QUOTA', 'Raw fm quota-usage output or unavailable reason.', 'Mostly relevant for Private Cloud Compute.'),
|
|
94
|
+
entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.'),
|
|
95
|
+
entry('colors', 'GREEN', 'Passing, steadier, faster, or better within this benchmark context.', ''),
|
|
96
|
+
entry('colors', 'YELLOW', 'Marginal, partial, near a budget, or middle-ranked within this benchmark context.', ''),
|
|
97
|
+
entry('colors', 'RED', 'Failed budget, unstable, slower, lower, or worse within this benchmark context.', ''),
|
|
98
|
+
entry('colors', 'MUTED', 'Unavailable model, skipped value, or descriptive text.', '')
|
|
99
|
+
];
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function renderLegend(options = {}) {
|
|
103
|
+
const width = terminalWidth(options);
|
|
104
|
+
const rows = legendEntries().map((item) => [
|
|
105
|
+
item.table,
|
|
106
|
+
item.column,
|
|
107
|
+
item.definition,
|
|
108
|
+
item.rule || '-'
|
|
109
|
+
]);
|
|
110
|
+
|
|
111
|
+
if (options.compact || width < 88) {
|
|
112
|
+
return legendEntries().map((item) => {
|
|
113
|
+
const suffix = item.rule ? ` Rule: ${item.rule}` : '';
|
|
114
|
+
return truncate(`${item.table.toUpperCase()} ${item.column}: ${item.definition}${suffix}`, width);
|
|
115
|
+
}).join('\n');
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return renderTable(['table', 'column', 'definition', 'rule'], rows, {
|
|
119
|
+
...options,
|
|
120
|
+
maxCellWidth: Math.max(20, Math.floor((width - 48) / 2))
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
|
|
65
124
|
export function formatMs(value) {
|
|
66
125
|
if (value == null || !Number.isFinite(value)) return '-';
|
|
67
126
|
if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
|
|
@@ -250,11 +309,6 @@ function terminalWidth(options = {}) {
|
|
|
250
309
|
return options.width || process.stdout.columns || 120;
|
|
251
310
|
}
|
|
252
311
|
|
|
253
|
-
function compactLegend(width) {
|
|
254
|
-
const text = 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence. CV = lower is steadier.';
|
|
255
|
-
return truncate(text, width);
|
|
256
|
-
}
|
|
257
|
-
|
|
258
312
|
function formatSlo(slo = {}) {
|
|
259
313
|
const parts = [];
|
|
260
314
|
if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
|
|
@@ -263,6 +317,15 @@ function formatSlo(slo = {}) {
|
|
|
263
317
|
return parts.length ? `SLO ${parts.join(',')}` : '';
|
|
264
318
|
}
|
|
265
319
|
|
|
320
|
+
function entry(table, column, definition, rule) {
|
|
321
|
+
return {
|
|
322
|
+
table,
|
|
323
|
+
column,
|
|
324
|
+
definition,
|
|
325
|
+
rule
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
|
|
266
329
|
const ASCII_TABLE = {
|
|
267
330
|
topLeft: '+',
|
|
268
331
|
topJoin: '+',
|