fm-bench 0.2.0 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +23 -0
- package/docs/methodology.md +13 -1
- package/package.json +1 -1
- package/src/bench.js +94 -16
- package/src/cli.js +73 -3
- package/src/report.js +2 -0
- package/src/stats.js +83 -13
- package/src/table.js +334 -68
package/README.md
CHANGED
|
@@ -71,6 +71,8 @@ fm-bench doctor [options]
|
|
|
71
71
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
72
72
|
fm-bench --models system --runs 5 --profile interactive
|
|
73
73
|
fm-bench --models system --runs 3 --profile throughput --warmup 1
|
|
74
|
+
fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
|
|
75
|
+
fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
74
76
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
75
77
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
76
78
|
fm-bench --format csv --out reports/bench.csv
|
|
@@ -82,7 +84,9 @@ Useful flags:
|
|
|
82
84
|
- `--runs <n>`: measured runs per prompt/model.
|
|
83
85
|
- `--warmup <n>`: warmup runs per model before measurement.
|
|
84
86
|
- `--concurrency <n>`: parallel `fm` processes.
|
|
87
|
+
- `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
|
|
85
88
|
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
89
|
+
- `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
|
|
86
90
|
- `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
|
|
87
91
|
- `--prompt <text>`: custom prompt, repeatable.
|
|
88
92
|
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
@@ -91,6 +95,9 @@ Useful flags:
|
|
|
91
95
|
- `--capture-output`: include raw model output in JSON reports.
|
|
92
96
|
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
93
97
|
- `--ascii`: use plain ASCII table borders.
|
|
98
|
+
- `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
|
|
99
|
+
- `--compact`: force the narrow terminal layout.
|
|
100
|
+
- `--width <n>`: render as if the terminal has `n` columns.
|
|
94
101
|
- `--out <file>`: save a report.
|
|
95
102
|
|
|
96
103
|
## Prompt Files
|
|
@@ -123,6 +130,8 @@ Plain text files are split on blank lines.
|
|
|
123
130
|
- output tokens per second per request.
|
|
124
131
|
- total output token throughput across the measured window.
|
|
125
132
|
- requests per second across the measured window.
|
|
133
|
+
- goodput percentage and goodput RPS when SLO flags are set.
|
|
134
|
+
- coefficient of variation (CV) and confidence interval context for stability.
|
|
126
135
|
- prompt and output token counts.
|
|
127
136
|
- p50, p95, and p99 tail latency views.
|
|
128
137
|
- repeatability across repeated runs of the same prompt.
|
|
@@ -133,6 +142,20 @@ Token counts come from `fm token-count --quiet`. If `fm` cannot count a response
|
|
|
133
142
|
|
|
134
143
|
Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
|
|
135
144
|
|
|
145
|
+
Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
|
|
146
|
+
|
|
147
|
+
## Terminal Colors
|
|
148
|
+
|
|
149
|
+
Table output uses semantic ANSI color on interactive terminals:
|
|
150
|
+
|
|
151
|
+
- green: passing, steadier, or better than the current comparison set.
|
|
152
|
+
- yellow: marginal, partial, or near a budget.
|
|
153
|
+
- red: failing a budget, unstable, or slower/lower than peers.
|
|
154
|
+
|
|
155
|
+
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
156
|
+
|
|
157
|
+
Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
|
|
158
|
+
|
|
136
159
|
See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
|
|
137
160
|
|
|
138
161
|
## Requirements
|
package/docs/methodology.md
CHANGED
|
@@ -9,8 +9,9 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
9
9
|
- Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
|
|
10
10
|
- NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
|
|
11
11
|
- NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
|
|
12
|
-
- vLLM benchmark tooling reports
|
|
12
|
+
- vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
|
|
13
13
|
- MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
|
|
14
|
+
- MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
|
|
14
15
|
|
|
15
16
|
## Metrics
|
|
16
17
|
|
|
@@ -22,10 +23,21 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
22
23
|
- `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
|
|
23
24
|
- `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
|
|
24
25
|
- `RPS`: successful requests divided by that model's measured wall-clock window.
|
|
26
|
+
- `goodput`: successful requests that also satisfy all provided SLO thresholds.
|
|
25
27
|
- `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
|
|
28
|
+
- `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
|
|
29
|
+
- `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
|
|
30
|
+
|
|
31
|
+
## Operating Points
|
|
32
|
+
|
|
33
|
+
Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
|
|
34
|
+
|
|
35
|
+
`fm-bench` does not interpolate between operating points. It reports only what was actually measured.
|
|
26
36
|
|
|
27
37
|
## Caveats
|
|
28
38
|
|
|
29
39
|
`fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
|
|
30
40
|
|
|
31
41
|
Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
|
|
42
|
+
|
|
43
|
+
For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
|
package/package.json
CHANGED
package/src/bench.js
CHANGED
|
@@ -44,15 +44,69 @@ export async function runBenchmark(options = {}) {
|
|
|
44
44
|
const runnableModels = modelStatuses.filter((model) => model.available);
|
|
45
45
|
const environment = await collectEnvironment(inspection.fmBin);
|
|
46
46
|
const promptTokenCounts = new Map();
|
|
47
|
+
const concurrencies = normalizeConcurrencySweep(options);
|
|
47
48
|
|
|
48
49
|
for (const prompt of prompts) {
|
|
49
50
|
const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
|
|
50
51
|
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
51
52
|
}
|
|
52
53
|
|
|
54
|
+
const results = [];
|
|
55
|
+
const scenarios = [];
|
|
56
|
+
for (const concurrency of concurrencies) {
|
|
57
|
+
const scenario = await runScenario({
|
|
58
|
+
fmBin: inspection.fmBin,
|
|
59
|
+
prompts,
|
|
60
|
+
runnableModels,
|
|
61
|
+
modelStatuses,
|
|
62
|
+
promptTokenCounts,
|
|
63
|
+
options,
|
|
64
|
+
concurrency
|
|
65
|
+
});
|
|
66
|
+
scenarios.push(scenario);
|
|
67
|
+
results.push(...scenario.results);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
71
|
+
|| (a.concurrency ?? 0) - (b.concurrency ?? 0)
|
|
72
|
+
|| a.promptId.localeCompare(b.promptId)
|
|
73
|
+
|| a.run - b.run);
|
|
74
|
+
|
|
75
|
+
const summary = summarizeByModel(results, modelStatuses, { concurrencies });
|
|
76
|
+
return {
|
|
77
|
+
tool: 'fm-bench',
|
|
78
|
+
version: options.version,
|
|
79
|
+
startedAt,
|
|
80
|
+
finishedAt: new Date().toISOString(),
|
|
81
|
+
options: publicOptions(options),
|
|
82
|
+
environment,
|
|
83
|
+
prompts: prompts.map((prompt) => ({
|
|
84
|
+
id: prompt.id,
|
|
85
|
+
prompt: prompt.prompt,
|
|
86
|
+
promptTokens: promptTokenCounts.get(prompt.id)
|
|
87
|
+
})),
|
|
88
|
+
models: modelStatuses,
|
|
89
|
+
scenarios,
|
|
90
|
+
summary,
|
|
91
|
+
results
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
async function runScenario(context) {
|
|
96
|
+
const {
|
|
97
|
+
fmBin,
|
|
98
|
+
prompts,
|
|
99
|
+
runnableModels,
|
|
100
|
+
modelStatuses,
|
|
101
|
+
promptTokenCounts,
|
|
102
|
+
options,
|
|
103
|
+
concurrency
|
|
104
|
+
} = context;
|
|
105
|
+
const startedAt = new Date().toISOString();
|
|
106
|
+
|
|
53
107
|
for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
|
|
54
108
|
for (const model of runnableModels) {
|
|
55
|
-
await respond(
|
|
109
|
+
await respond(fmBin, model.name, prompts[0].prompt, {
|
|
56
110
|
...options,
|
|
57
111
|
stream: false
|
|
58
112
|
});
|
|
@@ -64,14 +118,14 @@ export async function runBenchmark(options = {}) {
|
|
|
64
118
|
for (const model of runnableModels) {
|
|
65
119
|
for (const prompt of prompts) {
|
|
66
120
|
for (let run = 1; run <= options.runs; run += 1) {
|
|
67
|
-
jobs.push({ model, prompt, run });
|
|
121
|
+
jobs.push({ model, prompt, run, concurrency });
|
|
68
122
|
}
|
|
69
123
|
}
|
|
70
124
|
}
|
|
71
125
|
|
|
72
126
|
const results = [];
|
|
73
|
-
await runLimited(jobs,
|
|
74
|
-
const result = await runSingleBenchmark(
|
|
127
|
+
await runLimited(jobs, concurrency, async (job) => {
|
|
128
|
+
const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
|
|
75
129
|
results.push(result);
|
|
76
130
|
if (!result.ok && options.failFast) {
|
|
77
131
|
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
@@ -84,21 +138,11 @@ export async function runBenchmark(options = {}) {
|
|
|
84
138
|
|| a.promptId.localeCompare(b.promptId)
|
|
85
139
|
|| a.run - b.run);
|
|
86
140
|
|
|
87
|
-
const summary = summarizeByModel(results, modelStatuses);
|
|
88
141
|
return {
|
|
89
|
-
|
|
90
|
-
version: options.version,
|
|
142
|
+
concurrency,
|
|
91
143
|
startedAt,
|
|
92
144
|
finishedAt: new Date().toISOString(),
|
|
93
|
-
|
|
94
|
-
environment,
|
|
95
|
-
prompts: prompts.map((prompt) => ({
|
|
96
|
-
id: prompt.id,
|
|
97
|
-
prompt: prompt.prompt,
|
|
98
|
-
promptTokens: promptTokenCounts.get(prompt.id)
|
|
99
|
-
})),
|
|
100
|
-
models: modelStatuses,
|
|
101
|
-
summary,
|
|
145
|
+
summary: summarizeByModel(results, modelStatuses, { concurrencies: [concurrency] }),
|
|
102
146
|
results
|
|
103
147
|
};
|
|
104
148
|
}
|
|
@@ -130,6 +174,7 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
130
174
|
|
|
131
175
|
return {
|
|
132
176
|
model: job.model.name,
|
|
177
|
+
concurrency: job.concurrency,
|
|
133
178
|
promptId: job.prompt.id,
|
|
134
179
|
run: job.run,
|
|
135
180
|
ok: response.ok,
|
|
@@ -149,6 +194,11 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
149
194
|
streamed: response.streamed,
|
|
150
195
|
stdoutChunks: response.stdoutChunks,
|
|
151
196
|
outputHash: response.ok ? hashOutput(response.output) : null,
|
|
197
|
+
good: response.ok ? evaluateSlo({
|
|
198
|
+
firstTokenMs,
|
|
199
|
+
durationMs: response.durationMs,
|
|
200
|
+
tpotMs
|
|
201
|
+
}, options) : false,
|
|
152
202
|
output: options.captureOutput ? response.output : undefined,
|
|
153
203
|
error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
|
|
154
204
|
};
|
|
@@ -175,23 +225,51 @@ function normalizeModelSelection(models) {
|
|
|
175
225
|
}
|
|
176
226
|
|
|
177
227
|
function publicOptions(options) {
|
|
228
|
+
const concurrencies = normalizeConcurrencySweep(options);
|
|
178
229
|
return {
|
|
179
230
|
models: normalizeModelSelection(options.models),
|
|
180
231
|
runs: options.runs,
|
|
181
232
|
warmup: options.warmup,
|
|
182
233
|
concurrency: options.concurrency,
|
|
234
|
+
sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
|
|
183
235
|
timeoutMs: options.timeoutMs,
|
|
184
236
|
profile: options.profile,
|
|
185
237
|
promptCount: options.promptCount,
|
|
186
238
|
greedy: options.greedy,
|
|
187
239
|
stream: options.stream,
|
|
240
|
+
slo: {
|
|
241
|
+
ttftMs: options.sloTtftMs || null,
|
|
242
|
+
e2eMs: options.sloE2eMs || null,
|
|
243
|
+
tpotMs: options.sloTpotMs || null
|
|
244
|
+
},
|
|
188
245
|
instructions: options.instructions ? '[set]' : ''
|
|
189
246
|
};
|
|
190
247
|
}
|
|
191
248
|
|
|
249
|
+
function normalizeConcurrencySweep(options) {
|
|
250
|
+
if (options.sweepConcurrency?.length) {
|
|
251
|
+
return [...new Set(options.sweepConcurrency)]
|
|
252
|
+
.filter((value) => Number.isInteger(value) && value > 0)
|
|
253
|
+
.sort((a, b) => a - b);
|
|
254
|
+
}
|
|
255
|
+
return [Math.max(1, options.concurrency || 1)];
|
|
256
|
+
}
|
|
257
|
+
|
|
192
258
|
function hashOutput(output) {
|
|
193
259
|
return crypto.createHash('sha256')
|
|
194
260
|
.update(output.replace(/\s+/g, ' ').trim())
|
|
195
261
|
.digest('hex')
|
|
196
262
|
.slice(0, 16);
|
|
197
263
|
}
|
|
264
|
+
|
|
265
|
+
function evaluateSlo(metrics, options) {
|
|
266
|
+
const thresholds = [
|
|
267
|
+
['firstTokenMs', options.sloTtftMs],
|
|
268
|
+
['durationMs', options.sloE2eMs],
|
|
269
|
+
['tpotMs', options.sloTpotMs]
|
|
270
|
+
].filter(([, threshold]) => Number.isFinite(threshold));
|
|
271
|
+
|
|
272
|
+
if (thresholds.length === 0) return null;
|
|
273
|
+
|
|
274
|
+
return thresholds.every(([field, threshold]) => metrics[field] != null && metrics[field] <= threshold);
|
|
275
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -31,7 +31,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
31
31
|
if (parsed.format === 'json') {
|
|
32
32
|
console.log(JSON.stringify(inspection.models, null, 2));
|
|
33
33
|
} else {
|
|
34
|
-
console.log(renderModelsTable(inspection.models,
|
|
34
|
+
console.log(renderModelsTable(inspection.models, renderOptions(parsed)));
|
|
35
35
|
}
|
|
36
36
|
return;
|
|
37
37
|
}
|
|
@@ -46,7 +46,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
46
46
|
} else if (parsed.format === 'csv') {
|
|
47
47
|
console.log(toCsv(flattenResults(payload.results)));
|
|
48
48
|
} else {
|
|
49
|
-
console.log(renderBenchmarkReport(payload,
|
|
49
|
+
console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
|
|
50
50
|
if (parsed.verbose) {
|
|
51
51
|
console.log();
|
|
52
52
|
console.log(toCsv(flattenResults(payload.results)));
|
|
@@ -70,16 +70,23 @@ export function parseArgs(argv) {
|
|
|
70
70
|
runs: 1,
|
|
71
71
|
warmup: 0,
|
|
72
72
|
concurrency: 1,
|
|
73
|
+
sweepConcurrency: [],
|
|
73
74
|
timeoutMs: 60_000,
|
|
74
75
|
profile: 'standard',
|
|
75
76
|
greedy: true,
|
|
76
77
|
stream: true,
|
|
78
|
+
sloTtftMs: null,
|
|
79
|
+
sloE2eMs: null,
|
|
80
|
+
sloTpotMs: null,
|
|
77
81
|
format: 'table',
|
|
78
82
|
captureOutput: false,
|
|
79
83
|
availableOnly: false,
|
|
80
84
|
failFast: false,
|
|
81
85
|
verbose: false,
|
|
82
|
-
ascii: false
|
|
86
|
+
ascii: false,
|
|
87
|
+
color: 'auto',
|
|
88
|
+
compact: false,
|
|
89
|
+
width: null
|
|
83
90
|
};
|
|
84
91
|
|
|
85
92
|
const args = [...argv];
|
|
@@ -118,10 +125,25 @@ export function parseArgs(argv) {
|
|
|
118
125
|
case '--concurrency':
|
|
119
126
|
options.concurrency = parsePositiveInt(requireValue(arg, args), arg);
|
|
120
127
|
break;
|
|
128
|
+
case '--sweep-concurrency':
|
|
129
|
+
options.sweepConcurrency = parsePositiveIntList(requireValue(arg, args), arg);
|
|
130
|
+
if (options.sweepConcurrency.length > 0) {
|
|
131
|
+
options.concurrency = options.sweepConcurrency[0];
|
|
132
|
+
}
|
|
133
|
+
break;
|
|
121
134
|
case '--timeout':
|
|
122
135
|
case '--timeout-ms':
|
|
123
136
|
options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
124
137
|
break;
|
|
138
|
+
case '--slo-ttft-ms':
|
|
139
|
+
options.sloTtftMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
140
|
+
break;
|
|
141
|
+
case '--slo-e2e-ms':
|
|
142
|
+
options.sloE2eMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
143
|
+
break;
|
|
144
|
+
case '--slo-tpot-ms':
|
|
145
|
+
options.sloTpotMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
146
|
+
break;
|
|
125
147
|
case '-p':
|
|
126
148
|
case '--prompt':
|
|
127
149
|
options.prompts.push(requireValue(arg, args));
|
|
@@ -172,6 +194,18 @@ export function parseArgs(argv) {
|
|
|
172
194
|
case '--ascii':
|
|
173
195
|
options.ascii = true;
|
|
174
196
|
break;
|
|
197
|
+
case '--color':
|
|
198
|
+
options.color = 'always';
|
|
199
|
+
break;
|
|
200
|
+
case '--no-color':
|
|
201
|
+
options.color = 'never';
|
|
202
|
+
break;
|
|
203
|
+
case '--compact':
|
|
204
|
+
options.compact = true;
|
|
205
|
+
break;
|
|
206
|
+
case '--width':
|
|
207
|
+
options.width = parsePositiveInt(requireValue(arg, args), arg);
|
|
208
|
+
break;
|
|
175
209
|
case '-o':
|
|
176
210
|
case '--out':
|
|
177
211
|
options.out = requireValue(arg, args);
|
|
@@ -251,6 +285,33 @@ function parseNonNegativeInt(value, option) {
|
|
|
251
285
|
return parsed;
|
|
252
286
|
}
|
|
253
287
|
|
|
288
|
+
function parsePositiveIntList(value, option) {
|
|
289
|
+
const parsed = String(value)
|
|
290
|
+
.split(',')
|
|
291
|
+
.map((item) => item.trim())
|
|
292
|
+
.filter(Boolean)
|
|
293
|
+
.map((item) => parsePositiveInt(item, option));
|
|
294
|
+
if (parsed.length === 0) throw new Error(`${option} requires at least one positive integer`);
|
|
295
|
+
return parsed;
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
function renderOptions(parsed) {
|
|
299
|
+
return {
|
|
300
|
+
ascii: parsed.ascii,
|
|
301
|
+
color: resolveColor(parsed.color),
|
|
302
|
+
compact: parsed.compact,
|
|
303
|
+
width: parsed.width
|
|
304
|
+
};
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
function resolveColor(value) {
|
|
308
|
+
if (value === 'always') return true;
|
|
309
|
+
if (value === 'never') return false;
|
|
310
|
+
if (process.env.NO_COLOR) return false;
|
|
311
|
+
if (process.env.FORCE_COLOR && process.env.FORCE_COLOR !== '0') return true;
|
|
312
|
+
return Boolean(process.stdout.isTTY);
|
|
313
|
+
}
|
|
314
|
+
|
|
254
315
|
function helpText() {
|
|
255
316
|
return `fm-bench ${packageJson.version}
|
|
256
317
|
|
|
@@ -266,7 +327,12 @@ Run options:
|
|
|
266
327
|
-r, --runs <n> Runs per prompt/model (default: 1)
|
|
267
328
|
--warmup <n> Warmup runs per model before measurement
|
|
268
329
|
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
330
|
+
--sweep-concurrency <list>
|
|
331
|
+
Run separate operating points, e.g. 1,2,4
|
|
269
332
|
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
333
|
+
--slo-ttft-ms <n> Count request as good only if TTFT is <= n
|
|
334
|
+
--slo-e2e-ms <n> Count request as good only if E2E latency is <= n
|
|
335
|
+
--slo-tpot-ms <n> Count request as good only if TPOT is <= n
|
|
270
336
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
271
337
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
272
338
|
--profile <name> quick, standard, interactive, throughput, or stress
|
|
@@ -286,6 +352,10 @@ Output:
|
|
|
286
352
|
--json Alias for --format json
|
|
287
353
|
--csv Alias for --format csv
|
|
288
354
|
--ascii Use plain ASCII tables instead of Unicode
|
|
355
|
+
--color Force ANSI colors in table output
|
|
356
|
+
--no-color Disable ANSI colors in table output
|
|
357
|
+
--compact Force compact terminal layout
|
|
358
|
+
--width <n> Render for a specific terminal width
|
|
289
359
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
290
360
|
-v, --verbose Include per-run CSV after the summary table
|
|
291
361
|
|
package/src/report.js
CHANGED
|
@@ -13,6 +13,7 @@ export function toCsv(rows) {
|
|
|
13
13
|
export function flattenResults(results) {
|
|
14
14
|
return results.map((result) => ({
|
|
15
15
|
model: result.model,
|
|
16
|
+
concurrency: result.concurrency ?? '',
|
|
16
17
|
prompt_id: result.promptId,
|
|
17
18
|
run: result.run,
|
|
18
19
|
ok: result.ok,
|
|
@@ -30,6 +31,7 @@ export function flattenResults(results) {
|
|
|
30
31
|
streamed: result.streamed,
|
|
31
32
|
stdout_chunks: result.stdoutChunks,
|
|
32
33
|
output_hash: result.outputHash || '',
|
|
34
|
+
good: result.good == null ? '' : result.good,
|
|
33
35
|
error: result.error || ''
|
|
34
36
|
}));
|
|
35
37
|
}
|
package/src/stats.js
CHANGED
|
@@ -7,6 +7,10 @@ export function summarizeNumbers(values) {
|
|
|
7
7
|
max: null,
|
|
8
8
|
avg: null,
|
|
9
9
|
sum: 0,
|
|
10
|
+
stddev: null,
|
|
11
|
+
cv: null,
|
|
12
|
+
ci95Low: null,
|
|
13
|
+
ci95High: null,
|
|
10
14
|
p50: null,
|
|
11
15
|
p90: null,
|
|
12
16
|
p95: null,
|
|
@@ -15,12 +19,22 @@ export function summarizeNumbers(values) {
|
|
|
15
19
|
}
|
|
16
20
|
|
|
17
21
|
const total = clean.reduce((sum, value) => sum + value, 0);
|
|
22
|
+
const avg = total / clean.length;
|
|
23
|
+
const variance = clean.length > 1
|
|
24
|
+
? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
|
|
25
|
+
: 0;
|
|
26
|
+
const stddev = Math.sqrt(variance);
|
|
27
|
+
const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
|
|
18
28
|
return {
|
|
19
29
|
count: clean.length,
|
|
20
30
|
min: clean[0],
|
|
21
31
|
max: clean[clean.length - 1],
|
|
22
|
-
avg
|
|
32
|
+
avg,
|
|
23
33
|
sum: total,
|
|
34
|
+
stddev,
|
|
35
|
+
cv: avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
36
|
+
ci95Low: avg - margin,
|
|
37
|
+
ci95High: avg + margin,
|
|
24
38
|
p50: percentile(clean, 50),
|
|
25
39
|
p90: percentile(clean, 90),
|
|
26
40
|
p95: percentile(clean, 95),
|
|
@@ -40,35 +54,44 @@ export function percentile(sortedValues, percentileValue) {
|
|
|
40
54
|
return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
|
|
41
55
|
}
|
|
42
56
|
|
|
43
|
-
export function summarizeByModel(results, modelStatuses = []) {
|
|
57
|
+
export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
44
58
|
const byModel = new Map();
|
|
59
|
+
const concurrencies = options.concurrencies?.length ? options.concurrencies : [undefined];
|
|
45
60
|
|
|
46
|
-
for (const
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
61
|
+
for (const concurrency of concurrencies) {
|
|
62
|
+
for (const status of modelStatuses) {
|
|
63
|
+
const key = summaryKey(status.name, concurrency);
|
|
64
|
+
byModel.set(key, {
|
|
65
|
+
model: status.name,
|
|
66
|
+
concurrency,
|
|
67
|
+
description: status.description,
|
|
68
|
+
available: status.available,
|
|
69
|
+
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
70
|
+
results: []
|
|
71
|
+
});
|
|
72
|
+
}
|
|
54
73
|
}
|
|
55
74
|
|
|
56
75
|
for (const result of results) {
|
|
57
|
-
|
|
58
|
-
|
|
76
|
+
const key = summaryKey(result.model, result.concurrency);
|
|
77
|
+
if (!byModel.has(key)) {
|
|
78
|
+
byModel.set(key, {
|
|
59
79
|
model: result.model,
|
|
80
|
+
concurrency: result.concurrency,
|
|
60
81
|
description: '',
|
|
61
82
|
available: true,
|
|
62
83
|
skippedReason: '',
|
|
63
84
|
results: []
|
|
64
85
|
});
|
|
65
86
|
}
|
|
66
|
-
byModel.get(
|
|
87
|
+
byModel.get(key).results.push(result);
|
|
67
88
|
}
|
|
68
89
|
|
|
69
90
|
return [...byModel.values()].map((entry) => {
|
|
70
91
|
const successes = entry.results.filter((result) => result.ok);
|
|
71
92
|
const failures = entry.results.filter((result) => !result.ok);
|
|
93
|
+
const goodResults = successes.filter((result) => result.good === true);
|
|
94
|
+
const goodMeasured = successes.filter((result) => result.good != null);
|
|
72
95
|
const latency = summarizeNumbers(successes.map((result) => result.durationMs));
|
|
73
96
|
const ttft = summarizeNumbers(successes.map((result) => result.firstTokenMs).filter((value) => value != null));
|
|
74
97
|
const generation = summarizeNumbers(successes.map((result) => result.generationMs).filter((value) => value != null));
|
|
@@ -80,10 +103,12 @@ export function summarizeByModel(results, modelStatuses = []) {
|
|
|
80
103
|
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
81
104
|
const windowMs = modelWindowMs(successes);
|
|
82
105
|
const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
|
|
106
|
+
const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
|
|
83
107
|
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
84
108
|
|
|
85
109
|
return {
|
|
86
110
|
model: entry.model,
|
|
111
|
+
concurrency: entry.concurrency,
|
|
87
112
|
description: entry.description,
|
|
88
113
|
available: entry.available,
|
|
89
114
|
skippedReason: entry.skippedReason,
|
|
@@ -91,7 +116,9 @@ export function summarizeByModel(results, modelStatuses = []) {
|
|
|
91
116
|
successes: successes.length,
|
|
92
117
|
failures: failures.length,
|
|
93
118
|
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
119
|
+
goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
|
|
94
120
|
rps,
|
|
121
|
+
goodputRps,
|
|
95
122
|
outputTokenThroughput,
|
|
96
123
|
repeatability: summarizeRepeatability(successes),
|
|
97
124
|
latency,
|
|
@@ -107,6 +134,49 @@ export function summarizeByModel(results, modelStatuses = []) {
|
|
|
107
134
|
});
|
|
108
135
|
}
|
|
109
136
|
|
|
137
|
+
function summaryKey(model, concurrency) {
|
|
138
|
+
return `${model}::${concurrency ?? 'default'}`;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function tCritical95(n) {
|
|
142
|
+
const df = Math.max(1, n - 1);
|
|
143
|
+
const table = {
|
|
144
|
+
1: 12.706,
|
|
145
|
+
2: 4.303,
|
|
146
|
+
3: 3.182,
|
|
147
|
+
4: 2.776,
|
|
148
|
+
5: 2.571,
|
|
149
|
+
6: 2.447,
|
|
150
|
+
7: 2.365,
|
|
151
|
+
8: 2.306,
|
|
152
|
+
9: 2.262,
|
|
153
|
+
10: 2.228,
|
|
154
|
+
11: 2.201,
|
|
155
|
+
12: 2.179,
|
|
156
|
+
13: 2.16,
|
|
157
|
+
14: 2.145,
|
|
158
|
+
15: 2.131,
|
|
159
|
+
16: 2.12,
|
|
160
|
+
17: 2.11,
|
|
161
|
+
18: 2.101,
|
|
162
|
+
19: 2.093,
|
|
163
|
+
20: 2.086,
|
|
164
|
+
21: 2.08,
|
|
165
|
+
22: 2.074,
|
|
166
|
+
23: 2.069,
|
|
167
|
+
24: 2.064,
|
|
168
|
+
25: 2.06,
|
|
169
|
+
26: 2.056,
|
|
170
|
+
27: 2.052,
|
|
171
|
+
28: 2.048,
|
|
172
|
+
29: 2.045,
|
|
173
|
+
30: 2.042
|
|
174
|
+
};
|
|
175
|
+
if (df <= 30) return table[df];
|
|
176
|
+
if (df <= 60) return 2;
|
|
177
|
+
return 1.96;
|
|
178
|
+
}
|
|
179
|
+
|
|
110
180
|
function modelWindowMs(results) {
|
|
111
181
|
const starts = results.map((result) => result.startOffsetMs).filter((value) => Number.isFinite(value));
|
|
112
182
|
const ends = results.map((result) => result.endOffsetMs).filter((value) => Number.isFinite(value));
|
package/src/table.js
CHANGED
|
@@ -1,8 +1,17 @@
|
|
|
1
|
+
import { stripAnsi } from './ansi.js';
|
|
2
|
+
|
|
1
3
|
export function renderTable(headers, rows, options = {}) {
|
|
2
4
|
const ascii = Boolean(options.ascii);
|
|
3
|
-
const
|
|
5
|
+
const maxCellWidth = options.maxCellWidth || 60;
|
|
6
|
+
const normalizedRows = rows.map((row) => row.map((cell) => {
|
|
7
|
+
const normalized = normalizeCell(cell);
|
|
8
|
+
return {
|
|
9
|
+
...normalized,
|
|
10
|
+
text: truncate(normalized.text, maxCellWidth)
|
|
11
|
+
};
|
|
12
|
+
}));
|
|
4
13
|
const widths = headers.map((header, index) => {
|
|
5
|
-
const values = [header, ...
|
|
14
|
+
const values = [truncate(header, maxCellWidth), ...normalizedRows.map((row) => row[index]?.text ?? '')];
|
|
6
15
|
return Math.max(...values.map(visibleLength));
|
|
7
16
|
});
|
|
8
17
|
const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
|
|
@@ -10,27 +19,45 @@ export function renderTable(headers, rows, options = {}) {
|
|
|
10
19
|
const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
|
|
11
20
|
const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
|
|
12
21
|
const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
|
|
13
|
-
const headerLine = rowLine(headers, widths, style, true);
|
|
14
|
-
const bodyLines =
|
|
22
|
+
const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true, options);
|
|
23
|
+
const bodyLines = normalizedRows.map((row) => rowLine(row, widths, style, false, options));
|
|
15
24
|
|
|
16
25
|
return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
|
|
17
26
|
}
|
|
18
27
|
|
|
19
28
|
export function renderBenchmarkReport(payload, options = {}) {
|
|
29
|
+
const width = terminalWidth(options);
|
|
30
|
+
const mode = options.compact || width < 88
|
|
31
|
+
? 'compact'
|
|
32
|
+
: width < 140
|
|
33
|
+
? 'medium'
|
|
34
|
+
: 'wide';
|
|
20
35
|
const lines = [];
|
|
21
36
|
const elapsedMs = Date.parse(payload.finishedAt) - Date.parse(payload.startedAt);
|
|
22
37
|
const skipped = payload.summary.filter((item) => !item.available).length;
|
|
23
38
|
const measured = payload.summary.reduce((sum, item) => sum + item.successes, 0);
|
|
24
39
|
const failed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
|
|
40
|
+
const concurrencies = payload.options.sweepConcurrency?.length
|
|
41
|
+
? payload.options.sweepConcurrency.join(',')
|
|
42
|
+
: String(payload.options.concurrency);
|
|
43
|
+
const slo = formatSlo(payload.options.slo);
|
|
25
44
|
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
lines.push(
|
|
29
|
-
lines.push(
|
|
45
|
+
const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
|
|
46
|
+
const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
|
|
47
|
+
lines.push(truncate(title, width));
|
|
48
|
+
lines.push(truncate(meta, width));
|
|
30
49
|
lines.push('');
|
|
31
|
-
|
|
50
|
+
|
|
51
|
+
if (mode === 'compact') {
|
|
52
|
+
lines.push(renderCompactSummary(payload.summary, { ...options, width, slo: payload.options.slo }));
|
|
53
|
+
} else {
|
|
54
|
+
lines.push(renderSummaryTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
|
|
55
|
+
lines.push('');
|
|
56
|
+
lines.push(renderDetailTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
|
|
57
|
+
}
|
|
58
|
+
|
|
32
59
|
lines.push('');
|
|
33
|
-
lines.push(
|
|
60
|
+
lines.push(compactLegend(width));
|
|
34
61
|
|
|
35
62
|
return lines.join('\n');
|
|
36
63
|
}
|
|
@@ -53,77 +80,181 @@ export function formatPercent(value, digits = 0) {
|
|
|
53
80
|
}
|
|
54
81
|
|
|
55
82
|
export function renderSummaryTable(summary, options = {}) {
|
|
83
|
+
const mode = options.mode || 'wide';
|
|
84
|
+
const hasGoodput = summary.some((item) => item.goodputRate != null);
|
|
85
|
+
const ranks = rankSummary(summary);
|
|
56
86
|
const rows = summary.map((item) => {
|
|
57
87
|
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
item.
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
formatMs(item.
|
|
67
|
-
formatMs(item.
|
|
68
|
-
formatMs(item.
|
|
69
|
-
|
|
70
|
-
formatNumber(item.
|
|
71
|
-
item.
|
|
88
|
+
const tones = metricTones(item, ranks, options.slo);
|
|
89
|
+
const base = [
|
|
90
|
+
cell(formatConcurrency(item.concurrency)),
|
|
91
|
+
cell(item.model, item.available ? null : 'muted'),
|
|
92
|
+
cell(status, statusTone(status)),
|
|
93
|
+
cell(item.attempted ? `${item.successes}/${item.attempted}` : '-', item.failures > 0 ? 'yellow' : item.available ? 'green' : 'muted'),
|
|
94
|
+
cell(formatPercent(item.successRate), percentTone(item.successRate, 0.95, 1)),
|
|
95
|
+
cell(formatPercent(item.goodputRate), percentTone(item.goodputRate, 0.8, 1)),
|
|
96
|
+
cell(formatMs(item.ttft.p50), tones.ttft),
|
|
97
|
+
cell(formatMs(item.ttft.p95), tones.ttftP95),
|
|
98
|
+
cell(formatMs(item.latency.p50), tones.e2e),
|
|
99
|
+
cell(formatMs(item.latency.p95), tones.e2eP95),
|
|
100
|
+
cell(formatNumber(item.tokensPerSecond.avg), tones.userTps),
|
|
101
|
+
cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
|
|
102
|
+
cell(formatNumber(item.rps), tones.rps),
|
|
103
|
+
cell(formatPercent(item.latency.cv), cvTone(item.latency.cv)),
|
|
104
|
+
cell(item.available ? '' : compactReason(item.skippedReason), item.available ? null : 'yellow')
|
|
72
105
|
];
|
|
106
|
+
|
|
107
|
+
if (mode === 'medium') {
|
|
108
|
+
const medium = [
|
|
109
|
+
base[0],
|
|
110
|
+
base[1],
|
|
111
|
+
base[2],
|
|
112
|
+
base[3]
|
|
113
|
+
];
|
|
114
|
+
if (hasGoodput) medium.push(base[5]);
|
|
115
|
+
medium.push(base[6], base[8], base[9], base[10], base[11], base[13], base[14]);
|
|
116
|
+
return medium;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const wide = [base[0], base[1], base[2], base[3], base[4]];
|
|
120
|
+
if (hasGoodput) wide.push(base[5]);
|
|
121
|
+
wide.push(
|
|
122
|
+
base[6],
|
|
123
|
+
base[7],
|
|
124
|
+
base[8],
|
|
125
|
+
base[9],
|
|
126
|
+
cell(formatMs(item.tpot.p50), tones.tpot),
|
|
127
|
+
cell(formatMs(item.tpot.p95), tones.tpotP95),
|
|
128
|
+
base[10],
|
|
129
|
+
base[11],
|
|
130
|
+
base[12],
|
|
131
|
+
base[13],
|
|
132
|
+
base[14]
|
|
133
|
+
);
|
|
134
|
+
return wide;
|
|
73
135
|
});
|
|
74
136
|
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
'
|
|
79
|
-
'
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
'ttft p95',
|
|
83
|
-
'e2e p50',
|
|
84
|
-
'e2e p95',
|
|
85
|
-
'tpot p50',
|
|
86
|
-
'tok/s',
|
|
87
|
-
'rps',
|
|
88
|
-
'note'
|
|
89
|
-
], rows, options);
|
|
137
|
+
const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
|
|
138
|
+
const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
|
|
139
|
+
const headers = mode === 'medium'
|
|
140
|
+
? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
|
|
141
|
+
: (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
|
|
142
|
+
|
|
143
|
+
return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
|
|
90
144
|
}
|
|
91
145
|
|
|
92
146
|
export function renderDetailTable(summary, options = {}) {
|
|
93
|
-
const
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
'
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
147
|
+
const mode = options.mode || 'wide';
|
|
148
|
+
const ranks = rankSummary(summary);
|
|
149
|
+
const rows = summary.map((item) => {
|
|
150
|
+
const tones = metricTones(item, ranks, options.slo);
|
|
151
|
+
const base = [
|
|
152
|
+
cell(formatConcurrency(item.concurrency)),
|
|
153
|
+
cell(item.model, item.available ? null : 'muted'),
|
|
154
|
+
cell(formatNumber(item.promptTokens.avg, 0)),
|
|
155
|
+
cell(formatNumber(item.outputTokens.avg, 0)),
|
|
156
|
+
cell(formatNumber(item.decodeTokensPerSecond.avg), tones.decodeTps),
|
|
157
|
+
cell(formatMs(item.latency.p99), tones.e2eP99),
|
|
158
|
+
cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
|
|
159
|
+
cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
|
|
160
|
+
cell(item.description || '-', 'muted')
|
|
161
|
+
];
|
|
162
|
+
|
|
163
|
+
if (mode === 'medium') {
|
|
164
|
+
return base.slice(0, 8);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
return base;
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
const headers = mode === 'medium'
|
|
171
|
+
? ['c', 'model', 'in avg', 'out avg', 'decode t/s', 'e2e p99', 'e2e 95% ci', 'repeat']
|
|
172
|
+
: [
|
|
173
|
+
'c',
|
|
174
|
+
'model',
|
|
175
|
+
'in tok avg',
|
|
176
|
+
'out tok avg',
|
|
177
|
+
'decode tok/s',
|
|
178
|
+
'e2e p99',
|
|
179
|
+
'e2e 95% ci',
|
|
180
|
+
'repeat',
|
|
181
|
+
'description'
|
|
182
|
+
];
|
|
183
|
+
|
|
184
|
+
return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52 });
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
export function renderCompactSummary(summary, options = {}) {
|
|
188
|
+
const width = options.width || 80;
|
|
189
|
+
const separator = options.ascii ? '-' : '─';
|
|
190
|
+
const lines = [];
|
|
191
|
+
const ranks = rankSummary(summary);
|
|
192
|
+
|
|
193
|
+
for (const item of summary) {
|
|
194
|
+
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
195
|
+
const title = `${item.model} c${formatConcurrency(item.concurrency)} ${status} ${item.attempted ? `${item.successes}/${item.attempted}` : '-'}`;
|
|
196
|
+
lines.push(toneText(truncate(title, width), statusTone(status), options));
|
|
197
|
+
|
|
198
|
+
if (item.available) {
|
|
199
|
+
const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
|
|
200
|
+
const tones = metricTones(item, ranks, options.slo);
|
|
201
|
+
lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
|
|
202
|
+
lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
|
|
203
|
+
lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width), worstTone(tones.tpot, percentTone(item.repeatability, 0.5, 0.9)), options));
|
|
204
|
+
} else {
|
|
205
|
+
lines.push(toneText(truncate(` ${compactReason(item.skippedReason)}`, width), 'yellow', options));
|
|
206
|
+
}
|
|
207
|
+
lines.push(separator.repeat(Math.min(width, 72)));
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
if (lines.at(-1)?.startsWith(separator)) lines.pop();
|
|
211
|
+
return lines.join('\n');
|
|
116
212
|
}
|
|
117
213
|
|
|
118
214
|
export function renderModelsTable(models, options = {}) {
|
|
215
|
+
const width = terminalWidth(options);
|
|
216
|
+
const compact = options.compact || width < 88;
|
|
217
|
+
if (compact) {
|
|
218
|
+
return models.map((model) => {
|
|
219
|
+
const status = model.available ? 'yes' : 'no';
|
|
220
|
+
return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(model.reason || model.description || '-')}`;
|
|
221
|
+
}).join('\n');
|
|
222
|
+
}
|
|
223
|
+
|
|
119
224
|
return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
|
|
120
|
-
model.name,
|
|
121
|
-
model.available ? 'yes' : 'no',
|
|
225
|
+
cell(model.name),
|
|
226
|
+
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
|
|
122
227
|
model.description || '-',
|
|
123
228
|
compactReason(model.quota || model.reason || '-')
|
|
124
229
|
]), options);
|
|
125
230
|
}
|
|
126
231
|
|
|
232
|
+
function formatRangeMs(low, high) {
|
|
233
|
+
if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
|
|
234
|
+
return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function formatConcurrency(value) {
|
|
238
|
+
return value == null ? '1' : String(value);
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
function terminalWidth(options = {}) {
|
|
242
|
+
return options.width || process.stdout.columns || 120;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
function compactLegend(width) {
|
|
246
|
+
const text = 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence. CV = lower is steadier.';
|
|
247
|
+
return truncate(text, width);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function formatSlo(slo = {}) {
|
|
251
|
+
const parts = [];
|
|
252
|
+
if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
|
|
253
|
+
if (slo.e2eMs) parts.push(`E2E<=${formatMs(slo.e2eMs)}`);
|
|
254
|
+
if (slo.tpotMs) parts.push(`TPOT<=${formatMs(slo.tpotMs)}`);
|
|
255
|
+
return parts.length ? `SLO ${parts.join(',')}` : '';
|
|
256
|
+
}
|
|
257
|
+
|
|
127
258
|
const ASCII_TABLE = {
|
|
128
259
|
topLeft: '+',
|
|
129
260
|
topJoin: '+',
|
|
@@ -156,13 +287,32 @@ function rule(left, join, right, horizontal, widths) {
|
|
|
156
287
|
return `${left}${widths.map((width) => horizontal.repeat(width + 2)).join(join)}${right}`;
|
|
157
288
|
}
|
|
158
289
|
|
|
159
|
-
function rowLine(row, widths, style, header = false) {
|
|
290
|
+
function rowLine(row, widths, style, header = false, options = {}) {
|
|
160
291
|
return `${style.vertical}${row.map((cell, index) => {
|
|
161
|
-
const
|
|
162
|
-
|
|
292
|
+
const normalized = header ? { text: String(cell).toUpperCase(), tone: 'header' } : normalizeCell(cell);
|
|
293
|
+
const value = normalized.text;
|
|
294
|
+
const padded = pad(value, widths[index], !header && isNumericCell(value));
|
|
295
|
+
return ` ${toneText(padded, normalized.tone, options)} `;
|
|
163
296
|
}).join(style.vertical)}${style.vertical}`;
|
|
164
297
|
}
|
|
165
298
|
|
|
299
|
+
function cell(text, tone = null) {
|
|
300
|
+
return { text, tone };
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
function normalizeCell(value) {
|
|
304
|
+
if (value && typeof value === 'object' && Object.hasOwn(value, 'text')) {
|
|
305
|
+
return {
|
|
306
|
+
text: formatCell(value.text),
|
|
307
|
+
tone: value.tone || null
|
|
308
|
+
};
|
|
309
|
+
}
|
|
310
|
+
return {
|
|
311
|
+
text: formatCell(value),
|
|
312
|
+
tone: null
|
|
313
|
+
};
|
|
314
|
+
}
|
|
315
|
+
|
|
166
316
|
function formatCell(value) {
|
|
167
317
|
if (value == null) return '';
|
|
168
318
|
return String(value);
|
|
@@ -179,7 +329,7 @@ function isNumericCell(value) {
|
|
|
179
329
|
}
|
|
180
330
|
|
|
181
331
|
function visibleLength(value) {
|
|
182
|
-
return
|
|
332
|
+
return stripAnsi(value).length;
|
|
183
333
|
}
|
|
184
334
|
|
|
185
335
|
function compactReason(value) {
|
|
@@ -187,3 +337,119 @@ function compactReason(value) {
|
|
|
187
337
|
if (clean.length <= 58) return clean;
|
|
188
338
|
return `${clean.slice(0, 55)}...`;
|
|
189
339
|
}
|
|
340
|
+
|
|
341
|
+
function truncate(value, width) {
|
|
342
|
+
const text = String(value ?? '');
|
|
343
|
+
if (visibleLength(text) <= width) return text;
|
|
344
|
+
if (width <= 1) return '…';
|
|
345
|
+
return `${text.slice(0, width - 1)}…`;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
const TONES = {
|
|
349
|
+
header: ['\x1b[1m', '\x1b[0m'],
|
|
350
|
+
green: ['\x1b[32m', '\x1b[0m'],
|
|
351
|
+
yellow: ['\x1b[33m', '\x1b[0m'],
|
|
352
|
+
red: ['\x1b[31m', '\x1b[0m'],
|
|
353
|
+
muted: ['\x1b[2m', '\x1b[0m']
|
|
354
|
+
};
|
|
355
|
+
|
|
356
|
+
function toneText(text, tone, options = {}) {
|
|
357
|
+
if (!options.color || !tone || !TONES[tone]) return text;
|
|
358
|
+
const [open, close] = TONES[tone];
|
|
359
|
+
return `${open}${text}${close}`;
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
function statusTone(status) {
|
|
363
|
+
if (status === 'ok') return 'green';
|
|
364
|
+
if (status === 'partial') return 'yellow';
|
|
365
|
+
if (status === 'skipped') return 'yellow';
|
|
366
|
+
return 'red';
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
function percentTone(value, yellowAt, greenAt) {
|
|
370
|
+
if (value == null || !Number.isFinite(value)) return null;
|
|
371
|
+
if (value >= greenAt) return 'green';
|
|
372
|
+
if (value >= yellowAt) return 'yellow';
|
|
373
|
+
return 'red';
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
function cvTone(value) {
|
|
377
|
+
if (value == null || !Number.isFinite(value)) return null;
|
|
378
|
+
if (value <= 0.1) return 'green';
|
|
379
|
+
if (value <= 0.25) return 'yellow';
|
|
380
|
+
return 'red';
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
function thresholdTone(value, threshold) {
|
|
384
|
+
if (value == null || !Number.isFinite(value) || !Number.isFinite(threshold)) return null;
|
|
385
|
+
if (value <= threshold * 0.8) return 'green';
|
|
386
|
+
if (value <= threshold) return 'yellow';
|
|
387
|
+
return 'red';
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
function rankSummary(summary) {
|
|
391
|
+
return {
|
|
392
|
+
ttft: collectMetric(summary, (item) => item.ttft?.p50),
|
|
393
|
+
ttftP95: collectMetric(summary, (item) => item.ttft?.p95),
|
|
394
|
+
e2e: collectMetric(summary, (item) => item.latency?.p50),
|
|
395
|
+
e2eP95: collectMetric(summary, (item) => item.latency?.p95),
|
|
396
|
+
e2eP99: collectMetric(summary, (item) => item.latency?.p99),
|
|
397
|
+
tpot: collectMetric(summary, (item) => item.tpot?.p50),
|
|
398
|
+
tpotP95: collectMetric(summary, (item) => item.tpot?.p95),
|
|
399
|
+
userTps: collectMetric(summary, (item) => item.tokensPerSecond?.avg),
|
|
400
|
+
systemTps: collectMetric(summary, (item) => item.outputTokenThroughput),
|
|
401
|
+
rps: collectMetric(summary, (item) => item.rps),
|
|
402
|
+
decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg)
|
|
403
|
+
};
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
function collectMetric(summary, accessor) {
|
|
407
|
+
return summary.map(accessor).filter((value) => Number.isFinite(value));
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
function rankTone(value, values) {
|
|
411
|
+
if (value == null || !Number.isFinite(value) || values.length === 0) return null;
|
|
412
|
+
if (values.length === 1) return 'green';
|
|
413
|
+
const min = Math.min(...values);
|
|
414
|
+
const max = Math.max(...values);
|
|
415
|
+
if (max === min) return 'green';
|
|
416
|
+
const position = (value - min) / (max - min);
|
|
417
|
+
if (position >= 0.67) return 'green';
|
|
418
|
+
if (position >= 0.34) return 'yellow';
|
|
419
|
+
return 'red';
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
function rankToneLower(value, values) {
|
|
423
|
+
if (value == null || !Number.isFinite(value) || values.length === 0) return null;
|
|
424
|
+
if (values.length === 1) return 'green';
|
|
425
|
+
const min = Math.min(...values);
|
|
426
|
+
const max = Math.max(...values);
|
|
427
|
+
if (max === min) return 'green';
|
|
428
|
+
const position = (max - value) / (max - min);
|
|
429
|
+
if (position >= 0.67) return 'green';
|
|
430
|
+
if (position >= 0.34) return 'yellow';
|
|
431
|
+
return 'red';
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
function metricTones(item, ranks, slo = {}) {
|
|
435
|
+
return {
|
|
436
|
+
ttft: thresholdTone(item.ttft?.p50, slo.ttftMs) || rankToneLower(item.ttft?.p50, ranks.ttft),
|
|
437
|
+
ttftP95: thresholdTone(item.ttft?.p95, slo.ttftMs) || rankToneLower(item.ttft?.p95, ranks.ttftP95),
|
|
438
|
+
e2e: thresholdTone(item.latency?.p50, slo.e2eMs) || rankToneLower(item.latency?.p50, ranks.e2e),
|
|
439
|
+
e2eP95: thresholdTone(item.latency?.p95, slo.e2eMs) || rankToneLower(item.latency?.p95, ranks.e2eP95),
|
|
440
|
+
e2eP99: thresholdTone(item.latency?.p99, slo.e2eMs) || rankToneLower(item.latency?.p99, ranks.e2eP99),
|
|
441
|
+
tpot: thresholdTone(item.tpot?.p50, slo.tpotMs) || rankToneLower(item.tpot?.p50, ranks.tpot),
|
|
442
|
+
tpotP95: thresholdTone(item.tpot?.p95, slo.tpotMs) || rankToneLower(item.tpot?.p95, ranks.tpotP95),
|
|
443
|
+
userTps: rankTone(item.tokensPerSecond?.avg, ranks.userTps),
|
|
444
|
+
systemTps: rankTone(item.outputTokenThroughput, ranks.systemTps),
|
|
445
|
+
rps: rankTone(item.rps, ranks.rps),
|
|
446
|
+
decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps)
|
|
447
|
+
};
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
function worstTone(...tones) {
|
|
451
|
+
if (tones.includes('red')) return 'red';
|
|
452
|
+
if (tones.includes('yellow')) return 'yellow';
|
|
453
|
+
if (tones.includes('green')) return 'green';
|
|
454
|
+
return null;
|
|
455
|
+
}
|