fm-bench 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -17
- package/docs/methodology.md +15 -1
- package/package.json +1 -1
- package/src/bench.js +150 -7
- package/src/cli.js +48 -4
- package/src/fm.js +2 -1
- package/src/process.js +6 -1
- package/src/progress.js +190 -0
- package/src/prompts.js +22 -0
- package/src/report.js +4 -0
- package/src/stats.js +11 -2
- package/src/table.js +40 -16
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
`fm-bench` is a dynamic benchmark CLI for Apple's `fm` command on macOS 27 and newer.
|
|
4
4
|
|
|
5
|
-
It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, and prints
|
|
5
|
+
It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, shows live progress while it works, and prints terminal tables with latency, throughput, stability, goodput, and streaming-quality stats.
|
|
6
6
|
|
|
7
7
|
Apple introduced the preinstalled `fm` command for macOS 27 as part of the Foundation Models tooling. `fm-bench` intentionally shells out to the system `fm` binary instead of linking private APIs, so it can adapt as Apple adds models or changes availability.
|
|
8
8
|
|
|
@@ -35,20 +35,15 @@ fm-bench
|
|
|
35
35
|
Example output:
|
|
36
36
|
|
|
37
37
|
```text
|
|
38
|
-
fm-bench 0.
|
|
39
|
-
prompts
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
│ MODEL │ STATUS │ RUNS │
|
|
43
|
-
|
|
44
|
-
│ system │ ok │
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
|
|
48
|
-
│ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
|
|
49
|
-
├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
|
|
50
|
-
│ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
|
|
51
|
-
└────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
|
|
38
|
+
fm-bench 0.4.0 | darwin/arm64 | fm
|
|
39
|
+
prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
|
|
40
|
+
|
|
41
|
+
┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
|
|
42
|
+
│ C │ MODEL │ STATUS │ OK/RUNS │ SUCC │ GOOD │ GOOD RPS │ TTFT │ E2E P95 │ SYS │ CV │
|
|
43
|
+
├───┼────────┼────────┼─────────┼──────┼──────┼──────────┼──────┼──────────┼─────┼─────┤
|
|
44
|
+
│ 1 │ system │ ok │ 15/15 │ 100% │ 93% │ 0.4 │ 318ms│ 3.20s │ 42 │ 12% │
|
|
45
|
+
│ 2 │ system │ ok │ 15/15 │ 100% │ 80% │ 0.7 │ 501ms│ 4.40s │ 68 │ 21% │
|
|
46
|
+
└───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
|
|
52
47
|
```
|
|
53
48
|
|
|
54
49
|
## Commands
|
|
@@ -72,6 +67,7 @@ fm-bench --models system,pcc --runs 3 --profile stress
|
|
|
72
67
|
fm-bench --models system --runs 5 --profile interactive
|
|
73
68
|
fm-bench --models system --runs 3 --profile throughput --warmup 1
|
|
74
69
|
fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
|
|
70
|
+
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5 --ramp-up-ms 2000
|
|
75
71
|
fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
76
72
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
77
73
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
@@ -85,9 +81,11 @@ Useful flags:
|
|
|
85
81
|
- `--warmup <n>`: warmup runs per model before measurement.
|
|
86
82
|
- `--concurrency <n>`: parallel `fm` processes.
|
|
87
83
|
- `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
|
|
84
|
+
- `--request-rate <rps>`: pace request starts at a target requests-per-second rate.
|
|
85
|
+
- `--ramp-up-ms <n>`: gradually ramp request pacing over `n` milliseconds.
|
|
88
86
|
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
89
87
|
- `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
|
|
90
|
-
- `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
|
|
88
|
+
- `--profile quick|standard|interactive|throughput|client|stress`: built-in prompt suite.
|
|
91
89
|
- `--prompt <text>`: custom prompt, repeatable.
|
|
92
90
|
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
93
91
|
- `--instructions <text>`: passed to `fm respond`.
|
|
@@ -96,6 +94,7 @@ Useful flags:
|
|
|
96
94
|
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
97
95
|
- `--ascii`: use plain ASCII table borders.
|
|
98
96
|
- `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
|
|
97
|
+
- `--progress`, `--no-progress`: force or disable the live progress status line on stderr.
|
|
99
98
|
- `--compact`: force the narrow terminal layout.
|
|
100
99
|
- `--width <n>`: render as if the terminal has `n` columns.
|
|
101
100
|
- `--out <file>`: save a report.
|
|
@@ -127,8 +126,11 @@ Plain text files are split on blank lines.
|
|
|
127
126
|
- TTFT, or time to first streamed output.
|
|
128
127
|
- E2E latency, or full response wall-clock latency.
|
|
129
128
|
- TPOT, or decode time per output token after the first output token.
|
|
129
|
+
- second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
|
|
130
|
+
- prefill tokens per second, or prompt tokens divided by TTFT.
|
|
130
131
|
- output tokens per second per request.
|
|
131
132
|
- total output token throughput across the measured window.
|
|
133
|
+
- total token throughput, including prompt and output tokens.
|
|
132
134
|
- requests per second across the measured window.
|
|
133
135
|
- goodput percentage and goodput RPS when SLO flags are set.
|
|
134
136
|
- coefficient of variation (CV) and confidence interval context for stability.
|
|
@@ -140,10 +142,16 @@ Plain text files are split on blank lines.
|
|
|
140
142
|
|
|
141
143
|
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
|
|
142
144
|
|
|
143
|
-
Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and
|
|
145
|
+
Measured runs stream by default so `fm-bench` can capture TTFT and streaming smoothness. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT, TPOT, second-chunk, and chunk-gap fields that depend on streaming will be blank.
|
|
144
146
|
|
|
145
147
|
Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
|
|
146
148
|
|
|
149
|
+
## Live Progress
|
|
150
|
+
|
|
151
|
+
Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
|
|
152
|
+
|
|
153
|
+
Progress is automatic for table output on TTYs. Use `--progress` to force it or `--no-progress` to keep the terminal completely quiet until the report is ready.
|
|
154
|
+
|
|
147
155
|
## Terminal Colors
|
|
148
156
|
|
|
149
157
|
Table output uses semantic ANSI color on interactive terminals:
|
package/docs/methodology.md
CHANGED
|
@@ -9,6 +9,7 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
9
9
|
- Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
|
|
10
10
|
- NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
|
|
11
11
|
- NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
|
|
12
|
+
- NVIDIA AIPerf documents time to second token, inter-token latency, inter-chunk latency, per-user output throughput, and prefill throughput: <https://docs.nvidia.com/aiperf/reference/ai-perf-metrics-reference>
|
|
12
13
|
- vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
|
|
13
14
|
- MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
|
|
14
15
|
- MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
|
|
@@ -19,11 +20,16 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
19
20
|
- `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
|
|
20
21
|
- `generation_ms`: `E2E - TTFT`.
|
|
21
22
|
- `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
|
|
23
|
+
- `second_chunk_ms`: time between the first and second streamed stdout chunks. This is a terminal-side proxy for time-to-second-token style startup smoothness.
|
|
24
|
+
- `chunk_gap`: the distribution of time between consecutive streamed stdout chunks. It is useful for spotting streaming jitter, but it is chunk-based rather than token-based because the `fm` CLI writes stdout chunks, not token timestamp events.
|
|
25
|
+
- `prefill_tokens_per_second`: input prompt tokens divided by TTFT seconds. This estimates prompt-processing speed for streaming runs.
|
|
22
26
|
- `tokens_per_second`: output tokens divided by E2E seconds for one request.
|
|
23
27
|
- `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
|
|
24
28
|
- `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
|
|
29
|
+
- `total token throughput`: successful prompt and output tokens divided by that model's measured wall-clock window.
|
|
25
30
|
- `RPS`: successful requests divided by that model's measured wall-clock window.
|
|
26
31
|
- `goodput`: successful requests that also satisfy all provided SLO thresholds.
|
|
32
|
+
- `goodput RPS`: SLO-passing requests divided by that model's measured wall-clock window. If SLOs are set and no requests pass, this is reported as zero.
|
|
27
33
|
- `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
|
|
28
34
|
- `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
|
|
29
35
|
- `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
|
|
@@ -32,12 +38,20 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
32
38
|
|
|
33
39
|
Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
|
|
34
40
|
|
|
41
|
+
Use `--request-rate <rps>` to pace request starts independently of concurrency. Concurrency limits how many `fm respond` processes can be active at once; request rate controls how quickly new work is admitted. Use `--ramp-up-ms` to avoid instantly shocking a model or quota path when you start a higher-rate run.
|
|
42
|
+
|
|
35
43
|
`fm-bench` does not interpolate between operating points. It reports only what was actually measured.
|
|
36
44
|
|
|
45
|
+
## Prompt Profiles
|
|
46
|
+
|
|
47
|
+
The `client` profile is a pragmatic local-machine mix inspired by MLPerf Client's emphasis on multiple task categories and prompt/response lengths. It includes short chat, content generation, structured extraction, light summarization, and code analysis prompts. It is not a formal MLPerf submission suite; it is a convenient built-in workload for comparing your own Mac, OS build, and `fm` models over time.
|
|
48
|
+
|
|
37
49
|
## Caveats
|
|
38
50
|
|
|
39
51
|
`fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
|
|
40
52
|
|
|
41
53
|
Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
|
|
42
54
|
|
|
43
|
-
|
|
55
|
+
Stream smoothness metrics use stdout chunk arrival times. A chunk can contain more than one token, and terminal or pipe buffering can affect chunk boundaries. Treat `second_chunk_ms` and `chunk_gap` as user-visible streaming diagnostics, not raw decoder telemetry.
|
|
56
|
+
|
|
57
|
+
For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput or client profiles, compare models at the same concurrency operating points, set SLOs that match your real UX budget, and save JSON reports for later analysis.
|
package/package.json
CHANGED
package/src/bench.js
CHANGED
|
@@ -36,7 +36,9 @@ export async function inspectModels(options = {}) {
|
|
|
36
36
|
|
|
37
37
|
export async function runBenchmark(options = {}) {
|
|
38
38
|
const startedAt = new Date().toISOString();
|
|
39
|
+
notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
|
|
39
40
|
const prompts = await loadPrompts(options);
|
|
41
|
+
notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
|
|
40
42
|
const inspection = await inspectModels(options);
|
|
41
43
|
const modelStatuses = options.availableOnly
|
|
42
44
|
? inspection.models.filter((model) => model.available)
|
|
@@ -45,15 +47,36 @@ export async function runBenchmark(options = {}) {
|
|
|
45
47
|
const environment = await collectEnvironment(inspection.fmBin);
|
|
46
48
|
const promptTokenCounts = new Map();
|
|
47
49
|
const concurrencies = normalizeConcurrencySweep(options);
|
|
50
|
+
const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
|
|
48
51
|
|
|
52
|
+
notify(options, {
|
|
53
|
+
type: 'tokens:start',
|
|
54
|
+
total: prompts.length,
|
|
55
|
+
message: 'counting prompt tokens'
|
|
56
|
+
});
|
|
49
57
|
for (const prompt of prompts) {
|
|
50
58
|
const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
|
|
51
59
|
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
60
|
+
notify(options, {
|
|
61
|
+
type: 'tokens:progress',
|
|
62
|
+
completed: promptTokenCounts.size,
|
|
63
|
+
total: prompts.length,
|
|
64
|
+
promptId: prompt.id
|
|
65
|
+
});
|
|
52
66
|
}
|
|
53
67
|
|
|
54
68
|
const results = [];
|
|
55
69
|
const scenarios = [];
|
|
56
|
-
|
|
70
|
+
let completedRuns = 0;
|
|
71
|
+
let failedRuns = 0;
|
|
72
|
+
notify(options, {
|
|
73
|
+
type: 'benchmark:start',
|
|
74
|
+
total: totalRuns,
|
|
75
|
+
modelCount: runnableModels.length,
|
|
76
|
+
promptCount: prompts.length,
|
|
77
|
+
scenarioCount: concurrencies.length
|
|
78
|
+
});
|
|
79
|
+
for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
|
|
57
80
|
const scenario = await runScenario({
|
|
58
81
|
fmBin: inspection.fmBin,
|
|
59
82
|
prompts,
|
|
@@ -61,7 +84,26 @@ export async function runBenchmark(options = {}) {
|
|
|
61
84
|
modelStatuses,
|
|
62
85
|
promptTokenCounts,
|
|
63
86
|
options,
|
|
64
|
-
concurrency
|
|
87
|
+
concurrency,
|
|
88
|
+
scenarioIndex: scenarioIndex + 1,
|
|
89
|
+
scenarioCount: concurrencies.length,
|
|
90
|
+
onMeasuredResult: (result) => {
|
|
91
|
+
completedRuns += 1;
|
|
92
|
+
if (!result.ok) failedRuns += 1;
|
|
93
|
+
notify(options, {
|
|
94
|
+
type: 'benchmark:progress',
|
|
95
|
+
completed: completedRuns,
|
|
96
|
+
failed: failedRuns,
|
|
97
|
+
total: totalRuns,
|
|
98
|
+
concurrency,
|
|
99
|
+
model: result.model,
|
|
100
|
+
promptId: result.promptId,
|
|
101
|
+
run: result.run,
|
|
102
|
+
ok: result.ok,
|
|
103
|
+
durationMs: result.durationMs,
|
|
104
|
+
firstTokenMs: result.firstTokenMs
|
|
105
|
+
});
|
|
106
|
+
}
|
|
65
107
|
});
|
|
66
108
|
scenarios.push(scenario);
|
|
67
109
|
results.push(...scenario.results);
|
|
@@ -73,7 +115,7 @@ export async function runBenchmark(options = {}) {
|
|
|
73
115
|
|| a.run - b.run);
|
|
74
116
|
|
|
75
117
|
const summary = summarizeByModel(results, modelStatuses, { concurrencies });
|
|
76
|
-
|
|
118
|
+
const payload = {
|
|
77
119
|
tool: 'fm-bench',
|
|
78
120
|
version: options.version,
|
|
79
121
|
startedAt,
|
|
@@ -90,6 +132,13 @@ export async function runBenchmark(options = {}) {
|
|
|
90
132
|
summary,
|
|
91
133
|
results
|
|
92
134
|
};
|
|
135
|
+
notify(options, {
|
|
136
|
+
type: 'benchmark:complete',
|
|
137
|
+
completed: completedRuns,
|
|
138
|
+
failed: failedRuns,
|
|
139
|
+
total: totalRuns
|
|
140
|
+
});
|
|
141
|
+
return payload;
|
|
93
142
|
}
|
|
94
143
|
|
|
95
144
|
async function runScenario(context) {
|
|
@@ -100,16 +149,40 @@ async function runScenario(context) {
|
|
|
100
149
|
modelStatuses,
|
|
101
150
|
promptTokenCounts,
|
|
102
151
|
options,
|
|
103
|
-
concurrency
|
|
152
|
+
concurrency,
|
|
153
|
+
scenarioIndex,
|
|
154
|
+
scenarioCount,
|
|
155
|
+
onMeasuredResult
|
|
104
156
|
} = context;
|
|
105
157
|
const startedAt = new Date().toISOString();
|
|
106
158
|
|
|
159
|
+
const warmupTotal = options.warmup * runnableModels.length;
|
|
160
|
+
if (warmupTotal > 0) {
|
|
161
|
+
notify(options, {
|
|
162
|
+
type: 'warmup:start',
|
|
163
|
+
concurrency,
|
|
164
|
+
scenarioIndex,
|
|
165
|
+
scenarioCount,
|
|
166
|
+
total: warmupTotal
|
|
167
|
+
});
|
|
168
|
+
}
|
|
169
|
+
let warmupCompleted = 0;
|
|
107
170
|
for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
|
|
108
171
|
for (const model of runnableModels) {
|
|
109
172
|
await respond(fmBin, model.name, prompts[0].prompt, {
|
|
110
173
|
...options,
|
|
111
174
|
stream: false
|
|
112
175
|
});
|
|
176
|
+
warmupCompleted += 1;
|
|
177
|
+
notify(options, {
|
|
178
|
+
type: 'warmup:progress',
|
|
179
|
+
concurrency,
|
|
180
|
+
scenarioIndex,
|
|
181
|
+
scenarioCount,
|
|
182
|
+
completed: warmupCompleted,
|
|
183
|
+
total: warmupTotal,
|
|
184
|
+
model: model.name
|
|
185
|
+
});
|
|
113
186
|
}
|
|
114
187
|
}
|
|
115
188
|
|
|
@@ -124,15 +197,23 @@ async function runScenario(context) {
|
|
|
124
197
|
}
|
|
125
198
|
|
|
126
199
|
const results = [];
|
|
200
|
+
notify(options, {
|
|
201
|
+
type: 'scenario:start',
|
|
202
|
+
concurrency,
|
|
203
|
+
scenarioIndex,
|
|
204
|
+
scenarioCount,
|
|
205
|
+
total: jobs.length
|
|
206
|
+
});
|
|
127
207
|
await runLimited(jobs, concurrency, async (job) => {
|
|
128
208
|
const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
|
|
129
209
|
results.push(result);
|
|
210
|
+
if (onMeasuredResult) onMeasuredResult(result);
|
|
130
211
|
if (!result.ok && options.failFast) {
|
|
131
212
|
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
132
213
|
error.exitCode = 1;
|
|
133
214
|
throw error;
|
|
134
215
|
}
|
|
135
|
-
});
|
|
216
|
+
}, options);
|
|
136
217
|
|
|
137
218
|
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
138
219
|
|| a.promptId.localeCompare(b.promptId)
|
|
@@ -171,6 +252,12 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
171
252
|
: null;
|
|
172
253
|
const chars = response.output.length;
|
|
173
254
|
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
255
|
+
const promptTokens = promptTokenCounts.get(job.prompt.id);
|
|
256
|
+
const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
|
|
257
|
+
? promptTokens / (firstTokenMs / 1000)
|
|
258
|
+
: null;
|
|
259
|
+
const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
|
|
260
|
+
const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
|
|
174
261
|
|
|
175
262
|
return {
|
|
176
263
|
model: job.model.name,
|
|
@@ -182,17 +269,22 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
182
269
|
firstTokenMs,
|
|
183
270
|
generationMs,
|
|
184
271
|
tpotMs,
|
|
185
|
-
promptTokens
|
|
272
|
+
promptTokens,
|
|
186
273
|
outputTokens: countedOutputTokens,
|
|
187
274
|
chars,
|
|
188
275
|
words,
|
|
189
276
|
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
190
277
|
decodeTokensPerSecond,
|
|
278
|
+
prefillTokensPerSecond,
|
|
191
279
|
charsPerSecond: seconds > 0 ? chars / seconds : 0,
|
|
192
280
|
startOffsetMs,
|
|
193
281
|
endOffsetMs,
|
|
194
282
|
streamed: response.streamed,
|
|
195
283
|
stdoutChunks: response.stdoutChunks,
|
|
284
|
+
secondChunkMs,
|
|
285
|
+
chunkGapsMs,
|
|
286
|
+
chunkGapAvgMs: average(chunkGapsMs),
|
|
287
|
+
chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
|
|
196
288
|
outputHash: response.ok ? hashOutput(response.output) : null,
|
|
197
289
|
good: response.ok ? evaluateSlo({
|
|
198
290
|
firstTokenMs,
|
|
@@ -204,12 +296,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
204
296
|
};
|
|
205
297
|
}
|
|
206
298
|
|
|
207
|
-
async function runLimited(items, concurrency, worker) {
|
|
299
|
+
async function runLimited(items, concurrency, worker, options = {}) {
|
|
208
300
|
let nextIndex = 0;
|
|
301
|
+
const waitForSlot = createPacer(options.requestRate, options.rampUpMs);
|
|
209
302
|
const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
|
|
210
303
|
while (nextIndex < items.length) {
|
|
211
304
|
const index = nextIndex;
|
|
212
305
|
nextIndex += 1;
|
|
306
|
+
await waitForSlot(index);
|
|
213
307
|
await worker(items[index]);
|
|
214
308
|
}
|
|
215
309
|
});
|
|
@@ -233,6 +327,8 @@ function publicOptions(options) {
|
|
|
233
327
|
concurrency: options.concurrency,
|
|
234
328
|
sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
|
|
235
329
|
timeoutMs: options.timeoutMs,
|
|
330
|
+
requestRate: options.requestRate || null,
|
|
331
|
+
rampUpMs: options.rampUpMs || null,
|
|
236
332
|
profile: options.profile,
|
|
237
333
|
promptCount: options.promptCount,
|
|
238
334
|
greedy: options.greedy,
|
|
@@ -262,6 +358,53 @@ function hashOutput(output) {
|
|
|
262
358
|
.slice(0, 16);
|
|
263
359
|
}
|
|
264
360
|
|
|
361
|
+
function chunkGaps(times = []) {
|
|
362
|
+
const gaps = [];
|
|
363
|
+
for (let index = 1; index < times.length; index += 1) {
|
|
364
|
+
gaps.push(Math.max(0, times[index] - times[index - 1]));
|
|
365
|
+
}
|
|
366
|
+
return gaps;
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
function average(values) {
|
|
370
|
+
const clean = values.filter((value) => Number.isFinite(value));
|
|
371
|
+
if (clean.length === 0) return null;
|
|
372
|
+
return clean.reduce((sum, value) => sum + value, 0) / clean.length;
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
function createPacer(requestRate, rampUpMs = 0) {
|
|
376
|
+
if (!Number.isFinite(requestRate) || requestRate <= 0) {
|
|
377
|
+
return async () => {};
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
const startedAt = process.hrtime.bigint();
|
|
381
|
+
const offsets = [];
|
|
382
|
+
const steadyIntervalMs = 1000 / requestRate;
|
|
383
|
+
const warmIntervalMs = steadyIntervalMs * 4;
|
|
384
|
+
|
|
385
|
+
return async (index) => {
|
|
386
|
+
while (offsets.length <= index) {
|
|
387
|
+
const previousOffset = offsets.length === 0 ? 0 : offsets[offsets.length - 1];
|
|
388
|
+
const fraction = rampUpMs > 0 ? Math.min(1, previousOffset / rampUpMs) : 1;
|
|
389
|
+
const interval = warmIntervalMs + ((steadyIntervalMs - warmIntervalMs) * fraction);
|
|
390
|
+
offsets.push(offsets.length === 0 ? 0 : previousOffset + interval);
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
const targetMs = offsets[index];
|
|
394
|
+
const elapsedMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
395
|
+
const waitMs = targetMs - elapsedMs;
|
|
396
|
+
if (waitMs > 0) {
|
|
397
|
+
await new Promise((resolve) => setTimeout(resolve, waitMs));
|
|
398
|
+
}
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
function notify(options, event) {
|
|
403
|
+
if (typeof options.onProgress === 'function') {
|
|
404
|
+
options.onProgress(event);
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
265
408
|
function evaluateSlo(metrics, options) {
|
|
266
409
|
const thresholds = [
|
|
267
410
|
['firstTokenMs', options.sloTtftMs],
|
package/src/cli.js
CHANGED
|
@@ -2,6 +2,7 @@ import fs from 'node:fs/promises';
|
|
|
2
2
|
import { createRequire } from 'node:module';
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { runProcess } from './process.js';
|
|
5
|
+
import { createProgress } from './progress.js';
|
|
5
6
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
6
7
|
import { renderBenchmarkReport, renderModelsTable } from './table.js';
|
|
7
8
|
|
|
@@ -36,10 +37,21 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
36
37
|
return;
|
|
37
38
|
}
|
|
38
39
|
|
|
39
|
-
const
|
|
40
|
-
...parsed,
|
|
41
|
-
|
|
40
|
+
const progress = createProgress({
|
|
41
|
+
...renderOptions(parsed),
|
|
42
|
+
enabled: resolveProgress(parsed),
|
|
43
|
+
stream: process.stderr
|
|
42
44
|
});
|
|
45
|
+
let payload;
|
|
46
|
+
try {
|
|
47
|
+
payload = await runBenchmark({
|
|
48
|
+
...parsed,
|
|
49
|
+
version: packageJson.version,
|
|
50
|
+
onProgress: (event) => progress.update(event)
|
|
51
|
+
});
|
|
52
|
+
} finally {
|
|
53
|
+
progress.stop();
|
|
54
|
+
}
|
|
43
55
|
|
|
44
56
|
if (parsed.format === 'json') {
|
|
45
57
|
console.log(JSON.stringify(payload, null, 2));
|
|
@@ -71,6 +83,8 @@ export function parseArgs(argv) {
|
|
|
71
83
|
warmup: 0,
|
|
72
84
|
concurrency: 1,
|
|
73
85
|
sweepConcurrency: [],
|
|
86
|
+
requestRate: null,
|
|
87
|
+
rampUpMs: 0,
|
|
74
88
|
timeoutMs: 60_000,
|
|
75
89
|
profile: 'standard',
|
|
76
90
|
greedy: true,
|
|
@@ -85,6 +99,7 @@ export function parseArgs(argv) {
|
|
|
85
99
|
verbose: false,
|
|
86
100
|
ascii: false,
|
|
87
101
|
color: 'auto',
|
|
102
|
+
progress: 'auto',
|
|
88
103
|
compact: false,
|
|
89
104
|
width: null
|
|
90
105
|
};
|
|
@@ -131,6 +146,12 @@ export function parseArgs(argv) {
|
|
|
131
146
|
options.concurrency = options.sweepConcurrency[0];
|
|
132
147
|
}
|
|
133
148
|
break;
|
|
149
|
+
case '--request-rate':
|
|
150
|
+
options.requestRate = parsePositiveNumber(requireValue(arg, args), arg);
|
|
151
|
+
break;
|
|
152
|
+
case '--ramp-up-ms':
|
|
153
|
+
options.rampUpMs = parseNonNegativeInt(requireValue(arg, args), arg);
|
|
154
|
+
break;
|
|
134
155
|
case '--timeout':
|
|
135
156
|
case '--timeout-ms':
|
|
136
157
|
options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
@@ -200,6 +221,12 @@ export function parseArgs(argv) {
|
|
|
200
221
|
case '--no-color':
|
|
201
222
|
options.color = 'never';
|
|
202
223
|
break;
|
|
224
|
+
case '--progress':
|
|
225
|
+
options.progress = 'always';
|
|
226
|
+
break;
|
|
227
|
+
case '--no-progress':
|
|
228
|
+
options.progress = 'never';
|
|
229
|
+
break;
|
|
203
230
|
case '--compact':
|
|
204
231
|
options.compact = true;
|
|
205
232
|
break;
|
|
@@ -279,6 +306,12 @@ function parsePositiveInt(value, option) {
|
|
|
279
306
|
return parsed;
|
|
280
307
|
}
|
|
281
308
|
|
|
309
|
+
function parsePositiveNumber(value, option) {
|
|
310
|
+
const parsed = Number.parseFloat(value);
|
|
311
|
+
if (!Number.isFinite(parsed) || parsed <= 0) throw new Error(`${option} must be a positive number`);
|
|
312
|
+
return parsed;
|
|
313
|
+
}
|
|
314
|
+
|
|
282
315
|
function parseNonNegativeInt(value, option) {
|
|
283
316
|
const parsed = Number.parseInt(value, 10);
|
|
284
317
|
if (!Number.isInteger(parsed) || parsed < 0) throw new Error(`${option} must be a non-negative integer`);
|
|
@@ -312,6 +345,12 @@ function resolveColor(value) {
|
|
|
312
345
|
return Boolean(process.stdout.isTTY);
|
|
313
346
|
}
|
|
314
347
|
|
|
348
|
+
function resolveProgress(parsed) {
|
|
349
|
+
if (parsed.progress === 'always') return true;
|
|
350
|
+
if (parsed.progress === 'never') return false;
|
|
351
|
+
return parsed.format === 'table' ? 'auto' : false;
|
|
352
|
+
}
|
|
353
|
+
|
|
315
354
|
function helpText() {
|
|
316
355
|
return `fm-bench ${packageJson.version}
|
|
317
356
|
|
|
@@ -329,13 +368,15 @@ Run options:
|
|
|
329
368
|
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
330
369
|
--sweep-concurrency <list>
|
|
331
370
|
Run separate operating points, e.g. 1,2,4
|
|
371
|
+
--request-rate <rps> Pace request starts at a target requests/sec
|
|
372
|
+
--ramp-up-ms <n> Gradually ramp request pacing over n ms
|
|
332
373
|
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
333
374
|
--slo-ttft-ms <n> Count request as good only if TTFT is <= n
|
|
334
375
|
--slo-e2e-ms <n> Count request as good only if E2E latency is <= n
|
|
335
376
|
--slo-tpot-ms <n> Count request as good only if TPOT is <= n
|
|
336
377
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
337
378
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
338
|
-
--profile <name> quick, standard, interactive, throughput, or stress
|
|
379
|
+
--profile <name> quick, standard, interactive, throughput, client, or stress
|
|
339
380
|
-i, --instructions <text> Instructions passed to fm respond
|
|
340
381
|
--use-case <case> Pass a system model use case through to fm
|
|
341
382
|
--guardrails <level> Pass a system model guardrail level through to fm
|
|
@@ -354,6 +395,8 @@ Output:
|
|
|
354
395
|
--ascii Use plain ASCII tables instead of Unicode
|
|
355
396
|
--color Force ANSI colors in table output
|
|
356
397
|
--no-color Disable ANSI colors in table output
|
|
398
|
+
--progress Force live progress on stderr
|
|
399
|
+
--no-progress Disable live progress on stderr
|
|
357
400
|
--compact Force compact terminal layout
|
|
358
401
|
--width <n> Render for a specific terminal width
|
|
359
402
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
@@ -367,6 +410,7 @@ Environment:
|
|
|
367
410
|
Examples:
|
|
368
411
|
fm-bench
|
|
369
412
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
413
|
+
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
370
414
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
371
415
|
fm-bench models
|
|
372
416
|
fm-bench doctor
|
package/src/fm.js
CHANGED
|
@@ -176,7 +176,8 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
176
176
|
durationMs: result.durationMs,
|
|
177
177
|
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
178
178
|
streamed,
|
|
179
|
-
stdoutChunks: result.stdoutChunks
|
|
179
|
+
stdoutChunks: result.stdoutChunks,
|
|
180
|
+
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
|
|
180
181
|
};
|
|
181
182
|
}
|
|
182
183
|
|
package/src/process.js
CHANGED
|
@@ -20,6 +20,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
20
20
|
let stderr = '';
|
|
21
21
|
let stdoutChunks = 0;
|
|
22
22
|
let stderrChunks = 0;
|
|
23
|
+
const stdoutChunkTimesMs = [];
|
|
23
24
|
let firstStdoutMs = null;
|
|
24
25
|
let firstStderrMs = null;
|
|
25
26
|
let timedOut = false;
|
|
@@ -38,10 +39,12 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
38
39
|
child.stdout.setEncoding('utf8');
|
|
39
40
|
child.stderr.setEncoding('utf8');
|
|
40
41
|
child.stdout.on('data', (chunk) => {
|
|
42
|
+
const chunkAtMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
41
43
|
stdoutChunks += 1;
|
|
42
44
|
if (firstStdoutMs == null && chunk.length > 0) {
|
|
43
|
-
firstStdoutMs =
|
|
45
|
+
firstStdoutMs = chunkAtMs;
|
|
44
46
|
}
|
|
47
|
+
if (chunk.length > 0) stdoutChunkTimesMs.push(chunkAtMs);
|
|
45
48
|
stdout += chunk;
|
|
46
49
|
});
|
|
47
50
|
child.stderr.on('data', (chunk) => {
|
|
@@ -65,6 +68,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
65
68
|
stderr: stderr || error.message,
|
|
66
69
|
stdoutChunks,
|
|
67
70
|
stderrChunks,
|
|
71
|
+
stdoutChunkTimesMs,
|
|
68
72
|
firstStdoutMs,
|
|
69
73
|
firstStderrMs,
|
|
70
74
|
error,
|
|
@@ -86,6 +90,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
86
90
|
stderr,
|
|
87
91
|
stdoutChunks,
|
|
88
92
|
stderrChunks,
|
|
93
|
+
stdoutChunkTimesMs,
|
|
89
94
|
firstStdoutMs,
|
|
90
95
|
firstStderrMs,
|
|
91
96
|
timedOut,
|
package/src/progress.js
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
import { formatMs } from './table.js';
|
|
2
|
+
|
|
3
|
+
const UNICODE_FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'];
|
|
4
|
+
const ASCII_FRAMES = ['-', '\\', '|', '/'];
|
|
5
|
+
|
|
6
|
+
const TONES = {
|
|
7
|
+
green: ['\x1b[32m', '\x1b[0m'],
|
|
8
|
+
yellow: ['\x1b[33m', '\x1b[0m'],
|
|
9
|
+
red: ['\x1b[31m', '\x1b[0m'],
|
|
10
|
+
dim: ['\x1b[2m', '\x1b[0m']
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
export function createProgress(options = {}) {
|
|
14
|
+
const stream = options.stream || process.stderr;
|
|
15
|
+
const enabled = options.enabled === true || (options.enabled === 'auto' && Boolean(stream.isTTY));
|
|
16
|
+
if (!enabled) return noopProgress();
|
|
17
|
+
return new StatusLine({ ...options, stream });
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
function noopProgress() {
|
|
21
|
+
return {
|
|
22
|
+
update() {},
|
|
23
|
+
stop() {}
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
class StatusLine {
|
|
28
|
+
constructor(options = {}) {
|
|
29
|
+
this.stream = options.stream || process.stderr;
|
|
30
|
+
this.color = Boolean(options.color);
|
|
31
|
+
this.ascii = Boolean(options.ascii);
|
|
32
|
+
this.frames = this.ascii ? ASCII_FRAMES : UNICODE_FRAMES;
|
|
33
|
+
this.frameIndex = 0;
|
|
34
|
+
this.startedAt = Date.now();
|
|
35
|
+
this.state = {
|
|
36
|
+
phase: 'starting',
|
|
37
|
+
message: 'starting',
|
|
38
|
+
completed: 0,
|
|
39
|
+
failed: 0,
|
|
40
|
+
total: null
|
|
41
|
+
};
|
|
42
|
+
this.timer = setInterval(() => this.render(), 90);
|
|
43
|
+
this.timer.unref?.();
|
|
44
|
+
this.render();
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
update(event = {}) {
|
|
48
|
+
if (event.type === 'phase') {
|
|
49
|
+
this.state.phase = event.phase || 'working';
|
|
50
|
+
this.state.message = event.message || this.state.message;
|
|
51
|
+
} else if (event.type === 'tokens:start') {
|
|
52
|
+
this.state.phase = 'tokens';
|
|
53
|
+
this.state.message = event.message || 'counting prompt tokens';
|
|
54
|
+
this.state.completed = 0;
|
|
55
|
+
this.state.total = event.total;
|
|
56
|
+
} else if (event.type === 'tokens:progress') {
|
|
57
|
+
this.state.phase = 'tokens';
|
|
58
|
+
this.state.message = `counted ${event.promptId || 'prompt'}`;
|
|
59
|
+
this.state.completed = event.completed;
|
|
60
|
+
this.state.total = event.total;
|
|
61
|
+
} else if (event.type === 'benchmark:start') {
|
|
62
|
+
this.state.phase = 'benchmark';
|
|
63
|
+
this.state.message = `running ${event.modelCount} model(s), ${event.promptCount} prompt(s)`;
|
|
64
|
+
this.state.completed = 0;
|
|
65
|
+
this.state.failed = 0;
|
|
66
|
+
this.state.total = event.total;
|
|
67
|
+
} else if (event.type === 'warmup:start') {
|
|
68
|
+
this.state.phase = 'warmup';
|
|
69
|
+
this.state.message = `warming c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
|
|
70
|
+
this.state.completed = 0;
|
|
71
|
+
this.state.total = event.total;
|
|
72
|
+
} else if (event.type === 'warmup:progress') {
|
|
73
|
+
this.state.phase = 'warmup';
|
|
74
|
+
this.state.message = `warmed ${event.model || 'model'} at c${event.concurrency}`;
|
|
75
|
+
this.state.completed = event.completed;
|
|
76
|
+
this.state.total = event.total;
|
|
77
|
+
} else if (event.type === 'scenario:start') {
|
|
78
|
+
this.state.phase = 'benchmark';
|
|
79
|
+
this.state.message = `measuring c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
|
|
80
|
+
this.state.total = event.total ?? this.state.total;
|
|
81
|
+
} else if (event.type === 'benchmark:progress') {
|
|
82
|
+
this.state.phase = 'benchmark';
|
|
83
|
+
this.state.message = `${event.model} ${event.promptId} run ${event.run} ${event.ok ? 'ok' : 'failed'} (${formatMs(event.durationMs)})`;
|
|
84
|
+
this.state.completed = event.completed;
|
|
85
|
+
this.state.failed = event.failed;
|
|
86
|
+
this.state.total = event.total;
|
|
87
|
+
} else if (event.type === 'benchmark:complete') {
|
|
88
|
+
this.state.phase = 'complete';
|
|
89
|
+
this.state.message = `complete ${event.completed}/${event.total}`;
|
|
90
|
+
this.state.completed = event.completed;
|
|
91
|
+
this.state.failed = event.failed;
|
|
92
|
+
this.state.total = event.total;
|
|
93
|
+
}
|
|
94
|
+
this.render();
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
stop(finalMessage = '') {
|
|
98
|
+
if (this.timer) clearInterval(this.timer);
|
|
99
|
+
this.timer = null;
|
|
100
|
+
if (this.stream.clearLine && this.stream.cursorTo) {
|
|
101
|
+
this.clearLine();
|
|
102
|
+
} else {
|
|
103
|
+
this.stream.write('\n');
|
|
104
|
+
}
|
|
105
|
+
if (finalMessage) this.stream.write(`${finalMessage}\n`);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
render() {
|
|
109
|
+
const width = this.stream.columns || process.stderr.columns || 100;
|
|
110
|
+
const elapsed = Date.now() - this.startedAt;
|
|
111
|
+
const frame = this.frames[this.frameIndex % this.frames.length];
|
|
112
|
+
this.frameIndex += 1;
|
|
113
|
+
const progress = progressText(this.state);
|
|
114
|
+
const eta = etaText(this.state, elapsed);
|
|
115
|
+
const failures = this.state.failed > 0 ? this.tone(`fail ${this.state.failed}`, 'red') : '';
|
|
116
|
+
const parts = [
|
|
117
|
+
this.tone(frame, 'green'),
|
|
118
|
+
'fm-bench',
|
|
119
|
+
this.state.phase,
|
|
120
|
+
progress,
|
|
121
|
+
eta,
|
|
122
|
+
failures,
|
|
123
|
+
this.tone(this.state.message, 'dim')
|
|
124
|
+
].filter(Boolean);
|
|
125
|
+
this.writeLine(truncateVisible(parts.join(' '), Math.max(20, width - 1)));
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
writeLine(line) {
|
|
129
|
+
if (this.stream.clearLine && this.stream.cursorTo) {
|
|
130
|
+
this.stream.clearLine(0);
|
|
131
|
+
this.stream.cursorTo(0);
|
|
132
|
+
this.stream.write(line);
|
|
133
|
+
} else {
|
|
134
|
+
this.stream.write(`\r${line}`);
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
clearLine() {
|
|
139
|
+
if (this.stream.clearLine && this.stream.cursorTo) {
|
|
140
|
+
this.stream.clearLine(0);
|
|
141
|
+
this.stream.cursorTo(0);
|
|
142
|
+
} else {
|
|
143
|
+
this.stream.write('\r');
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
tone(text, tone) {
|
|
148
|
+
if (!this.color || !TONES[tone]) return text;
|
|
149
|
+
const [open, close] = TONES[tone];
|
|
150
|
+
return `${open}${text}${close}`;
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function progressText(state) {
|
|
155
|
+
if (!Number.isFinite(state.total) || state.total <= 0) return '';
|
|
156
|
+
const completed = Math.min(state.completed || 0, state.total);
|
|
157
|
+
const percent = Math.round((completed / state.total) * 100);
|
|
158
|
+
return `${completed}/${state.total} ${percent}%`;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function etaText(state, elapsedMs) {
|
|
162
|
+
if (!Number.isFinite(state.total) || state.total <= 0 || !Number.isFinite(state.completed) || state.completed <= 0) {
|
|
163
|
+
return '';
|
|
164
|
+
}
|
|
165
|
+
const remaining = Math.max(0, state.total - state.completed);
|
|
166
|
+
if (remaining === 0) return `elapsed ${formatMs(elapsedMs)}`;
|
|
167
|
+
const perItemMs = elapsedMs / state.completed;
|
|
168
|
+
return `eta ${formatMs(perItemMs * remaining)}`;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
function truncateVisible(value, width) {
|
|
172
|
+
const text = String(value);
|
|
173
|
+
const clean = text.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, '');
|
|
174
|
+
if (clean.length <= width) return text;
|
|
175
|
+
let visible = 0;
|
|
176
|
+
let output = '';
|
|
177
|
+
for (let index = 0; index < text.length && visible < width - 1; index += 1) {
|
|
178
|
+
if (text[index] === '\x1b') {
|
|
179
|
+
const match = text.slice(index).match(/^\u001b\[[0-?]*[ -/]*[@-~]/);
|
|
180
|
+
if (match) {
|
|
181
|
+
output += match[0];
|
|
182
|
+
index += match[0].length - 1;
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
output += text[index];
|
|
187
|
+
visible += 1;
|
|
188
|
+
}
|
|
189
|
+
return `${output}…`;
|
|
190
|
+
}
|
package/src/prompts.js
CHANGED
|
@@ -50,6 +50,28 @@ const PROFILES = {
|
|
|
50
50
|
prompt: 'Create a compact test plan for benchmarking a local foundation model across short, medium, and long prompts.'
|
|
51
51
|
}
|
|
52
52
|
],
|
|
53
|
+
client: [
|
|
54
|
+
{
|
|
55
|
+
id: 'short-chat',
|
|
56
|
+
prompt: 'In one sentence, explain why time to first token matters for an interactive assistant.'
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
id: 'content-generation',
|
|
60
|
+
prompt: 'Write a practical 180-word product update for developers explaining a new terminal benchmark feature. Keep it specific and avoid marketing fluff.'
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
id: 'structured-extraction',
|
|
64
|
+
prompt: 'Return compact valid JSON with keys "risk", "owner", "deadline", and "next_step" from this note: The benchmark release is blocked by flaky p95 latency on the PCC model. Maya owns the investigation and needs a fix before Friday.'
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
id: 'summarization-light',
|
|
68
|
+
prompt: 'Summarize this in three bullets: A serious local LLM benchmark should separate time to first token from total latency, report tail percentiles, include prompt and output token counts, measure throughput at multiple concurrency operating points, and preserve the raw prompt suite so future runs are comparable.'
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
id: 'code-analysis',
|
|
72
|
+
prompt: 'Review this JavaScript function for one correctness issue and one readability improvement: function p(v){let s=0;for(let i=0;i<=v.length;i++)s+=v[i];return s/v.length}'
|
|
73
|
+
}
|
|
74
|
+
],
|
|
53
75
|
stress: [
|
|
54
76
|
{
|
|
55
77
|
id: 'interactive-short',
|
package/src/report.js
CHANGED
|
@@ -27,9 +27,13 @@ export function flattenResults(results) {
|
|
|
27
27
|
words: result.words,
|
|
28
28
|
tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
|
|
29
29
|
decode_tokens_per_second: result.decodeTokensPerSecond == null ? '' : round(result.decodeTokensPerSecond),
|
|
30
|
+
prefill_tokens_per_second: result.prefillTokensPerSecond == null ? '' : round(result.prefillTokensPerSecond),
|
|
30
31
|
chars_per_second: round(result.charsPerSecond),
|
|
31
32
|
streamed: result.streamed,
|
|
32
33
|
stdout_chunks: result.stdoutChunks,
|
|
34
|
+
second_chunk_ms: round(result.secondChunkMs),
|
|
35
|
+
chunk_gap_avg_ms: round(result.chunkGapAvgMs),
|
|
36
|
+
chunk_gap_max_ms: round(result.chunkGapMaxMs),
|
|
33
37
|
output_hash: result.outputHash || '',
|
|
34
38
|
good: result.good == null ? '' : result.good,
|
|
35
39
|
error: result.error || ''
|
package/src/stats.js
CHANGED
|
@@ -101,10 +101,15 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
101
101
|
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
102
102
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
103
103
|
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
104
|
+
const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
|
|
105
|
+
const secondChunk = summarizeNumbers(successes.map((result) => result.secondChunkMs).filter((value) => value != null));
|
|
106
|
+
const chunkGap = summarizeNumbers(successes.flatMap((result) => result.chunkGapsMs || []));
|
|
104
107
|
const windowMs = modelWindowMs(successes);
|
|
105
108
|
const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
|
|
106
|
-
const goodputRps =
|
|
109
|
+
const goodputRps = goodMeasured.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
|
|
107
110
|
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
111
|
+
const totalTokens = promptTokens.sum + outputTokens.sum;
|
|
112
|
+
const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
|
|
108
113
|
|
|
109
114
|
return {
|
|
110
115
|
model: entry.model,
|
|
@@ -120,6 +125,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
120
125
|
rps,
|
|
121
126
|
goodputRps,
|
|
122
127
|
outputTokenThroughput,
|
|
128
|
+
totalTokenThroughput,
|
|
123
129
|
repeatability: summarizeRepeatability(successes),
|
|
124
130
|
latency,
|
|
125
131
|
ttft,
|
|
@@ -129,7 +135,10 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
129
135
|
outputTokens,
|
|
130
136
|
charsPerSecond,
|
|
131
137
|
tokensPerSecond,
|
|
132
|
-
decodeTokensPerSecond
|
|
138
|
+
decodeTokensPerSecond,
|
|
139
|
+
prefillTokensPerSecond,
|
|
140
|
+
secondChunk,
|
|
141
|
+
chunkGap
|
|
133
142
|
};
|
|
134
143
|
});
|
|
135
144
|
}
|
package/src/table.js
CHANGED
|
@@ -93,6 +93,7 @@ export function renderSummaryTable(summary, options = {}) {
|
|
|
93
93
|
cell(item.attempted ? `${item.successes}/${item.attempted}` : '-', item.failures > 0 ? 'yellow' : item.available ? 'green' : 'muted'),
|
|
94
94
|
cell(formatPercent(item.successRate), percentTone(item.successRate, 0.95, 1)),
|
|
95
95
|
cell(formatPercent(item.goodputRate), percentTone(item.goodputRate, 0.8, 1)),
|
|
96
|
+
cell(formatNumber(item.goodputRps), tones.goodputRps),
|
|
96
97
|
cell(formatMs(item.ttft.p50), tones.ttft),
|
|
97
98
|
cell(formatMs(item.ttft.p95), tones.ttftP95),
|
|
98
99
|
cell(formatMs(item.latency.p50), tones.e2e),
|
|
@@ -111,34 +112,34 @@ export function renderSummaryTable(summary, options = {}) {
|
|
|
111
112
|
base[2],
|
|
112
113
|
base[3]
|
|
113
114
|
];
|
|
114
|
-
if (hasGoodput) medium.push(base[5]);
|
|
115
|
-
medium.push(base[
|
|
115
|
+
if (hasGoodput) medium.push(base[5], base[6]);
|
|
116
|
+
medium.push(base[7], base[9], base[10], base[11], base[12], base[14], base[15]);
|
|
116
117
|
return medium;
|
|
117
118
|
}
|
|
118
119
|
|
|
119
120
|
const wide = [base[0], base[1], base[2], base[3], base[4]];
|
|
120
|
-
if (hasGoodput) wide.push(base[5]);
|
|
121
|
+
if (hasGoodput) wide.push(base[5], base[6]);
|
|
121
122
|
wide.push(
|
|
122
|
-
base[6],
|
|
123
123
|
base[7],
|
|
124
124
|
base[8],
|
|
125
125
|
base[9],
|
|
126
|
+
base[10],
|
|
126
127
|
cell(formatMs(item.tpot.p50), tones.tpot),
|
|
127
128
|
cell(formatMs(item.tpot.p95), tones.tpotP95),
|
|
128
|
-
base[10],
|
|
129
129
|
base[11],
|
|
130
130
|
base[12],
|
|
131
131
|
base[13],
|
|
132
|
-
base[14]
|
|
132
|
+
base[14],
|
|
133
|
+
base[15]
|
|
133
134
|
);
|
|
134
135
|
return wide;
|
|
135
136
|
});
|
|
136
137
|
|
|
137
|
-
const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
|
|
138
|
-
const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
|
|
138
|
+
const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'good rps', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
|
|
139
|
+
const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'good rps', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
|
|
139
140
|
const headers = mode === 'medium'
|
|
140
|
-
? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
|
|
141
|
-
: (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
|
|
141
|
+
? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good' && header !== 'good rps'))
|
|
142
|
+
: (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good' && header !== 'good rps'));
|
|
142
143
|
|
|
143
144
|
return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
|
|
144
145
|
}
|
|
@@ -153,7 +154,10 @@ export function renderDetailTable(summary, options = {}) {
|
|
|
153
154
|
cell(item.model, item.available ? null : 'muted'),
|
|
154
155
|
cell(formatNumber(item.promptTokens.avg, 0)),
|
|
155
156
|
cell(formatNumber(item.outputTokens.avg, 0)),
|
|
156
|
-
cell(formatNumber(item.
|
|
157
|
+
cell(formatNumber(item.prefillTokensPerSecond?.avg), tones.prefillTps),
|
|
158
|
+
cell(formatNumber(item.decodeTokensPerSecond?.avg), tones.decodeTps),
|
|
159
|
+
cell(formatMs(item.secondChunk?.p50), tones.secondChunk),
|
|
160
|
+
cell(formatMs(item.chunkGap?.p95), tones.chunkGapP95),
|
|
157
161
|
cell(formatMs(item.latency.p99), tones.e2eP99),
|
|
158
162
|
cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
|
|
159
163
|
cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
|
|
@@ -161,20 +165,23 @@ export function renderDetailTable(summary, options = {}) {
|
|
|
161
165
|
];
|
|
162
166
|
|
|
163
167
|
if (mode === 'medium') {
|
|
164
|
-
return base
|
|
168
|
+
return [base[0], base[1], base[2], base[3], base[4], base[5], base[7], base[8], base[10]];
|
|
165
169
|
}
|
|
166
170
|
|
|
167
171
|
return base;
|
|
168
172
|
});
|
|
169
173
|
|
|
170
174
|
const headers = mode === 'medium'
|
|
171
|
-
? ['c', 'model', 'in avg', 'out avg', 'decode
|
|
175
|
+
? ['c', 'model', 'in avg', 'out avg', 'prefill/s', 'decode/s', 'chunk p95', 'e2e p99', 'repeat']
|
|
172
176
|
: [
|
|
173
177
|
'c',
|
|
174
178
|
'model',
|
|
175
179
|
'in tok avg',
|
|
176
180
|
'out tok avg',
|
|
181
|
+
'prefill tok/s',
|
|
177
182
|
'decode tok/s',
|
|
183
|
+
'2nd chunk',
|
|
184
|
+
'chunk p95',
|
|
178
185
|
'e2e p99',
|
|
179
186
|
'e2e 95% ci',
|
|
180
187
|
'repeat',
|
|
@@ -200,7 +207,8 @@ export function renderCompactSummary(summary, options = {}) {
|
|
|
200
207
|
const tones = metricTones(item, ranks, options.slo);
|
|
201
208
|
lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
|
|
202
209
|
lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
|
|
203
|
-
lines.push(toneText(truncate(`
|
|
210
|
+
lines.push(toneText(truncate(` prefill ${formatNumber(item.prefillTokensPerSecond?.avg)} tok/s | TPOT ${formatMs(item.tpot.p50)} | chunk p95 ${formatMs(item.chunkGap?.p95)}`, width), worstTone(tones.prefillTps, tones.tpot, tones.chunkGapP95), options));
|
|
211
|
+
lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | repeat ${formatPercent(item.repeatability)}`, width), percentTone(item.repeatability, 0.5, 0.9), options));
|
|
204
212
|
} else {
|
|
205
213
|
lines.push(toneText(truncate(` ${compactReason(item.skippedReason)}`, width), 'yellow', options));
|
|
206
214
|
}
|
|
@@ -396,10 +404,15 @@ function rankSummary(summary) {
|
|
|
396
404
|
e2eP99: collectMetric(summary, (item) => item.latency?.p99),
|
|
397
405
|
tpot: collectMetric(summary, (item) => item.tpot?.p50),
|
|
398
406
|
tpotP95: collectMetric(summary, (item) => item.tpot?.p95),
|
|
407
|
+
secondChunk: collectMetric(summary, (item) => item.secondChunk?.p50),
|
|
408
|
+
chunkGapP95: collectMetric(summary, (item) => item.chunkGap?.p95),
|
|
399
409
|
userTps: collectMetric(summary, (item) => item.tokensPerSecond?.avg),
|
|
400
410
|
systemTps: collectMetric(summary, (item) => item.outputTokenThroughput),
|
|
411
|
+
totalTps: collectMetric(summary, (item) => item.totalTokenThroughput),
|
|
401
412
|
rps: collectMetric(summary, (item) => item.rps),
|
|
402
|
-
|
|
413
|
+
goodputRps: collectMetric(summary, (item) => item.goodputRps),
|
|
414
|
+
decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg),
|
|
415
|
+
prefillTps: collectMetric(summary, (item) => item.prefillTokensPerSecond?.avg)
|
|
403
416
|
};
|
|
404
417
|
}
|
|
405
418
|
|
|
@@ -440,10 +453,15 @@ function metricTones(item, ranks, slo = {}) {
|
|
|
440
453
|
e2eP99: thresholdTone(item.latency?.p99, slo.e2eMs) || rankToneLower(item.latency?.p99, ranks.e2eP99),
|
|
441
454
|
tpot: thresholdTone(item.tpot?.p50, slo.tpotMs) || rankToneLower(item.tpot?.p50, ranks.tpot),
|
|
442
455
|
tpotP95: thresholdTone(item.tpot?.p95, slo.tpotMs) || rankToneLower(item.tpot?.p95, ranks.tpotP95),
|
|
456
|
+
secondChunk: rankToneLower(item.secondChunk?.p50, ranks.secondChunk),
|
|
457
|
+
chunkGapP95: rankToneLower(item.chunkGap?.p95, ranks.chunkGapP95),
|
|
443
458
|
userTps: rankTone(item.tokensPerSecond?.avg, ranks.userTps),
|
|
444
459
|
systemTps: rankTone(item.outputTokenThroughput, ranks.systemTps),
|
|
460
|
+
totalTps: rankTone(item.totalTokenThroughput, ranks.totalTps),
|
|
445
461
|
rps: rankTone(item.rps, ranks.rps),
|
|
446
|
-
|
|
462
|
+
goodputRps: goodputRpsTone(item, ranks.goodputRps),
|
|
463
|
+
decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps),
|
|
464
|
+
prefillTps: rankTone(item.prefillTokensPerSecond?.avg, ranks.prefillTps)
|
|
447
465
|
};
|
|
448
466
|
}
|
|
449
467
|
|
|
@@ -453,3 +471,9 @@ function worstTone(...tones) {
|
|
|
453
471
|
if (tones.includes('green')) return 'green';
|
|
454
472
|
return null;
|
|
455
473
|
}
|
|
474
|
+
|
|
475
|
+
function goodputRpsTone(item, values) {
|
|
476
|
+
const passTone = percentTone(item.goodputRate, 0.8, 1);
|
|
477
|
+
if (passTone && passTone !== 'green') return passTone;
|
|
478
|
+
return rankTone(item.goodputRps, values) || passTone;
|
|
479
|
+
}
|