fm-bench 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/docs/methodology.md +15 -1
- package/package.json +1 -1
- package/src/bench.js +150 -7
- package/src/cli.js +66 -4
- package/src/fm.js +2 -1
- package/src/process.js +6 -1
- package/src/progress.js +190 -0
- package/src/prompts.js +22 -0
- package/src/report.js +4 -0
- package/src/stats.js +11 -2
- package/src/table.js +225 -56
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
`fm-bench` is a dynamic benchmark CLI for Apple's `fm` command on macOS 27 and newer.
|
|
4
4
|
|
|
5
|
-
It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, and prints
|
|
5
|
+
It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, shows live progress while it works, and prints terminal tables with latency, throughput, stability, goodput, and streaming-quality stats.
|
|
6
6
|
|
|
7
7
|
Apple introduced the preinstalled `fm` command for macOS 27 as part of the Foundation Models tooling. `fm-bench` intentionally shells out to the system `fm` binary instead of linking private APIs, so it can adapt as Apple adds models or changes availability.
|
|
8
8
|
|
|
@@ -35,20 +35,15 @@ fm-bench
|
|
|
35
35
|
Example output:
|
|
36
36
|
|
|
37
37
|
```text
|
|
38
|
-
fm-bench 0.
|
|
39
|
-
prompts
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
│ MODEL │ STATUS │ RUNS │
|
|
43
|
-
|
|
44
|
-
│ system │ ok │
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
|
|
48
|
-
│ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
|
|
49
|
-
├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
|
|
50
|
-
│ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
|
|
51
|
-
└────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
|
|
38
|
+
fm-bench 0.4.0 | darwin/arm64 | fm
|
|
39
|
+
prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
|
|
40
|
+
|
|
41
|
+
┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
|
|
42
|
+
│ C │ MODEL │ STATUS │ OK/RUNS │ SUCC │ GOOD │ GOOD RPS │ TTFT │ E2E P95 │ SYS │ CV │
|
|
43
|
+
├───┼────────┼────────┼─────────┼──────┼──────┼──────────┼──────┼──────────┼─────┼─────┤
|
|
44
|
+
│ 1 │ system │ ok │ 15/15 │ 100% │ 93% │ 0.4 │ 318ms│ 3.20s │ 42 │ 12% │
|
|
45
|
+
│ 2 │ system │ ok │ 15/15 │ 100% │ 80% │ 0.7 │ 501ms│ 4.40s │ 68 │ 21% │
|
|
46
|
+
└───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
|
|
52
47
|
```
|
|
53
48
|
|
|
54
49
|
## Commands
|
|
@@ -72,6 +67,7 @@ fm-bench --models system,pcc --runs 3 --profile stress
|
|
|
72
67
|
fm-bench --models system --runs 5 --profile interactive
|
|
73
68
|
fm-bench --models system --runs 3 --profile throughput --warmup 1
|
|
74
69
|
fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
|
|
70
|
+
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5 --ramp-up-ms 2000
|
|
75
71
|
fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
76
72
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
77
73
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
@@ -85,9 +81,11 @@ Useful flags:
|
|
|
85
81
|
- `--warmup <n>`: warmup runs per model before measurement.
|
|
86
82
|
- `--concurrency <n>`: parallel `fm` processes.
|
|
87
83
|
- `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
|
|
84
|
+
- `--request-rate <rps>`: pace request starts at a target requests-per-second rate.
|
|
85
|
+
- `--ramp-up-ms <n>`: gradually ramp request pacing over `n` milliseconds.
|
|
88
86
|
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
89
87
|
- `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
|
|
90
|
-
- `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
|
|
88
|
+
- `--profile quick|standard|interactive|throughput|client|stress`: built-in prompt suite.
|
|
91
89
|
- `--prompt <text>`: custom prompt, repeatable.
|
|
92
90
|
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
93
91
|
- `--instructions <text>`: passed to `fm respond`.
|
|
@@ -95,6 +93,8 @@ Useful flags:
|
|
|
95
93
|
- `--capture-output`: include raw model output in JSON reports.
|
|
96
94
|
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
97
95
|
- `--ascii`: use plain ASCII table borders.
|
|
96
|
+
- `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
|
|
97
|
+
- `--progress`, `--no-progress`: force or disable the live progress status line on stderr.
|
|
98
98
|
- `--compact`: force the narrow terminal layout.
|
|
99
99
|
- `--width <n>`: render as if the terminal has `n` columns.
|
|
100
100
|
- `--out <file>`: save a report.
|
|
@@ -126,8 +126,11 @@ Plain text files are split on blank lines.
|
|
|
126
126
|
- TTFT, or time to first streamed output.
|
|
127
127
|
- E2E latency, or full response wall-clock latency.
|
|
128
128
|
- TPOT, or decode time per output token after the first output token.
|
|
129
|
+
- second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
|
|
130
|
+
- prefill tokens per second, or prompt tokens divided by TTFT.
|
|
129
131
|
- output tokens per second per request.
|
|
130
132
|
- total output token throughput across the measured window.
|
|
133
|
+
- total token throughput, including prompt and output tokens.
|
|
131
134
|
- requests per second across the measured window.
|
|
132
135
|
- goodput percentage and goodput RPS when SLO flags are set.
|
|
133
136
|
- coefficient of variation (CV) and confidence interval context for stability.
|
|
@@ -139,10 +142,28 @@ Plain text files are split on blank lines.
|
|
|
139
142
|
|
|
140
143
|
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
|
|
141
144
|
|
|
142
|
-
Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and
|
|
145
|
+
Measured runs stream by default so `fm-bench` can capture TTFT and streaming smoothness. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT, TPOT, second-chunk, and chunk-gap fields that depend on streaming will be blank.
|
|
143
146
|
|
|
144
147
|
Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
|
|
145
148
|
|
|
149
|
+
## Live Progress
|
|
150
|
+
|
|
151
|
+
Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
|
|
152
|
+
|
|
153
|
+
Progress is automatic for table output on TTYs. Use `--progress` to force it or `--no-progress` to keep the terminal completely quiet until the report is ready.
|
|
154
|
+
|
|
155
|
+
## Terminal Colors
|
|
156
|
+
|
|
157
|
+
Table output uses semantic ANSI color on interactive terminals:
|
|
158
|
+
|
|
159
|
+
- green: passing, steadier, or better than the current comparison set.
|
|
160
|
+
- yellow: marginal, partial, or near a budget.
|
|
161
|
+
- red: failing a budget, unstable, or slower/lower than peers.
|
|
162
|
+
|
|
163
|
+
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
164
|
+
|
|
165
|
+
Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
|
|
166
|
+
|
|
146
167
|
See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
|
|
147
168
|
|
|
148
169
|
## Requirements
|
package/docs/methodology.md
CHANGED
|
@@ -9,6 +9,7 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
9
9
|
- Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
|
|
10
10
|
- NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
|
|
11
11
|
- NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
|
|
12
|
+
- NVIDIA AIPerf documents time to second token, inter-token latency, inter-chunk latency, per-user output throughput, and prefill throughput: <https://docs.nvidia.com/aiperf/reference/ai-perf-metrics-reference>
|
|
12
13
|
- vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
|
|
13
14
|
- MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
|
|
14
15
|
- MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
|
|
@@ -19,11 +20,16 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
19
20
|
- `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
|
|
20
21
|
- `generation_ms`: `E2E - TTFT`.
|
|
21
22
|
- `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
|
|
23
|
+
- `second_chunk_ms`: time between the first and second streamed stdout chunks. This is a terminal-side proxy for time-to-second-token style startup smoothness.
|
|
24
|
+
- `chunk_gap`: the distribution of time between consecutive streamed stdout chunks. It is useful for spotting streaming jitter, but it is chunk-based rather than token-based because the `fm` CLI writes stdout chunks, not token timestamp events.
|
|
25
|
+
- `prefill_tokens_per_second`: input prompt tokens divided by TTFT seconds. This estimates prompt-processing speed for streaming runs.
|
|
22
26
|
- `tokens_per_second`: output tokens divided by E2E seconds for one request.
|
|
23
27
|
- `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
|
|
24
28
|
- `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
|
|
29
|
+
- `total token throughput`: successful prompt and output tokens divided by that model's measured wall-clock window.
|
|
25
30
|
- `RPS`: successful requests divided by that model's measured wall-clock window.
|
|
26
31
|
- `goodput`: successful requests that also satisfy all provided SLO thresholds.
|
|
32
|
+
- `goodput RPS`: SLO-passing requests divided by that model's measured wall-clock window. If SLOs are set and no requests pass, this is reported as zero.
|
|
27
33
|
- `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
|
|
28
34
|
- `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
|
|
29
35
|
- `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
|
|
@@ -32,12 +38,20 @@ The metric set follows common LLM inference benchmark practice:
|
|
|
32
38
|
|
|
33
39
|
Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
|
|
34
40
|
|
|
41
|
+
Use `--request-rate <rps>` to pace request starts independently of concurrency. Concurrency limits how many `fm respond` processes can be active at once; request rate controls how quickly new work is admitted. Use `--ramp-up-ms` to avoid instantly shocking a model or quota path when you start a higher-rate run.
|
|
42
|
+
|
|
35
43
|
`fm-bench` does not interpolate between operating points. It reports only what was actually measured.
|
|
36
44
|
|
|
45
|
+
## Prompt Profiles
|
|
46
|
+
|
|
47
|
+
The `client` profile is a pragmatic local-machine mix inspired by MLPerf Client's emphasis on multiple task categories and prompt/response lengths. It includes short chat, content generation, structured extraction, light summarization, and code analysis prompts. It is not a formal MLPerf submission suite; it is a convenient built-in workload for comparing your own Mac, OS build, and `fm` models over time.
|
|
48
|
+
|
|
37
49
|
## Caveats
|
|
38
50
|
|
|
39
51
|
`fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
|
|
40
52
|
|
|
41
53
|
Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
|
|
42
54
|
|
|
43
|
-
|
|
55
|
+
Stream smoothness metrics use stdout chunk arrival times. A chunk can contain more than one token, and terminal or pipe buffering can affect chunk boundaries. Treat `second_chunk_ms` and `chunk_gap` as user-visible streaming diagnostics, not raw decoder telemetry.
|
|
56
|
+
|
|
57
|
+
For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput or client profiles, compare models at the same concurrency operating points, set SLOs that match your real UX budget, and save JSON reports for later analysis.
|
package/package.json
CHANGED
package/src/bench.js
CHANGED
|
@@ -36,7 +36,9 @@ export async function inspectModels(options = {}) {
|
|
|
36
36
|
|
|
37
37
|
export async function runBenchmark(options = {}) {
|
|
38
38
|
const startedAt = new Date().toISOString();
|
|
39
|
+
notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
|
|
39
40
|
const prompts = await loadPrompts(options);
|
|
41
|
+
notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
|
|
40
42
|
const inspection = await inspectModels(options);
|
|
41
43
|
const modelStatuses = options.availableOnly
|
|
42
44
|
? inspection.models.filter((model) => model.available)
|
|
@@ -45,15 +47,36 @@ export async function runBenchmark(options = {}) {
|
|
|
45
47
|
const environment = await collectEnvironment(inspection.fmBin);
|
|
46
48
|
const promptTokenCounts = new Map();
|
|
47
49
|
const concurrencies = normalizeConcurrencySweep(options);
|
|
50
|
+
const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
|
|
48
51
|
|
|
52
|
+
notify(options, {
|
|
53
|
+
type: 'tokens:start',
|
|
54
|
+
total: prompts.length,
|
|
55
|
+
message: 'counting prompt tokens'
|
|
56
|
+
});
|
|
49
57
|
for (const prompt of prompts) {
|
|
50
58
|
const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
|
|
51
59
|
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
60
|
+
notify(options, {
|
|
61
|
+
type: 'tokens:progress',
|
|
62
|
+
completed: promptTokenCounts.size,
|
|
63
|
+
total: prompts.length,
|
|
64
|
+
promptId: prompt.id
|
|
65
|
+
});
|
|
52
66
|
}
|
|
53
67
|
|
|
54
68
|
const results = [];
|
|
55
69
|
const scenarios = [];
|
|
56
|
-
|
|
70
|
+
let completedRuns = 0;
|
|
71
|
+
let failedRuns = 0;
|
|
72
|
+
notify(options, {
|
|
73
|
+
type: 'benchmark:start',
|
|
74
|
+
total: totalRuns,
|
|
75
|
+
modelCount: runnableModels.length,
|
|
76
|
+
promptCount: prompts.length,
|
|
77
|
+
scenarioCount: concurrencies.length
|
|
78
|
+
});
|
|
79
|
+
for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
|
|
57
80
|
const scenario = await runScenario({
|
|
58
81
|
fmBin: inspection.fmBin,
|
|
59
82
|
prompts,
|
|
@@ -61,7 +84,26 @@ export async function runBenchmark(options = {}) {
|
|
|
61
84
|
modelStatuses,
|
|
62
85
|
promptTokenCounts,
|
|
63
86
|
options,
|
|
64
|
-
concurrency
|
|
87
|
+
concurrency,
|
|
88
|
+
scenarioIndex: scenarioIndex + 1,
|
|
89
|
+
scenarioCount: concurrencies.length,
|
|
90
|
+
onMeasuredResult: (result) => {
|
|
91
|
+
completedRuns += 1;
|
|
92
|
+
if (!result.ok) failedRuns += 1;
|
|
93
|
+
notify(options, {
|
|
94
|
+
type: 'benchmark:progress',
|
|
95
|
+
completed: completedRuns,
|
|
96
|
+
failed: failedRuns,
|
|
97
|
+
total: totalRuns,
|
|
98
|
+
concurrency,
|
|
99
|
+
model: result.model,
|
|
100
|
+
promptId: result.promptId,
|
|
101
|
+
run: result.run,
|
|
102
|
+
ok: result.ok,
|
|
103
|
+
durationMs: result.durationMs,
|
|
104
|
+
firstTokenMs: result.firstTokenMs
|
|
105
|
+
});
|
|
106
|
+
}
|
|
65
107
|
});
|
|
66
108
|
scenarios.push(scenario);
|
|
67
109
|
results.push(...scenario.results);
|
|
@@ -73,7 +115,7 @@ export async function runBenchmark(options = {}) {
|
|
|
73
115
|
|| a.run - b.run);
|
|
74
116
|
|
|
75
117
|
const summary = summarizeByModel(results, modelStatuses, { concurrencies });
|
|
76
|
-
|
|
118
|
+
const payload = {
|
|
77
119
|
tool: 'fm-bench',
|
|
78
120
|
version: options.version,
|
|
79
121
|
startedAt,
|
|
@@ -90,6 +132,13 @@ export async function runBenchmark(options = {}) {
|
|
|
90
132
|
summary,
|
|
91
133
|
results
|
|
92
134
|
};
|
|
135
|
+
notify(options, {
|
|
136
|
+
type: 'benchmark:complete',
|
|
137
|
+
completed: completedRuns,
|
|
138
|
+
failed: failedRuns,
|
|
139
|
+
total: totalRuns
|
|
140
|
+
});
|
|
141
|
+
return payload;
|
|
93
142
|
}
|
|
94
143
|
|
|
95
144
|
async function runScenario(context) {
|
|
@@ -100,16 +149,40 @@ async function runScenario(context) {
|
|
|
100
149
|
modelStatuses,
|
|
101
150
|
promptTokenCounts,
|
|
102
151
|
options,
|
|
103
|
-
concurrency
|
|
152
|
+
concurrency,
|
|
153
|
+
scenarioIndex,
|
|
154
|
+
scenarioCount,
|
|
155
|
+
onMeasuredResult
|
|
104
156
|
} = context;
|
|
105
157
|
const startedAt = new Date().toISOString();
|
|
106
158
|
|
|
159
|
+
const warmupTotal = options.warmup * runnableModels.length;
|
|
160
|
+
if (warmupTotal > 0) {
|
|
161
|
+
notify(options, {
|
|
162
|
+
type: 'warmup:start',
|
|
163
|
+
concurrency,
|
|
164
|
+
scenarioIndex,
|
|
165
|
+
scenarioCount,
|
|
166
|
+
total: warmupTotal
|
|
167
|
+
});
|
|
168
|
+
}
|
|
169
|
+
let warmupCompleted = 0;
|
|
107
170
|
for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
|
|
108
171
|
for (const model of runnableModels) {
|
|
109
172
|
await respond(fmBin, model.name, prompts[0].prompt, {
|
|
110
173
|
...options,
|
|
111
174
|
stream: false
|
|
112
175
|
});
|
|
176
|
+
warmupCompleted += 1;
|
|
177
|
+
notify(options, {
|
|
178
|
+
type: 'warmup:progress',
|
|
179
|
+
concurrency,
|
|
180
|
+
scenarioIndex,
|
|
181
|
+
scenarioCount,
|
|
182
|
+
completed: warmupCompleted,
|
|
183
|
+
total: warmupTotal,
|
|
184
|
+
model: model.name
|
|
185
|
+
});
|
|
113
186
|
}
|
|
114
187
|
}
|
|
115
188
|
|
|
@@ -124,15 +197,23 @@ async function runScenario(context) {
|
|
|
124
197
|
}
|
|
125
198
|
|
|
126
199
|
const results = [];
|
|
200
|
+
notify(options, {
|
|
201
|
+
type: 'scenario:start',
|
|
202
|
+
concurrency,
|
|
203
|
+
scenarioIndex,
|
|
204
|
+
scenarioCount,
|
|
205
|
+
total: jobs.length
|
|
206
|
+
});
|
|
127
207
|
await runLimited(jobs, concurrency, async (job) => {
|
|
128
208
|
const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
|
|
129
209
|
results.push(result);
|
|
210
|
+
if (onMeasuredResult) onMeasuredResult(result);
|
|
130
211
|
if (!result.ok && options.failFast) {
|
|
131
212
|
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
132
213
|
error.exitCode = 1;
|
|
133
214
|
throw error;
|
|
134
215
|
}
|
|
135
|
-
});
|
|
216
|
+
}, options);
|
|
136
217
|
|
|
137
218
|
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
138
219
|
|| a.promptId.localeCompare(b.promptId)
|
|
@@ -171,6 +252,12 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
171
252
|
: null;
|
|
172
253
|
const chars = response.output.length;
|
|
173
254
|
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
255
|
+
const promptTokens = promptTokenCounts.get(job.prompt.id);
|
|
256
|
+
const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
|
|
257
|
+
? promptTokens / (firstTokenMs / 1000)
|
|
258
|
+
: null;
|
|
259
|
+
const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
|
|
260
|
+
const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
|
|
174
261
|
|
|
175
262
|
return {
|
|
176
263
|
model: job.model.name,
|
|
@@ -182,17 +269,22 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
182
269
|
firstTokenMs,
|
|
183
270
|
generationMs,
|
|
184
271
|
tpotMs,
|
|
185
|
-
promptTokens
|
|
272
|
+
promptTokens,
|
|
186
273
|
outputTokens: countedOutputTokens,
|
|
187
274
|
chars,
|
|
188
275
|
words,
|
|
189
276
|
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
190
277
|
decodeTokensPerSecond,
|
|
278
|
+
prefillTokensPerSecond,
|
|
191
279
|
charsPerSecond: seconds > 0 ? chars / seconds : 0,
|
|
192
280
|
startOffsetMs,
|
|
193
281
|
endOffsetMs,
|
|
194
282
|
streamed: response.streamed,
|
|
195
283
|
stdoutChunks: response.stdoutChunks,
|
|
284
|
+
secondChunkMs,
|
|
285
|
+
chunkGapsMs,
|
|
286
|
+
chunkGapAvgMs: average(chunkGapsMs),
|
|
287
|
+
chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
|
|
196
288
|
outputHash: response.ok ? hashOutput(response.output) : null,
|
|
197
289
|
good: response.ok ? evaluateSlo({
|
|
198
290
|
firstTokenMs,
|
|
@@ -204,12 +296,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
|
|
|
204
296
|
};
|
|
205
297
|
}
|
|
206
298
|
|
|
207
|
-
async function runLimited(items, concurrency, worker) {
|
|
299
|
+
async function runLimited(items, concurrency, worker, options = {}) {
|
|
208
300
|
let nextIndex = 0;
|
|
301
|
+
const waitForSlot = createPacer(options.requestRate, options.rampUpMs);
|
|
209
302
|
const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
|
|
210
303
|
while (nextIndex < items.length) {
|
|
211
304
|
const index = nextIndex;
|
|
212
305
|
nextIndex += 1;
|
|
306
|
+
await waitForSlot(index);
|
|
213
307
|
await worker(items[index]);
|
|
214
308
|
}
|
|
215
309
|
});
|
|
@@ -233,6 +327,8 @@ function publicOptions(options) {
|
|
|
233
327
|
concurrency: options.concurrency,
|
|
234
328
|
sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
|
|
235
329
|
timeoutMs: options.timeoutMs,
|
|
330
|
+
requestRate: options.requestRate || null,
|
|
331
|
+
rampUpMs: options.rampUpMs || null,
|
|
236
332
|
profile: options.profile,
|
|
237
333
|
promptCount: options.promptCount,
|
|
238
334
|
greedy: options.greedy,
|
|
@@ -262,6 +358,53 @@ function hashOutput(output) {
|
|
|
262
358
|
.slice(0, 16);
|
|
263
359
|
}
|
|
264
360
|
|
|
361
|
+
function chunkGaps(times = []) {
|
|
362
|
+
const gaps = [];
|
|
363
|
+
for (let index = 1; index < times.length; index += 1) {
|
|
364
|
+
gaps.push(Math.max(0, times[index] - times[index - 1]));
|
|
365
|
+
}
|
|
366
|
+
return gaps;
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
function average(values) {
|
|
370
|
+
const clean = values.filter((value) => Number.isFinite(value));
|
|
371
|
+
if (clean.length === 0) return null;
|
|
372
|
+
return clean.reduce((sum, value) => sum + value, 0) / clean.length;
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
function createPacer(requestRate, rampUpMs = 0) {
|
|
376
|
+
if (!Number.isFinite(requestRate) || requestRate <= 0) {
|
|
377
|
+
return async () => {};
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
const startedAt = process.hrtime.bigint();
|
|
381
|
+
const offsets = [];
|
|
382
|
+
const steadyIntervalMs = 1000 / requestRate;
|
|
383
|
+
const warmIntervalMs = steadyIntervalMs * 4;
|
|
384
|
+
|
|
385
|
+
return async (index) => {
|
|
386
|
+
while (offsets.length <= index) {
|
|
387
|
+
const previousOffset = offsets.length === 0 ? 0 : offsets[offsets.length - 1];
|
|
388
|
+
const fraction = rampUpMs > 0 ? Math.min(1, previousOffset / rampUpMs) : 1;
|
|
389
|
+
const interval = warmIntervalMs + ((steadyIntervalMs - warmIntervalMs) * fraction);
|
|
390
|
+
offsets.push(offsets.length === 0 ? 0 : previousOffset + interval);
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
const targetMs = offsets[index];
|
|
394
|
+
const elapsedMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
395
|
+
const waitMs = targetMs - elapsedMs;
|
|
396
|
+
if (waitMs > 0) {
|
|
397
|
+
await new Promise((resolve) => setTimeout(resolve, waitMs));
|
|
398
|
+
}
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
function notify(options, event) {
|
|
403
|
+
if (typeof options.onProgress === 'function') {
|
|
404
|
+
options.onProgress(event);
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
265
408
|
function evaluateSlo(metrics, options) {
|
|
266
409
|
const thresholds = [
|
|
267
410
|
['firstTokenMs', options.sloTtftMs],
|
package/src/cli.js
CHANGED
|
@@ -2,6 +2,7 @@ import fs from 'node:fs/promises';
|
|
|
2
2
|
import { createRequire } from 'node:module';
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { runProcess } from './process.js';
|
|
5
|
+
import { createProgress } from './progress.js';
|
|
5
6
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
6
7
|
import { renderBenchmarkReport, renderModelsTable } from './table.js';
|
|
7
8
|
|
|
@@ -36,10 +37,21 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
36
37
|
return;
|
|
37
38
|
}
|
|
38
39
|
|
|
39
|
-
const
|
|
40
|
-
...parsed,
|
|
41
|
-
|
|
40
|
+
const progress = createProgress({
|
|
41
|
+
...renderOptions(parsed),
|
|
42
|
+
enabled: resolveProgress(parsed),
|
|
43
|
+
stream: process.stderr
|
|
42
44
|
});
|
|
45
|
+
let payload;
|
|
46
|
+
try {
|
|
47
|
+
payload = await runBenchmark({
|
|
48
|
+
...parsed,
|
|
49
|
+
version: packageJson.version,
|
|
50
|
+
onProgress: (event) => progress.update(event)
|
|
51
|
+
});
|
|
52
|
+
} finally {
|
|
53
|
+
progress.stop();
|
|
54
|
+
}
|
|
43
55
|
|
|
44
56
|
if (parsed.format === 'json') {
|
|
45
57
|
console.log(JSON.stringify(payload, null, 2));
|
|
@@ -71,6 +83,8 @@ export function parseArgs(argv) {
|
|
|
71
83
|
warmup: 0,
|
|
72
84
|
concurrency: 1,
|
|
73
85
|
sweepConcurrency: [],
|
|
86
|
+
requestRate: null,
|
|
87
|
+
rampUpMs: 0,
|
|
74
88
|
timeoutMs: 60_000,
|
|
75
89
|
profile: 'standard',
|
|
76
90
|
greedy: true,
|
|
@@ -84,6 +98,8 @@ export function parseArgs(argv) {
|
|
|
84
98
|
failFast: false,
|
|
85
99
|
verbose: false,
|
|
86
100
|
ascii: false,
|
|
101
|
+
color: 'auto',
|
|
102
|
+
progress: 'auto',
|
|
87
103
|
compact: false,
|
|
88
104
|
width: null
|
|
89
105
|
};
|
|
@@ -130,6 +146,12 @@ export function parseArgs(argv) {
|
|
|
130
146
|
options.concurrency = options.sweepConcurrency[0];
|
|
131
147
|
}
|
|
132
148
|
break;
|
|
149
|
+
case '--request-rate':
|
|
150
|
+
options.requestRate = parsePositiveNumber(requireValue(arg, args), arg);
|
|
151
|
+
break;
|
|
152
|
+
case '--ramp-up-ms':
|
|
153
|
+
options.rampUpMs = parseNonNegativeInt(requireValue(arg, args), arg);
|
|
154
|
+
break;
|
|
133
155
|
case '--timeout':
|
|
134
156
|
case '--timeout-ms':
|
|
135
157
|
options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
@@ -193,6 +215,18 @@ export function parseArgs(argv) {
|
|
|
193
215
|
case '--ascii':
|
|
194
216
|
options.ascii = true;
|
|
195
217
|
break;
|
|
218
|
+
case '--color':
|
|
219
|
+
options.color = 'always';
|
|
220
|
+
break;
|
|
221
|
+
case '--no-color':
|
|
222
|
+
options.color = 'never';
|
|
223
|
+
break;
|
|
224
|
+
case '--progress':
|
|
225
|
+
options.progress = 'always';
|
|
226
|
+
break;
|
|
227
|
+
case '--no-progress':
|
|
228
|
+
options.progress = 'never';
|
|
229
|
+
break;
|
|
196
230
|
case '--compact':
|
|
197
231
|
options.compact = true;
|
|
198
232
|
break;
|
|
@@ -272,6 +306,12 @@ function parsePositiveInt(value, option) {
|
|
|
272
306
|
return parsed;
|
|
273
307
|
}
|
|
274
308
|
|
|
309
|
+
function parsePositiveNumber(value, option) {
|
|
310
|
+
const parsed = Number.parseFloat(value);
|
|
311
|
+
if (!Number.isFinite(parsed) || parsed <= 0) throw new Error(`${option} must be a positive number`);
|
|
312
|
+
return parsed;
|
|
313
|
+
}
|
|
314
|
+
|
|
275
315
|
function parseNonNegativeInt(value, option) {
|
|
276
316
|
const parsed = Number.parseInt(value, 10);
|
|
277
317
|
if (!Number.isInteger(parsed) || parsed < 0) throw new Error(`${option} must be a non-negative integer`);
|
|
@@ -291,11 +331,26 @@ function parsePositiveIntList(value, option) {
|
|
|
291
331
|
function renderOptions(parsed) {
|
|
292
332
|
return {
|
|
293
333
|
ascii: parsed.ascii,
|
|
334
|
+
color: resolveColor(parsed.color),
|
|
294
335
|
compact: parsed.compact,
|
|
295
336
|
width: parsed.width
|
|
296
337
|
};
|
|
297
338
|
}
|
|
298
339
|
|
|
340
|
+
function resolveColor(value) {
|
|
341
|
+
if (value === 'always') return true;
|
|
342
|
+
if (value === 'never') return false;
|
|
343
|
+
if (process.env.NO_COLOR) return false;
|
|
344
|
+
if (process.env.FORCE_COLOR && process.env.FORCE_COLOR !== '0') return true;
|
|
345
|
+
return Boolean(process.stdout.isTTY);
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
function resolveProgress(parsed) {
|
|
349
|
+
if (parsed.progress === 'always') return true;
|
|
350
|
+
if (parsed.progress === 'never') return false;
|
|
351
|
+
return parsed.format === 'table' ? 'auto' : false;
|
|
352
|
+
}
|
|
353
|
+
|
|
299
354
|
function helpText() {
|
|
300
355
|
return `fm-bench ${packageJson.version}
|
|
301
356
|
|
|
@@ -313,13 +368,15 @@ Run options:
|
|
|
313
368
|
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
314
369
|
--sweep-concurrency <list>
|
|
315
370
|
Run separate operating points, e.g. 1,2,4
|
|
371
|
+
--request-rate <rps> Pace request starts at a target requests/sec
|
|
372
|
+
--ramp-up-ms <n> Gradually ramp request pacing over n ms
|
|
316
373
|
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
317
374
|
--slo-ttft-ms <n> Count request as good only if TTFT is <= n
|
|
318
375
|
--slo-e2e-ms <n> Count request as good only if E2E latency is <= n
|
|
319
376
|
--slo-tpot-ms <n> Count request as good only if TPOT is <= n
|
|
320
377
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
321
378
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
322
|
-
--profile <name> quick, standard, interactive, throughput, or stress
|
|
379
|
+
--profile <name> quick, standard, interactive, throughput, client, or stress
|
|
323
380
|
-i, --instructions <text> Instructions passed to fm respond
|
|
324
381
|
--use-case <case> Pass a system model use case through to fm
|
|
325
382
|
--guardrails <level> Pass a system model guardrail level through to fm
|
|
@@ -336,6 +393,10 @@ Output:
|
|
|
336
393
|
--json Alias for --format json
|
|
337
394
|
--csv Alias for --format csv
|
|
338
395
|
--ascii Use plain ASCII tables instead of Unicode
|
|
396
|
+
--color Force ANSI colors in table output
|
|
397
|
+
--no-color Disable ANSI colors in table output
|
|
398
|
+
--progress Force live progress on stderr
|
|
399
|
+
--no-progress Disable live progress on stderr
|
|
339
400
|
--compact Force compact terminal layout
|
|
340
401
|
--width <n> Render for a specific terminal width
|
|
341
402
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
@@ -349,6 +410,7 @@ Environment:
|
|
|
349
410
|
Examples:
|
|
350
411
|
fm-bench
|
|
351
412
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
413
|
+
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
352
414
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
353
415
|
fm-bench models
|
|
354
416
|
fm-bench doctor
|
package/src/fm.js
CHANGED
|
@@ -176,7 +176,8 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
176
176
|
durationMs: result.durationMs,
|
|
177
177
|
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
178
178
|
streamed,
|
|
179
|
-
stdoutChunks: result.stdoutChunks
|
|
179
|
+
stdoutChunks: result.stdoutChunks,
|
|
180
|
+
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
|
|
180
181
|
};
|
|
181
182
|
}
|
|
182
183
|
|
package/src/process.js
CHANGED
|
@@ -20,6 +20,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
20
20
|
let stderr = '';
|
|
21
21
|
let stdoutChunks = 0;
|
|
22
22
|
let stderrChunks = 0;
|
|
23
|
+
const stdoutChunkTimesMs = [];
|
|
23
24
|
let firstStdoutMs = null;
|
|
24
25
|
let firstStderrMs = null;
|
|
25
26
|
let timedOut = false;
|
|
@@ -38,10 +39,12 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
38
39
|
child.stdout.setEncoding('utf8');
|
|
39
40
|
child.stderr.setEncoding('utf8');
|
|
40
41
|
child.stdout.on('data', (chunk) => {
|
|
42
|
+
const chunkAtMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
41
43
|
stdoutChunks += 1;
|
|
42
44
|
if (firstStdoutMs == null && chunk.length > 0) {
|
|
43
|
-
firstStdoutMs =
|
|
45
|
+
firstStdoutMs = chunkAtMs;
|
|
44
46
|
}
|
|
47
|
+
if (chunk.length > 0) stdoutChunkTimesMs.push(chunkAtMs);
|
|
45
48
|
stdout += chunk;
|
|
46
49
|
});
|
|
47
50
|
child.stderr.on('data', (chunk) => {
|
|
@@ -65,6 +68,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
65
68
|
stderr: stderr || error.message,
|
|
66
69
|
stdoutChunks,
|
|
67
70
|
stderrChunks,
|
|
71
|
+
stdoutChunkTimesMs,
|
|
68
72
|
firstStdoutMs,
|
|
69
73
|
firstStderrMs,
|
|
70
74
|
error,
|
|
@@ -86,6 +90,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
86
90
|
stderr,
|
|
87
91
|
stdoutChunks,
|
|
88
92
|
stderrChunks,
|
|
93
|
+
stdoutChunkTimesMs,
|
|
89
94
|
firstStdoutMs,
|
|
90
95
|
firstStderrMs,
|
|
91
96
|
timedOut,
|