fm-bench 0.4.4 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +190 -130
- package/package.json +1 -1
- package/src/bench.js +15 -5
- package/src/cli.js +217 -7
- package/src/compare.js +241 -0
- package/src/history.js +95 -0
- package/src/prompts.js +66 -0
- package/src/table.js +47 -0
package/README.md
CHANGED
|
@@ -1,41 +1,34 @@
|
|
|
1
1
|
# fm-bench
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Benchmark Apple's `fm` command on macOS 27+.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Measure latency, throughput, streaming smoothness, stability, and goodput across Apple Foundation Models — with repeatable prompt suites and JSON/CSV reports for automation.
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
## Why fm-bench exists
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Apple's Foundation Models can run on-device, through Private Cloud Compute, and through model adapters. That makes raw model quality only half the story.
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
npm install -g fm-bench
|
|
13
|
-
```
|
|
11
|
+
For real apps, the important questions are:
|
|
14
12
|
|
|
15
|
-
|
|
13
|
+
- How fast does the first token arrive?
|
|
14
|
+
- Does streaming stay smooth?
|
|
15
|
+
- How stable is latency over repeated runs?
|
|
16
|
+
- What happens under concurrency?
|
|
17
|
+
- Which model/hardware pair meets an interactive SLO?
|
|
16
18
|
|
|
17
|
-
|
|
18
|
-
npm install -g --install-links git+https://github.com/devinoldenburg/fm-bench.git
|
|
19
|
-
```
|
|
20
|
-
|
|
21
|
-
For local development from this repository:
|
|
22
|
-
|
|
23
|
-
```sh
|
|
24
|
-
npm install
|
|
25
|
-
npm link
|
|
26
|
-
fm-bench doctor
|
|
27
|
-
```
|
|
19
|
+
`fm-bench` answers those questions with repeatable local benchmarks. Think of it as GeekBench for Apple Foundation Models — run it, get numbers, compare across hardware, models, and macOS updates.
|
|
28
20
|
|
|
29
21
|
## Quick Start
|
|
30
22
|
|
|
31
23
|
```sh
|
|
24
|
+
npm install -g fm-bench
|
|
32
25
|
fm-bench
|
|
33
26
|
```
|
|
34
27
|
|
|
35
|
-
|
|
28
|
+
One command discovers your models, runs the standard prompt suite, and prints a full benchmark report:
|
|
36
29
|
|
|
37
30
|
```text
|
|
38
|
-
fm-bench 0.
|
|
31
|
+
fm-bench 0.5.0 | darwin/arm64 | fm
|
|
39
32
|
prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
|
|
40
33
|
|
|
41
34
|
┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
|
|
@@ -46,63 +39,173 @@ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skip
|
|
|
46
39
|
└───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
|
|
47
40
|
```
|
|
48
41
|
|
|
49
|
-
|
|
42
|
+
Wide terminals add TTFT P95, TPOT, decode/prefill throughput, chunk-gap smoothness, and 95% CI columns. Narrow terminals switch to compact model cards automatically.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
50
45
|
|
|
51
46
|
```sh
|
|
52
|
-
fm-bench
|
|
53
|
-
fm-bench models [options]
|
|
54
|
-
fm-bench legend [options]
|
|
55
|
-
fm-bench doctor [options]
|
|
47
|
+
npm install -g fm-bench
|
|
56
48
|
```
|
|
57
49
|
|
|
58
|
-
|
|
50
|
+
Install directly from GitHub (always latest):
|
|
59
51
|
|
|
60
|
-
|
|
52
|
+
```sh
|
|
53
|
+
npm install -g --install-links git+https://github.com/devinoldenburg/fm-bench.git
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Local development:
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
npm install && npm link
|
|
60
|
+
fm-bench doctor # verify your setup
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**Requirements:** macOS 27+, Node.js 20+, Apple Intelligence enabled.
|
|
61
64
|
|
|
62
|
-
|
|
65
|
+
## Commands
|
|
63
66
|
|
|
64
|
-
|
|
67
|
+
| Command | What it does |
|
|
68
|
+
|---------|-------------|
|
|
69
|
+
| `fm-bench` | Run the full benchmark (default) |
|
|
70
|
+
| `fm-bench models` | List discovered models, availability, and quota |
|
|
71
|
+
| `fm-bench compare <a.json> <b.json>` | Regression diff: before/after metrics with color-coded deltas |
|
|
72
|
+
| `fm-bench history [dir]` | Chronological trend table from a directory of saved reports |
|
|
73
|
+
| `fm-bench legend` | Definitions for every table column and color rule |
|
|
74
|
+
| `fm-bench doctor` | Environment check: Node, macOS, `fm`, CPU, memory, thermals, battery |
|
|
65
75
|
|
|
66
|
-
##
|
|
76
|
+
## Common Recipes
|
|
67
77
|
|
|
68
78
|
```sh
|
|
79
|
+
# Quick smoke test
|
|
80
|
+
fm-bench --profile quick
|
|
81
|
+
|
|
82
|
+
# Standard 5-run benchmark with SLO budgets
|
|
83
|
+
fm-bench --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
84
|
+
|
|
85
|
+
# Sweep concurrency to find your throughput ceiling
|
|
86
|
+
fm-bench --sweep-concurrency 1,2,4 --runs 3
|
|
87
|
+
|
|
88
|
+
# Stress both on-device and PCC models
|
|
69
89
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
fm-bench --
|
|
73
|
-
fm-bench --profile
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
fm-bench --
|
|
77
|
-
fm-bench --
|
|
78
|
-
fm-bench
|
|
79
|
-
|
|
90
|
+
|
|
91
|
+
# Reasoning and coding workloads
|
|
92
|
+
fm-bench --profile reasoning --runs 5
|
|
93
|
+
fm-bench --profile coding --runs 3 --histogram
|
|
94
|
+
|
|
95
|
+
# Archive runs and compare before/after a macOS update
|
|
96
|
+
fm-bench --output-dir reports/ --tag before-update
|
|
97
|
+
fm-bench --output-dir reports/ --tag after-update
|
|
98
|
+
fm-bench compare reports/fm-bench_*before*.json reports/fm-bench_*after*.json
|
|
99
|
+
|
|
100
|
+
# Fail CI when SLOs regress
|
|
101
|
+
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
|
|
102
|
+
|
|
103
|
+
# Save JSON for automation
|
|
104
|
+
fm-bench --json --out bench.json
|
|
105
|
+
fm-bench --format csv --out bench.csv
|
|
80
106
|
```
|
|
81
107
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
108
|
+
## Options Reference
|
|
109
|
+
|
|
110
|
+
**Workload**
|
|
111
|
+
|
|
112
|
+
| Flag | Default | Description |
|
|
113
|
+
|------|---------|-------------|
|
|
114
|
+
| `-m, --models <list>` | all | Comma-separated or repeated model names |
|
|
115
|
+
| `-r, --runs <n>` | 1 | Measured runs per prompt/model |
|
|
116
|
+
| `--warmup <n>` | 0 | Warmup runs per model before measurement |
|
|
117
|
+
| `-c, --concurrency <n>` | 1 | Parallel `fm` processes |
|
|
118
|
+
| `--sweep-concurrency <list>` | — | Separate operating points, e.g. `1,2,4` |
|
|
119
|
+
| `--request-rate <rps>` | — | Pace request starts at a target rate |
|
|
120
|
+
| `--ramp-up-ms <n>` | 0 | Gradually ramp pacing over `n` ms |
|
|
121
|
+
| `--timeout-ms <n>` | 60000 | Timeout per `fm` call |
|
|
122
|
+
| `--retry <n>` | 0 | Retry failed calls with exponential backoff (500ms–4s) |
|
|
123
|
+
| `--profile <name>` | standard | Built-in prompt suite (see Profiles) |
|
|
124
|
+
| `-p, --prompt <text>` | — | Custom prompt, repeatable |
|
|
125
|
+
| `--prompt-file <file>` | — | JSON, JSONL, or blank-line separated prompts |
|
|
126
|
+
| `-i, --instructions <text>` | — | Passed to `fm respond` |
|
|
127
|
+
|
|
128
|
+
**Quality Gates**
|
|
129
|
+
|
|
130
|
+
| Flag | Description |
|
|
131
|
+
|------|-------------|
|
|
132
|
+
| `--slo-ttft-ms <n>` | Count a run as good only if TTFT ≤ n ms |
|
|
133
|
+
| `--slo-e2e-ms <n>` | Count a run as good only if E2E latency ≤ n ms |
|
|
134
|
+
| `--slo-tpot-ms <n>` | Count a run as good only if TPOT ≤ n ms |
|
|
135
|
+
| `--ci` | Exit 1 if any run fails or any SLO is violated (for pipelines) |
|
|
136
|
+
| `--fail-fast` | Stop after the first failed run |
|
|
137
|
+
|
|
138
|
+
**Output**
|
|
139
|
+
|
|
140
|
+
| Flag | Description |
|
|
141
|
+
|------|-------------|
|
|
142
|
+
| `--json` / `--csv` | Output format (also `--format table\|json\|csv`) |
|
|
143
|
+
| `-o, --out <file>` | Save a report to a file |
|
|
144
|
+
| `--output-dir <dir>` | Auto-save a timestamped JSON report to a directory |
|
|
145
|
+
| `--tag <name>` | Label this run; repeatable; appears in payload and header |
|
|
146
|
+
| `--note <text>` | Freeform annotation in payload and header |
|
|
147
|
+
| `--histogram` | Print ASCII latency distribution chart after the report |
|
|
148
|
+
| `--capture-output` | Include raw model output in JSON reports |
|
|
149
|
+
| `-v, --verbose` | Append per-run CSV after the summary table |
|
|
150
|
+
|
|
151
|
+
**Display**
|
|
152
|
+
|
|
153
|
+
| Flag | Description |
|
|
154
|
+
|------|-------------|
|
|
155
|
+
| `--color` / `--no-color` | Force or disable ANSI colors (auto on TTYs) |
|
|
156
|
+
| `--ascii` | Plain ASCII table borders instead of Unicode |
|
|
157
|
+
| `--compact` | Force narrow terminal layout |
|
|
158
|
+
| `--width <n>` | Render as if the terminal is `n` columns wide |
|
|
159
|
+
| `--progress` / `--no-progress` | Force or disable the live progress line |
|
|
160
|
+
|
|
161
|
+
## Prompt Profiles
|
|
162
|
+
|
|
163
|
+
Nine built-in suites, choose the one that matches your use case:
|
|
164
|
+
|
|
165
|
+
| Profile | Prompts | Best for |
|
|
166
|
+
|---------|---------|----------|
|
|
167
|
+
| `quick` | 1 | Smoke test, fast health check |
|
|
168
|
+
| `standard` | 3 | Default — short chat, JSON generation, medium output |
|
|
169
|
+
| `interactive` | 3 | Conversational latency (TTFT-heavy) |
|
|
170
|
+
| `throughput` | 3 | Longer generation, token throughput signal |
|
|
171
|
+
| `client` | 5 | Real-world mix: chat, content, extraction, summarization, code |
|
|
172
|
+
| `stress` | 5 | High-load mix with math and reasoning |
|
|
173
|
+
| `reasoning` | 5 | Multi-step logic, estimation, debugging — capability + speed |
|
|
174
|
+
| `coding` | 5 | Code review, refactoring, algorithms, system design |
|
|
175
|
+
| `creative` | 5 | Product copy, analogies, commit messages, docs |
|
|
176
|
+
|
|
177
|
+
## Regression Tracking
|
|
178
|
+
|
|
179
|
+
Track performance across macOS updates, model changes, or hardware swaps:
|
|
180
|
+
|
|
181
|
+
```sh
|
|
182
|
+
# Before
|
|
183
|
+
fm-bench --profile coding --runs 5 --output-dir reports/ --tag before
|
|
184
|
+
|
|
185
|
+
# After the change
|
|
186
|
+
fm-bench --profile coding --runs 5 --output-dir reports/ --tag after
|
|
187
|
+
|
|
188
|
+
# See what changed
|
|
189
|
+
fm-bench compare reports/fm-bench_*before*.json reports/fm-bench_*after*.json
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
The compare output shows each model/concurrency row with the before value, a color-coded percent delta (green = improvement, red = regression), and the after value — for TTFT, E2E, TPOT, tokens/s, RPS, success rate, and CV.
|
|
193
|
+
|
|
194
|
+
```sh
|
|
195
|
+
# View the full trend over time
|
|
196
|
+
fm-bench history reports/
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## CI Integration
|
|
200
|
+
|
|
201
|
+
Gate deployments or model updates on benchmark quality:
|
|
202
|
+
|
|
203
|
+
```sh
|
|
204
|
+
# Fails with exit code 1 if TTFT > 750ms or E2E > 4s on any run
|
|
205
|
+
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Prints `fm-bench ci: PASS` or `fm-bench ci: FAIL — <reason>` to stderr. Designed for GitHub Actions, Buildkite, or any shell-based pipeline.
|
|
106
209
|
|
|
107
210
|
## Prompt Files
|
|
108
211
|
|
|
@@ -126,95 +229,52 @@ Plain text files are split on blank lines.
|
|
|
126
229
|
|
|
127
230
|
## Metrics
|
|
128
231
|
|
|
129
|
-
|
|
232
|
+
**Latency** — TTFT (p50/p95), E2E (p50/p95/p99), TPOT (p50/p95), 95% confidence interval, coefficient of variation (CV).
|
|
130
233
|
|
|
131
|
-
|
|
132
|
-
- E2E latency, or full response wall-clock latency.
|
|
133
|
-
- TPOT, or decode time per output token after the first output token.
|
|
134
|
-
- second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
|
|
135
|
-
- prefill tokens per second, or prompt tokens divided by TTFT.
|
|
136
|
-
- output tokens per second per request.
|
|
137
|
-
- total output token throughput across the measured window.
|
|
138
|
-
- total token throughput, including prompt and output tokens.
|
|
139
|
-
- requests per second across the measured window.
|
|
140
|
-
- goodput percentage and goodput RPS when SLO flags are set.
|
|
141
|
-
- coefficient of variation (CV) and confidence interval context for stability.
|
|
142
|
-
- prompt and output token counts.
|
|
143
|
-
- p50, p95, and p99 tail latency views.
|
|
144
|
-
- repeatability across repeated runs of the same prompt.
|
|
145
|
-
- success and failure counts.
|
|
146
|
-
- unavailable model notes.
|
|
234
|
+
**Throughput** — prefill tokens/s, decode tokens/s, output tokens/s per request, aggregate system tokens/s, requests per second.
|
|
147
235
|
|
|
148
|
-
|
|
236
|
+
**Streaming quality** — second-chunk delay, chunk-gap p95. Captured from `stdout` chunks during streaming runs.
|
|
149
237
|
|
|
150
|
-
|
|
238
|
+
**Reliability** — success rate, goodput rate and RPS against SLO budgets, repeatability (most common output hash frequency across repeated runs).
|
|
151
239
|
|
|
152
|
-
|
|
240
|
+
**Stability** — CV (stddev/mean for E2E latency); green ≤10%, yellow ≤25%, red >25%.
|
|
153
241
|
|
|
154
|
-
|
|
242
|
+
Token counts come from `fm token-count --quiet`. If `fm` cannot count tokens, those fields are blank while character throughput is still reported.
|
|
155
243
|
|
|
156
|
-
|
|
244
|
+
Terminal layout is responsive: wide → full scoreboard + detail tables, medium → tighter single table, narrow → compact model cards. Use `--width` to preview any layout and `--ascii` for log-friendly output.
|
|
157
245
|
|
|
158
|
-
|
|
159
|
-
fm-bench legend
|
|
160
|
-
fm-bench legend --json
|
|
161
|
-
fm-bench legend --csv
|
|
162
|
-
```
|
|
246
|
+
## Colors and Legend
|
|
163
247
|
|
|
164
|
-
|
|
248
|
+
Table output is color-coded on interactive terminals — **green** is better/passing, **yellow** is marginal/partial, **red** is failing/unstable. Fixed thresholds apply to success rate, goodput, CV, and repeatability. Latency uses SLO thresholds when set, otherwise lower-is-better relative ranking. Throughput uses higher-is-better relative ranking.
|
|
165
249
|
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
Progress is automatic for table output on TTYs. Use `--progress` to force it or `--no-progress` to keep the terminal completely quiet until the report is ready.
|
|
171
|
-
|
|
172
|
-
## Terminal Colors
|
|
173
|
-
|
|
174
|
-
Table output uses semantic ANSI color on interactive terminals:
|
|
175
|
-
|
|
176
|
-
- green: passing, steadier, or better than the current comparison set.
|
|
177
|
-
- yellow: marginal, partial, or near a budget.
|
|
178
|
-
- red: failing a budget, unstable, or slower/lower than peers.
|
|
179
|
-
|
|
180
|
-
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. CV is green at `<=10%`, yellow at `<=25%`, and red above `25%` because higher CV means less steady latency. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
250
|
+
```sh
|
|
251
|
+
fm-bench legend # full column definitions and color rules
|
|
252
|
+
fm-bench legend --json # machine-readable
|
|
253
|
+
```
|
|
181
254
|
|
|
182
|
-
|
|
255
|
+
`NO_COLOR=1` disables color; `FORCE_COLOR=1` or `--color` enables it. `--ascii` switches to plain ASCII borders for log systems.
|
|
183
256
|
|
|
184
|
-
|
|
257
|
+
A live single-line progress indicator runs on stderr during interactive sessions. The final report always goes to stdout — `--json`, `--csv`, and `--out` stay automation-friendly.
|
|
185
258
|
|
|
186
259
|
## Requirements
|
|
187
260
|
|
|
188
|
-
- macOS 27 or newer
|
|
261
|
+
- macOS 27 or newer (Apple's `fm` CLI is preinstalled).
|
|
189
262
|
- Node.js 20 or newer.
|
|
190
|
-
- Apple Intelligence
|
|
263
|
+
- Apple Intelligence enabled on the device.
|
|
191
264
|
|
|
192
|
-
Private Cloud Compute
|
|
265
|
+
`pcc` (Private Cloud Compute) availability depends on Apple's current eligibility. `fm-bench` shows it as skipped if `fm available --model pcc` reports unavailable.
|
|
266
|
+
|
|
267
|
+
See [docs/methodology.md](docs/methodology.md) for benchmark methodology and metric references.
|
|
193
268
|
|
|
194
269
|
## Development
|
|
195
270
|
|
|
196
271
|
```sh
|
|
197
272
|
npm install
|
|
198
|
-
npm test
|
|
273
|
+
npm test # node --test
|
|
199
274
|
npm run lint
|
|
200
|
-
npm pack
|
|
201
|
-
```
|
|
202
|
-
|
|
203
|
-
The package has no runtime npm dependencies.
|
|
204
|
-
|
|
205
|
-
## Releases
|
|
206
|
-
|
|
207
|
-
Releases are tag-driven:
|
|
208
|
-
|
|
209
|
-
```sh
|
|
210
|
-
npm run release:patch
|
|
211
|
-
npm run release:minor
|
|
212
|
-
npm run release:major
|
|
213
275
|
```
|
|
214
276
|
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
Maintainers can also run the **Version** workflow manually from GitHub Actions to bump the version and push the tag.
|
|
277
|
+
No runtime npm dependencies.
|
|
218
278
|
|
|
219
279
|
## License
|
|
220
280
|
|
package/package.json
CHANGED
package/src/bench.js
CHANGED
|
@@ -229,11 +229,18 @@ async function runScenario(context) {
|
|
|
229
229
|
}
|
|
230
230
|
|
|
231
231
|
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt) {
|
|
232
|
+
const maxAttempts = 1 + Math.max(0, options.retry ?? 0);
|
|
232
233
|
const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
234
|
+
let response;
|
|
235
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
236
|
+
response = await respond(fmBin, job.model.name, job.prompt.prompt, {
|
|
237
|
+
...options,
|
|
238
|
+
stream: options.stream
|
|
239
|
+
});
|
|
240
|
+
if (response.ok || attempt >= maxAttempts) break;
|
|
241
|
+
const backoffMs = Math.min(500 * 2 ** (attempt - 1), 4000);
|
|
242
|
+
await new Promise((resolve) => setTimeout(resolve, backoffMs));
|
|
243
|
+
}
|
|
237
244
|
const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
238
245
|
const outputTokens = response.ok
|
|
239
246
|
? await countTokens(fmBin, response.output, options)
|
|
@@ -338,7 +345,10 @@ function publicOptions(options) {
|
|
|
338
345
|
e2eMs: options.sloE2eMs || null,
|
|
339
346
|
tpotMs: options.sloTpotMs || null
|
|
340
347
|
},
|
|
341
|
-
instructions: options.instructions ? '[set]' : ''
|
|
348
|
+
instructions: options.instructions ? '[set]' : '',
|
|
349
|
+
retry: options.retry ?? 0,
|
|
350
|
+
tags: options.tags?.length ? options.tags : [],
|
|
351
|
+
note: options.note ?? null
|
|
342
352
|
};
|
|
343
353
|
}
|
|
344
354
|
|
package/src/cli.js
CHANGED
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import fs from 'node:fs/promises';
|
|
2
2
|
import { createRequire } from 'node:module';
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
|
+
import { diffReports, renderCompareReport } from './compare.js';
|
|
5
|
+
import { loadHistory, renderHistoryReport } from './history.js';
|
|
4
6
|
import { runProcess } from './process.js';
|
|
5
7
|
import { createProgress } from './progress.js';
|
|
6
8
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
7
|
-
import { legendEntries, renderBenchmarkReport, renderLegend, renderModelsTable } from './table.js';
|
|
9
|
+
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend, renderModelsTable } from './table.js';
|
|
8
10
|
|
|
9
11
|
const require = createRequire(import.meta.url);
|
|
10
12
|
const packageJson = require('../package.json');
|
|
@@ -38,6 +40,16 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
38
40
|
return;
|
|
39
41
|
}
|
|
40
42
|
|
|
43
|
+
if (parsed.command === 'compare') {
|
|
44
|
+
await runCompare(parsed, renderOptions(parsed));
|
|
45
|
+
return;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
if (parsed.command === 'history') {
|
|
49
|
+
await runHistory(parsed, renderOptions(parsed));
|
|
50
|
+
return;
|
|
51
|
+
}
|
|
52
|
+
|
|
41
53
|
if (parsed.command === 'models') {
|
|
42
54
|
const inspection = await inspectModels(parsed);
|
|
43
55
|
if (parsed.format === 'json') {
|
|
@@ -48,6 +60,11 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
48
60
|
return;
|
|
49
61
|
}
|
|
50
62
|
|
|
63
|
+
if (parsed.ci) {
|
|
64
|
+
if (parsed.color === 'auto') parsed.color = 'never';
|
|
65
|
+
if (parsed.progress === 'auto') parsed.progress = 'never';
|
|
66
|
+
}
|
|
67
|
+
|
|
51
68
|
const progress = createProgress({
|
|
52
69
|
...renderOptions(parsed),
|
|
53
70
|
enabled: resolveProgress(parsed),
|
|
@@ -70,6 +87,10 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
70
87
|
console.log(toCsv(flattenResults(payload.results)));
|
|
71
88
|
} else {
|
|
72
89
|
console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
|
|
90
|
+
if (parsed.histogram) {
|
|
91
|
+
console.log();
|
|
92
|
+
console.log(renderLatencyHistogram(payload.results, renderOptions(parsed)));
|
|
93
|
+
}
|
|
73
94
|
if (parsed.verbose) {
|
|
74
95
|
console.log();
|
|
75
96
|
console.log(toCsv(flattenResults(payload.results)));
|
|
@@ -83,11 +104,62 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
83
104
|
console.error(`Saved ${reportFormat.toUpperCase()} report to ${written}`);
|
|
84
105
|
}
|
|
85
106
|
}
|
|
107
|
+
|
|
108
|
+
if (parsed.outputDir) {
|
|
109
|
+
const stamp = payload.startedAt.replace(/[:.]/g, '-').replace('T', '_').slice(0, 19);
|
|
110
|
+
const firstModel = parsed.models?.flatMap((m) => String(m).split(',')).map((m) => m.trim()).filter(Boolean)[0] || 'all';
|
|
111
|
+
const modelSlug = firstModel.replace(/[^a-z0-9]/gi, '_');
|
|
112
|
+
const filename = `fm-bench_${stamp}_${modelSlug}.json`;
|
|
113
|
+
const filePath = `${parsed.outputDir}/${filename}`;
|
|
114
|
+
const written = await writeReport(filePath, payload, 'json');
|
|
115
|
+
if (parsed.format !== 'json') {
|
|
116
|
+
console.error(`Saved JSON report to ${written}`);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
if (parsed.ci) {
|
|
121
|
+
const ciResult = evaluateCi(payload);
|
|
122
|
+
if (!ciResult.passed) {
|
|
123
|
+
const reasons = ciResult.reasons.join('; ');
|
|
124
|
+
console.error(`fm-bench ci: FAIL — ${reasons}`);
|
|
125
|
+
const error = new Error(`CI checks failed: ${reasons}`);
|
|
126
|
+
error.exitCode = 1;
|
|
127
|
+
throw error;
|
|
128
|
+
}
|
|
129
|
+
console.error(`fm-bench ci: PASS`);
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
function evaluateCi(payload) {
|
|
134
|
+
const reasons = [];
|
|
135
|
+
const totalFailed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
|
|
136
|
+
if (totalFailed > 0) {
|
|
137
|
+
reasons.push(`${totalFailed} run(s) failed`);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const hasSlo = payload.options?.slo && (
|
|
141
|
+
payload.options.slo.ttftMs || payload.options.slo.e2eMs || payload.options.slo.tpotMs
|
|
142
|
+
);
|
|
143
|
+
if (hasSlo) {
|
|
144
|
+
for (const item of payload.summary) {
|
|
145
|
+
if (!item.available) continue;
|
|
146
|
+
if (item.goodputRate != null && item.goodputRate < 1) {
|
|
147
|
+
const pct = Math.round((1 - item.goodputRate) * 100);
|
|
148
|
+
reasons.push(`${item.model} c${item.concurrency ?? 1}: ${pct}% of runs violated SLO`);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return {
|
|
154
|
+
passed: reasons.length === 0,
|
|
155
|
+
reasons
|
|
156
|
+
};
|
|
86
157
|
}
|
|
87
158
|
|
|
88
159
|
export function parseArgs(argv) {
|
|
89
160
|
const options = {
|
|
90
161
|
command: 'run',
|
|
162
|
+
compareFiles: [],
|
|
91
163
|
models: [],
|
|
92
164
|
prompts: [],
|
|
93
165
|
runs: 1,
|
|
@@ -107,16 +179,21 @@ export function parseArgs(argv) {
|
|
|
107
179
|
captureOutput: false,
|
|
108
180
|
availableOnly: false,
|
|
109
181
|
failFast: false,
|
|
182
|
+
retry: 0,
|
|
183
|
+
ci: false,
|
|
184
|
+
tags: [],
|
|
185
|
+
note: null,
|
|
110
186
|
verbose: false,
|
|
111
187
|
ascii: false,
|
|
112
188
|
color: 'auto',
|
|
113
189
|
progress: 'auto',
|
|
114
190
|
compact: false,
|
|
115
|
-
width: null
|
|
191
|
+
width: null,
|
|
192
|
+
histogram: false
|
|
116
193
|
};
|
|
117
194
|
|
|
118
195
|
const args = [...argv];
|
|
119
|
-
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'help'].includes(args[0])) {
|
|
196
|
+
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'legend', 'metrics', 'compare', 'history', 'help'].includes(args[0])) {
|
|
120
197
|
options.command = args.shift();
|
|
121
198
|
}
|
|
122
199
|
|
|
@@ -189,6 +266,9 @@ export function parseArgs(argv) {
|
|
|
189
266
|
break;
|
|
190
267
|
case '--profile':
|
|
191
268
|
options.profile = requireValue(arg, args);
|
|
269
|
+
if (!['quick', 'standard', 'interactive', 'throughput', 'client', 'stress', 'reasoning', 'coding', 'creative'].includes(options.profile)) {
|
|
270
|
+
throw new Error(`--profile must be one of: quick, standard, interactive, throughput, client, stress, reasoning, coding, creative`);
|
|
271
|
+
}
|
|
192
272
|
break;
|
|
193
273
|
case '-i':
|
|
194
274
|
case '--instructions':
|
|
@@ -248,10 +328,16 @@ export function parseArgs(argv) {
|
|
|
248
328
|
case '--width':
|
|
249
329
|
options.width = parsePositiveInt(requireValue(arg, args), arg);
|
|
250
330
|
break;
|
|
331
|
+
case '--histogram':
|
|
332
|
+
options.histogram = true;
|
|
333
|
+
break;
|
|
251
334
|
case '-o':
|
|
252
335
|
case '--out':
|
|
253
336
|
options.out = requireValue(arg, args);
|
|
254
337
|
break;
|
|
338
|
+
case '--output-dir':
|
|
339
|
+
options.outputDir = requireValue(arg, args);
|
|
340
|
+
break;
|
|
255
341
|
case '--capture-output':
|
|
256
342
|
options.captureOutput = true;
|
|
257
343
|
break;
|
|
@@ -261,6 +347,18 @@ export function parseArgs(argv) {
|
|
|
261
347
|
case '--fail-fast':
|
|
262
348
|
options.failFast = true;
|
|
263
349
|
break;
|
|
350
|
+
case '--retry':
|
|
351
|
+
options.retry = parseNonNegativeInt(requireValue(arg, args), arg);
|
|
352
|
+
break;
|
|
353
|
+
case '--ci':
|
|
354
|
+
options.ci = true;
|
|
355
|
+
break;
|
|
356
|
+
case '--tag':
|
|
357
|
+
options.tags.push(requireValue(arg, args));
|
|
358
|
+
break;
|
|
359
|
+
case '--note':
|
|
360
|
+
options.note = requireValue(arg, args);
|
|
361
|
+
break;
|
|
264
362
|
case '-v':
|
|
265
363
|
case '--verbose':
|
|
266
364
|
options.verbose = true;
|
|
@@ -275,8 +373,14 @@ export function parseArgs(argv) {
|
|
|
275
373
|
if (arg.startsWith('-')) {
|
|
276
374
|
throw new Error(`Unknown option: ${arg}`);
|
|
277
375
|
}
|
|
278
|
-
options.
|
|
279
|
-
|
|
376
|
+
if (options.command === 'compare') {
|
|
377
|
+
options.compareFiles.push(arg);
|
|
378
|
+
} else if (options.command === 'history') {
|
|
379
|
+
options.historyDir = arg;
|
|
380
|
+
} else {
|
|
381
|
+
options.prompts.push([arg, ...args].join(' '));
|
|
382
|
+
args.length = 0;
|
|
383
|
+
}
|
|
280
384
|
break;
|
|
281
385
|
}
|
|
282
386
|
}
|
|
@@ -284,6 +388,64 @@ export function parseArgs(argv) {
|
|
|
284
388
|
return options;
|
|
285
389
|
}
|
|
286
390
|
|
|
391
|
+
async function runHistory(options, renderOpts) {
|
|
392
|
+
const dir = options.historyDir || options.outputDir || '.';
|
|
393
|
+
const reports = await loadHistory(dir);
|
|
394
|
+
|
|
395
|
+
if (options.format === 'json') {
|
|
396
|
+
const data = reports.map(({ filePath, report }) => ({
|
|
397
|
+
file: filePath,
|
|
398
|
+
startedAt: report.startedAt,
|
|
399
|
+
version: report.version,
|
|
400
|
+
summary: report.summary
|
|
401
|
+
}));
|
|
402
|
+
console.log(JSON.stringify(data, null, 2));
|
|
403
|
+
} else {
|
|
404
|
+
console.log(renderHistoryReport(reports, renderOpts));
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
async function runCompare(options, renderOpts) {
|
|
409
|
+
const files = options.compareFiles;
|
|
410
|
+
if (files.length < 2) {
|
|
411
|
+
throw new Error('compare requires two JSON report files: fm-bench compare before.json after.json');
|
|
412
|
+
}
|
|
413
|
+
if (files.length > 2) {
|
|
414
|
+
throw new Error('compare accepts exactly two JSON report files');
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
const [beforePath, afterPath] = files;
|
|
418
|
+
const [beforeText, afterText] = await Promise.all([
|
|
419
|
+
fs.readFile(beforePath, 'utf8'),
|
|
420
|
+
fs.readFile(afterPath, 'utf8')
|
|
421
|
+
]);
|
|
422
|
+
|
|
423
|
+
let before, after;
|
|
424
|
+
try {
|
|
425
|
+
before = JSON.parse(beforeText);
|
|
426
|
+
} catch {
|
|
427
|
+
throw new Error(`Cannot parse ${beforePath} as JSON`);
|
|
428
|
+
}
|
|
429
|
+
try {
|
|
430
|
+
after = JSON.parse(afterText);
|
|
431
|
+
} catch {
|
|
432
|
+
throw new Error(`Cannot parse ${afterPath} as JSON`);
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
const diff = diffReports(before, after);
|
|
436
|
+
|
|
437
|
+
if (options.format === 'json') {
|
|
438
|
+
console.log(JSON.stringify(diff, null, 2));
|
|
439
|
+
} else {
|
|
440
|
+
console.log(renderCompareReport(diff, renderOpts));
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
if (options.out) {
|
|
444
|
+
await fs.writeFile(options.out, `${JSON.stringify(diff, null, 2)}\n`, 'utf8');
|
|
445
|
+
console.error(`Saved compare report to ${options.out}`);
|
|
446
|
+
}
|
|
447
|
+
}
|
|
448
|
+
|
|
287
449
|
async function runDoctor(options) {
|
|
288
450
|
const checks = [];
|
|
289
451
|
checks.push(['node', process.version, true]);
|
|
@@ -295,13 +457,46 @@ async function runDoctor(options) {
|
|
|
295
457
|
const major = versionMatch ? Number.parseInt(versionMatch[1].split('.')[0], 10) : null;
|
|
296
458
|
checks.push(['macOS', versionMatch?.[1] || 'unknown', major == null || major >= 27]);
|
|
297
459
|
|
|
460
|
+
const hwModel = await runProcess('sysctl', ['-n', 'hw.model'], { timeoutMs: 3_000 });
|
|
461
|
+
const hwModelStr = (hwModel.stdout || '').trim();
|
|
462
|
+
if (hwModelStr) checks.push(['hw.model', hwModelStr, true]);
|
|
463
|
+
|
|
464
|
+
const cpuBrand = await runProcess('sysctl', ['-n', 'machdep.cpu.brand_string'], { timeoutMs: 3_000 });
|
|
465
|
+
const cpuBrandStr = (cpuBrand.stdout || '').trim();
|
|
466
|
+
if (cpuBrandStr) checks.push(['cpu', cpuBrandStr, true]);
|
|
467
|
+
|
|
468
|
+
const memBytes = await runProcess('sysctl', ['-n', 'hw.memsize'], { timeoutMs: 3_000 });
|
|
469
|
+
const memRaw = (memBytes.stdout || '').trim();
|
|
470
|
+
if (memRaw) {
|
|
471
|
+
const gb = (Number(memRaw) / (1024 ** 3)).toFixed(0);
|
|
472
|
+
checks.push(['memory', `${gb} GB`, true]);
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
const thermalResult = await runProcess('pmset', ['-g', 'therm'], { timeoutMs: 5_000 });
|
|
476
|
+
const thermalOut = (thermalResult.stdout || thermalResult.stderr || '').trim();
|
|
477
|
+
const thermalMatch = thermalOut.match(/CPU_Scheduler_Limit\s*=\s*(\d+)/);
|
|
478
|
+
if (thermalMatch) {
|
|
479
|
+
const limit = Number.parseInt(thermalMatch[1], 10);
|
|
480
|
+
checks.push(['thermal limit', `${limit}%`, limit >= 100]);
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
|
|
484
|
+
const batteryOut = (batteryResult.stdout || '').trim();
|
|
485
|
+
const batteryLine = batteryOut.split('\n').find((line) => line.includes('%'));
|
|
486
|
+
if (batteryLine) {
|
|
487
|
+
const pctMatch = batteryLine.match(/(\d+)%/);
|
|
488
|
+
const charging = /charging|AC Power|charged/i.test(batteryLine);
|
|
489
|
+
const pct = pctMatch ? `${pctMatch[1]}%` : '?%';
|
|
490
|
+
checks.push(['battery', `${pct}${charging ? ' (charging/AC)' : ' (battery)'}`, charging || Number.parseInt(pctMatch?.[1], 10) >= 20]);
|
|
491
|
+
}
|
|
492
|
+
|
|
298
493
|
const inspection = await inspectModels(options);
|
|
299
494
|
checks.push(['fm', inspection.fmBin, inspection.models.length > 0]);
|
|
300
495
|
for (const model of inspection.models) {
|
|
301
496
|
checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
|
|
302
497
|
}
|
|
303
498
|
|
|
304
|
-
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(
|
|
499
|
+
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(16)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
|
|
305
500
|
console.log(lines.join('\n'));
|
|
306
501
|
|
|
307
502
|
if (options.out) {
|
|
@@ -374,12 +569,16 @@ Dynamic benchmark CLI for Apple's fm command on macOS 27+.
|
|
|
374
569
|
Usage:
|
|
375
570
|
fm-bench [run] [options]
|
|
376
571
|
fm-bench models [options]
|
|
572
|
+
fm-bench compare <before.json> <after.json> [options]
|
|
573
|
+
fm-bench history [dir] [options]
|
|
377
574
|
fm-bench legend [options]
|
|
378
575
|
fm-bench doctor [options]
|
|
379
576
|
|
|
380
577
|
Commands:
|
|
381
578
|
run Benchmark discovered or selected fm models
|
|
382
579
|
models List discovered models and availability
|
|
580
|
+
compare Compare two saved JSON reports and show metric deltas
|
|
581
|
+
history Show a trend table from all fm-bench JSON reports in a directory
|
|
383
582
|
legend Explain every terminal table column and color rule
|
|
384
583
|
doctor Check Node, macOS, fm, and model availability
|
|
385
584
|
|
|
@@ -398,7 +597,7 @@ Run options:
|
|
|
398
597
|
--slo-tpot-ms <n> Count request as good only if TPOT is <= n
|
|
399
598
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
400
599
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
401
|
-
--profile <name> quick, standard, interactive, throughput, client, or
|
|
600
|
+
--profile <name> quick, standard, interactive, throughput, client, stress, reasoning, coding, or creative
|
|
402
601
|
-i, --instructions <text> Instructions passed to fm respond
|
|
403
602
|
--use-case <case> Pass a system model use case through to fm
|
|
404
603
|
--guardrails <level> Pass a system model guardrail level through to fm
|
|
@@ -409,6 +608,10 @@ Run options:
|
|
|
409
608
|
--available-only Hide unavailable discovered models
|
|
410
609
|
--capture-output Include raw model output in JSON reports
|
|
411
610
|
--fail-fast Stop after the first failed measured run
|
|
611
|
+
--retry <n> Retry failed fm calls up to n times with exponential backoff
|
|
612
|
+
--ci Exit 1 if any run fails or any SLO is violated; disables color and progress
|
|
613
|
+
--tag <name> Tag this run; repeatable; included in JSON payload and report header
|
|
614
|
+
--note <text> Freeform note included in JSON payload and report header
|
|
412
615
|
|
|
413
616
|
Output:
|
|
414
617
|
--format <type> table, json, or csv (default: table)
|
|
@@ -421,7 +624,9 @@ Output:
|
|
|
421
624
|
--no-progress Disable live progress on stderr
|
|
422
625
|
--compact Force compact terminal layout
|
|
423
626
|
--width <n> Render for a specific terminal width
|
|
627
|
+
--histogram Print an ASCII latency distribution histogram after the report
|
|
424
628
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
629
|
+
--output-dir <dir> Save a timestamped JSON report to a directory automatically
|
|
425
630
|
-v, --verbose Include per-run CSV after the summary table
|
|
426
631
|
|
|
427
632
|
Environment:
|
|
@@ -434,6 +639,11 @@ Examples:
|
|
|
434
639
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
435
640
|
fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
|
|
436
641
|
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
642
|
+
fm-bench --profile reasoning --runs 5 --retry 2
|
|
643
|
+
fm-bench compare before.json after.json
|
|
644
|
+
fm-bench compare before.json after.json --json
|
|
645
|
+
fm-bench history ./reports
|
|
646
|
+
fm-bench history ./reports --json
|
|
437
647
|
fm-bench legend
|
|
438
648
|
fm-bench models
|
|
439
649
|
fm-bench doctor
|
package/src/compare.js
ADDED
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
2
|
+
|
|
3
|
+
export function diffReports(before, after) {
|
|
4
|
+
const beforeByKey = indexSummary(before.summary ?? []);
|
|
5
|
+
const afterByKey = indexSummary(after.summary ?? []);
|
|
6
|
+
|
|
7
|
+
const keys = new Set([...beforeByKey.keys(), ...afterByKey.keys()]);
|
|
8
|
+
const rows = [];
|
|
9
|
+
|
|
10
|
+
for (const key of [...keys].sort()) {
|
|
11
|
+
const b = beforeByKey.get(key);
|
|
12
|
+
const a = afterByKey.get(key);
|
|
13
|
+
rows.push(buildDiffRow(key, b ?? null, a ?? null));
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
return {
|
|
17
|
+
before: reportMeta(before),
|
|
18
|
+
after: reportMeta(after),
|
|
19
|
+
rows
|
|
20
|
+
};
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function reportMeta(report) {
|
|
24
|
+
return {
|
|
25
|
+
version: report.version ?? '?',
|
|
26
|
+
startedAt: report.startedAt ?? '',
|
|
27
|
+
finishedAt: report.finishedAt ?? '',
|
|
28
|
+
runs: report.options?.runs ?? null,
|
|
29
|
+
profile: report.options?.profile ?? null,
|
|
30
|
+
concurrency: report.options?.concurrency ?? null
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function indexSummary(summary) {
|
|
35
|
+
const map = new Map();
|
|
36
|
+
for (const item of summary) {
|
|
37
|
+
const key = `${item.model}::${item.concurrency ?? 1}`;
|
|
38
|
+
map.set(key, item);
|
|
39
|
+
}
|
|
40
|
+
return map;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function buildDiffRow(key, before, after) {
|
|
44
|
+
const [model, concurrencyStr] = key.split('::');
|
|
45
|
+
return {
|
|
46
|
+
model,
|
|
47
|
+
concurrency: Number.parseInt(concurrencyStr, 10) || 1,
|
|
48
|
+
available: {
|
|
49
|
+
before: before?.available ?? null,
|
|
50
|
+
after: after?.available ?? null
|
|
51
|
+
},
|
|
52
|
+
ttftP50: diffMs(before?.ttft?.p50, after?.ttft?.p50),
|
|
53
|
+
ttftP95: diffMs(before?.ttft?.p95, after?.ttft?.p95),
|
|
54
|
+
e2eP50: diffMs(before?.latency?.p50, after?.latency?.p50),
|
|
55
|
+
e2eP95: diffMs(before?.latency?.p95, after?.latency?.p95),
|
|
56
|
+
e2eP99: diffMs(before?.latency?.p99, after?.latency?.p99),
|
|
57
|
+
tpotP50: diffMs(before?.tpot?.p50, after?.tpot?.p50),
|
|
58
|
+
successRate: diffPercent(before?.successRate, after?.successRate),
|
|
59
|
+
goodputRate: diffPercent(before?.goodputRate, after?.goodputRate),
|
|
60
|
+
tokensPerSecond: diffNumber(before?.tokensPerSecond?.avg, after?.tokensPerSecond?.avg, false),
|
|
61
|
+
rps: diffNumber(before?.rps, after?.rps, false),
|
|
62
|
+
cv: diffPercent(before?.latency?.cv, after?.latency?.cv)
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function diffMs(before, after) {
|
|
67
|
+
return {
|
|
68
|
+
before: before ?? null,
|
|
69
|
+
after: after ?? null,
|
|
70
|
+
delta: numDelta(before, after),
|
|
71
|
+
deltaPercent: percentDelta(before, after)
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function diffNumber(before, after, lowerIsBetter = true) {
|
|
76
|
+
return {
|
|
77
|
+
before: before ?? null,
|
|
78
|
+
after: after ?? null,
|
|
79
|
+
delta: numDelta(before, after),
|
|
80
|
+
deltaPercent: percentDelta(before, after),
|
|
81
|
+
lowerIsBetter
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function diffPercent(before, after) {
|
|
86
|
+
return {
|
|
87
|
+
before: before ?? null,
|
|
88
|
+
after: after ?? null,
|
|
89
|
+
delta: numDelta(before, after),
|
|
90
|
+
deltaPercent: percentDelta(before, after)
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
function numDelta(before, after) {
|
|
95
|
+
if (!Number.isFinite(before) || !Number.isFinite(after)) return null;
|
|
96
|
+
return after - before;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function percentDelta(before, after) {
|
|
100
|
+
if (!Number.isFinite(before) || !Number.isFinite(after) || before === 0) return null;
|
|
101
|
+
return (after - before) / Math.abs(before);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function renderCompareReport(diff, options = {}) {
|
|
105
|
+
const { color = false, ascii = false } = options;
|
|
106
|
+
const width = options.width || process.stdout.columns || 120;
|
|
107
|
+
const V = ascii ? '|' : '│';
|
|
108
|
+
const H = ascii ? '-' : '─';
|
|
109
|
+
const TL = ascii ? '+' : '┌';
|
|
110
|
+
const TR = ascii ? '+' : '┐';
|
|
111
|
+
const BL = ascii ? '+' : '└';
|
|
112
|
+
const BR = ascii ? '+' : '┘';
|
|
113
|
+
const ML = ascii ? '+' : '├';
|
|
114
|
+
const MR = ascii ? '+' : '┤';
|
|
115
|
+
const TJ = ascii ? '+' : '┬';
|
|
116
|
+
const MJ = ascii ? '+' : '┼';
|
|
117
|
+
const BJ = ascii ? '+' : '┴';
|
|
118
|
+
|
|
119
|
+
const lines = [];
|
|
120
|
+
|
|
121
|
+
const beforeLabel = `${diff.before.version} ${diff.before.startedAt ? diff.before.startedAt.slice(0, 19).replace('T', ' ') : '?'}`;
|
|
122
|
+
const afterLabel = `${diff.after.version} ${diff.after.startedAt ? diff.after.startedAt.slice(0, 19).replace('T', ' ') : '?'}`;
|
|
123
|
+
|
|
124
|
+
lines.push(`fm-bench compare`);
|
|
125
|
+
lines.push(` before: ${beforeLabel}`);
|
|
126
|
+
lines.push(` after: ${afterLabel}`);
|
|
127
|
+
lines.push('');
|
|
128
|
+
|
|
129
|
+
const colWidths = [8, 3, 10, 10, 10, 10, 10, 10, 8, 8, 9, 7, 7];
|
|
130
|
+
const headers = ['MODEL', 'C', 'TTFT P50', 'TTFT P95', 'E2E P50', 'E2E P95', 'E2E P99', 'TPOT P50', 'SUCC', 'GOOD', 'USER T/S', 'RPS', 'CV'];
|
|
131
|
+
|
|
132
|
+
const renderRule = (l, j, r) => `${l}${colWidths.map((w) => H.repeat(w + 2)).join(j)}${r}`;
|
|
133
|
+
const renderRow = (cells) => `${V}${cells.map((cell, i) => ` ${fitCell(cell, colWidths[i])} `).join(V)}${V}`;
|
|
134
|
+
|
|
135
|
+
lines.push(renderRule(TL, TJ, TR));
|
|
136
|
+
lines.push(renderRow(headers.map((h, i) => pad(h, colWidths[i]))));
|
|
137
|
+
lines.push(renderRule(ML, MJ, MR));
|
|
138
|
+
|
|
139
|
+
for (const row of diff.rows) {
|
|
140
|
+
const bLine = renderRow([
|
|
141
|
+
fitCell(row.model, colWidths[0]),
|
|
142
|
+
fitCell(String(row.concurrency), colWidths[1]),
|
|
143
|
+
fmtDiffMs(row.ttftP50, 'before', color),
|
|
144
|
+
fmtDiffMs(row.ttftP95, 'before', color),
|
|
145
|
+
fmtDiffMs(row.e2eP50, 'before', color),
|
|
146
|
+
fmtDiffMs(row.e2eP95, 'before', color),
|
|
147
|
+
fmtDiffMs(row.e2eP99, 'before', color),
|
|
148
|
+
fmtDiffMs(row.tpotP50, 'before', color),
|
|
149
|
+
fitCell(row.successRate.before != null ? formatPercent(row.successRate.before) : '-', colWidths[8]),
|
|
150
|
+
fitCell(row.goodputRate.before != null ? formatPercent(row.goodputRate.before) : '-', colWidths[9]),
|
|
151
|
+
fitCell(row.tokensPerSecond.before != null ? formatNumber(row.tokensPerSecond.before) : '-', colWidths[10]),
|
|
152
|
+
fitCell(row.rps.before != null ? formatNumber(row.rps.before) : '-', colWidths[11]),
|
|
153
|
+
fitCell(row.cv.before != null ? formatPercent(row.cv.before) : '-', colWidths[12])
|
|
154
|
+
]);
|
|
155
|
+
|
|
156
|
+
const deltaLine = renderRow([
|
|
157
|
+
fitCell('', colWidths[0]),
|
|
158
|
+
fitCell('', colWidths[1]),
|
|
159
|
+
fmtDelta(row.ttftP50, true, color, colWidths[2]),
|
|
160
|
+
fmtDelta(row.ttftP95, true, color, colWidths[3]),
|
|
161
|
+
fmtDelta(row.e2eP50, true, color, colWidths[4]),
|
|
162
|
+
fmtDelta(row.e2eP95, true, color, colWidths[5]),
|
|
163
|
+
fmtDelta(row.e2eP99, true, color, colWidths[6]),
|
|
164
|
+
fmtDelta(row.tpotP50, true, color, colWidths[7]),
|
|
165
|
+
fmtDelta(row.successRate, false, color, colWidths[8]),
|
|
166
|
+
fmtDelta(row.goodputRate, false, color, colWidths[9]),
|
|
167
|
+
fmtDelta(row.tokensPerSecond, false, color, colWidths[10]),
|
|
168
|
+
fmtDelta(row.rps, false, color, colWidths[11]),
|
|
169
|
+
fmtDelta(row.cv, true, color, colWidths[12])
|
|
170
|
+
]);
|
|
171
|
+
|
|
172
|
+
const aLine = renderRow([
|
|
173
|
+
fitCell('', colWidths[0]),
|
|
174
|
+
fitCell('', colWidths[1]),
|
|
175
|
+
fmtDiffMs(row.ttftP50, 'after', color),
|
|
176
|
+
fmtDiffMs(row.ttftP95, 'after', color),
|
|
177
|
+
fmtDiffMs(row.e2eP50, 'after', color),
|
|
178
|
+
fmtDiffMs(row.e2eP95, 'after', color),
|
|
179
|
+
fmtDiffMs(row.e2eP99, 'after', color),
|
|
180
|
+
fmtDiffMs(row.tpotP50, 'after', color),
|
|
181
|
+
fitCell(row.successRate.after != null ? formatPercent(row.successRate.after) : '-', colWidths[8]),
|
|
182
|
+
fitCell(row.goodputRate.after != null ? formatPercent(row.goodputRate.after) : '-', colWidths[9]),
|
|
183
|
+
fitCell(row.tokensPerSecond.after != null ? formatNumber(row.tokensPerSecond.after) : '-', colWidths[10]),
|
|
184
|
+
fitCell(row.rps.after != null ? formatNumber(row.rps.after) : '-', colWidths[11]),
|
|
185
|
+
fitCell(row.cv.after != null ? formatPercent(row.cv.after) : '-', colWidths[12])
|
|
186
|
+
]);
|
|
187
|
+
|
|
188
|
+
lines.push(bLine);
|
|
189
|
+
lines.push(deltaLine);
|
|
190
|
+
lines.push(aLine);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
lines.push(renderRule(BL, BJ, BR));
|
|
194
|
+
lines.push('');
|
|
195
|
+
lines.push('Rows: before → delta (% change) → after | Lower is better for latency and CV; higher for throughput and success.');
|
|
196
|
+
|
|
197
|
+
return lines.join('\n');
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
function fmtDiffMs(diff, which, color) {
|
|
201
|
+
const value = diff[which];
|
|
202
|
+
if (value == null) return '-';
|
|
203
|
+
return formatMs(value);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function fmtDelta(diff, lowerIsBetter, color, width) {
|
|
207
|
+
const { delta, deltaPercent } = diff;
|
|
208
|
+
if (delta == null || deltaPercent == null) return fitCell('n/a', width ?? 10);
|
|
209
|
+
|
|
210
|
+
const pct = Math.round(deltaPercent * 100);
|
|
211
|
+
const sign = delta >= 0 ? '+' : '';
|
|
212
|
+
const label = `${sign}${pct}%`;
|
|
213
|
+
|
|
214
|
+
let tone = null;
|
|
215
|
+
if (lowerIsBetter) {
|
|
216
|
+
tone = pct < -5 ? 'green' : pct > 5 ? 'red' : 'yellow';
|
|
217
|
+
} else {
|
|
218
|
+
tone = pct > 5 ? 'green' : pct < -5 ? 'red' : 'yellow';
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
const text = fitCell(label, width ?? 10);
|
|
222
|
+
if (color) return applyTone(text, tone);
|
|
223
|
+
return text;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
function applyTone(text, tone) {
|
|
227
|
+
const TONES = { green: '\x1b[32m', yellow: '\x1b[33m', red: '\x1b[31m' };
|
|
228
|
+
const reset = '\x1b[0m';
|
|
229
|
+
return tone && TONES[tone] ? `${TONES[tone]}${text}${reset}` : text;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
function pad(text, width) {
|
|
233
|
+
const len = String(text ?? '').length;
|
|
234
|
+
return String(text ?? '') + ' '.repeat(Math.max(0, width - len));
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function fitCell(text, width) {
|
|
238
|
+
const str = String(text ?? '');
|
|
239
|
+
if (str.length <= width) return str + ' '.repeat(width - str.length);
|
|
240
|
+
return `${str.slice(0, width - 1)}…`;
|
|
241
|
+
}
|
package/src/history.js
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
4
|
+
|
|
5
|
+
export async function loadHistory(dir) {
|
|
6
|
+
const absDir = path.resolve(dir);
|
|
7
|
+
let entries;
|
|
8
|
+
try {
|
|
9
|
+
entries = await fs.readdir(absDir);
|
|
10
|
+
} catch {
|
|
11
|
+
throw new Error(`Cannot read directory: ${absDir}`);
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
const jsonFiles = entries
|
|
15
|
+
.filter((name) => name.endsWith('.json'))
|
|
16
|
+
.map((name) => path.join(absDir, name))
|
|
17
|
+
.sort();
|
|
18
|
+
|
|
19
|
+
const reports = [];
|
|
20
|
+
for (const filePath of jsonFiles) {
|
|
21
|
+
try {
|
|
22
|
+
const text = await fs.readFile(filePath, 'utf8');
|
|
23
|
+
const parsed = JSON.parse(text);
|
|
24
|
+
if (parsed.tool === 'fm-bench' && parsed.summary) {
|
|
25
|
+
reports.push({ filePath, report: parsed });
|
|
26
|
+
}
|
|
27
|
+
} catch {
|
|
28
|
+
// skip unparseable files
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
return reports;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export function renderHistoryReport(reports, options = {}) {
|
|
36
|
+
if (reports.length === 0) {
|
|
37
|
+
return 'No fm-bench JSON reports found in the given directory.';
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const { color = false, ascii = false } = options;
|
|
41
|
+
const V = ascii ? '|' : '│';
|
|
42
|
+
const H = ascii ? '-' : '─';
|
|
43
|
+
const TL = ascii ? '+' : '┌';
|
|
44
|
+
const TR = ascii ? '+' : '┐';
|
|
45
|
+
const BL = ascii ? '+' : '└';
|
|
46
|
+
const BR = ascii ? '+' : '┘';
|
|
47
|
+
const ML = ascii ? '+' : '├';
|
|
48
|
+
const MR = ascii ? '+' : '┤';
|
|
49
|
+
const TJ = ascii ? '+' : '┬';
|
|
50
|
+
const MJ = ascii ? '+' : '┼';
|
|
51
|
+
const BJ = ascii ? '+' : '┴';
|
|
52
|
+
|
|
53
|
+
const colWidths = [19, 8, 3, 6, 9, 9, 9, 9, 5];
|
|
54
|
+
const headers = ['STARTED AT', 'MODEL', 'C', 'RUNS', 'TTFT P50', 'E2E P50', 'E2E P95', 'USER T/S', 'SUCC'];
|
|
55
|
+
|
|
56
|
+
const renderRule = (l, j, r) => `${l}${colWidths.map((w) => H.repeat(w + 2)).join(j)}${r}`;
|
|
57
|
+
const renderRowLine = (cells) => `${V}${cells.map((text, i) => ` ${fit(text, colWidths[i])} `).join(V)}${V}`;
|
|
58
|
+
|
|
59
|
+
const lines = [];
|
|
60
|
+
lines.push(`fm-bench history (${reports.length} report${reports.length === 1 ? '' : 's'})`);
|
|
61
|
+
lines.push('');
|
|
62
|
+
lines.push(renderRule(TL, TJ, TR));
|
|
63
|
+
lines.push(renderRowLine(headers.map((h, i) => h)));
|
|
64
|
+
lines.push(renderRule(ML, MJ, MR));
|
|
65
|
+
|
|
66
|
+
for (const { filePath, report } of reports) {
|
|
67
|
+
const started = report.startedAt ? report.startedAt.slice(0, 19).replace('T', ' ') : '?';
|
|
68
|
+
const modelRows = report.summary ?? [];
|
|
69
|
+
|
|
70
|
+
for (const item of modelRows) {
|
|
71
|
+
if (!item.available) continue;
|
|
72
|
+
const row = [
|
|
73
|
+
started,
|
|
74
|
+
item.model ?? '-',
|
|
75
|
+
String(item.concurrency ?? 1),
|
|
76
|
+
String(item.successes ?? '-'),
|
|
77
|
+
item.ttft?.p50 != null ? formatMs(item.ttft.p50) : '-',
|
|
78
|
+
item.latency?.p50 != null ? formatMs(item.latency.p50) : '-',
|
|
79
|
+
item.latency?.p95 != null ? formatMs(item.latency.p95) : '-',
|
|
80
|
+
item.tokensPerSecond?.avg != null ? formatNumber(item.tokensPerSecond.avg) : '-',
|
|
81
|
+
item.successRate != null ? formatPercent(item.successRate) : '-'
|
|
82
|
+
];
|
|
83
|
+
lines.push(renderRowLine(row));
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
lines.push(renderRule(BL, BJ, BR));
|
|
88
|
+
return lines.join('\n');
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function fit(text, width) {
|
|
92
|
+
const str = String(text ?? '');
|
|
93
|
+
if (str.length <= width) return str + ' '.repeat(width - str.length);
|
|
94
|
+
return `${str.slice(0, width - 1)}…`;
|
|
95
|
+
}
|
package/src/prompts.js
CHANGED
|
@@ -93,6 +93,72 @@ const PROFILES = {
|
|
|
93
93
|
id: 'summarize',
|
|
94
94
|
prompt: 'Summarize this in two bullets: local model benchmarks should measure first-token latency, total latency, throughput, failures, and the exact prompt suite so results can be compared later.'
|
|
95
95
|
}
|
|
96
|
+
],
|
|
97
|
+
reasoning: [
|
|
98
|
+
{
|
|
99
|
+
id: 'math-word',
|
|
100
|
+
prompt: 'A server processes 1,200 requests per minute at peak. Each request uses 0.8 ms of CPU time on average. How many CPU cores are needed to handle peak load at 70% utilization? Show your reasoning step-by-step, then give a single final answer.'
|
|
101
|
+
},
|
|
102
|
+
{
|
|
103
|
+
id: 'logic-sequence',
|
|
104
|
+
prompt: 'Five engineers — Alice, Bob, Carol, Dave, Eve — each deploy one service. Alice deploys before Bob. Carol deploys after Dave but before Eve. Bob deploys before Dave. List the deployment order from first to last.'
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
id: 'causal-chain',
|
|
108
|
+
prompt: 'A CI pipeline has: lint (2 min), unit tests (5 min, parallel), integration tests (8 min, depends on unit), build (3 min, depends on integration), deploy (1 min, depends on build). What is the minimum wall-clock time from start to deployed? Explain each step.'
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
id: 'estimation',
|
|
112
|
+
prompt: 'Estimate how many tokens per day a popular AI coding assistant might process if it has 500,000 daily active users, each averaging 30 completions of 200 output tokens. Show your calculation and state any assumptions.'
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
id: 'debug-logic',
|
|
116
|
+
prompt: 'This function should return the median of a list: def median(lst): lst.sort(); n=len(lst); return lst[n//2] if n%2 else (lst[n//2-1]+lst[n//2])/2. Find all bugs and explain why each is a bug.'
|
|
117
|
+
}
|
|
118
|
+
],
|
|
119
|
+
coding: [
|
|
120
|
+
{
|
|
121
|
+
id: 'code-review',
|
|
122
|
+
prompt: 'Review this TypeScript for correctness, performance, and readability issues: async function fetchAll(urls: string[]) { const results = []; for (const url of urls) { const r = await fetch(url); results.push(await r.json()); } return results; } Give three specific improvements with brief explanations.'
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
id: 'refactor',
|
|
126
|
+
prompt: 'Refactor this JavaScript to be cleaner and handle edge cases: function getUser(id, cb) { db.query("SELECT * FROM users WHERE id=" + id, function(err, rows) { if (err) { cb(null, err); } else { cb(rows[0]); } }); } Return only the improved code and a two-sentence explanation.'
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
id: 'algorithm',
|
|
130
|
+
prompt: 'Write a JavaScript function findDuplicates(arr) that returns all duplicate values in O(n) time and O(n) space. Include a brief complexity explanation and two edge-case examples.'
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
id: 'explain-code',
|
|
134
|
+
prompt: 'Explain what this code does, why it might be used, and one potential problem: const cache = new WeakMap(); function memoize(fn) { return function(...args) { if (!cache.has(this)) cache.set(this, new Map()); const key = JSON.stringify(args); if (!cache.get(this).has(key)) cache.get(this).set(key, fn.apply(this, args)); return cache.get(this).get(key); }; }'
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
id: 'system-design',
|
|
138
|
+
prompt: 'Design a rate limiter in 120 words or fewer: specify the data structure, the algorithm (token bucket, sliding window, or fixed window), and how you handle distributed deployments. Be precise and practical.'
|
|
139
|
+
}
|
|
140
|
+
],
|
|
141
|
+
creative: [
|
|
142
|
+
{
|
|
143
|
+
id: 'product-announcement',
|
|
144
|
+
prompt: 'Write a 100-word product announcement for a developer tool called "PulseDB" that shows real-time query performance heatmaps in the terminal. Tone: enthusiastic but not hypey. Include one concrete example of what a user would see.'
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
id: 'error-message',
|
|
148
|
+
prompt: 'Rewrite this cryptic error into a helpful, actionable message a junior developer could act on: "Error: ECONNREFUSED 127.0.0.1:5432 errno: -111 syscall: connect code: ECONNREFUSED". Include what likely caused it and the first two things to check.'
|
|
149
|
+
},
|
|
150
|
+
{
|
|
151
|
+
id: 'technical-analogy',
|
|
152
|
+
prompt: 'Explain CPU context switching to someone who has never programmed, using a single concrete analogy from everyday life. Keep it under 80 words and make sure the analogy captures the performance cost.'
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
id: 'commit-message',
|
|
156
|
+
prompt: 'Write a clear and conventional git commit message for a change that adds retry logic with exponential backoff and jitter to the HTTP client. Include a subject line and a 3-bullet body.'
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
id: 'doc-summary',
|
|
160
|
+
prompt: 'Write a one-paragraph README introduction for an open-source CLI tool called "logslice" that extracts time-bounded log windows from large log files without loading them fully into memory. Target audience: backend engineers.'
|
|
161
|
+
}
|
|
96
162
|
]
|
|
97
163
|
};
|
|
98
164
|
|
package/src/table.js
CHANGED
|
@@ -50,8 +50,12 @@ export function renderBenchmarkReport(payload, options = {}) {
|
|
|
50
50
|
|
|
51
51
|
const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
|
|
52
52
|
const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
|
|
53
|
+
const tags = payload.options?.tags?.length ? payload.options.tags : [];
|
|
54
|
+
const note = payload.options?.note ?? null;
|
|
53
55
|
lines.push(truncate(title, width));
|
|
54
56
|
lines.push(truncate(meta, width));
|
|
57
|
+
if (tags.length > 0) lines.push(truncate(`tags: ${tags.join(', ')}`, width));
|
|
58
|
+
if (note) lines.push(truncate(`note: ${note}`, width));
|
|
55
59
|
lines.push('');
|
|
56
60
|
|
|
57
61
|
if (mode === 'compact') {
|
|
@@ -121,6 +125,49 @@ export function renderLegend(options = {}) {
|
|
|
121
125
|
return renderWrappedTable(['table', 'column', 'definition', 'rule'], rows, legendColumnWidths(width), options);
|
|
122
126
|
}
|
|
123
127
|
|
|
128
|
+
export function renderLatencyHistogram(results, options = {}) {
|
|
129
|
+
const { ascii = false, color = false, width: termWidth = 80 } = options;
|
|
130
|
+
const successes = results.filter((r) => r.ok && Number.isFinite(r.durationMs));
|
|
131
|
+
if (successes.length === 0) return 'No successful results to histogram.';
|
|
132
|
+
|
|
133
|
+
const values = successes.map((r) => r.durationMs).sort((a, b) => a - b);
|
|
134
|
+
const min = values[0];
|
|
135
|
+
const max = values[values.length - 1];
|
|
136
|
+
|
|
137
|
+
if (min === max) {
|
|
138
|
+
return `E2E latency histogram: all ${values.length} values = ${formatMs(min)}`;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const numBuckets = Math.min(20, Math.max(5, Math.floor((termWidth - 20) / 4)));
|
|
142
|
+
const bucketSize = (max - min) / numBuckets;
|
|
143
|
+
const counts = new Array(numBuckets).fill(0);
|
|
144
|
+
|
|
145
|
+
for (const v of values) {
|
|
146
|
+
const idx = Math.min(numBuckets - 1, Math.floor((v - min) / bucketSize));
|
|
147
|
+
counts[idx] += 1;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const maxCount = Math.max(...counts);
|
|
151
|
+
const barWidth = Math.max(1, termWidth - 24);
|
|
152
|
+
const lines = [];
|
|
153
|
+
const bar = ascii ? '#' : '█';
|
|
154
|
+
const halfBar = ascii ? ':' : '▌';
|
|
155
|
+
|
|
156
|
+
lines.push(`E2E latency distribution (n=${values.length}, ${formatMs(min)}…${formatMs(max)}):`);
|
|
157
|
+
|
|
158
|
+
for (let i = 0; i < numBuckets; i += 1) {
|
|
159
|
+
const bucketMin = min + i * bucketSize;
|
|
160
|
+
const bucketMax = bucketMin + bucketSize;
|
|
161
|
+
const label = `${formatMs(bucketMin)}`.padStart(7);
|
|
162
|
+
const fillLength = Math.round((counts[i] / maxCount) * barWidth);
|
|
163
|
+
const filled = bar.repeat(fillLength);
|
|
164
|
+
const countStr = counts[i] > 0 ? String(counts[i]) : '';
|
|
165
|
+
lines.push(`${label} ${filled}${countStr ? ` ${countStr}` : ''}`);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
return lines.join('\n');
|
|
169
|
+
}
|
|
170
|
+
|
|
124
171
|
export function formatMs(value) {
|
|
125
172
|
if (value == null || !Number.isFinite(value)) return '-';
|
|
126
173
|
if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
|