fm-bench 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -174
- package/package.json +1 -1
- package/src/cli.js +14 -11
- package/src/system.js +65 -0
package/README.md
CHANGED
|
@@ -1,41 +1,34 @@
|
|
|
1
1
|
# fm-bench
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Benchmark Apple's `fm` command on macOS 27+.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Measure latency, throughput, streaming smoothness, stability, and goodput across Apple Foundation Models — with repeatable prompt suites and JSON/CSV reports for automation.
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
## Why fm-bench exists
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Apple's Foundation Models can run on-device, through Private Cloud Compute, and through model adapters. That makes raw model quality only half the story.
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
npm install -g fm-bench
|
|
13
|
-
```
|
|
14
|
-
|
|
15
|
-
You can also install directly from GitHub:
|
|
16
|
-
|
|
17
|
-
```sh
|
|
18
|
-
npm install -g --install-links git+https://github.com/devinoldenburg/fm-bench.git
|
|
19
|
-
```
|
|
11
|
+
For real apps, the important questions are:
|
|
20
12
|
|
|
21
|
-
|
|
13
|
+
- How fast does the first token arrive?
|
|
14
|
+
- Does streaming stay smooth?
|
|
15
|
+
- How stable is latency over repeated runs?
|
|
16
|
+
- What happens under concurrency?
|
|
17
|
+
- Which model/hardware pair meets an interactive SLO?
|
|
22
18
|
|
|
23
|
-
|
|
24
|
-
npm install
|
|
25
|
-
npm link
|
|
26
|
-
fm-bench doctor
|
|
27
|
-
```
|
|
19
|
+
`fm-bench` answers those questions with repeatable local benchmarks. Think of it as GeekBench for Apple Foundation Models — run it, get numbers, compare across hardware, models, and macOS updates.
|
|
28
20
|
|
|
29
21
|
## Quick Start
|
|
30
22
|
|
|
31
23
|
```sh
|
|
24
|
+
npm install -g fm-bench
|
|
32
25
|
fm-bench
|
|
33
26
|
```
|
|
34
27
|
|
|
35
|
-
|
|
28
|
+
One command discovers your models, runs the standard prompt suite, and prints a full benchmark report:
|
|
36
29
|
|
|
37
30
|
```text
|
|
38
|
-
fm-bench 0.
|
|
31
|
+
fm-bench 0.5.0 | darwin/arm64 | fm
|
|
39
32
|
prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
|
|
40
33
|
|
|
41
34
|
┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
|
|
@@ -46,132 +39,173 @@ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skip
|
|
|
46
39
|
└───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
|
|
47
40
|
```
|
|
48
41
|
|
|
49
|
-
|
|
42
|
+
Wide terminals add TTFT P95, TPOT, decode/prefill throughput, chunk-gap smoothness, and 95% CI columns. Narrow terminals switch to compact model cards automatically.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
50
45
|
|
|
51
46
|
```sh
|
|
52
|
-
fm-bench
|
|
53
|
-
fm-bench models [options]
|
|
54
|
-
fm-bench compare <before.json> <after.json> [options]
|
|
55
|
-
fm-bench history [dir] [options]
|
|
56
|
-
fm-bench legend [options]
|
|
57
|
-
fm-bench doctor [options]
|
|
47
|
+
npm install -g fm-bench
|
|
58
48
|
```
|
|
59
49
|
|
|
60
|
-
|
|
50
|
+
Install directly from GitHub (always latest):
|
|
61
51
|
|
|
62
|
-
|
|
52
|
+
```sh
|
|
53
|
+
npm install -g --install-links git+https://github.com/devinoldenburg/fm-bench.git
|
|
54
|
+
```
|
|
63
55
|
|
|
64
|
-
|
|
56
|
+
Local development:
|
|
65
57
|
|
|
66
|
-
|
|
58
|
+
```sh
|
|
59
|
+
npm install && npm link
|
|
60
|
+
fm-bench doctor # verify your setup
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**Requirements:** macOS 27+, Node.js 20+, Apple Intelligence enabled.
|
|
67
64
|
|
|
68
|
-
|
|
65
|
+
## Commands
|
|
69
66
|
|
|
70
|
-
|
|
67
|
+
| Command | What it does |
|
|
68
|
+
|---------|-------------|
|
|
69
|
+
| `fm-bench` | Run the full benchmark (default) |
|
|
70
|
+
| `fm-bench models` | List discovered models, availability, and quota |
|
|
71
|
+
| `fm-bench compare <a.json> <b.json>` | Regression diff: before/after metrics with color-coded deltas |
|
|
72
|
+
| `fm-bench history [dir]` | Chronological trend table from a directory of saved reports |
|
|
73
|
+
| `fm-bench legend` | Definitions for every table column and color rule |
|
|
74
|
+
| `fm-bench doctor` | Environment check: Node, macOS, `fm`, CPU, memory, thermals, battery |
|
|
71
75
|
|
|
72
|
-
##
|
|
76
|
+
## Common Recipes
|
|
73
77
|
|
|
74
78
|
```sh
|
|
79
|
+
# Quick smoke test
|
|
80
|
+
fm-bench --profile quick
|
|
81
|
+
|
|
82
|
+
# Standard 5-run benchmark with SLO budgets
|
|
83
|
+
fm-bench --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
84
|
+
|
|
85
|
+
# Sweep concurrency to find your throughput ceiling
|
|
86
|
+
fm-bench --sweep-concurrency 1,2,4 --runs 3
|
|
87
|
+
|
|
88
|
+
# Stress both on-device and PCC models
|
|
75
89
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
fm-bench --
|
|
79
|
-
fm-bench --profile
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
fm-bench --
|
|
83
|
-
fm-bench --
|
|
84
|
-
fm-bench
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
fm-bench
|
|
91
|
-
fm-bench
|
|
90
|
+
|
|
91
|
+
# Reasoning and coding workloads
|
|
92
|
+
fm-bench --profile reasoning --runs 5
|
|
93
|
+
fm-bench --profile coding --runs 3 --histogram
|
|
94
|
+
|
|
95
|
+
# Archive runs and compare before/after a macOS update
|
|
96
|
+
fm-bench --output-dir reports/ --tag before-update
|
|
97
|
+
fm-bench --output-dir reports/ --tag after-update
|
|
98
|
+
fm-bench compare reports/fm-bench_*before*.json reports/fm-bench_*after*.json
|
|
99
|
+
|
|
100
|
+
# Fail CI when SLOs regress
|
|
101
|
+
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
|
|
102
|
+
|
|
103
|
+
# Save JSON for automation
|
|
104
|
+
fm-bench --json --out bench.json
|
|
105
|
+
fm-bench --format csv --out bench.csv
|
|
92
106
|
```
|
|
93
107
|
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
108
|
+
## Options Reference
|
|
109
|
+
|
|
110
|
+
**Workload**
|
|
111
|
+
|
|
112
|
+
| Flag | Default | Description |
|
|
113
|
+
|------|---------|-------------|
|
|
114
|
+
| `-m, --models <list>` | all | Comma-separated or repeated model names |
|
|
115
|
+
| `-r, --runs <n>` | 1 | Measured runs per prompt/model |
|
|
116
|
+
| `--warmup <n>` | 0 | Warmup runs per model before measurement |
|
|
117
|
+
| `-c, --concurrency <n>` | 1 | Parallel `fm` processes |
|
|
118
|
+
| `--sweep-concurrency <list>` | — | Separate operating points, e.g. `1,2,4` |
|
|
119
|
+
| `--request-rate <rps>` | — | Pace request starts at a target rate |
|
|
120
|
+
| `--ramp-up-ms <n>` | 0 | Gradually ramp pacing over `n` ms |
|
|
121
|
+
| `--timeout-ms <n>` | 60000 | Timeout per `fm` call |
|
|
122
|
+
| `--retry <n>` | 0 | Retry failed calls with exponential backoff (500ms–4s) |
|
|
123
|
+
| `--profile <name>` | standard | Built-in prompt suite (see Profiles) |
|
|
124
|
+
| `-p, --prompt <text>` | — | Custom prompt, repeatable |
|
|
125
|
+
| `--prompt-file <file>` | — | JSON, JSONL, or blank-line separated prompts |
|
|
126
|
+
| `-i, --instructions <text>` | — | Passed to `fm respond` |
|
|
127
|
+
|
|
128
|
+
**Quality Gates**
|
|
129
|
+
|
|
130
|
+
| Flag | Description |
|
|
131
|
+
|------|-------------|
|
|
132
|
+
| `--slo-ttft-ms <n>` | Count a run as good only if TTFT ≤ n ms |
|
|
133
|
+
| `--slo-e2e-ms <n>` | Count a run as good only if E2E latency ≤ n ms |
|
|
134
|
+
| `--slo-tpot-ms <n>` | Count a run as good only if TPOT ≤ n ms |
|
|
135
|
+
| `--ci` | Exit 1 if any run fails or any SLO is violated (for pipelines) |
|
|
136
|
+
| `--fail-fast` | Stop after the first failed run |
|
|
137
|
+
|
|
138
|
+
**Output**
|
|
139
|
+
|
|
140
|
+
| Flag | Description |
|
|
141
|
+
|------|-------------|
|
|
142
|
+
| `--json` / `--csv` | Output format (also `--format table\|json\|csv`) |
|
|
143
|
+
| `-o, --out <file>` | Save a report to a file |
|
|
144
|
+
| `--output-dir <dir>` | Auto-save a timestamped JSON report to a directory |
|
|
145
|
+
| `--tag <name>` | Label this run; repeatable; appears in payload and header |
|
|
146
|
+
| `--note <text>` | Freeform annotation in payload and header |
|
|
147
|
+
| `--histogram` | Print ASCII latency distribution chart after the report |
|
|
148
|
+
| `--capture-output` | Include raw model output in JSON reports |
|
|
149
|
+
| `-v, --verbose` | Append per-run CSV after the summary table |
|
|
150
|
+
|
|
151
|
+
**Display**
|
|
152
|
+
|
|
153
|
+
| Flag | Description |
|
|
154
|
+
|------|-------------|
|
|
155
|
+
| `--color` / `--no-color` | Force or disable ANSI colors (auto on TTYs) |
|
|
156
|
+
| `--ascii` | Plain ASCII table borders instead of Unicode |
|
|
157
|
+
| `--compact` | Force narrow terminal layout |
|
|
158
|
+
| `--width <n>` | Render as if the terminal is `n` columns wide |
|
|
159
|
+
| `--progress` / `--no-progress` | Force or disable the live progress line |
|
|
124
160
|
|
|
125
161
|
## Prompt Profiles
|
|
126
162
|
|
|
127
|
-
Nine built-in
|
|
128
|
-
|
|
129
|
-
| Profile | Prompts | Focus |
|
|
130
|
-
|---------|---------|-------|
|
|
131
|
-
| `quick` | 1 | Single-prompt smoke test |
|
|
132
|
-
| `standard` | 3 | Short chat, structured JSON, medium generation |
|
|
133
|
-
| `interactive` | 3 | Short conversational turns |
|
|
134
|
-
| `throughput` | 3 | Longer generation and transformation tasks |
|
|
135
|
-
| `client` | 5 | Broad real-world mix: chat, content, extraction, summarization, code review |
|
|
136
|
-
| `stress` | 5 | Diverse stress mix with math and reasoning |
|
|
137
|
-
| `reasoning` | 5 | Multi-step math, logic, causal chains, estimation, debugging |
|
|
138
|
-
| `coding` | 5 | Code review, refactoring, algorithms, code explanation, system design |
|
|
139
|
-
| `creative` | 5 | Product copy, error messages, analogies, commit messages, doc writing |
|
|
163
|
+
Nine built-in suites, choose the one that matches your use case:
|
|
140
164
|
|
|
141
|
-
|
|
165
|
+
| Profile | Prompts | Best for |
|
|
166
|
+
|---------|---------|----------|
|
|
167
|
+
| `quick` | 1 | Smoke test, fast health check |
|
|
168
|
+
| `standard` | 3 | Default — short chat, JSON generation, medium output |
|
|
169
|
+
| `interactive` | 3 | Conversational latency (TTFT-heavy) |
|
|
170
|
+
| `throughput` | 3 | Longer generation, token throughput signal |
|
|
171
|
+
| `client` | 5 | Real-world mix: chat, content, extraction, summarization, code |
|
|
172
|
+
| `stress` | 5 | High-load mix with math and reasoning |
|
|
173
|
+
| `reasoning` | 5 | Multi-step logic, estimation, debugging — capability + speed |
|
|
174
|
+
| `coding` | 5 | Code review, refactoring, algorithms, system design |
|
|
175
|
+
| `creative` | 5 | Product copy, analogies, commit messages, docs |
|
|
142
176
|
|
|
143
|
-
##
|
|
177
|
+
## Regression Tracking
|
|
144
178
|
|
|
145
|
-
|
|
179
|
+
Track performance across macOS updates, model changes, or hardware swaps:
|
|
146
180
|
|
|
147
181
|
```sh
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
fm-bench --runs 5 --json --out after.json
|
|
151
|
-
fm-bench compare before.json after.json
|
|
152
|
-
```
|
|
182
|
+
# Before
|
|
183
|
+
fm-bench --profile coding --runs 5 --output-dir reports/ --tag before
|
|
153
184
|
|
|
154
|
-
|
|
185
|
+
# After the change
|
|
186
|
+
fm-bench --profile coding --runs 5 --output-dir reports/ --tag after
|
|
155
187
|
|
|
156
|
-
|
|
188
|
+
# See what changed
|
|
189
|
+
fm-bench compare reports/fm-bench_*before*.json reports/fm-bench_*after*.json
|
|
190
|
+
```
|
|
157
191
|
|
|
158
|
-
|
|
192
|
+
The compare output shows each model/concurrency row with the before value, a color-coded percent delta (green = improvement, red = regression), and the after value — for TTFT, E2E, TPOT, tokens/s, RPS, success rate, and CV.
|
|
159
193
|
|
|
160
194
|
```sh
|
|
161
|
-
|
|
162
|
-
fm-bench --output-dir reports/ --runs 3
|
|
195
|
+
# View the full trend over time
|
|
163
196
|
fm-bench history reports/
|
|
164
197
|
```
|
|
165
198
|
|
|
166
199
|
## CI Integration
|
|
167
200
|
|
|
168
|
-
|
|
201
|
+
Gate deployments or model updates on benchmark quality:
|
|
169
202
|
|
|
170
203
|
```sh
|
|
204
|
+
# Fails with exit code 1 if TTFT > 750ms or E2E > 4s on any run
|
|
171
205
|
fm-bench --ci --slo-ttft-ms 750 --slo-e2e-ms 4000 --runs 5
|
|
172
206
|
```
|
|
173
207
|
|
|
174
|
-
|
|
208
|
+
Prints `fm-bench ci: PASS` or `fm-bench ci: FAIL — <reason>` to stderr. Designed for GitHub Actions, Buildkite, or any shell-based pipeline.
|
|
175
209
|
|
|
176
210
|
## Prompt Files
|
|
177
211
|
|
|
@@ -195,95 +229,52 @@ Plain text files are split on blank lines.
|
|
|
195
229
|
|
|
196
230
|
## Metrics
|
|
197
231
|
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
- TTFT, or time to first streamed output.
|
|
201
|
-
- E2E latency, or full response wall-clock latency.
|
|
202
|
-
- TPOT, or decode time per output token after the first output token.
|
|
203
|
-
- second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
|
|
204
|
-
- prefill tokens per second, or prompt tokens divided by TTFT.
|
|
205
|
-
- output tokens per second per request.
|
|
206
|
-
- total output token throughput across the measured window.
|
|
207
|
-
- total token throughput, including prompt and output tokens.
|
|
208
|
-
- requests per second across the measured window.
|
|
209
|
-
- goodput percentage and goodput RPS when SLO flags are set.
|
|
210
|
-
- coefficient of variation (CV) and confidence interval context for stability.
|
|
211
|
-
- prompt and output token counts.
|
|
212
|
-
- p50, p95, and p99 tail latency views.
|
|
213
|
-
- repeatability across repeated runs of the same prompt.
|
|
214
|
-
- success and failure counts.
|
|
215
|
-
- unavailable model notes.
|
|
216
|
-
|
|
217
|
-
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
|
|
232
|
+
**Latency** — TTFT (p50/p95), E2E (p50/p95/p99), TPOT (p50/p95), 95% confidence interval, coefficient of variation (CV).
|
|
218
233
|
|
|
219
|
-
|
|
234
|
+
**Throughput** — prefill tokens/s, decode tokens/s, output tokens/s per request, aggregate system tokens/s, requests per second.
|
|
220
235
|
|
|
221
|
-
|
|
236
|
+
**Streaming quality** — second-chunk delay, chunk-gap p95. Captured from `stdout` chunks during streaming runs.
|
|
222
237
|
|
|
223
|
-
|
|
238
|
+
**Reliability** — success rate, goodput rate and RPS against SLO budgets, repeatability (most common output hash frequency across repeated runs).
|
|
224
239
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
```sh
|
|
228
|
-
fm-bench legend
|
|
229
|
-
fm-bench legend --json
|
|
230
|
-
fm-bench legend --csv
|
|
231
|
-
```
|
|
240
|
+
**Stability** — CV (stddev/mean for E2E latency); green ≤10%, yellow ≤25%, red >25%.
|
|
232
241
|
|
|
233
|
-
|
|
242
|
+
Token counts come from `fm token-count --quiet`. If `fm` cannot count tokens, those fields are blank while character throughput is still reported.
|
|
234
243
|
|
|
235
|
-
|
|
244
|
+
Terminal layout is responsive: wide → full scoreboard + detail tables, medium → tighter single table, narrow → compact model cards. Use `--width` to preview any layout and `--ascii` for log-friendly output.
|
|
236
245
|
|
|
237
|
-
|
|
246
|
+
## Colors and Legend
|
|
238
247
|
|
|
239
|
-
|
|
248
|
+
Table output is color-coded on interactive terminals — **green** is better/passing, **yellow** is marginal/partial, **red** is failing/unstable. Fixed thresholds apply to success rate, goodput, CV, and repeatability. Latency uses SLO thresholds when set, otherwise lower-is-better relative ranking. Throughput uses higher-is-better relative ranking.
|
|
240
249
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
- green: passing, steadier, or better than the current comparison set.
|
|
246
|
-
- yellow: marginal, partial, or near a budget.
|
|
247
|
-
- red: failing a budget, unstable, or slower/lower than peers.
|
|
248
|
-
|
|
249
|
-
Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. CV is green at `<=10%`, yellow at `<=25%`, and red above `25%` because higher CV means less steady latency. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
|
|
250
|
+
```sh
|
|
251
|
+
fm-bench legend # full column definitions and color rules
|
|
252
|
+
fm-bench legend --json # machine-readable
|
|
253
|
+
```
|
|
250
254
|
|
|
251
|
-
|
|
255
|
+
`NO_COLOR=1` disables color; `FORCE_COLOR=1` or `--color` enables it. `--ascii` switches to plain ASCII borders for log systems.
|
|
252
256
|
|
|
253
|
-
|
|
257
|
+
A live single-line progress indicator runs on stderr during interactive sessions. The final report always goes to stdout — `--json`, `--csv`, and `--out` stay automation-friendly.
|
|
254
258
|
|
|
255
259
|
## Requirements
|
|
256
260
|
|
|
257
|
-
- macOS 27 or newer
|
|
261
|
+
- macOS 27 or newer (Apple's `fm` CLI is preinstalled).
|
|
258
262
|
- Node.js 20 or newer.
|
|
259
|
-
- Apple Intelligence
|
|
263
|
+
- Apple Intelligence enabled on the device.
|
|
264
|
+
|
|
265
|
+
`pcc` (Private Cloud Compute) availability depends on Apple's current eligibility. `fm-bench` shows it as skipped if `fm available --model pcc` reports unavailable.
|
|
260
266
|
|
|
261
|
-
|
|
267
|
+
See [docs/methodology.md](docs/methodology.md) for benchmark methodology and metric references.
|
|
262
268
|
|
|
263
269
|
## Development
|
|
264
270
|
|
|
265
271
|
```sh
|
|
266
272
|
npm install
|
|
267
|
-
npm test
|
|
273
|
+
npm test # node --test
|
|
268
274
|
npm run lint
|
|
269
|
-
npm pack
|
|
270
275
|
```
|
|
271
276
|
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
## Releases
|
|
275
|
-
|
|
276
|
-
Releases are tag-driven:
|
|
277
|
-
|
|
278
|
-
```sh
|
|
279
|
-
npm run release:patch
|
|
280
|
-
npm run release:minor
|
|
281
|
-
npm run release:major
|
|
282
|
-
```
|
|
283
|
-
|
|
284
|
-
Pushing a `v*.*.*` tag runs the GitHub Release workflow, publishes to npm using the repository `NPM_TOKEN` secret, and creates a GitHub Release with generated notes.
|
|
285
|
-
|
|
286
|
-
Maintainers can also run the **Version** workflow manually from GitHub Actions to bump the version and push the tag.
|
|
277
|
+
No runtime npm dependencies.
|
|
287
278
|
|
|
288
279
|
## License
|
|
289
280
|
|
package/package.json
CHANGED
package/src/cli.js
CHANGED
|
@@ -6,6 +6,7 @@ import { loadHistory, renderHistoryReport } from './history.js';
|
|
|
6
6
|
import { runProcess } from './process.js';
|
|
7
7
|
import { createProgress } from './progress.js';
|
|
8
8
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
9
|
+
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
9
10
|
import { legendEntries, renderBenchmarkReport, renderLatencyHistogram, renderLegend, renderModelsTable } from './table.js';
|
|
10
11
|
|
|
11
12
|
const require = createRequire(import.meta.url);
|
|
@@ -473,21 +474,23 @@ async function runDoctor(options) {
|
|
|
473
474
|
}
|
|
474
475
|
|
|
475
476
|
const thermalResult = await runProcess('pmset', ['-g', 'therm'], { timeoutMs: 5_000 });
|
|
476
|
-
const
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
const limit = Number.parseInt(thermalMatch[1], 10);
|
|
477
|
+
const thermal = parseThermalOutput(`${thermalResult.stdout || ''}${thermalResult.stderr || ''}`);
|
|
478
|
+
if (thermal.available && thermal.schedulerLimit != null) {
|
|
479
|
+
const limit = thermal.schedulerLimit;
|
|
480
480
|
checks.push(['thermal limit', `${limit}%`, limit >= 100]);
|
|
481
|
+
} else if (thermal.available && thermal.healthyIdle) {
|
|
482
|
+
checks.push(['thermal', 'no thermal pressure', true]);
|
|
483
|
+
} else if (!thermal.available) {
|
|
484
|
+
checks.push(['thermal', 'unavailable', true]);
|
|
481
485
|
}
|
|
482
486
|
|
|
483
487
|
const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
|
|
484
|
-
const
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
const
|
|
488
|
-
const
|
|
489
|
-
|
|
490
|
-
checks.push(['battery', `${pct}${charging ? ' (charging/AC)' : ' (battery)'}`, charging || Number.parseInt(pctMatch?.[1], 10) >= 20]);
|
|
488
|
+
const battery = parseBatteryOutput(`${batteryResult.stdout || ''}${batteryResult.stderr || ''}`);
|
|
489
|
+
if (battery.present) {
|
|
490
|
+
const pct = battery.pct != null ? `${battery.pct}%` : '?%';
|
|
491
|
+
const source = battery.onAC ? 'charging/AC' : 'battery';
|
|
492
|
+
const ok = battery.onAC || (battery.pct != null && battery.pct >= 20);
|
|
493
|
+
checks.push(['battery', `${pct} (${source})`, ok]);
|
|
491
494
|
}
|
|
492
495
|
|
|
493
496
|
const inspection = await inspectModels(options);
|
package/src/system.js
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
// Parsers for macOS system probes used by `fm-bench doctor`.
|
|
2
|
+
//
|
|
3
|
+
// macOS 27 changed the text emitted by `pmset -g therm` and `pmset -g batt`.
|
|
4
|
+
// These helpers accept the raw stdout/stderr of those commands and return
|
|
5
|
+
// structured values so the doctor command and its tests are not coupled to
|
|
6
|
+
// the exact wording of any one macOS release.
|
|
7
|
+
|
|
8
|
+
export function parseThermalOutput(output = '') {
|
|
9
|
+
const text = String(output).trim();
|
|
10
|
+
if (!text) {
|
|
11
|
+
return { available: false, schedulerLimit: null, raw: '' };
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
const limitMatch = text.match(/CPU_Scheduler_Limit\s*=\s*(-?\d+)/);
|
|
15
|
+
if (limitMatch) {
|
|
16
|
+
const schedulerLimit = Number.parseInt(limitMatch[1], 10);
|
|
17
|
+
return {
|
|
18
|
+
available: true,
|
|
19
|
+
schedulerLimit,
|
|
20
|
+
raw: text
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// macOS 27 (and recent releases) prints these informational "Note:" lines
|
|
25
|
+
// when the system is NOT under thermal pressure. Treat that as a healthy,
|
|
26
|
+
// explicitly-reported state rather than silently dropping the check.
|
|
27
|
+
const hasNote = /^Note:/m.test(text);
|
|
28
|
+
if (hasNote) {
|
|
29
|
+
return {
|
|
30
|
+
available: true,
|
|
31
|
+
schedulerLimit: null,
|
|
32
|
+
raw: text,
|
|
33
|
+
healthyIdle: true
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
return { available: false, schedulerLimit: null, raw: text };
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function parseBatteryOutput(output = '') {
|
|
41
|
+
const text = String(output).trim();
|
|
42
|
+
if (!text) {
|
|
43
|
+
return { present: false, pct: null, onAC: false, raw: '' };
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
const lines = text.split(/\r?\n/);
|
|
47
|
+
const batteryLine = lines.find((line) => line.includes('%')) || '';
|
|
48
|
+
const pctMatch = batteryLine.match(/(\d+)%/);
|
|
49
|
+
const pct = pctMatch ? Number.parseInt(pctMatch[1], 10) : null;
|
|
50
|
+
|
|
51
|
+
// macOS 27 writes "Now drawing from 'AC Power'" on its own line and
|
|
52
|
+
// "AC attached" on the battery line. Older releases used "AC Power" in
|
|
53
|
+
// the battery line itself. Accept either form.
|
|
54
|
+
const onAC = /AC Power|AC attached|AC attached;/i.test(text)
|
|
55
|
+
|| /;\s*charging\b/i.test(batteryLine);
|
|
56
|
+
|
|
57
|
+
const present = /InternalBattery|present:\s*true/i.test(text);
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
present: present || pct != null,
|
|
61
|
+
pct,
|
|
62
|
+
onAC,
|
|
63
|
+
raw: text
|
|
64
|
+
};
|
|
65
|
+
}
|