fm-bench 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +153 -0
- package/bin/fm-bench.js +8 -0
- package/package.json +49 -0
- package/src/ansi.js +5 -0
- package/src/bench.js +161 -0
- package/src/cli.js +290 -0
- package/src/fm.js +197 -0
- package/src/process.js +82 -0
- package/src/prompts.js +120 -0
- package/src/report.js +51 -0
- package/src/stats.js +85 -0
- package/src/table.js +94 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Devin Oldenburg
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# fm-bench
|
|
2
|
+
|
|
3
|
+
`fm-bench` is a dynamic benchmark CLI for Apple's `fm` command on macOS 27 and newer.
|
|
4
|
+
|
|
5
|
+
It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, and prints a terminal table with latency and throughput stats.
|
|
6
|
+
|
|
7
|
+
Apple introduced the preinstalled `fm` command for macOS 27 as part of the Foundation Models tooling. `fm-bench` intentionally shells out to the system `fm` binary instead of linking private APIs, so it can adapt as Apple adds models or changes availability.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```sh
|
|
12
|
+
npm install -g fm-bench
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
You can also install directly from GitHub:
|
|
16
|
+
|
|
17
|
+
```sh
|
|
18
|
+
npm install -g --install-links git+https://github.com/devinoldenburg/fm-bench.git
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
For local development from this repository:
|
|
22
|
+
|
|
23
|
+
```sh
|
|
24
|
+
npm install
|
|
25
|
+
npm link
|
|
26
|
+
fm-bench doctor
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## Quick Start
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
fm-bench
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Example output:
|
|
36
|
+
|
|
37
|
+
```text
|
|
38
|
+
+--------+---------+------+----+------+-------+-------+-------+-------+--------+---------+------+
|
|
39
|
+
| model | status | runs | ok | fail | p50 | p95 | avg | tok/s | char/s | out tok | note |
|
|
40
|
+
+--------+---------+------+----+------+-------+-------+-------+-------+--------+---------+------+
|
|
41
|
+
| system | ok | 3 | 3 | - | 1.21s | 1.85s | 1.34s | 18.7 | 83 | 24 | |
|
|
42
|
+
| pcc | skipped | - | - | - | - | - | - | - | - | - | PCC inference is not available... |
|
|
43
|
+
+--------+---------+------+----+------+-------+-------+-------+-------+--------+---------+------+
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Commands
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
fm-bench [run] [options]
|
|
50
|
+
fm-bench models [options]
|
|
51
|
+
fm-bench doctor [options]
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
`run` is the default command. It benchmarks all models discovered from `fm` and skips models that are currently unavailable.
|
|
55
|
+
|
|
56
|
+
`models` lists discovered models, availability, descriptions, and quota output.
|
|
57
|
+
|
|
58
|
+
`doctor` checks Node, macOS, `fm`, and model availability.
|
|
59
|
+
|
|
60
|
+
## Benchmark Options
|
|
61
|
+
|
|
62
|
+
```sh
|
|
63
|
+
fm-bench --models system,pcc --runs 3 --profile stress
|
|
64
|
+
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
65
|
+
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
66
|
+
fm-bench --format csv --out reports/bench.csv
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Useful flags:
|
|
70
|
+
|
|
71
|
+
- `--models <list>`: comma-separated or repeated model names.
|
|
72
|
+
- `--runs <n>`: measured runs per prompt/model.
|
|
73
|
+
- `--warmup <n>`: warmup runs per model before measurement.
|
|
74
|
+
- `--concurrency <n>`: parallel `fm` processes.
|
|
75
|
+
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
76
|
+
- `--profile quick|standard|stress`: built-in prompt suite.
|
|
77
|
+
- `--prompt <text>`: custom prompt, repeatable.
|
|
78
|
+
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
79
|
+
- `--instructions <text>`: passed to `fm respond`.
|
|
80
|
+
- `--available-only`: hide unavailable discovered models.
|
|
81
|
+
- `--capture-output`: include raw model output in JSON reports.
|
|
82
|
+
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
83
|
+
- `--out <file>`: save a report.
|
|
84
|
+
|
|
85
|
+
## Prompt Files
|
|
86
|
+
|
|
87
|
+
JSON array:
|
|
88
|
+
|
|
89
|
+
```json
|
|
90
|
+
[
|
|
91
|
+
{ "id": "tiny", "prompt": "Reply with exactly: ok" },
|
|
92
|
+
{ "id": "json", "prompt": "Convert alpha, beta, gamma into JSON." }
|
|
93
|
+
]
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
JSONL:
|
|
97
|
+
|
|
98
|
+
```jsonl
|
|
99
|
+
{"id":"tiny","prompt":"Reply with exactly: ok"}
|
|
100
|
+
{"id":"latency","prompt":"Explain p95 latency in one sentence."}
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Plain text files are split on blank lines.
|
|
104
|
+
|
|
105
|
+
## Metrics
|
|
106
|
+
|
|
107
|
+
`fm-bench` reports:
|
|
108
|
+
|
|
109
|
+
- `p50`, `p95`, and average wall-clock latency.
|
|
110
|
+
- average output tokens per second.
|
|
111
|
+
- average characters per second.
|
|
112
|
+
- average output tokens.
|
|
113
|
+
- success and failure counts.
|
|
114
|
+
- unavailable model notes.
|
|
115
|
+
|
|
116
|
+
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, the token fields are left blank while character throughput is still reported.
|
|
117
|
+
|
|
118
|
+
## Requirements
|
|
119
|
+
|
|
120
|
+
- macOS 27 or newer for Apple's `fm` CLI.
|
|
121
|
+
- Node.js 20 or newer.
|
|
122
|
+
- Apple Intelligence and model availability configured for the machine.
|
|
123
|
+
|
|
124
|
+
Private Cloud Compute (`pcc`) availability depends on Apple's current eligibility and context. If `fm available --model pcc` reports unavailable, `fm-bench` will show it as skipped.
|
|
125
|
+
|
|
126
|
+
## Development
|
|
127
|
+
|
|
128
|
+
```sh
|
|
129
|
+
npm install
|
|
130
|
+
npm test
|
|
131
|
+
npm run lint
|
|
132
|
+
npm pack
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
The package has no runtime npm dependencies.
|
|
136
|
+
|
|
137
|
+
## Releases
|
|
138
|
+
|
|
139
|
+
Releases are tag-driven:
|
|
140
|
+
|
|
141
|
+
```sh
|
|
142
|
+
npm run release:patch
|
|
143
|
+
npm run release:minor
|
|
144
|
+
npm run release:major
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Pushing a `v*.*.*` tag runs the GitHub Release workflow, publishes to npm using the repository `NPM_TOKEN` secret, and creates a GitHub Release with generated notes.
|
|
148
|
+
|
|
149
|
+
Maintainers can also run the **Version** workflow manually from GitHub Actions to bump the version and push the tag.
|
|
150
|
+
|
|
151
|
+
## License
|
|
152
|
+
|
|
153
|
+
MIT
|
package/bin/fm-bench.js
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { runCli } from '../src/cli.js';
|
|
3
|
+
|
|
4
|
+
runCli(process.argv.slice(2)).catch((error) => {
|
|
5
|
+
const message = error?.message || String(error);
|
|
6
|
+
console.error(`fm-bench: ${message}`);
|
|
7
|
+
process.exitCode = typeof error?.exitCode === 'number' ? error.exitCode : 1;
|
|
8
|
+
});
|
package/package.json
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "fm-bench",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": {
|
|
7
|
+
"fm-bench": "./bin/fm-bench.js"
|
|
8
|
+
},
|
|
9
|
+
"files": [
|
|
10
|
+
"bin",
|
|
11
|
+
"src",
|
|
12
|
+
"README.md",
|
|
13
|
+
"LICENSE"
|
|
14
|
+
],
|
|
15
|
+
"scripts": {
|
|
16
|
+
"test": "node --test",
|
|
17
|
+
"lint": "node --check bin/fm-bench.js && find src test -name '*.js' -print0 | xargs -0 -n1 node --check",
|
|
18
|
+
"prepack": "npm test && npm run lint",
|
|
19
|
+
"release:patch": "npm version patch && git push --follow-tags",
|
|
20
|
+
"release:minor": "npm version minor && git push --follow-tags",
|
|
21
|
+
"release:major": "npm version major && git push --follow-tags",
|
|
22
|
+
"publish:dry-run": "npm publish --dry-run --access public"
|
|
23
|
+
},
|
|
24
|
+
"keywords": [
|
|
25
|
+
"apple",
|
|
26
|
+
"foundation-models",
|
|
27
|
+
"fm",
|
|
28
|
+
"benchmark",
|
|
29
|
+
"macos",
|
|
30
|
+
"cli"
|
|
31
|
+
],
|
|
32
|
+
"author": "Devin Oldenburg",
|
|
33
|
+
"license": "MIT",
|
|
34
|
+
"engines": {
|
|
35
|
+
"node": ">=20"
|
|
36
|
+
},
|
|
37
|
+
"repository": {
|
|
38
|
+
"type": "git",
|
|
39
|
+
"url": "git+https://github.com/devinoldenburg/fm-bench.git"
|
|
40
|
+
},
|
|
41
|
+
"bugs": {
|
|
42
|
+
"url": "https://github.com/devinoldenburg/fm-bench/issues"
|
|
43
|
+
},
|
|
44
|
+
"homepage": "https://github.com/devinoldenburg/fm-bench#readme",
|
|
45
|
+
"publishConfig": {
|
|
46
|
+
"access": "public",
|
|
47
|
+
"registry": "https://registry.npmjs.org/"
|
|
48
|
+
}
|
|
49
|
+
}
|
package/src/ansi.js
ADDED
package/src/bench.js
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
import { checkModelAvailability, collectEnvironment, countTokens, discoverModels, getQuotaUsage, respond } from './fm.js';
|
|
2
|
+
import { loadPrompts } from './prompts.js';
|
|
3
|
+
import { summarizeByModel } from './stats.js';
|
|
4
|
+
|
|
5
|
+
export async function inspectModels(options = {}) {
|
|
6
|
+
const discovered = await discoverModels(options);
|
|
7
|
+
const requested = normalizeModelSelection(options.models);
|
|
8
|
+
const models = requested.length > 0
|
|
9
|
+
? discovered.models.filter((model) => requested.includes(model.name))
|
|
10
|
+
: discovered.models;
|
|
11
|
+
|
|
12
|
+
const missing = requested.filter((name) => !models.some((model) => model.name === name));
|
|
13
|
+
for (const name of missing) {
|
|
14
|
+
models.push({ name, description: 'Requested model not reported by fm --help' });
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
const inspected = [];
|
|
18
|
+
for (const model of models) {
|
|
19
|
+
const availability = await checkModelAvailability(discovered.fmBin, model.name, options);
|
|
20
|
+
const quota = await getQuotaUsage(discovered.fmBin, model.name, options);
|
|
21
|
+
inspected.push({
|
|
22
|
+
...model,
|
|
23
|
+
available: availability.available,
|
|
24
|
+
reason: availability.reason || availability.raw,
|
|
25
|
+
quota: quota.raw
|
|
26
|
+
});
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
return {
|
|
30
|
+
fmBin: discovered.fmBin,
|
|
31
|
+
models: inspected,
|
|
32
|
+
help: discovered.help
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export async function runBenchmark(options = {}) {
|
|
37
|
+
const startedAt = new Date().toISOString();
|
|
38
|
+
const prompts = await loadPrompts(options);
|
|
39
|
+
const inspection = await inspectModels(options);
|
|
40
|
+
const modelStatuses = options.availableOnly
|
|
41
|
+
? inspection.models.filter((model) => model.available)
|
|
42
|
+
: inspection.models;
|
|
43
|
+
const runnableModels = modelStatuses.filter((model) => model.available);
|
|
44
|
+
const environment = await collectEnvironment(inspection.fmBin);
|
|
45
|
+
const promptTokenCounts = new Map();
|
|
46
|
+
|
|
47
|
+
for (const prompt of prompts) {
|
|
48
|
+
const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
|
|
49
|
+
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
|
|
53
|
+
for (const model of runnableModels) {
|
|
54
|
+
await respond(inspection.fmBin, model.name, prompts[0].prompt, {
|
|
55
|
+
...options,
|
|
56
|
+
stream: false
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const jobs = [];
|
|
62
|
+
for (const model of runnableModels) {
|
|
63
|
+
for (const prompt of prompts) {
|
|
64
|
+
for (let run = 1; run <= options.runs; run += 1) {
|
|
65
|
+
jobs.push({ model, prompt, run });
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
const results = [];
|
|
71
|
+
await runLimited(jobs, Math.max(1, options.concurrency), async (job) => {
|
|
72
|
+
const result = await runSingleBenchmark(inspection.fmBin, job, promptTokenCounts, options);
|
|
73
|
+
results.push(result);
|
|
74
|
+
if (!result.ok && options.failFast) {
|
|
75
|
+
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
76
|
+
error.exitCode = 1;
|
|
77
|
+
throw error;
|
|
78
|
+
}
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
const summary = summarizeByModel(results, modelStatuses);
|
|
82
|
+
return {
|
|
83
|
+
tool: 'fm-bench',
|
|
84
|
+
version: options.version,
|
|
85
|
+
startedAt,
|
|
86
|
+
finishedAt: new Date().toISOString(),
|
|
87
|
+
options: publicOptions(options),
|
|
88
|
+
environment,
|
|
89
|
+
prompts: prompts.map((prompt) => ({
|
|
90
|
+
id: prompt.id,
|
|
91
|
+
prompt: prompt.prompt,
|
|
92
|
+
promptTokens: promptTokenCounts.get(prompt.id)
|
|
93
|
+
})),
|
|
94
|
+
models: modelStatuses,
|
|
95
|
+
summary,
|
|
96
|
+
results
|
|
97
|
+
};
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options) {
|
|
101
|
+
const response = await respond(fmBin, job.model.name, job.prompt.prompt, {
|
|
102
|
+
...options,
|
|
103
|
+
stream: false
|
|
104
|
+
});
|
|
105
|
+
const outputTokens = response.ok
|
|
106
|
+
? await countTokens(fmBin, response.output, options)
|
|
107
|
+
: { ok: false, count: null };
|
|
108
|
+
const seconds = response.durationMs / 1000;
|
|
109
|
+
const chars = response.output.length;
|
|
110
|
+
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
111
|
+
|
|
112
|
+
return {
|
|
113
|
+
model: job.model.name,
|
|
114
|
+
promptId: job.prompt.id,
|
|
115
|
+
run: job.run,
|
|
116
|
+
ok: response.ok,
|
|
117
|
+
durationMs: response.durationMs,
|
|
118
|
+
promptTokens: promptTokenCounts.get(job.prompt.id),
|
|
119
|
+
outputTokens: outputTokens.ok ? outputTokens.count : null,
|
|
120
|
+
chars,
|
|
121
|
+
words,
|
|
122
|
+
tokensPerSecond: outputTokens.ok && seconds > 0 ? outputTokens.count / seconds : null,
|
|
123
|
+
charsPerSecond: seconds > 0 ? chars / seconds : 0,
|
|
124
|
+
output: options.captureOutput ? response.output : undefined,
|
|
125
|
+
error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
async function runLimited(items, concurrency, worker) {
|
|
130
|
+
let nextIndex = 0;
|
|
131
|
+
const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
|
|
132
|
+
while (nextIndex < items.length) {
|
|
133
|
+
const index = nextIndex;
|
|
134
|
+
nextIndex += 1;
|
|
135
|
+
await worker(items[index]);
|
|
136
|
+
}
|
|
137
|
+
});
|
|
138
|
+
await Promise.all(workers);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function normalizeModelSelection(models) {
|
|
142
|
+
if (!models) return [];
|
|
143
|
+
const values = Array.isArray(models) ? models : [models];
|
|
144
|
+
return values.flatMap((value) => String(value).split(','))
|
|
145
|
+
.map((value) => value.trim())
|
|
146
|
+
.filter(Boolean);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
function publicOptions(options) {
|
|
150
|
+
return {
|
|
151
|
+
models: normalizeModelSelection(options.models),
|
|
152
|
+
runs: options.runs,
|
|
153
|
+
warmup: options.warmup,
|
|
154
|
+
concurrency: options.concurrency,
|
|
155
|
+
timeoutMs: options.timeoutMs,
|
|
156
|
+
profile: options.profile,
|
|
157
|
+
promptCount: options.promptCount,
|
|
158
|
+
greedy: options.greedy,
|
|
159
|
+
instructions: options.instructions ? '[set]' : ''
|
|
160
|
+
};
|
|
161
|
+
}
|
package/src/cli.js
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import { createRequire } from 'node:module';
|
|
3
|
+
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
|
+
import { runProcess } from './process.js';
|
|
5
|
+
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
6
|
+
import { renderModelsTable, renderSummaryTable } from './table.js';
|
|
7
|
+
|
|
8
|
+
const require = createRequire(import.meta.url);
|
|
9
|
+
const packageJson = require('../package.json');
|
|
10
|
+
|
|
11
|
+
export async function runCli(argv = process.argv.slice(2)) {
|
|
12
|
+
const parsed = parseArgs(argv);
|
|
13
|
+
|
|
14
|
+
if (parsed.help) {
|
|
15
|
+
console.log(helpText());
|
|
16
|
+
return;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
if (parsed.versionOnly) {
|
|
20
|
+
console.log(packageJson.version);
|
|
21
|
+
return;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
if (parsed.command === 'doctor') {
|
|
25
|
+
await runDoctor(parsed);
|
|
26
|
+
return;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
if (parsed.command === 'models') {
|
|
30
|
+
const inspection = await inspectModels(parsed);
|
|
31
|
+
if (parsed.format === 'json') {
|
|
32
|
+
console.log(JSON.stringify(inspection.models, null, 2));
|
|
33
|
+
} else {
|
|
34
|
+
console.log(renderModelsTable(inspection.models));
|
|
35
|
+
}
|
|
36
|
+
return;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const payload = await runBenchmark({
|
|
40
|
+
...parsed,
|
|
41
|
+
version: packageJson.version
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
if (parsed.format === 'json') {
|
|
45
|
+
console.log(JSON.stringify(payload, null, 2));
|
|
46
|
+
} else if (parsed.format === 'csv') {
|
|
47
|
+
console.log(toCsv(flattenResults(payload.results)));
|
|
48
|
+
} else {
|
|
49
|
+
console.log(renderSummaryTable(payload.summary));
|
|
50
|
+
if (parsed.verbose) {
|
|
51
|
+
console.log();
|
|
52
|
+
console.log(toCsv(flattenResults(payload.results)));
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
if (parsed.out) {
|
|
57
|
+
const reportFormat = parsed.out.endsWith('.csv') ? 'csv' : 'json';
|
|
58
|
+
const written = await writeReport(parsed.out, payload, reportFormat);
|
|
59
|
+
if (parsed.format !== 'json') {
|
|
60
|
+
console.error(`Saved ${reportFormat.toUpperCase()} report to ${written}`);
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function parseArgs(argv) {
|
|
66
|
+
const options = {
|
|
67
|
+
command: 'run',
|
|
68
|
+
models: [],
|
|
69
|
+
prompts: [],
|
|
70
|
+
runs: 1,
|
|
71
|
+
warmup: 0,
|
|
72
|
+
concurrency: 1,
|
|
73
|
+
timeoutMs: 60_000,
|
|
74
|
+
profile: 'standard',
|
|
75
|
+
greedy: true,
|
|
76
|
+
format: 'table',
|
|
77
|
+
captureOutput: false,
|
|
78
|
+
availableOnly: false,
|
|
79
|
+
failFast: false,
|
|
80
|
+
verbose: false
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
const args = [...argv];
|
|
84
|
+
if (args[0] && !args[0].startsWith('-') && ['run', 'models', 'doctor', 'help'].includes(args[0])) {
|
|
85
|
+
options.command = args.shift();
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
if (options.command === 'help') {
|
|
89
|
+
options.help = true;
|
|
90
|
+
return options;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
while (args.length > 0) {
|
|
94
|
+
const arg = args.shift();
|
|
95
|
+
switch (arg) {
|
|
96
|
+
case '-h':
|
|
97
|
+
case '--help':
|
|
98
|
+
options.help = true;
|
|
99
|
+
break;
|
|
100
|
+
case '--version':
|
|
101
|
+
options.versionOnly = true;
|
|
102
|
+
break;
|
|
103
|
+
case '-m':
|
|
104
|
+
case '--model':
|
|
105
|
+
case '--models':
|
|
106
|
+
options.models.push(requireValue(arg, args));
|
|
107
|
+
break;
|
|
108
|
+
case '-r':
|
|
109
|
+
case '--runs':
|
|
110
|
+
options.runs = parsePositiveInt(requireValue(arg, args), arg);
|
|
111
|
+
break;
|
|
112
|
+
case '--warmup':
|
|
113
|
+
options.warmup = parseNonNegativeInt(requireValue(arg, args), arg);
|
|
114
|
+
break;
|
|
115
|
+
case '-c':
|
|
116
|
+
case '--concurrency':
|
|
117
|
+
options.concurrency = parsePositiveInt(requireValue(arg, args), arg);
|
|
118
|
+
break;
|
|
119
|
+
case '--timeout':
|
|
120
|
+
case '--timeout-ms':
|
|
121
|
+
options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
122
|
+
break;
|
|
123
|
+
case '-p':
|
|
124
|
+
case '--prompt':
|
|
125
|
+
options.prompts.push(requireValue(arg, args));
|
|
126
|
+
break;
|
|
127
|
+
case '--prompt-file':
|
|
128
|
+
options.promptFile = requireValue(arg, args);
|
|
129
|
+
break;
|
|
130
|
+
case '--profile':
|
|
131
|
+
options.profile = requireValue(arg, args);
|
|
132
|
+
break;
|
|
133
|
+
case '-i':
|
|
134
|
+
case '--instructions':
|
|
135
|
+
options.instructions = requireValue(arg, args);
|
|
136
|
+
break;
|
|
137
|
+
case '--fm-bin':
|
|
138
|
+
options.fmBin = requireValue(arg, args);
|
|
139
|
+
break;
|
|
140
|
+
case '--use-case':
|
|
141
|
+
options.useCase = requireValue(arg, args);
|
|
142
|
+
break;
|
|
143
|
+
case '--guardrails':
|
|
144
|
+
options.guardrails = requireValue(arg, args);
|
|
145
|
+
break;
|
|
146
|
+
case '--greedy':
|
|
147
|
+
options.greedy = true;
|
|
148
|
+
break;
|
|
149
|
+
case '--no-greedy':
|
|
150
|
+
options.greedy = false;
|
|
151
|
+
break;
|
|
152
|
+
case '--json':
|
|
153
|
+
options.format = 'json';
|
|
154
|
+
break;
|
|
155
|
+
case '--csv':
|
|
156
|
+
options.format = 'csv';
|
|
157
|
+
break;
|
|
158
|
+
case '--format':
|
|
159
|
+
options.format = requireValue(arg, args);
|
|
160
|
+
if (!['table', 'json', 'csv'].includes(options.format)) {
|
|
161
|
+
throw new Error('--format must be one of: table, json, csv');
|
|
162
|
+
}
|
|
163
|
+
break;
|
|
164
|
+
case '-o':
|
|
165
|
+
case '--out':
|
|
166
|
+
options.out = requireValue(arg, args);
|
|
167
|
+
break;
|
|
168
|
+
case '--capture-output':
|
|
169
|
+
options.captureOutput = true;
|
|
170
|
+
break;
|
|
171
|
+
case '--available-only':
|
|
172
|
+
options.availableOnly = true;
|
|
173
|
+
break;
|
|
174
|
+
case '--fail-fast':
|
|
175
|
+
options.failFast = true;
|
|
176
|
+
break;
|
|
177
|
+
case '-v':
|
|
178
|
+
case '--verbose':
|
|
179
|
+
options.verbose = true;
|
|
180
|
+
break;
|
|
181
|
+
case '--':
|
|
182
|
+
if (args.length > 0) {
|
|
183
|
+
options.prompts.push(args.join(' '));
|
|
184
|
+
args.length = 0;
|
|
185
|
+
}
|
|
186
|
+
break;
|
|
187
|
+
default:
|
|
188
|
+
if (arg.startsWith('-')) {
|
|
189
|
+
throw new Error(`Unknown option: ${arg}`);
|
|
190
|
+
}
|
|
191
|
+
options.prompts.push([arg, ...args].join(' '));
|
|
192
|
+
args.length = 0;
|
|
193
|
+
break;
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
return options;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
async function runDoctor(options) {
|
|
201
|
+
const checks = [];
|
|
202
|
+
checks.push(['node', process.version, true]);
|
|
203
|
+
checks.push(['platform', `${process.platform}/${process.arch}`, process.platform === 'darwin']);
|
|
204
|
+
|
|
205
|
+
const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
|
|
206
|
+
const macOS = swVers.stdout || swVers.stderr;
|
|
207
|
+
const versionMatch = macOS.match(/ProductVersion:\s*([0-9.]+)/);
|
|
208
|
+
const major = versionMatch ? Number.parseInt(versionMatch[1].split('.')[0], 10) : null;
|
|
209
|
+
checks.push(['macOS', versionMatch?.[1] || 'unknown', major == null || major >= 27]);
|
|
210
|
+
|
|
211
|
+
const inspection = await inspectModels(options);
|
|
212
|
+
checks.push(['fm', inspection.fmBin, inspection.models.length > 0]);
|
|
213
|
+
for (const model of inspection.models) {
|
|
214
|
+
checks.push([`model:${model.name}`, model.available ? 'available' : model.reason || 'unavailable', model.available]);
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
const lines = checks.map(([name, detail, ok]) => `${ok ? 'ok ' : 'warn'} ${name.padEnd(14)} ${String(detail).replace(/\s+/g, ' ').trim()}`);
|
|
218
|
+
console.log(lines.join('\n'));
|
|
219
|
+
|
|
220
|
+
if (options.out) {
|
|
221
|
+
await fs.writeFile(options.out, `${JSON.stringify({ checks, models: inspection.models }, null, 2)}\n`, 'utf8');
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
function requireValue(option, args) {
|
|
226
|
+
const value = args.shift();
|
|
227
|
+
if (value == null || value === '') throw new Error(`${option} requires a value`);
|
|
228
|
+
return value;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
function parsePositiveInt(value, option) {
|
|
232
|
+
const parsed = Number.parseInt(value, 10);
|
|
233
|
+
if (!Number.isInteger(parsed) || parsed < 1) throw new Error(`${option} must be a positive integer`);
|
|
234
|
+
return parsed;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function parseNonNegativeInt(value, option) {
|
|
238
|
+
const parsed = Number.parseInt(value, 10);
|
|
239
|
+
if (!Number.isInteger(parsed) || parsed < 0) throw new Error(`${option} must be a non-negative integer`);
|
|
240
|
+
return parsed;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function helpText() {
|
|
244
|
+
return `fm-bench ${packageJson.version}
|
|
245
|
+
|
|
246
|
+
Dynamic benchmark CLI for Apple's fm command on macOS 27+.
|
|
247
|
+
|
|
248
|
+
Usage:
|
|
249
|
+
fm-bench [run] [options]
|
|
250
|
+
fm-bench models [options]
|
|
251
|
+
fm-bench doctor [options]
|
|
252
|
+
|
|
253
|
+
Run options:
|
|
254
|
+
-m, --models <list> Models to benchmark, comma-separated or repeated
|
|
255
|
+
-r, --runs <n> Runs per prompt/model (default: 1)
|
|
256
|
+
--warmup <n> Warmup runs per model before measurement
|
|
257
|
+
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
258
|
+
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
259
|
+
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
260
|
+
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
261
|
+
--profile <name> quick, standard, or stress (default: standard)
|
|
262
|
+
-i, --instructions <text> Instructions passed to fm respond
|
|
263
|
+
--use-case <case> Pass a system model use case through to fm
|
|
264
|
+
--guardrails <level> Pass a system model guardrail level through to fm
|
|
265
|
+
--greedy Use greedy sampling (default)
|
|
266
|
+
--no-greedy Do not request greedy sampling
|
|
267
|
+
--available-only Hide unavailable discovered models
|
|
268
|
+
--capture-output Include raw model output in JSON reports
|
|
269
|
+
--fail-fast Stop after the first failed measured run
|
|
270
|
+
|
|
271
|
+
Output:
|
|
272
|
+
--format <type> table, json, or csv (default: table)
|
|
273
|
+
--json Alias for --format json
|
|
274
|
+
--csv Alias for --format csv
|
|
275
|
+
-o, --out <file> Save JSON or CSV report based on file extension
|
|
276
|
+
-v, --verbose Include per-run CSV after the summary table
|
|
277
|
+
|
|
278
|
+
Environment:
|
|
279
|
+
--fm-bin <path> fm binary to execute (default: FM_BIN or fm)
|
|
280
|
+
-h, --help Show this help
|
|
281
|
+
--version Print version
|
|
282
|
+
|
|
283
|
+
Examples:
|
|
284
|
+
fm-bench
|
|
285
|
+
fm-bench --models system,pcc --runs 3 --profile stress
|
|
286
|
+
fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
|
|
287
|
+
fm-bench models
|
|
288
|
+
fm-bench doctor
|
|
289
|
+
`;
|
|
290
|
+
}
|
package/src/fm.js
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
import os from 'node:os';
|
|
2
|
+
import { stripAnsi } from './ansi.js';
|
|
3
|
+
import { runProcess } from './process.js';
|
|
4
|
+
|
|
5
|
+
const DEFAULT_MODELS = [
|
|
6
|
+
{ name: 'system', description: 'On-device Apple Foundation Model' },
|
|
7
|
+
{ name: 'pcc', description: 'Apple Foundation Model on Private Cloud Compute' }
|
|
8
|
+
];
|
|
9
|
+
|
|
10
|
+
export function fmBinaryFromOptions(options = {}) {
|
|
11
|
+
return options.fmBin || process.env.FM_BIN || 'fm';
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export function parseModelsFromHelp(helpText) {
|
|
15
|
+
const clean = stripAnsi(helpText);
|
|
16
|
+
const lines = clean.split(/\r?\n/);
|
|
17
|
+
const models = new Map();
|
|
18
|
+
let inModels = false;
|
|
19
|
+
|
|
20
|
+
for (const line of lines) {
|
|
21
|
+
if (/^\s*MODELS\s*$/.test(line)) {
|
|
22
|
+
inModels = true;
|
|
23
|
+
continue;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
if (inModels && /^\s*[A-Z][A-Z -]+\s*$/.test(line) && !/^\s*MODELS\s*$/.test(line)) {
|
|
27
|
+
inModels = false;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
if (inModels) {
|
|
31
|
+
const match = line.match(/^\s*([A-Za-z0-9._:-]+)\s{2,}(.+?)\s*$/);
|
|
32
|
+
if (match) {
|
|
33
|
+
models.set(match[1], {
|
|
34
|
+
name: match[1],
|
|
35
|
+
description: match[2].replace(/\s*\(default\)\s*$/, '').trim()
|
|
36
|
+
});
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const optionMatch = /--model\b/.test(line)
|
|
41
|
+
? line.match(/\bmodel\b.*?\(([^)]+)\)/i)
|
|
42
|
+
: null;
|
|
43
|
+
if (optionMatch) {
|
|
44
|
+
for (const raw of optionMatch[1].split(',')) {
|
|
45
|
+
const name = raw.trim();
|
|
46
|
+
if (/^[A-Za-z0-9._:-]+$/.test(name) && !models.has(name)) {
|
|
47
|
+
models.set(name, { name, description: '' });
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
return [...models.values()];
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
57
|
+
const result = await runProcess(fmBin, ['--help'], { timeoutMs });
|
|
58
|
+
if (result.error) {
|
|
59
|
+
const error = new Error(`Unable to execute ${fmBin}: ${result.stderr || result.error.message}`);
|
|
60
|
+
error.exitCode = 2;
|
|
61
|
+
throw error;
|
|
62
|
+
}
|
|
63
|
+
return {
|
|
64
|
+
ok: result.code === 0,
|
|
65
|
+
text: `${result.stdout}${result.stderr}`,
|
|
66
|
+
result
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export async function discoverModels(options = {}) {
|
|
71
|
+
const fmBin = fmBinaryFromOptions(options);
|
|
72
|
+
const help = await getFmHelp(fmBin, options.timeoutMs ?? 10_000);
|
|
73
|
+
let models = parseModelsFromHelp(help.text);
|
|
74
|
+
|
|
75
|
+
if (models.length === 0 && /Apple Foundation Models CLI/i.test(stripAnsi(help.text))) {
|
|
76
|
+
models = DEFAULT_MODELS;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
return {
|
|
80
|
+
fmBin,
|
|
81
|
+
models,
|
|
82
|
+
help: stripAnsi(help.text)
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export function parseAvailabilityOutput(model, output, code) {
|
|
87
|
+
const clean = stripAnsi(output).trim();
|
|
88
|
+
const lower = clean.toLowerCase();
|
|
89
|
+
const modelLower = model.toLowerCase();
|
|
90
|
+
const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b/.test(lower);
|
|
91
|
+
const hasAvailable = new RegExp(`\\b${escapeRegExp(modelLower)}\\b[\\s\\S]{0,80}\\bavailable\\b|\\bavailable\\b[\\s\\S]{0,80}\\b${escapeRegExp(modelLower)}\\b`).test(lower)
|
|
92
|
+
|| lower.includes(`${modelLower} model available`)
|
|
93
|
+
|| lower.includes(`${titleCase(modelLower)} model available`.toLowerCase());
|
|
94
|
+
|
|
95
|
+
return {
|
|
96
|
+
model,
|
|
97
|
+
available: code === 0 && hasAvailable && !hasError,
|
|
98
|
+
raw: clean,
|
|
99
|
+
reason: hasError ? clean : ''
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
104
|
+
const result = await runProcess(fmBin, ['available', '--model', model], {
|
|
105
|
+
timeoutMs: options.timeoutMs ?? 15_000
|
|
106
|
+
});
|
|
107
|
+
const output = `${result.stdout}${result.stderr}`;
|
|
108
|
+
const parsed = parseAvailabilityOutput(model, output, result.code);
|
|
109
|
+
if (result.error) {
|
|
110
|
+
parsed.available = false;
|
|
111
|
+
parsed.reason = result.stderr || result.error.message;
|
|
112
|
+
}
|
|
113
|
+
return parsed;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export async function getQuotaUsage(fmBin, model, options = {}) {
|
|
117
|
+
const result = await runProcess(fmBin, ['quota-usage', '--model', model], {
|
|
118
|
+
timeoutMs: options.timeoutMs ?? 15_000
|
|
119
|
+
});
|
|
120
|
+
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
121
|
+
return {
|
|
122
|
+
model,
|
|
123
|
+
ok: result.code === 0,
|
|
124
|
+
raw: output,
|
|
125
|
+
unavailable: /\bunavailable\b|\bnot available\b|\berror:/i.test(output)
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export async function countTokens(fmBin, text, options = {}) {
|
|
130
|
+
const result = await runProcess(fmBin, ['token-count', '--quiet'], {
|
|
131
|
+
input: text,
|
|
132
|
+
timeoutMs: options.timeoutMs ?? 15_000
|
|
133
|
+
});
|
|
134
|
+
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
135
|
+
const match = output.match(/-?\d+/);
|
|
136
|
+
if (result.code !== 0 || !match) {
|
|
137
|
+
return {
|
|
138
|
+
ok: false,
|
|
139
|
+
count: null,
|
|
140
|
+
raw: output
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
return {
|
|
144
|
+
ok: true,
|
|
145
|
+
count: Number.parseInt(match[0], 10),
|
|
146
|
+
raw: output
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
export async function respond(fmBin, model, prompt, options = {}) {
|
|
151
|
+
const args = ['respond', '--model', model];
|
|
152
|
+
|
|
153
|
+
if (options.stream === false) args.push('--no-stream');
|
|
154
|
+
if (options.greedy) args.push('--greedy');
|
|
155
|
+
if (options.instructions) args.push('--instructions', options.instructions);
|
|
156
|
+
if (options.useCase) args.push('--use-case', options.useCase);
|
|
157
|
+
if (options.guardrails) args.push('--guardrails', options.guardrails);
|
|
158
|
+
|
|
159
|
+
const result = await runProcess(fmBin, args, {
|
|
160
|
+
input: prompt,
|
|
161
|
+
timeoutMs: options.timeoutMs ?? 60_000
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
const output = stripAnsi(result.stdout).trim();
|
|
165
|
+
const errorText = stripAnsi(result.stderr).trim();
|
|
166
|
+
return {
|
|
167
|
+
ok: result.code === 0 && !result.timedOut,
|
|
168
|
+
model,
|
|
169
|
+
prompt,
|
|
170
|
+
output,
|
|
171
|
+
stderr: errorText,
|
|
172
|
+
code: result.code,
|
|
173
|
+
signal: result.signal,
|
|
174
|
+
timedOut: result.timedOut,
|
|
175
|
+
durationMs: result.durationMs
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
export async function collectEnvironment(fmBin) {
|
|
180
|
+
const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
|
|
181
|
+
return {
|
|
182
|
+
platform: process.platform,
|
|
183
|
+
arch: process.arch,
|
|
184
|
+
node: process.version,
|
|
185
|
+
host: os.hostname(),
|
|
186
|
+
fmBin,
|
|
187
|
+
macOS: stripAnsi(swVers.stdout).trim() || null
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
function escapeRegExp(value) {
|
|
192
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function titleCase(value) {
|
|
196
|
+
return value.slice(0, 1).toUpperCase() + value.slice(1);
|
|
197
|
+
}
|
package/src/process.js
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import { spawn } from 'node:child_process';
|
|
2
|
+
|
|
3
|
+
export function runProcess(command, args = [], options = {}) {
|
|
4
|
+
const {
|
|
5
|
+
input,
|
|
6
|
+
timeoutMs = 30_000,
|
|
7
|
+
env = process.env,
|
|
8
|
+
cwd = process.cwd()
|
|
9
|
+
} = options;
|
|
10
|
+
|
|
11
|
+
return new Promise((resolve) => {
|
|
12
|
+
const startedAt = process.hrtime.bigint();
|
|
13
|
+
const child = spawn(command, args, {
|
|
14
|
+
cwd,
|
|
15
|
+
env,
|
|
16
|
+
stdio: ['pipe', 'pipe', 'pipe']
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
let stdout = '';
|
|
20
|
+
let stderr = '';
|
|
21
|
+
let timedOut = false;
|
|
22
|
+
let settled = false;
|
|
23
|
+
|
|
24
|
+
const timer = timeoutMs > 0
|
|
25
|
+
? setTimeout(() => {
|
|
26
|
+
timedOut = true;
|
|
27
|
+
child.kill('SIGTERM');
|
|
28
|
+
setTimeout(() => {
|
|
29
|
+
if (!settled) child.kill('SIGKILL');
|
|
30
|
+
}, 1_000).unref();
|
|
31
|
+
}, timeoutMs)
|
|
32
|
+
: null;
|
|
33
|
+
|
|
34
|
+
child.stdout.setEncoding('utf8');
|
|
35
|
+
child.stderr.setEncoding('utf8');
|
|
36
|
+
child.stdout.on('data', (chunk) => {
|
|
37
|
+
stdout += chunk;
|
|
38
|
+
});
|
|
39
|
+
child.stderr.on('data', (chunk) => {
|
|
40
|
+
stderr += chunk;
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
child.on('error', (error) => {
|
|
44
|
+
settled = true;
|
|
45
|
+
if (timer) clearTimeout(timer);
|
|
46
|
+
const endedAt = process.hrtime.bigint();
|
|
47
|
+
resolve({
|
|
48
|
+
command,
|
|
49
|
+
args,
|
|
50
|
+
code: null,
|
|
51
|
+
signal: null,
|
|
52
|
+
stdout,
|
|
53
|
+
stderr: stderr || error.message,
|
|
54
|
+
error,
|
|
55
|
+
timedOut,
|
|
56
|
+
durationMs: Number(endedAt - startedAt) / 1e6
|
|
57
|
+
});
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
child.on('close', (code, signal) => {
|
|
61
|
+
settled = true;
|
|
62
|
+
if (timer) clearTimeout(timer);
|
|
63
|
+
const endedAt = process.hrtime.bigint();
|
|
64
|
+
resolve({
|
|
65
|
+
command,
|
|
66
|
+
args,
|
|
67
|
+
code,
|
|
68
|
+
signal,
|
|
69
|
+
stdout,
|
|
70
|
+
stderr,
|
|
71
|
+
timedOut,
|
|
72
|
+
durationMs: Number(endedAt - startedAt) / 1e6
|
|
73
|
+
});
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
if (input != null) {
|
|
77
|
+
child.stdin.end(input);
|
|
78
|
+
} else {
|
|
79
|
+
child.stdin.end();
|
|
80
|
+
}
|
|
81
|
+
});
|
|
82
|
+
}
|
package/src/prompts.js
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
|
|
4
|
+
const PROFILES = {
|
|
5
|
+
quick: [
|
|
6
|
+
{
|
|
7
|
+
id: 'echo-ok',
|
|
8
|
+
prompt: 'Reply with exactly: ok'
|
|
9
|
+
}
|
|
10
|
+
],
|
|
11
|
+
standard: [
|
|
12
|
+
{
|
|
13
|
+
id: 'echo-ok',
|
|
14
|
+
prompt: 'Reply with exactly: ok'
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
id: 'explain-latency',
|
|
18
|
+
prompt: 'In one concise paragraph, explain why AI model benchmarks should report latency percentiles.'
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
id: 'json-transform',
|
|
22
|
+
prompt: 'Convert this list into a valid compact JSON array of strings: alpha, beta, gamma.'
|
|
23
|
+
}
|
|
24
|
+
],
|
|
25
|
+
stress: [
|
|
26
|
+
{
|
|
27
|
+
id: 'echo-ok',
|
|
28
|
+
prompt: 'Reply with exactly: ok'
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
id: 'explain-latency',
|
|
32
|
+
prompt: 'In one concise paragraph, explain why AI model benchmarks should report latency percentiles.'
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
id: 'json-transform',
|
|
36
|
+
prompt: 'Convert this list into a valid compact JSON array of strings: alpha, beta, gamma.'
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
id: 'reasoning',
|
|
40
|
+
prompt: 'A build starts at 09:12, takes 17 minutes, waits 8 minutes for review, then takes another 11 minutes. What time does it finish? Show only the answer and one short explanation.'
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
id: 'summarize',
|
|
44
|
+
prompt: 'Summarize this in two bullets: local model benchmarks should measure first-token latency, total latency, throughput, failures, and the exact prompt suite so results can be compared later.'
|
|
45
|
+
}
|
|
46
|
+
]
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
export function getPromptProfile(name = 'standard') {
|
|
50
|
+
if (!PROFILES[name]) {
|
|
51
|
+
throw new Error(`Unknown prompt profile "${name}". Use one of: ${Object.keys(PROFILES).join(', ')}`);
|
|
52
|
+
}
|
|
53
|
+
return PROFILES[name].map((prompt) => ({ ...prompt }));
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export async function loadPrompts(options = {}) {
|
|
57
|
+
const prompts = [];
|
|
58
|
+
|
|
59
|
+
if (options.promptFile) {
|
|
60
|
+
prompts.push(...await loadPromptFile(options.promptFile));
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
for (const prompt of options.prompts || []) {
|
|
64
|
+
prompts.push({
|
|
65
|
+
id: `custom-${prompts.length + 1}`,
|
|
66
|
+
prompt
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
if (prompts.length === 0) {
|
|
71
|
+
prompts.push(...getPromptProfile(options.profile || 'standard'));
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
return prompts.map((entry, index) => ({
|
|
75
|
+
id: entry.id || `prompt-${index + 1}`,
|
|
76
|
+
prompt: String(entry.prompt ?? entry.text ?? '').trim()
|
|
77
|
+
})).filter((entry) => entry.prompt.length > 0);
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
async function loadPromptFile(filePath) {
|
|
81
|
+
const absolutePath = path.resolve(filePath);
|
|
82
|
+
const content = await fs.readFile(absolutePath, 'utf8');
|
|
83
|
+
const trimmed = content.trim();
|
|
84
|
+
|
|
85
|
+
if (!trimmed) return [];
|
|
86
|
+
|
|
87
|
+
if (absolutePath.endsWith('.json')) {
|
|
88
|
+
const parsed = JSON.parse(trimmed);
|
|
89
|
+
const items = Array.isArray(parsed) ? parsed : parsed.prompts;
|
|
90
|
+
if (!Array.isArray(items)) {
|
|
91
|
+
throw new Error('Prompt JSON must be an array or an object with a prompts array');
|
|
92
|
+
}
|
|
93
|
+
return items.map((item, index) => normalizePromptItem(item, index));
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
if (absolutePath.endsWith('.jsonl')) {
|
|
97
|
+
return trimmed.split(/\r?\n/)
|
|
98
|
+
.filter(Boolean)
|
|
99
|
+
.map((line, index) => normalizePromptItem(JSON.parse(line), index));
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({
|
|
103
|
+
id: `file-${index + 1}`,
|
|
104
|
+
prompt: prompt.trim()
|
|
105
|
+
}));
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function normalizePromptItem(item, index) {
|
|
109
|
+
if (typeof item === 'string') {
|
|
110
|
+
return {
|
|
111
|
+
id: `file-${index + 1}`,
|
|
112
|
+
prompt: item
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return {
|
|
117
|
+
id: item.id || item.name || `file-${index + 1}`,
|
|
118
|
+
prompt: item.prompt || item.text || ''
|
|
119
|
+
};
|
|
120
|
+
}
|
package/src/report.js
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
|
|
4
|
+
export function toCsv(rows) {
|
|
5
|
+
if (rows.length === 0) return '';
|
|
6
|
+
const headers = Object.keys(rows[0]);
|
|
7
|
+
return [
|
|
8
|
+
headers.join(','),
|
|
9
|
+
...rows.map((row) => headers.map((header) => csvEscape(row[header])).join(','))
|
|
10
|
+
].join('\n');
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function flattenResults(results) {
|
|
14
|
+
return results.map((result) => ({
|
|
15
|
+
model: result.model,
|
|
16
|
+
prompt_id: result.promptId,
|
|
17
|
+
run: result.run,
|
|
18
|
+
ok: result.ok,
|
|
19
|
+
duration_ms: round(result.durationMs),
|
|
20
|
+
prompt_tokens: result.promptTokens ?? '',
|
|
21
|
+
output_tokens: result.outputTokens ?? '',
|
|
22
|
+
chars: result.chars,
|
|
23
|
+
words: result.words,
|
|
24
|
+
tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
|
|
25
|
+
chars_per_second: round(result.charsPerSecond),
|
|
26
|
+
error: result.error || ''
|
|
27
|
+
}));
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export async function writeReport(filePath, payload, format) {
|
|
31
|
+
const target = path.resolve(filePath);
|
|
32
|
+
await fs.mkdir(path.dirname(target), { recursive: true });
|
|
33
|
+
const content = format === 'csv'
|
|
34
|
+
? toCsv(flattenResults(payload.results))
|
|
35
|
+
: `${JSON.stringify(payload, null, 2)}\n`;
|
|
36
|
+
await fs.writeFile(target, content, 'utf8');
|
|
37
|
+
return target;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function csvEscape(value) {
|
|
41
|
+
const text = String(value ?? '');
|
|
42
|
+
if (/[",\n\r]/.test(text)) {
|
|
43
|
+
return `"${text.replaceAll('"', '""')}"`;
|
|
44
|
+
}
|
|
45
|
+
return text;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function round(value) {
|
|
49
|
+
if (!Number.isFinite(value)) return '';
|
|
50
|
+
return Math.round(value * 100) / 100;
|
|
51
|
+
}
|
package/src/stats.js
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
export function summarizeNumbers(values) {
|
|
2
|
+
const clean = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
|
|
3
|
+
if (clean.length === 0) {
|
|
4
|
+
return {
|
|
5
|
+
count: 0,
|
|
6
|
+
min: null,
|
|
7
|
+
max: null,
|
|
8
|
+
avg: null,
|
|
9
|
+
p50: null,
|
|
10
|
+
p95: null
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
const total = clean.reduce((sum, value) => sum + value, 0);
|
|
15
|
+
return {
|
|
16
|
+
count: clean.length,
|
|
17
|
+
min: clean[0],
|
|
18
|
+
max: clean[clean.length - 1],
|
|
19
|
+
avg: total / clean.length,
|
|
20
|
+
p50: percentile(clean, 50),
|
|
21
|
+
p95: percentile(clean, 95)
|
|
22
|
+
};
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export function percentile(sortedValues, percentileValue) {
|
|
26
|
+
if (sortedValues.length === 0) return null;
|
|
27
|
+
if (sortedValues.length === 1) return sortedValues[0];
|
|
28
|
+
|
|
29
|
+
const rank = (percentileValue / 100) * (sortedValues.length - 1);
|
|
30
|
+
const low = Math.floor(rank);
|
|
31
|
+
const high = Math.ceil(rank);
|
|
32
|
+
if (low === high) return sortedValues[low];
|
|
33
|
+
const weight = rank - low;
|
|
34
|
+
return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function summarizeByModel(results, modelStatuses = []) {
|
|
38
|
+
const byModel = new Map();
|
|
39
|
+
|
|
40
|
+
for (const status of modelStatuses) {
|
|
41
|
+
byModel.set(status.name, {
|
|
42
|
+
model: status.name,
|
|
43
|
+
description: status.description,
|
|
44
|
+
available: status.available,
|
|
45
|
+
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
46
|
+
results: []
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
for (const result of results) {
|
|
51
|
+
if (!byModel.has(result.model)) {
|
|
52
|
+
byModel.set(result.model, {
|
|
53
|
+
model: result.model,
|
|
54
|
+
description: '',
|
|
55
|
+
available: true,
|
|
56
|
+
skippedReason: '',
|
|
57
|
+
results: []
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
byModel.get(result.model).results.push(result);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
return [...byModel.values()].map((entry) => {
|
|
64
|
+
const successes = entry.results.filter((result) => result.ok);
|
|
65
|
+
const failures = entry.results.filter((result) => !result.ok);
|
|
66
|
+
const latency = summarizeNumbers(successes.map((result) => result.durationMs));
|
|
67
|
+
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
68
|
+
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
69
|
+
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
70
|
+
|
|
71
|
+
return {
|
|
72
|
+
model: entry.model,
|
|
73
|
+
description: entry.description,
|
|
74
|
+
available: entry.available,
|
|
75
|
+
skippedReason: entry.skippedReason,
|
|
76
|
+
attempted: entry.results.length,
|
|
77
|
+
successes: successes.length,
|
|
78
|
+
failures: failures.length,
|
|
79
|
+
latency,
|
|
80
|
+
outputTokens,
|
|
81
|
+
charsPerSecond,
|
|
82
|
+
tokensPerSecond
|
|
83
|
+
};
|
|
84
|
+
});
|
|
85
|
+
}
|
package/src/table.js
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
export function renderTable(headers, rows) {
|
|
2
|
+
const stringRows = rows.map((row) => row.map(formatCell));
|
|
3
|
+
const widths = headers.map((header, index) => {
|
|
4
|
+
const values = [header, ...stringRows.map((row) => row[index] ?? '')];
|
|
5
|
+
return Math.max(...values.map(visibleLength));
|
|
6
|
+
});
|
|
7
|
+
|
|
8
|
+
const separator = `+-${widths.map((width) => '-'.repeat(width)).join('-+-')}-+`;
|
|
9
|
+
const headerLine = `| ${headers.map((header, index) => pad(header, widths[index])).join(' | ')} |`;
|
|
10
|
+
const bodyLines = stringRows.map((row) => `| ${row.map((cell, index) => pad(cell, widths[index], isNumericCell(cell))).join(' | ')} |`);
|
|
11
|
+
|
|
12
|
+
return [separator, headerLine, separator, ...bodyLines, separator].join('\n');
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export function formatMs(value) {
|
|
16
|
+
if (value == null) return '-';
|
|
17
|
+
if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
|
|
18
|
+
return `${Math.round(value)}ms`;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function formatNumber(value, digits = 1) {
|
|
22
|
+
if (value == null || !Number.isFinite(value)) return '-';
|
|
23
|
+
if (Math.abs(value) >= 100) return Math.round(value).toLocaleString('en-US');
|
|
24
|
+
return value.toFixed(digits);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export function renderSummaryTable(summary) {
|
|
28
|
+
const rows = summary.map((item) => {
|
|
29
|
+
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
30
|
+
return [
|
|
31
|
+
item.model,
|
|
32
|
+
status,
|
|
33
|
+
item.attempted || '-',
|
|
34
|
+
item.successes || '-',
|
|
35
|
+
item.failures || '-',
|
|
36
|
+
formatMs(item.latency.p50),
|
|
37
|
+
formatMs(item.latency.p95),
|
|
38
|
+
formatMs(item.latency.avg),
|
|
39
|
+
formatNumber(item.tokensPerSecond.avg),
|
|
40
|
+
formatNumber(item.charsPerSecond.avg, 0),
|
|
41
|
+
formatNumber(item.outputTokens.avg, 0),
|
|
42
|
+
item.available ? '' : compactReason(item.skippedReason)
|
|
43
|
+
];
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
return renderTable([
|
|
47
|
+
'model',
|
|
48
|
+
'status',
|
|
49
|
+
'runs',
|
|
50
|
+
'ok',
|
|
51
|
+
'fail',
|
|
52
|
+
'p50',
|
|
53
|
+
'p95',
|
|
54
|
+
'avg',
|
|
55
|
+
'tok/s',
|
|
56
|
+
'char/s',
|
|
57
|
+
'out tok',
|
|
58
|
+
'note'
|
|
59
|
+
], rows);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function renderModelsTable(models) {
|
|
63
|
+
return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
|
|
64
|
+
model.name,
|
|
65
|
+
model.available ? 'yes' : 'no',
|
|
66
|
+
model.description || '-',
|
|
67
|
+
compactReason(model.quota || model.reason || '-')
|
|
68
|
+
]));
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function formatCell(value) {
|
|
72
|
+
if (value == null) return '';
|
|
73
|
+
return String(value);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function pad(value, width, left = false) {
|
|
77
|
+
const length = visibleLength(value);
|
|
78
|
+
const padding = ' '.repeat(Math.max(0, width - length));
|
|
79
|
+
return left ? `${padding}${value}` : `${value}${padding}`;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function isNumericCell(value) {
|
|
83
|
+
return /^-?$|^[\d,.]+(?:ms|s)?$/.test(value);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function visibleLength(value) {
|
|
87
|
+
return String(value).length;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function compactReason(value) {
|
|
91
|
+
const clean = String(value || '').replace(/\s+/g, ' ').trim();
|
|
92
|
+
if (clean.length <= 58) return clean;
|
|
93
|
+
return `${clean.slice(0, 55)}...`;
|
|
94
|
+
}
|