fm-bench 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -12
- package/docs/methodology.md +31 -0
- package/package.json +2 -1
- package/src/bench.js +41 -5
- package/src/cli.js +19 -5
- package/src/fm.js +6 -2
- package/src/process.js +20 -0
- package/src/prompts.js +36 -8
- package/src/report.js +7 -0
- package/src/stats.js +53 -3
- package/src/table.js +116 -21
package/README.md
CHANGED
|
@@ -35,12 +35,20 @@ fm-bench
|
|
|
35
35
|
Example output:
|
|
36
36
|
|
|
37
37
|
```text
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
38
|
+
fm-bench 0.2.0 | darwin/arm64 | fm
|
|
39
|
+
prompts 3 | runs 1 | concurrency 1 | stream on | measured 3 | failed 0 | skipped models 0 | elapsed 11.36s
|
|
40
|
+
|
|
41
|
+
┌────────┬────────┬──────┬────┬─────────┬──────────┬──────────┬─────────┬─────────┬──────────┬───────┬─────┬──────┐
|
|
42
|
+
│ MODEL │ STATUS │ RUNS │ OK │ SUCCESS │ TTFT P50 │ TTFT P95 │ E2E P50 │ E2E P95 │ TPOT P50 │ TOK/S │ RPS │ NOTE │
|
|
43
|
+
├────────┼────────┼──────┼────┼─────────┼──────────┼──────────┼─────────┼─────────┼──────────┼───────┼─────┼──────┤
|
|
44
|
+
│ system │ ok │ 3 │ 3 │ 100% │ 409ms │ 486ms │ 2.29s │ 5.97s │ 14ms │ 58.5 │ 0.3 │ │
|
|
45
|
+
└────────┴────────┴──────┴────┴─────────┴──────────┴──────────┴─────────┴─────────┴──────────┴───────┴─────┴──────┘
|
|
46
|
+
|
|
47
|
+
┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
|
|
48
|
+
│ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
|
|
49
|
+
├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
|
|
50
|
+
│ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
|
|
51
|
+
└────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
|
|
44
52
|
```
|
|
45
53
|
|
|
46
54
|
## Commands
|
|
@@ -61,6 +69,8 @@ fm-bench doctor [options]
|
|
|
61
69
|
|
|
62
70
|
```sh
|
|
63
71
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
72
|
+
fm-bench --models system --runs 5 --profile interactive
|
|
73
|
+
fm-bench --models system --runs 3 --profile throughput --warmup 1
|
|
64
74
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
65
75
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
66
76
|
fm-bench --format csv --out reports/bench.csv
|
|
@@ -73,13 +83,14 @@ Useful flags:
|
|
|
73
83
|
- `--warmup <n>`: warmup runs per model before measurement.
|
|
74
84
|
- `--concurrency <n>`: parallel `fm` processes.
|
|
75
85
|
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
76
|
-
- `--profile quick|standard|stress`: built-in prompt suite.
|
|
86
|
+
- `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
|
|
77
87
|
- `--prompt <text>`: custom prompt, repeatable.
|
|
78
88
|
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
79
89
|
- `--instructions <text>`: passed to `fm respond`.
|
|
80
90
|
- `--available-only`: hide unavailable discovered models.
|
|
81
91
|
- `--capture-output`: include raw model output in JSON reports.
|
|
82
92
|
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
93
|
+
- `--ascii`: use plain ASCII table borders.
|
|
83
94
|
- `--out <file>`: save a report.
|
|
84
95
|
|
|
85
96
|
## Prompt Files
|
|
@@ -106,14 +117,23 @@ Plain text files are split on blank lines.
|
|
|
106
117
|
|
|
107
118
|
`fm-bench` reports:
|
|
108
119
|
|
|
109
|
-
-
|
|
110
|
-
-
|
|
111
|
-
-
|
|
112
|
-
-
|
|
120
|
+
- TTFT, or time to first streamed output.
|
|
121
|
+
- E2E latency, or full response wall-clock latency.
|
|
122
|
+
- TPOT, or decode time per output token after the first output token.
|
|
123
|
+
- output tokens per second per request.
|
|
124
|
+
- total output token throughput across the measured window.
|
|
125
|
+
- requests per second across the measured window.
|
|
126
|
+
- prompt and output token counts.
|
|
127
|
+
- p50, p95, and p99 tail latency views.
|
|
128
|
+
- repeatability across repeated runs of the same prompt.
|
|
113
129
|
- success and failure counts.
|
|
114
130
|
- unavailable model notes.
|
|
115
131
|
|
|
116
|
-
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response,
|
|
132
|
+
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
|
|
133
|
+
|
|
134
|
+
Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
|
|
135
|
+
|
|
136
|
+
See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
|
|
117
137
|
|
|
118
138
|
## Requirements
|
|
119
139
|
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Methodology
|
|
2
|
+
|
|
3
|
+
`fm-bench` measures local `fm` command behavior from the client side. It is meant to answer: "What does this Mac deliver to a terminal user for this prompt suite right now?"
|
|
4
|
+
|
|
5
|
+
## Sources
|
|
6
|
+
|
|
7
|
+
The metric set follows common LLM inference benchmark practice:
|
|
8
|
+
|
|
9
|
+
- Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
|
|
10
|
+
- NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
|
|
11
|
+
- NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
|
|
12
|
+
- vLLM benchmark tooling reports end-to-end latency and configurable percentiles, and can save JSON results: <https://docs.vllm.ai/en/latest/benchmarking/cli/>
|
|
13
|
+
- MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
|
|
14
|
+
|
|
15
|
+
## Metrics
|
|
16
|
+
|
|
17
|
+
- `TTFT`: time from starting `fm respond` to the first streamed stdout chunk. This is a practical terminal-side proxy for time to first token.
|
|
18
|
+
- `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
|
|
19
|
+
- `generation_ms`: `E2E - TTFT`.
|
|
20
|
+
- `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
|
|
21
|
+
- `tokens_per_second`: output tokens divided by E2E seconds for one request.
|
|
22
|
+
- `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
|
|
23
|
+
- `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
|
|
24
|
+
- `RPS`: successful requests divided by that model's measured wall-clock window.
|
|
25
|
+
- `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
|
|
26
|
+
|
|
27
|
+
## Caveats
|
|
28
|
+
|
|
29
|
+
`fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
|
|
30
|
+
|
|
31
|
+
Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "fm-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.2.0",
|
|
4
4
|
"description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
},
|
|
9
9
|
"files": [
|
|
10
10
|
"bin",
|
|
11
|
+
"docs",
|
|
11
12
|
"src",
|
|
12
13
|
"README.md",
|
|
13
14
|
"LICENSE"
|
package/src/bench.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import crypto from 'node:crypto';
|
|
1
2
|
import { checkModelAvailability, collectEnvironment, countTokens, discoverModels, getQuotaUsage, respond } from './fm.js';
|
|
2
3
|
import { loadPrompts } from './prompts.js';
|
|
3
4
|
import { summarizeByModel } from './stats.js';
|
|
@@ -59,6 +60,7 @@ export async function runBenchmark(options = {}) {
|
|
|
59
60
|
}
|
|
60
61
|
|
|
61
62
|
const jobs = [];
|
|
63
|
+
const benchmarkStartedAt = process.hrtime.bigint();
|
|
62
64
|
for (const model of runnableModels) {
|
|
63
65
|
for (const prompt of prompts) {
|
|
64
66
|
for (let run = 1; run <= options.runs; run += 1) {
|
|
@@ -69,7 +71,7 @@ export async function runBenchmark(options = {}) {
|
|
|
69
71
|
|
|
70
72
|
const results = [];
|
|
71
73
|
await runLimited(jobs, Math.max(1, options.concurrency), async (job) => {
|
|
72
|
-
const result = await runSingleBenchmark(inspection.fmBin, job, promptTokenCounts, options);
|
|
74
|
+
const result = await runSingleBenchmark(inspection.fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
|
|
73
75
|
results.push(result);
|
|
74
76
|
if (!result.ok && options.failFast) {
|
|
75
77
|
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
@@ -78,6 +80,10 @@ export async function runBenchmark(options = {}) {
|
|
|
78
80
|
}
|
|
79
81
|
});
|
|
80
82
|
|
|
83
|
+
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
84
|
+
|| a.promptId.localeCompare(b.promptId)
|
|
85
|
+
|| a.run - b.run);
|
|
86
|
+
|
|
81
87
|
const summary = summarizeByModel(results, modelStatuses);
|
|
82
88
|
return {
|
|
83
89
|
tool: 'fm-bench',
|
|
@@ -97,15 +103,28 @@ export async function runBenchmark(options = {}) {
|
|
|
97
103
|
};
|
|
98
104
|
}
|
|
99
105
|
|
|
100
|
-
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options) {
|
|
106
|
+
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt) {
|
|
107
|
+
const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
101
108
|
const response = await respond(fmBin, job.model.name, job.prompt.prompt, {
|
|
102
109
|
...options,
|
|
103
|
-
stream:
|
|
110
|
+
stream: options.stream
|
|
104
111
|
});
|
|
112
|
+
const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
105
113
|
const outputTokens = response.ok
|
|
106
114
|
? await countTokens(fmBin, response.output, options)
|
|
107
115
|
: { ok: false, count: null };
|
|
108
116
|
const seconds = response.durationMs / 1000;
|
|
117
|
+
const firstTokenMs = response.firstOutputMs;
|
|
118
|
+
const generationMs = response.ok && firstTokenMs != null
|
|
119
|
+
? Math.max(0, response.durationMs - firstTokenMs)
|
|
120
|
+
: null;
|
|
121
|
+
const countedOutputTokens = outputTokens.ok ? outputTokens.count : null;
|
|
122
|
+
const decodeTokenCount = countedOutputTokens != null ? Math.max(0, countedOutputTokens - 1) : null;
|
|
123
|
+
const hasDecodeCadence = response.stdoutChunks > 2 && generationMs != null && generationMs > 0 && decodeTokenCount > 0;
|
|
124
|
+
const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
|
|
125
|
+
const decodeTokensPerSecond = hasDecodeCadence
|
|
126
|
+
? decodeTokenCount / (generationMs / 1000)
|
|
127
|
+
: null;
|
|
109
128
|
const chars = response.output.length;
|
|
110
129
|
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
111
130
|
|
|
@@ -115,12 +134,21 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options) {
|
|
|
115
134
|
run: job.run,
|
|
116
135
|
ok: response.ok,
|
|
117
136
|
durationMs: response.durationMs,
|
|
137
|
+
firstTokenMs,
|
|
138
|
+
generationMs,
|
|
139
|
+
tpotMs,
|
|
118
140
|
promptTokens: promptTokenCounts.get(job.prompt.id),
|
|
119
|
-
outputTokens:
|
|
141
|
+
outputTokens: countedOutputTokens,
|
|
120
142
|
chars,
|
|
121
143
|
words,
|
|
122
|
-
tokensPerSecond:
|
|
144
|
+
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
145
|
+
decodeTokensPerSecond,
|
|
123
146
|
charsPerSecond: seconds > 0 ? chars / seconds : 0,
|
|
147
|
+
startOffsetMs,
|
|
148
|
+
endOffsetMs,
|
|
149
|
+
streamed: response.streamed,
|
|
150
|
+
stdoutChunks: response.stdoutChunks,
|
|
151
|
+
outputHash: response.ok ? hashOutput(response.output) : null,
|
|
124
152
|
output: options.captureOutput ? response.output : undefined,
|
|
125
153
|
error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
|
|
126
154
|
};
|
|
@@ -156,6 +184,14 @@ function publicOptions(options) {
|
|
|
156
184
|
profile: options.profile,
|
|
157
185
|
promptCount: options.promptCount,
|
|
158
186
|
greedy: options.greedy,
|
|
187
|
+
stream: options.stream,
|
|
159
188
|
instructions: options.instructions ? '[set]' : ''
|
|
160
189
|
};
|
|
161
190
|
}
|
|
191
|
+
|
|
192
|
+
function hashOutput(output) {
|
|
193
|
+
return crypto.createHash('sha256')
|
|
194
|
+
.update(output.replace(/\s+/g, ' ').trim())
|
|
195
|
+
.digest('hex')
|
|
196
|
+
.slice(0, 16);
|
|
197
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -3,7 +3,7 @@ import { createRequire } from 'node:module';
|
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { runProcess } from './process.js';
|
|
5
5
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
6
|
-
import {
|
|
6
|
+
import { renderBenchmarkReport, renderModelsTable } from './table.js';
|
|
7
7
|
|
|
8
8
|
const require = createRequire(import.meta.url);
|
|
9
9
|
const packageJson = require('../package.json');
|
|
@@ -31,7 +31,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
31
31
|
if (parsed.format === 'json') {
|
|
32
32
|
console.log(JSON.stringify(inspection.models, null, 2));
|
|
33
33
|
} else {
|
|
34
|
-
console.log(renderModelsTable(inspection.models));
|
|
34
|
+
console.log(renderModelsTable(inspection.models, { ascii: parsed.ascii }));
|
|
35
35
|
}
|
|
36
36
|
return;
|
|
37
37
|
}
|
|
@@ -46,7 +46,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
46
46
|
} else if (parsed.format === 'csv') {
|
|
47
47
|
console.log(toCsv(flattenResults(payload.results)));
|
|
48
48
|
} else {
|
|
49
|
-
console.log(
|
|
49
|
+
console.log(renderBenchmarkReport(payload, { ascii: parsed.ascii }));
|
|
50
50
|
if (parsed.verbose) {
|
|
51
51
|
console.log();
|
|
52
52
|
console.log(toCsv(flattenResults(payload.results)));
|
|
@@ -73,11 +73,13 @@ export function parseArgs(argv) {
|
|
|
73
73
|
timeoutMs: 60_000,
|
|
74
74
|
profile: 'standard',
|
|
75
75
|
greedy: true,
|
|
76
|
+
stream: true,
|
|
76
77
|
format: 'table',
|
|
77
78
|
captureOutput: false,
|
|
78
79
|
availableOnly: false,
|
|
79
80
|
failFast: false,
|
|
80
|
-
verbose: false
|
|
81
|
+
verbose: false,
|
|
82
|
+
ascii: false
|
|
81
83
|
};
|
|
82
84
|
|
|
83
85
|
const args = [...argv];
|
|
@@ -149,6 +151,12 @@ export function parseArgs(argv) {
|
|
|
149
151
|
case '--no-greedy':
|
|
150
152
|
options.greedy = false;
|
|
151
153
|
break;
|
|
154
|
+
case '--stream':
|
|
155
|
+
options.stream = true;
|
|
156
|
+
break;
|
|
157
|
+
case '--no-stream':
|
|
158
|
+
options.stream = false;
|
|
159
|
+
break;
|
|
152
160
|
case '--json':
|
|
153
161
|
options.format = 'json';
|
|
154
162
|
break;
|
|
@@ -161,6 +169,9 @@ export function parseArgs(argv) {
|
|
|
161
169
|
throw new Error('--format must be one of: table, json, csv');
|
|
162
170
|
}
|
|
163
171
|
break;
|
|
172
|
+
case '--ascii':
|
|
173
|
+
options.ascii = true;
|
|
174
|
+
break;
|
|
164
175
|
case '-o':
|
|
165
176
|
case '--out':
|
|
166
177
|
options.out = requireValue(arg, args);
|
|
@@ -258,12 +269,14 @@ Run options:
|
|
|
258
269
|
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
259
270
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
260
271
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
261
|
-
--profile <name> quick, standard, or stress
|
|
272
|
+
--profile <name> quick, standard, interactive, throughput, or stress
|
|
262
273
|
-i, --instructions <text> Instructions passed to fm respond
|
|
263
274
|
--use-case <case> Pass a system model use case through to fm
|
|
264
275
|
--guardrails <level> Pass a system model guardrail level through to fm
|
|
265
276
|
--greedy Use greedy sampling (default)
|
|
266
277
|
--no-greedy Do not request greedy sampling
|
|
278
|
+
--stream Stream responses while measuring TTFT (default)
|
|
279
|
+
--no-stream Disable streaming; TTFT fields will be blank
|
|
267
280
|
--available-only Hide unavailable discovered models
|
|
268
281
|
--capture-output Include raw model output in JSON reports
|
|
269
282
|
--fail-fast Stop after the first failed measured run
|
|
@@ -272,6 +285,7 @@ Output:
|
|
|
272
285
|
--format <type> table, json, or csv (default: table)
|
|
273
286
|
--json Alias for --format json
|
|
274
287
|
--csv Alias for --format csv
|
|
288
|
+
--ascii Use plain ASCII tables instead of Unicode
|
|
275
289
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
276
290
|
-v, --verbose Include per-run CSV after the summary table
|
|
277
291
|
|
package/src/fm.js
CHANGED
|
@@ -149,8 +149,9 @@ export async function countTokens(fmBin, text, options = {}) {
|
|
|
149
149
|
|
|
150
150
|
export async function respond(fmBin, model, prompt, options = {}) {
|
|
151
151
|
const args = ['respond', '--model', model];
|
|
152
|
+
const streamed = options.stream !== false;
|
|
152
153
|
|
|
153
|
-
if (
|
|
154
|
+
if (!streamed) args.push('--no-stream');
|
|
154
155
|
if (options.greedy) args.push('--greedy');
|
|
155
156
|
if (options.instructions) args.push('--instructions', options.instructions);
|
|
156
157
|
if (options.useCase) args.push('--use-case', options.useCase);
|
|
@@ -172,7 +173,10 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
172
173
|
code: result.code,
|
|
173
174
|
signal: result.signal,
|
|
174
175
|
timedOut: result.timedOut,
|
|
175
|
-
durationMs: result.durationMs
|
|
176
|
+
durationMs: result.durationMs,
|
|
177
|
+
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
178
|
+
streamed,
|
|
179
|
+
stdoutChunks: result.stdoutChunks
|
|
176
180
|
};
|
|
177
181
|
}
|
|
178
182
|
|
package/src/process.js
CHANGED
|
@@ -18,6 +18,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
18
18
|
|
|
19
19
|
let stdout = '';
|
|
20
20
|
let stderr = '';
|
|
21
|
+
let stdoutChunks = 0;
|
|
22
|
+
let stderrChunks = 0;
|
|
23
|
+
let firstStdoutMs = null;
|
|
24
|
+
let firstStderrMs = null;
|
|
21
25
|
let timedOut = false;
|
|
22
26
|
let settled = false;
|
|
23
27
|
|
|
@@ -34,9 +38,17 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
34
38
|
child.stdout.setEncoding('utf8');
|
|
35
39
|
child.stderr.setEncoding('utf8');
|
|
36
40
|
child.stdout.on('data', (chunk) => {
|
|
41
|
+
stdoutChunks += 1;
|
|
42
|
+
if (firstStdoutMs == null && chunk.length > 0) {
|
|
43
|
+
firstStdoutMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
44
|
+
}
|
|
37
45
|
stdout += chunk;
|
|
38
46
|
});
|
|
39
47
|
child.stderr.on('data', (chunk) => {
|
|
48
|
+
stderrChunks += 1;
|
|
49
|
+
if (firstStderrMs == null && chunk.length > 0) {
|
|
50
|
+
firstStderrMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
51
|
+
}
|
|
40
52
|
stderr += chunk;
|
|
41
53
|
});
|
|
42
54
|
|
|
@@ -51,6 +63,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
51
63
|
signal: null,
|
|
52
64
|
stdout,
|
|
53
65
|
stderr: stderr || error.message,
|
|
66
|
+
stdoutChunks,
|
|
67
|
+
stderrChunks,
|
|
68
|
+
firstStdoutMs,
|
|
69
|
+
firstStderrMs,
|
|
54
70
|
error,
|
|
55
71
|
timedOut,
|
|
56
72
|
durationMs: Number(endedAt - startedAt) / 1e6
|
|
@@ -68,6 +84,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
68
84
|
signal,
|
|
69
85
|
stdout,
|
|
70
86
|
stderr,
|
|
87
|
+
stdoutChunks,
|
|
88
|
+
stderrChunks,
|
|
89
|
+
firstStdoutMs,
|
|
90
|
+
firstStderrMs,
|
|
71
91
|
timedOut,
|
|
72
92
|
durationMs: Number(endedAt - startedAt) / 1e6
|
|
73
93
|
});
|
package/src/prompts.js
CHANGED
|
@@ -10,22 +10,50 @@ const PROFILES = {
|
|
|
10
10
|
],
|
|
11
11
|
standard: [
|
|
12
12
|
{
|
|
13
|
-
id: '
|
|
14
|
-
prompt: '
|
|
13
|
+
id: 'interactive-short',
|
|
14
|
+
prompt: 'Answer in one sentence: why should an on-device model benchmark report p95 latency?'
|
|
15
15
|
},
|
|
16
16
|
{
|
|
17
|
-
id: '
|
|
18
|
-
prompt: '
|
|
17
|
+
id: 'structured-json',
|
|
18
|
+
prompt: 'Return compact valid JSON with keys "summary" and "risks" for this text: Local model benchmarks should measure latency, throughput, failures, prompt size, and output size.'
|
|
19
19
|
},
|
|
20
20
|
{
|
|
21
|
-
id: '
|
|
22
|
-
prompt: '
|
|
21
|
+
id: 'medium-generation',
|
|
22
|
+
prompt: 'Write a concise four-bullet checklist for evaluating whether a local AI model is fast enough for an interactive coding assistant.'
|
|
23
|
+
}
|
|
24
|
+
],
|
|
25
|
+
interactive: [
|
|
26
|
+
{
|
|
27
|
+
id: 'chat-short-1',
|
|
28
|
+
prompt: 'Reply in one sentence: what is time to first token?'
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
id: 'chat-short-2',
|
|
32
|
+
prompt: 'Give one practical reason to benchmark with multiple prompt lengths.'
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
id: 'chat-short-3',
|
|
36
|
+
prompt: 'In under 20 words, define throughput for text generation.'
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
throughput: [
|
|
40
|
+
{
|
|
41
|
+
id: 'long-explain',
|
|
42
|
+
prompt: 'Write six concise bullets explaining the tradeoff between latency and throughput in local LLM inference.'
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
id: 'long-transform',
|
|
46
|
+
prompt: 'Rewrite this note as a polished release note with a title and five bullets: fm-bench now measures TTFT, end-to-end latency, output tokens per second, failures, and repeatability.'
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
id: 'long-plan',
|
|
50
|
+
prompt: 'Create a compact test plan for benchmarking a local foundation model across short, medium, and long prompts.'
|
|
23
51
|
}
|
|
24
52
|
],
|
|
25
53
|
stress: [
|
|
26
54
|
{
|
|
27
|
-
id: '
|
|
28
|
-
prompt: '
|
|
55
|
+
id: 'interactive-short',
|
|
56
|
+
prompt: 'Answer in one sentence: why should an on-device model benchmark report p95 latency?'
|
|
29
57
|
},
|
|
30
58
|
{
|
|
31
59
|
id: 'explain-latency',
|
package/src/report.js
CHANGED
|
@@ -17,12 +17,19 @@ export function flattenResults(results) {
|
|
|
17
17
|
run: result.run,
|
|
18
18
|
ok: result.ok,
|
|
19
19
|
duration_ms: round(result.durationMs),
|
|
20
|
+
ttft_ms: round(result.firstTokenMs),
|
|
21
|
+
generation_ms: round(result.generationMs),
|
|
22
|
+
tpot_ms: round(result.tpotMs),
|
|
20
23
|
prompt_tokens: result.promptTokens ?? '',
|
|
21
24
|
output_tokens: result.outputTokens ?? '',
|
|
22
25
|
chars: result.chars,
|
|
23
26
|
words: result.words,
|
|
24
27
|
tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
|
|
28
|
+
decode_tokens_per_second: result.decodeTokensPerSecond == null ? '' : round(result.decodeTokensPerSecond),
|
|
25
29
|
chars_per_second: round(result.charsPerSecond),
|
|
30
|
+
streamed: result.streamed,
|
|
31
|
+
stdout_chunks: result.stdoutChunks,
|
|
32
|
+
output_hash: result.outputHash || '',
|
|
26
33
|
error: result.error || ''
|
|
27
34
|
}));
|
|
28
35
|
}
|
package/src/stats.js
CHANGED
|
@@ -6,8 +6,11 @@ export function summarizeNumbers(values) {
|
|
|
6
6
|
min: null,
|
|
7
7
|
max: null,
|
|
8
8
|
avg: null,
|
|
9
|
+
sum: 0,
|
|
9
10
|
p50: null,
|
|
10
|
-
|
|
11
|
+
p90: null,
|
|
12
|
+
p95: null,
|
|
13
|
+
p99: null
|
|
11
14
|
};
|
|
12
15
|
}
|
|
13
16
|
|
|
@@ -17,8 +20,11 @@ export function summarizeNumbers(values) {
|
|
|
17
20
|
min: clean[0],
|
|
18
21
|
max: clean[clean.length - 1],
|
|
19
22
|
avg: total / clean.length,
|
|
23
|
+
sum: total,
|
|
20
24
|
p50: percentile(clean, 50),
|
|
21
|
-
|
|
25
|
+
p90: percentile(clean, 90),
|
|
26
|
+
p95: percentile(clean, 95),
|
|
27
|
+
p99: percentile(clean, 99)
|
|
22
28
|
};
|
|
23
29
|
}
|
|
24
30
|
|
|
@@ -64,9 +70,17 @@ export function summarizeByModel(results, modelStatuses = []) {
|
|
|
64
70
|
const successes = entry.results.filter((result) => result.ok);
|
|
65
71
|
const failures = entry.results.filter((result) => !result.ok);
|
|
66
72
|
const latency = summarizeNumbers(successes.map((result) => result.durationMs));
|
|
73
|
+
const ttft = summarizeNumbers(successes.map((result) => result.firstTokenMs).filter((value) => value != null));
|
|
74
|
+
const generation = summarizeNumbers(successes.map((result) => result.generationMs).filter((value) => value != null));
|
|
75
|
+
const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
|
|
76
|
+
const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
|
|
67
77
|
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
68
78
|
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
69
79
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
80
|
+
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
81
|
+
const windowMs = modelWindowMs(successes);
|
|
82
|
+
const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
|
|
83
|
+
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
70
84
|
|
|
71
85
|
return {
|
|
72
86
|
model: entry.model,
|
|
@@ -76,10 +90,46 @@ export function summarizeByModel(results, modelStatuses = []) {
|
|
|
76
90
|
attempted: entry.results.length,
|
|
77
91
|
successes: successes.length,
|
|
78
92
|
failures: failures.length,
|
|
93
|
+
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
94
|
+
rps,
|
|
95
|
+
outputTokenThroughput,
|
|
96
|
+
repeatability: summarizeRepeatability(successes),
|
|
79
97
|
latency,
|
|
98
|
+
ttft,
|
|
99
|
+
generation,
|
|
100
|
+
tpot,
|
|
101
|
+
promptTokens,
|
|
80
102
|
outputTokens,
|
|
81
103
|
charsPerSecond,
|
|
82
|
-
tokensPerSecond
|
|
104
|
+
tokensPerSecond,
|
|
105
|
+
decodeTokensPerSecond
|
|
83
106
|
};
|
|
84
107
|
});
|
|
85
108
|
}
|
|
109
|
+
|
|
110
|
+
function modelWindowMs(results) {
|
|
111
|
+
const starts = results.map((result) => result.startOffsetMs).filter((value) => Number.isFinite(value));
|
|
112
|
+
const ends = results.map((result) => result.endOffsetMs).filter((value) => Number.isFinite(value));
|
|
113
|
+
if (starts.length === 0 || ends.length === 0) return null;
|
|
114
|
+
return Math.max(...ends) - Math.min(...starts);
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function summarizeRepeatability(results) {
|
|
118
|
+
const byPrompt = new Map();
|
|
119
|
+
for (const result of results) {
|
|
120
|
+
if (!result.outputHash) continue;
|
|
121
|
+
if (!byPrompt.has(result.promptId)) byPrompt.set(result.promptId, []);
|
|
122
|
+
byPrompt.get(result.promptId).push(result.outputHash);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const scores = [];
|
|
126
|
+
for (const hashes of byPrompt.values()) {
|
|
127
|
+
if (hashes.length < 2) continue;
|
|
128
|
+
const counts = new Map();
|
|
129
|
+
for (const hash of hashes) counts.set(hash, (counts.get(hash) || 0) + 1);
|
|
130
|
+
scores.push(Math.max(...counts.values()) / hashes.length);
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (scores.length === 0) return null;
|
|
134
|
+
return scores.reduce((sum, score) => sum + score, 0) / scores.length;
|
|
135
|
+
}
|
package/src/table.js
CHANGED
|
@@ -1,19 +1,42 @@
|
|
|
1
|
-
export function renderTable(headers, rows) {
|
|
1
|
+
export function renderTable(headers, rows, options = {}) {
|
|
2
|
+
const ascii = Boolean(options.ascii);
|
|
2
3
|
const stringRows = rows.map((row) => row.map(formatCell));
|
|
3
4
|
const widths = headers.map((header, index) => {
|
|
4
5
|
const values = [header, ...stringRows.map((row) => row[index] ?? '')];
|
|
5
6
|
return Math.max(...values.map(visibleLength));
|
|
6
7
|
});
|
|
8
|
+
const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
|
|
7
9
|
|
|
8
|
-
const
|
|
9
|
-
const
|
|
10
|
-
const
|
|
10
|
+
const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
|
|
11
|
+
const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
|
|
12
|
+
const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
|
|
13
|
+
const headerLine = rowLine(headers, widths, style, true);
|
|
14
|
+
const bodyLines = stringRows.map((row) => rowLine(row, widths, style));
|
|
11
15
|
|
|
12
|
-
return [
|
|
16
|
+
return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function renderBenchmarkReport(payload, options = {}) {
|
|
20
|
+
const lines = [];
|
|
21
|
+
const elapsedMs = Date.parse(payload.finishedAt) - Date.parse(payload.startedAt);
|
|
22
|
+
const skipped = payload.summary.filter((item) => !item.available).length;
|
|
23
|
+
const measured = payload.summary.reduce((sum, item) => sum + item.successes, 0);
|
|
24
|
+
const failed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
|
|
25
|
+
|
|
26
|
+
lines.push(`fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`);
|
|
27
|
+
lines.push(`prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${payload.options.concurrency} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped models ${skipped} | elapsed ${formatMs(elapsedMs)}`);
|
|
28
|
+
lines.push('');
|
|
29
|
+
lines.push(renderSummaryTable(payload.summary, options));
|
|
30
|
+
lines.push('');
|
|
31
|
+
lines.push(renderDetailTable(payload.summary, options));
|
|
32
|
+
lines.push('');
|
|
33
|
+
lines.push('TTFT = time to first streamed output, E2E = full response latency, TPOT = decode time per output token.');
|
|
34
|
+
|
|
35
|
+
return lines.join('\n');
|
|
13
36
|
}
|
|
14
37
|
|
|
15
38
|
export function formatMs(value) {
|
|
16
|
-
if (value == null) return '-';
|
|
39
|
+
if (value == null || !Number.isFinite(value)) return '-';
|
|
17
40
|
if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
|
|
18
41
|
return `${Math.round(value)}ms`;
|
|
19
42
|
}
|
|
@@ -24,7 +47,12 @@ export function formatNumber(value, digits = 1) {
|
|
|
24
47
|
return value.toFixed(digits);
|
|
25
48
|
}
|
|
26
49
|
|
|
27
|
-
export function
|
|
50
|
+
export function formatPercent(value, digits = 0) {
|
|
51
|
+
if (value == null || !Number.isFinite(value)) return '-';
|
|
52
|
+
return `${(value * 100).toFixed(digits)}%`;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function renderSummaryTable(summary, options = {}) {
|
|
28
56
|
const rows = summary.map((item) => {
|
|
29
57
|
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
30
58
|
return [
|
|
@@ -32,13 +60,14 @@ export function renderSummaryTable(summary) {
|
|
|
32
60
|
status,
|
|
33
61
|
item.attempted || '-',
|
|
34
62
|
item.successes || '-',
|
|
35
|
-
item.
|
|
63
|
+
formatPercent(item.successRate),
|
|
64
|
+
formatMs(item.ttft.p50),
|
|
65
|
+
formatMs(item.ttft.p95),
|
|
36
66
|
formatMs(item.latency.p50),
|
|
37
67
|
formatMs(item.latency.p95),
|
|
38
|
-
formatMs(item.
|
|
68
|
+
formatMs(item.tpot.p50),
|
|
39
69
|
formatNumber(item.tokensPerSecond.avg),
|
|
40
|
-
formatNumber(item.
|
|
41
|
-
formatNumber(item.outputTokens.avg, 0),
|
|
70
|
+
formatNumber(item.rps),
|
|
42
71
|
item.available ? '' : compactReason(item.skippedReason)
|
|
43
72
|
];
|
|
44
73
|
});
|
|
@@ -48,24 +77,90 @@ export function renderSummaryTable(summary) {
|
|
|
48
77
|
'status',
|
|
49
78
|
'runs',
|
|
50
79
|
'ok',
|
|
51
|
-
'
|
|
52
|
-
'p50',
|
|
53
|
-
'p95',
|
|
54
|
-
'
|
|
80
|
+
'success',
|
|
81
|
+
'ttft p50',
|
|
82
|
+
'ttft p95',
|
|
83
|
+
'e2e p50',
|
|
84
|
+
'e2e p95',
|
|
85
|
+
'tpot p50',
|
|
55
86
|
'tok/s',
|
|
56
|
-
'
|
|
57
|
-
'out tok',
|
|
87
|
+
'rps',
|
|
58
88
|
'note'
|
|
59
|
-
], rows);
|
|
89
|
+
], rows, options);
|
|
60
90
|
}
|
|
61
91
|
|
|
62
|
-
export function
|
|
92
|
+
export function renderDetailTable(summary, options = {}) {
|
|
93
|
+
const rows = summary.map((item) => [
|
|
94
|
+
item.model,
|
|
95
|
+
formatNumber(item.promptTokens.avg, 0),
|
|
96
|
+
formatNumber(item.outputTokens.avg, 0),
|
|
97
|
+
formatNumber(item.outputTokenThroughput),
|
|
98
|
+
formatNumber(item.decodeTokensPerSecond.avg),
|
|
99
|
+
formatMs(item.latency.p99),
|
|
100
|
+
formatMs(item.tpot.p95),
|
|
101
|
+
formatPercent(item.repeatability),
|
|
102
|
+
item.description || '-'
|
|
103
|
+
]);
|
|
104
|
+
|
|
105
|
+
return renderTable([
|
|
106
|
+
'model',
|
|
107
|
+
'in tok avg',
|
|
108
|
+
'out tok avg',
|
|
109
|
+
'total tok/s',
|
|
110
|
+
'decode tok/s',
|
|
111
|
+
'e2e p99',
|
|
112
|
+
'tpot p95',
|
|
113
|
+
'repeat',
|
|
114
|
+
'description'
|
|
115
|
+
], rows, options);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function renderModelsTable(models, options = {}) {
|
|
63
119
|
return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
|
|
64
120
|
model.name,
|
|
65
121
|
model.available ? 'yes' : 'no',
|
|
66
122
|
model.description || '-',
|
|
67
123
|
compactReason(model.quota || model.reason || '-')
|
|
68
|
-
]));
|
|
124
|
+
]), options);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const ASCII_TABLE = {
|
|
128
|
+
topLeft: '+',
|
|
129
|
+
topJoin: '+',
|
|
130
|
+
topRight: '+',
|
|
131
|
+
midLeft: '+',
|
|
132
|
+
midJoin: '+',
|
|
133
|
+
midRight: '+',
|
|
134
|
+
bottomLeft: '+',
|
|
135
|
+
bottomJoin: '+',
|
|
136
|
+
bottomRight: '+',
|
|
137
|
+
horizontal: '-',
|
|
138
|
+
vertical: '|'
|
|
139
|
+
};
|
|
140
|
+
|
|
141
|
+
const UNICODE_TABLE = {
|
|
142
|
+
topLeft: '┌',
|
|
143
|
+
topJoin: '┬',
|
|
144
|
+
topRight: '┐',
|
|
145
|
+
midLeft: '├',
|
|
146
|
+
midJoin: '┼',
|
|
147
|
+
midRight: '┤',
|
|
148
|
+
bottomLeft: '└',
|
|
149
|
+
bottomJoin: '┴',
|
|
150
|
+
bottomRight: '┘',
|
|
151
|
+
horizontal: '─',
|
|
152
|
+
vertical: '│'
|
|
153
|
+
};
|
|
154
|
+
|
|
155
|
+
function rule(left, join, right, horizontal, widths) {
|
|
156
|
+
return `${left}${widths.map((width) => horizontal.repeat(width + 2)).join(join)}${right}`;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function rowLine(row, widths, style, header = false) {
|
|
160
|
+
return `${style.vertical}${row.map((cell, index) => {
|
|
161
|
+
const value = header ? String(cell).toUpperCase() : formatCell(cell);
|
|
162
|
+
return ` ${pad(value, widths[index], !header && isNumericCell(value))} `;
|
|
163
|
+
}).join(style.vertical)}${style.vertical}`;
|
|
69
164
|
}
|
|
70
165
|
|
|
71
166
|
function formatCell(value) {
|
|
@@ -80,7 +175,7 @@ function pad(value, width, left = false) {
|
|
|
80
175
|
}
|
|
81
176
|
|
|
82
177
|
function isNumericCell(value) {
|
|
83
|
-
return /^-?$|^[\d,.]+(?:ms|s)?$/.test(value);
|
|
178
|
+
return /^-?$|^[\d,.]+(?:ms|s|%)?$/.test(value);
|
|
84
179
|
}
|
|
85
180
|
|
|
86
181
|
function visibleLength(value) {
|