fm-bench 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -12
- package/docs/methodology.md +43 -0
- package/package.json +2 -1
- package/src/bench.js +134 -20
- package/src/cli.js +71 -5
- package/src/fm.js +6 -2
- package/src/process.js +20 -0
- package/src/prompts.js +36 -8
- package/src/report.js +9 -0
- package/src/stats.js +136 -16
- package/src/table.js +251 -35
package/README.md
CHANGED
|
@@ -35,12 +35,20 @@ fm-bench
|
|
|
35
35
|
Example output:
|
|
36
36
|
|
|
37
37
|
```text
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
38
|
+
fm-bench 0.2.0 | darwin/arm64 | fm
|
|
39
|
+
prompts 3 | runs 1 | concurrency 1 | stream on | measured 3 | failed 0 | skipped models 0 | elapsed 11.36s
|
|
40
|
+
|
|
41
|
+
┌────────┬────────┬──────┬────┬─────────┬──────────┬──────────┬─────────┬─────────┬──────────┬───────┬─────┬──────┐
|
|
42
|
+
│ MODEL │ STATUS │ RUNS │ OK │ SUCCESS │ TTFT P50 │ TTFT P95 │ E2E P50 │ E2E P95 │ TPOT P50 │ TOK/S │ RPS │ NOTE │
|
|
43
|
+
├────────┼────────┼──────┼────┼─────────┼──────────┼──────────┼─────────┼─────────┼──────────┼───────┼─────┼──────┤
|
|
44
|
+
│ system │ ok │ 3 │ 3 │ 100% │ 409ms │ 486ms │ 2.29s │ 5.97s │ 14ms │ 58.5 │ 0.3 │ │
|
|
45
|
+
└────────┴────────┴──────┴────┴─────────┴──────────┴──────────┴─────────┴─────────┴──────────┴───────┴─────┴──────┘
|
|
46
|
+
|
|
47
|
+
┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
|
|
48
|
+
│ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
|
|
49
|
+
├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
|
|
50
|
+
│ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
|
|
51
|
+
└────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
|
|
44
52
|
```
|
|
45
53
|
|
|
46
54
|
## Commands
|
|
@@ -61,6 +69,10 @@ fm-bench doctor [options]
|
|
|
61
69
|
|
|
62
70
|
```sh
|
|
63
71
|
fm-bench --models system,pcc --runs 3 --profile stress
|
|
72
|
+
fm-bench --models system --runs 5 --profile interactive
|
|
73
|
+
fm-bench --models system --runs 3 --profile throughput --warmup 1
|
|
74
|
+
fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
|
|
75
|
+
fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
|
|
64
76
|
fm-bench --prompt "Reply with exactly: ok" --runs 5
|
|
65
77
|
fm-bench --prompt-file prompts.json --format json --out reports/bench.json
|
|
66
78
|
fm-bench --format csv --out reports/bench.csv
|
|
@@ -72,14 +84,19 @@ Useful flags:
|
|
|
72
84
|
- `--runs <n>`: measured runs per prompt/model.
|
|
73
85
|
- `--warmup <n>`: warmup runs per model before measurement.
|
|
74
86
|
- `--concurrency <n>`: parallel `fm` processes.
|
|
87
|
+
- `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
|
|
75
88
|
- `--timeout-ms <n>`: timeout per `fm` call.
|
|
76
|
-
- `--
|
|
89
|
+
- `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
|
|
90
|
+
- `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
|
|
77
91
|
- `--prompt <text>`: custom prompt, repeatable.
|
|
78
92
|
- `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
|
|
79
93
|
- `--instructions <text>`: passed to `fm respond`.
|
|
80
94
|
- `--available-only`: hide unavailable discovered models.
|
|
81
95
|
- `--capture-output`: include raw model output in JSON reports.
|
|
82
96
|
- `--json`, `--csv`, `--format table|json|csv`: choose output format.
|
|
97
|
+
- `--ascii`: use plain ASCII table borders.
|
|
98
|
+
- `--compact`: force the narrow terminal layout.
|
|
99
|
+
- `--width <n>`: render as if the terminal has `n` columns.
|
|
83
100
|
- `--out <file>`: save a report.
|
|
84
101
|
|
|
85
102
|
## Prompt Files
|
|
@@ -106,14 +123,27 @@ Plain text files are split on blank lines.
|
|
|
106
123
|
|
|
107
124
|
`fm-bench` reports:
|
|
108
125
|
|
|
109
|
-
-
|
|
110
|
-
-
|
|
111
|
-
-
|
|
112
|
-
-
|
|
126
|
+
- TTFT, or time to first streamed output.
|
|
127
|
+
- E2E latency, or full response wall-clock latency.
|
|
128
|
+
- TPOT, or decode time per output token after the first output token.
|
|
129
|
+
- output tokens per second per request.
|
|
130
|
+
- total output token throughput across the measured window.
|
|
131
|
+
- requests per second across the measured window.
|
|
132
|
+
- goodput percentage and goodput RPS when SLO flags are set.
|
|
133
|
+
- coefficient of variation (CV) and confidence interval context for stability.
|
|
134
|
+
- prompt and output token counts.
|
|
135
|
+
- p50, p95, and p99 tail latency views.
|
|
136
|
+
- repeatability across repeated runs of the same prompt.
|
|
113
137
|
- success and failure counts.
|
|
114
138
|
- unavailable model notes.
|
|
115
139
|
|
|
116
|
-
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response,
|
|
140
|
+
Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
|
|
141
|
+
|
|
142
|
+
Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
|
|
143
|
+
|
|
144
|
+
Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
|
|
145
|
+
|
|
146
|
+
See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
|
|
117
147
|
|
|
118
148
|
## Requirements
|
|
119
149
|
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Methodology
|
|
2
|
+
|
|
3
|
+
`fm-bench` measures local `fm` command behavior from the client side. It is meant to answer: "What does this Mac deliver to a terminal user for this prompt suite right now?"
|
|
4
|
+
|
|
5
|
+
## Sources
|
|
6
|
+
|
|
7
|
+
The metric set follows common LLM inference benchmark practice:
|
|
8
|
+
|
|
9
|
+
- Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
|
|
10
|
+
- NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
|
|
11
|
+
- NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
|
|
12
|
+
- vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
|
|
13
|
+
- MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
|
|
14
|
+
- MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
|
|
15
|
+
|
|
16
|
+
## Metrics
|
|
17
|
+
|
|
18
|
+
- `TTFT`: time from starting `fm respond` to the first streamed stdout chunk. This is a practical terminal-side proxy for time to first token.
|
|
19
|
+
- `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
|
|
20
|
+
- `generation_ms`: `E2E - TTFT`.
|
|
21
|
+
- `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
|
|
22
|
+
- `tokens_per_second`: output tokens divided by E2E seconds for one request.
|
|
23
|
+
- `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
|
|
24
|
+
- `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
|
|
25
|
+
- `RPS`: successful requests divided by that model's measured wall-clock window.
|
|
26
|
+
- `goodput`: successful requests that also satisfy all provided SLO thresholds.
|
|
27
|
+
- `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
|
|
28
|
+
- `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
|
|
29
|
+
- `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
|
|
30
|
+
|
|
31
|
+
## Operating Points
|
|
32
|
+
|
|
33
|
+
Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
|
|
34
|
+
|
|
35
|
+
`fm-bench` does not interpolate between operating points. It reports only what was actually measured.
|
|
36
|
+
|
|
37
|
+
## Caveats
|
|
38
|
+
|
|
39
|
+
`fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
|
|
40
|
+
|
|
41
|
+
Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
|
|
42
|
+
|
|
43
|
+
For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "fm-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
},
|
|
9
9
|
"files": [
|
|
10
10
|
"bin",
|
|
11
|
+
"docs",
|
|
11
12
|
"src",
|
|
12
13
|
"README.md",
|
|
13
14
|
"LICENSE"
|
package/src/bench.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import crypto from 'node:crypto';
|
|
1
2
|
import { checkModelAvailability, collectEnvironment, countTokens, discoverModels, getQuotaUsage, respond } from './fm.js';
|
|
2
3
|
import { loadPrompts } from './prompts.js';
|
|
3
4
|
import { summarizeByModel } from './stats.js';
|
|
@@ -43,15 +44,69 @@ export async function runBenchmark(options = {}) {
|
|
|
43
44
|
const runnableModels = modelStatuses.filter((model) => model.available);
|
|
44
45
|
const environment = await collectEnvironment(inspection.fmBin);
|
|
45
46
|
const promptTokenCounts = new Map();
|
|
47
|
+
const concurrencies = normalizeConcurrencySweep(options);
|
|
46
48
|
|
|
47
49
|
for (const prompt of prompts) {
|
|
48
50
|
const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
|
|
49
51
|
promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
|
|
50
52
|
}
|
|
51
53
|
|
|
54
|
+
const results = [];
|
|
55
|
+
const scenarios = [];
|
|
56
|
+
for (const concurrency of concurrencies) {
|
|
57
|
+
const scenario = await runScenario({
|
|
58
|
+
fmBin: inspection.fmBin,
|
|
59
|
+
prompts,
|
|
60
|
+
runnableModels,
|
|
61
|
+
modelStatuses,
|
|
62
|
+
promptTokenCounts,
|
|
63
|
+
options,
|
|
64
|
+
concurrency
|
|
65
|
+
});
|
|
66
|
+
scenarios.push(scenario);
|
|
67
|
+
results.push(...scenario.results);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
71
|
+
|| (a.concurrency ?? 0) - (b.concurrency ?? 0)
|
|
72
|
+
|| a.promptId.localeCompare(b.promptId)
|
|
73
|
+
|| a.run - b.run);
|
|
74
|
+
|
|
75
|
+
const summary = summarizeByModel(results, modelStatuses, { concurrencies });
|
|
76
|
+
return {
|
|
77
|
+
tool: 'fm-bench',
|
|
78
|
+
version: options.version,
|
|
79
|
+
startedAt,
|
|
80
|
+
finishedAt: new Date().toISOString(),
|
|
81
|
+
options: publicOptions(options),
|
|
82
|
+
environment,
|
|
83
|
+
prompts: prompts.map((prompt) => ({
|
|
84
|
+
id: prompt.id,
|
|
85
|
+
prompt: prompt.prompt,
|
|
86
|
+
promptTokens: promptTokenCounts.get(prompt.id)
|
|
87
|
+
})),
|
|
88
|
+
models: modelStatuses,
|
|
89
|
+
scenarios,
|
|
90
|
+
summary,
|
|
91
|
+
results
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
async function runScenario(context) {
|
|
96
|
+
const {
|
|
97
|
+
fmBin,
|
|
98
|
+
prompts,
|
|
99
|
+
runnableModels,
|
|
100
|
+
modelStatuses,
|
|
101
|
+
promptTokenCounts,
|
|
102
|
+
options,
|
|
103
|
+
concurrency
|
|
104
|
+
} = context;
|
|
105
|
+
const startedAt = new Date().toISOString();
|
|
106
|
+
|
|
52
107
|
for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
|
|
53
108
|
for (const model of runnableModels) {
|
|
54
|
-
await respond(
|
|
109
|
+
await respond(fmBin, model.name, prompts[0].prompt, {
|
|
55
110
|
...options,
|
|
56
111
|
stream: false
|
|
57
112
|
});
|
|
@@ -59,17 +114,18 @@ export async function runBenchmark(options = {}) {
|
|
|
59
114
|
}
|
|
60
115
|
|
|
61
116
|
const jobs = [];
|
|
117
|
+
const benchmarkStartedAt = process.hrtime.bigint();
|
|
62
118
|
for (const model of runnableModels) {
|
|
63
119
|
for (const prompt of prompts) {
|
|
64
120
|
for (let run = 1; run <= options.runs; run += 1) {
|
|
65
|
-
jobs.push({ model, prompt, run });
|
|
121
|
+
jobs.push({ model, prompt, run, concurrency });
|
|
66
122
|
}
|
|
67
123
|
}
|
|
68
124
|
}
|
|
69
125
|
|
|
70
126
|
const results = [];
|
|
71
|
-
await runLimited(jobs,
|
|
72
|
-
const result = await runSingleBenchmark(
|
|
127
|
+
await runLimited(jobs, concurrency, async (job) => {
|
|
128
|
+
const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
|
|
73
129
|
results.push(result);
|
|
74
130
|
if (!result.ok && options.failFast) {
|
|
75
131
|
const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
|
|
@@ -78,49 +134,71 @@ export async function runBenchmark(options = {}) {
|
|
|
78
134
|
}
|
|
79
135
|
});
|
|
80
136
|
|
|
81
|
-
|
|
137
|
+
results.sort((a, b) => a.model.localeCompare(b.model)
|
|
138
|
+
|| a.promptId.localeCompare(b.promptId)
|
|
139
|
+
|| a.run - b.run);
|
|
140
|
+
|
|
82
141
|
return {
|
|
83
|
-
|
|
84
|
-
version: options.version,
|
|
142
|
+
concurrency,
|
|
85
143
|
startedAt,
|
|
86
144
|
finishedAt: new Date().toISOString(),
|
|
87
|
-
|
|
88
|
-
environment,
|
|
89
|
-
prompts: prompts.map((prompt) => ({
|
|
90
|
-
id: prompt.id,
|
|
91
|
-
prompt: prompt.prompt,
|
|
92
|
-
promptTokens: promptTokenCounts.get(prompt.id)
|
|
93
|
-
})),
|
|
94
|
-
models: modelStatuses,
|
|
95
|
-
summary,
|
|
145
|
+
summary: summarizeByModel(results, modelStatuses, { concurrencies: [concurrency] }),
|
|
96
146
|
results
|
|
97
147
|
};
|
|
98
148
|
}
|
|
99
149
|
|
|
100
|
-
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options) {
|
|
150
|
+
async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt) {
|
|
151
|
+
const startOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
101
152
|
const response = await respond(fmBin, job.model.name, job.prompt.prompt, {
|
|
102
153
|
...options,
|
|
103
|
-
stream:
|
|
154
|
+
stream: options.stream
|
|
104
155
|
});
|
|
156
|
+
const endOffsetMs = Number(process.hrtime.bigint() - benchmarkStartedAt) / 1e6;
|
|
105
157
|
const outputTokens = response.ok
|
|
106
158
|
? await countTokens(fmBin, response.output, options)
|
|
107
159
|
: { ok: false, count: null };
|
|
108
160
|
const seconds = response.durationMs / 1000;
|
|
161
|
+
const firstTokenMs = response.firstOutputMs;
|
|
162
|
+
const generationMs = response.ok && firstTokenMs != null
|
|
163
|
+
? Math.max(0, response.durationMs - firstTokenMs)
|
|
164
|
+
: null;
|
|
165
|
+
const countedOutputTokens = outputTokens.ok ? outputTokens.count : null;
|
|
166
|
+
const decodeTokenCount = countedOutputTokens != null ? Math.max(0, countedOutputTokens - 1) : null;
|
|
167
|
+
const hasDecodeCadence = response.stdoutChunks > 2 && generationMs != null && generationMs > 0 && decodeTokenCount > 0;
|
|
168
|
+
const tpotMs = hasDecodeCadence ? generationMs / decodeTokenCount : null;
|
|
169
|
+
const decodeTokensPerSecond = hasDecodeCadence
|
|
170
|
+
? decodeTokenCount / (generationMs / 1000)
|
|
171
|
+
: null;
|
|
109
172
|
const chars = response.output.length;
|
|
110
173
|
const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
|
|
111
174
|
|
|
112
175
|
return {
|
|
113
176
|
model: job.model.name,
|
|
177
|
+
concurrency: job.concurrency,
|
|
114
178
|
promptId: job.prompt.id,
|
|
115
179
|
run: job.run,
|
|
116
180
|
ok: response.ok,
|
|
117
181
|
durationMs: response.durationMs,
|
|
182
|
+
firstTokenMs,
|
|
183
|
+
generationMs,
|
|
184
|
+
tpotMs,
|
|
118
185
|
promptTokens: promptTokenCounts.get(job.prompt.id),
|
|
119
|
-
outputTokens:
|
|
186
|
+
outputTokens: countedOutputTokens,
|
|
120
187
|
chars,
|
|
121
188
|
words,
|
|
122
|
-
tokensPerSecond:
|
|
189
|
+
tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
|
|
190
|
+
decodeTokensPerSecond,
|
|
123
191
|
charsPerSecond: seconds > 0 ? chars / seconds : 0,
|
|
192
|
+
startOffsetMs,
|
|
193
|
+
endOffsetMs,
|
|
194
|
+
streamed: response.streamed,
|
|
195
|
+
stdoutChunks: response.stdoutChunks,
|
|
196
|
+
outputHash: response.ok ? hashOutput(response.output) : null,
|
|
197
|
+
good: response.ok ? evaluateSlo({
|
|
198
|
+
firstTokenMs,
|
|
199
|
+
durationMs: response.durationMs,
|
|
200
|
+
tpotMs
|
|
201
|
+
}, options) : false,
|
|
124
202
|
output: options.captureOutput ? response.output : undefined,
|
|
125
203
|
error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
|
|
126
204
|
};
|
|
@@ -147,15 +225,51 @@ function normalizeModelSelection(models) {
|
|
|
147
225
|
}
|
|
148
226
|
|
|
149
227
|
function publicOptions(options) {
|
|
228
|
+
const concurrencies = normalizeConcurrencySweep(options);
|
|
150
229
|
return {
|
|
151
230
|
models: normalizeModelSelection(options.models),
|
|
152
231
|
runs: options.runs,
|
|
153
232
|
warmup: options.warmup,
|
|
154
233
|
concurrency: options.concurrency,
|
|
234
|
+
sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
|
|
155
235
|
timeoutMs: options.timeoutMs,
|
|
156
236
|
profile: options.profile,
|
|
157
237
|
promptCount: options.promptCount,
|
|
158
238
|
greedy: options.greedy,
|
|
239
|
+
stream: options.stream,
|
|
240
|
+
slo: {
|
|
241
|
+
ttftMs: options.sloTtftMs || null,
|
|
242
|
+
e2eMs: options.sloE2eMs || null,
|
|
243
|
+
tpotMs: options.sloTpotMs || null
|
|
244
|
+
},
|
|
159
245
|
instructions: options.instructions ? '[set]' : ''
|
|
160
246
|
};
|
|
161
247
|
}
|
|
248
|
+
|
|
249
|
+
function normalizeConcurrencySweep(options) {
|
|
250
|
+
if (options.sweepConcurrency?.length) {
|
|
251
|
+
return [...new Set(options.sweepConcurrency)]
|
|
252
|
+
.filter((value) => Number.isInteger(value) && value > 0)
|
|
253
|
+
.sort((a, b) => a - b);
|
|
254
|
+
}
|
|
255
|
+
return [Math.max(1, options.concurrency || 1)];
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
function hashOutput(output) {
|
|
259
|
+
return crypto.createHash('sha256')
|
|
260
|
+
.update(output.replace(/\s+/g, ' ').trim())
|
|
261
|
+
.digest('hex')
|
|
262
|
+
.slice(0, 16);
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
function evaluateSlo(metrics, options) {
|
|
266
|
+
const thresholds = [
|
|
267
|
+
['firstTokenMs', options.sloTtftMs],
|
|
268
|
+
['durationMs', options.sloE2eMs],
|
|
269
|
+
['tpotMs', options.sloTpotMs]
|
|
270
|
+
].filter(([, threshold]) => Number.isFinite(threshold));
|
|
271
|
+
|
|
272
|
+
if (thresholds.length === 0) return null;
|
|
273
|
+
|
|
274
|
+
return thresholds.every(([field, threshold]) => metrics[field] != null && metrics[field] <= threshold);
|
|
275
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -3,7 +3,7 @@ import { createRequire } from 'node:module';
|
|
|
3
3
|
import { inspectModels, runBenchmark } from './bench.js';
|
|
4
4
|
import { runProcess } from './process.js';
|
|
5
5
|
import { flattenResults, toCsv, writeReport } from './report.js';
|
|
6
|
-
import {
|
|
6
|
+
import { renderBenchmarkReport, renderModelsTable } from './table.js';
|
|
7
7
|
|
|
8
8
|
const require = createRequire(import.meta.url);
|
|
9
9
|
const packageJson = require('../package.json');
|
|
@@ -31,7 +31,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
31
31
|
if (parsed.format === 'json') {
|
|
32
32
|
console.log(JSON.stringify(inspection.models, null, 2));
|
|
33
33
|
} else {
|
|
34
|
-
console.log(renderModelsTable(inspection.models));
|
|
34
|
+
console.log(renderModelsTable(inspection.models, renderOptions(parsed)));
|
|
35
35
|
}
|
|
36
36
|
return;
|
|
37
37
|
}
|
|
@@ -46,7 +46,7 @@ export async function runCli(argv = process.argv.slice(2)) {
|
|
|
46
46
|
} else if (parsed.format === 'csv') {
|
|
47
47
|
console.log(toCsv(flattenResults(payload.results)));
|
|
48
48
|
} else {
|
|
49
|
-
console.log(
|
|
49
|
+
console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
|
|
50
50
|
if (parsed.verbose) {
|
|
51
51
|
console.log();
|
|
52
52
|
console.log(toCsv(flattenResults(payload.results)));
|
|
@@ -70,14 +70,22 @@ export function parseArgs(argv) {
|
|
|
70
70
|
runs: 1,
|
|
71
71
|
warmup: 0,
|
|
72
72
|
concurrency: 1,
|
|
73
|
+
sweepConcurrency: [],
|
|
73
74
|
timeoutMs: 60_000,
|
|
74
75
|
profile: 'standard',
|
|
75
76
|
greedy: true,
|
|
77
|
+
stream: true,
|
|
78
|
+
sloTtftMs: null,
|
|
79
|
+
sloE2eMs: null,
|
|
80
|
+
sloTpotMs: null,
|
|
76
81
|
format: 'table',
|
|
77
82
|
captureOutput: false,
|
|
78
83
|
availableOnly: false,
|
|
79
84
|
failFast: false,
|
|
80
|
-
verbose: false
|
|
85
|
+
verbose: false,
|
|
86
|
+
ascii: false,
|
|
87
|
+
compact: false,
|
|
88
|
+
width: null
|
|
81
89
|
};
|
|
82
90
|
|
|
83
91
|
const args = [...argv];
|
|
@@ -116,10 +124,25 @@ export function parseArgs(argv) {
|
|
|
116
124
|
case '--concurrency':
|
|
117
125
|
options.concurrency = parsePositiveInt(requireValue(arg, args), arg);
|
|
118
126
|
break;
|
|
127
|
+
case '--sweep-concurrency':
|
|
128
|
+
options.sweepConcurrency = parsePositiveIntList(requireValue(arg, args), arg);
|
|
129
|
+
if (options.sweepConcurrency.length > 0) {
|
|
130
|
+
options.concurrency = options.sweepConcurrency[0];
|
|
131
|
+
}
|
|
132
|
+
break;
|
|
119
133
|
case '--timeout':
|
|
120
134
|
case '--timeout-ms':
|
|
121
135
|
options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
122
136
|
break;
|
|
137
|
+
case '--slo-ttft-ms':
|
|
138
|
+
options.sloTtftMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
139
|
+
break;
|
|
140
|
+
case '--slo-e2e-ms':
|
|
141
|
+
options.sloE2eMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
142
|
+
break;
|
|
143
|
+
case '--slo-tpot-ms':
|
|
144
|
+
options.sloTpotMs = parsePositiveInt(requireValue(arg, args), arg);
|
|
145
|
+
break;
|
|
123
146
|
case '-p':
|
|
124
147
|
case '--prompt':
|
|
125
148
|
options.prompts.push(requireValue(arg, args));
|
|
@@ -149,6 +172,12 @@ export function parseArgs(argv) {
|
|
|
149
172
|
case '--no-greedy':
|
|
150
173
|
options.greedy = false;
|
|
151
174
|
break;
|
|
175
|
+
case '--stream':
|
|
176
|
+
options.stream = true;
|
|
177
|
+
break;
|
|
178
|
+
case '--no-stream':
|
|
179
|
+
options.stream = false;
|
|
180
|
+
break;
|
|
152
181
|
case '--json':
|
|
153
182
|
options.format = 'json';
|
|
154
183
|
break;
|
|
@@ -161,6 +190,15 @@ export function parseArgs(argv) {
|
|
|
161
190
|
throw new Error('--format must be one of: table, json, csv');
|
|
162
191
|
}
|
|
163
192
|
break;
|
|
193
|
+
case '--ascii':
|
|
194
|
+
options.ascii = true;
|
|
195
|
+
break;
|
|
196
|
+
case '--compact':
|
|
197
|
+
options.compact = true;
|
|
198
|
+
break;
|
|
199
|
+
case '--width':
|
|
200
|
+
options.width = parsePositiveInt(requireValue(arg, args), arg);
|
|
201
|
+
break;
|
|
164
202
|
case '-o':
|
|
165
203
|
case '--out':
|
|
166
204
|
options.out = requireValue(arg, args);
|
|
@@ -240,6 +278,24 @@ function parseNonNegativeInt(value, option) {
|
|
|
240
278
|
return parsed;
|
|
241
279
|
}
|
|
242
280
|
|
|
281
|
+
function parsePositiveIntList(value, option) {
|
|
282
|
+
const parsed = String(value)
|
|
283
|
+
.split(',')
|
|
284
|
+
.map((item) => item.trim())
|
|
285
|
+
.filter(Boolean)
|
|
286
|
+
.map((item) => parsePositiveInt(item, option));
|
|
287
|
+
if (parsed.length === 0) throw new Error(`${option} requires at least one positive integer`);
|
|
288
|
+
return parsed;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
function renderOptions(parsed) {
|
|
292
|
+
return {
|
|
293
|
+
ascii: parsed.ascii,
|
|
294
|
+
compact: parsed.compact,
|
|
295
|
+
width: parsed.width
|
|
296
|
+
};
|
|
297
|
+
}
|
|
298
|
+
|
|
243
299
|
function helpText() {
|
|
244
300
|
return `fm-bench ${packageJson.version}
|
|
245
301
|
|
|
@@ -255,15 +311,22 @@ Run options:
|
|
|
255
311
|
-r, --runs <n> Runs per prompt/model (default: 1)
|
|
256
312
|
--warmup <n> Warmup runs per model before measurement
|
|
257
313
|
-c, --concurrency <n> Parallel fm processes (default: 1)
|
|
314
|
+
--sweep-concurrency <list>
|
|
315
|
+
Run separate operating points, e.g. 1,2,4
|
|
258
316
|
--timeout-ms <n> Timeout per fm call in ms (default: 60000)
|
|
317
|
+
--slo-ttft-ms <n> Count request as good only if TTFT is <= n
|
|
318
|
+
--slo-e2e-ms <n> Count request as good only if E2E latency is <= n
|
|
319
|
+
--slo-tpot-ms <n> Count request as good only if TPOT is <= n
|
|
259
320
|
-p, --prompt <text> Prompt to benchmark; repeatable
|
|
260
321
|
--prompt-file <file> .json, .jsonl, or blank-line separated text prompts
|
|
261
|
-
--profile <name> quick, standard, or stress
|
|
322
|
+
--profile <name> quick, standard, interactive, throughput, or stress
|
|
262
323
|
-i, --instructions <text> Instructions passed to fm respond
|
|
263
324
|
--use-case <case> Pass a system model use case through to fm
|
|
264
325
|
--guardrails <level> Pass a system model guardrail level through to fm
|
|
265
326
|
--greedy Use greedy sampling (default)
|
|
266
327
|
--no-greedy Do not request greedy sampling
|
|
328
|
+
--stream Stream responses while measuring TTFT (default)
|
|
329
|
+
--no-stream Disable streaming; TTFT fields will be blank
|
|
267
330
|
--available-only Hide unavailable discovered models
|
|
268
331
|
--capture-output Include raw model output in JSON reports
|
|
269
332
|
--fail-fast Stop after the first failed measured run
|
|
@@ -272,6 +335,9 @@ Output:
|
|
|
272
335
|
--format <type> table, json, or csv (default: table)
|
|
273
336
|
--json Alias for --format json
|
|
274
337
|
--csv Alias for --format csv
|
|
338
|
+
--ascii Use plain ASCII tables instead of Unicode
|
|
339
|
+
--compact Force compact terminal layout
|
|
340
|
+
--width <n> Render for a specific terminal width
|
|
275
341
|
-o, --out <file> Save JSON or CSV report based on file extension
|
|
276
342
|
-v, --verbose Include per-run CSV after the summary table
|
|
277
343
|
|
package/src/fm.js
CHANGED
|
@@ -149,8 +149,9 @@ export async function countTokens(fmBin, text, options = {}) {
|
|
|
149
149
|
|
|
150
150
|
export async function respond(fmBin, model, prompt, options = {}) {
|
|
151
151
|
const args = ['respond', '--model', model];
|
|
152
|
+
const streamed = options.stream !== false;
|
|
152
153
|
|
|
153
|
-
if (
|
|
154
|
+
if (!streamed) args.push('--no-stream');
|
|
154
155
|
if (options.greedy) args.push('--greedy');
|
|
155
156
|
if (options.instructions) args.push('--instructions', options.instructions);
|
|
156
157
|
if (options.useCase) args.push('--use-case', options.useCase);
|
|
@@ -172,7 +173,10 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
172
173
|
code: result.code,
|
|
173
174
|
signal: result.signal,
|
|
174
175
|
timedOut: result.timedOut,
|
|
175
|
-
durationMs: result.durationMs
|
|
176
|
+
durationMs: result.durationMs,
|
|
177
|
+
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
178
|
+
streamed,
|
|
179
|
+
stdoutChunks: result.stdoutChunks
|
|
176
180
|
};
|
|
177
181
|
}
|
|
178
182
|
|
package/src/process.js
CHANGED
|
@@ -18,6 +18,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
18
18
|
|
|
19
19
|
let stdout = '';
|
|
20
20
|
let stderr = '';
|
|
21
|
+
let stdoutChunks = 0;
|
|
22
|
+
let stderrChunks = 0;
|
|
23
|
+
let firstStdoutMs = null;
|
|
24
|
+
let firstStderrMs = null;
|
|
21
25
|
let timedOut = false;
|
|
22
26
|
let settled = false;
|
|
23
27
|
|
|
@@ -34,9 +38,17 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
34
38
|
child.stdout.setEncoding('utf8');
|
|
35
39
|
child.stderr.setEncoding('utf8');
|
|
36
40
|
child.stdout.on('data', (chunk) => {
|
|
41
|
+
stdoutChunks += 1;
|
|
42
|
+
if (firstStdoutMs == null && chunk.length > 0) {
|
|
43
|
+
firstStdoutMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
44
|
+
}
|
|
37
45
|
stdout += chunk;
|
|
38
46
|
});
|
|
39
47
|
child.stderr.on('data', (chunk) => {
|
|
48
|
+
stderrChunks += 1;
|
|
49
|
+
if (firstStderrMs == null && chunk.length > 0) {
|
|
50
|
+
firstStderrMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
|
|
51
|
+
}
|
|
40
52
|
stderr += chunk;
|
|
41
53
|
});
|
|
42
54
|
|
|
@@ -51,6 +63,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
51
63
|
signal: null,
|
|
52
64
|
stdout,
|
|
53
65
|
stderr: stderr || error.message,
|
|
66
|
+
stdoutChunks,
|
|
67
|
+
stderrChunks,
|
|
68
|
+
firstStdoutMs,
|
|
69
|
+
firstStderrMs,
|
|
54
70
|
error,
|
|
55
71
|
timedOut,
|
|
56
72
|
durationMs: Number(endedAt - startedAt) / 1e6
|
|
@@ -68,6 +84,10 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
68
84
|
signal,
|
|
69
85
|
stdout,
|
|
70
86
|
stderr,
|
|
87
|
+
stdoutChunks,
|
|
88
|
+
stderrChunks,
|
|
89
|
+
firstStdoutMs,
|
|
90
|
+
firstStderrMs,
|
|
71
91
|
timedOut,
|
|
72
92
|
durationMs: Number(endedAt - startedAt) / 1e6
|
|
73
93
|
});
|
package/src/prompts.js
CHANGED
|
@@ -10,22 +10,50 @@ const PROFILES = {
|
|
|
10
10
|
],
|
|
11
11
|
standard: [
|
|
12
12
|
{
|
|
13
|
-
id: '
|
|
14
|
-
prompt: '
|
|
13
|
+
id: 'interactive-short',
|
|
14
|
+
prompt: 'Answer in one sentence: why should an on-device model benchmark report p95 latency?'
|
|
15
15
|
},
|
|
16
16
|
{
|
|
17
|
-
id: '
|
|
18
|
-
prompt: '
|
|
17
|
+
id: 'structured-json',
|
|
18
|
+
prompt: 'Return compact valid JSON with keys "summary" and "risks" for this text: Local model benchmarks should measure latency, throughput, failures, prompt size, and output size.'
|
|
19
19
|
},
|
|
20
20
|
{
|
|
21
|
-
id: '
|
|
22
|
-
prompt: '
|
|
21
|
+
id: 'medium-generation',
|
|
22
|
+
prompt: 'Write a concise four-bullet checklist for evaluating whether a local AI model is fast enough for an interactive coding assistant.'
|
|
23
|
+
}
|
|
24
|
+
],
|
|
25
|
+
interactive: [
|
|
26
|
+
{
|
|
27
|
+
id: 'chat-short-1',
|
|
28
|
+
prompt: 'Reply in one sentence: what is time to first token?'
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
id: 'chat-short-2',
|
|
32
|
+
prompt: 'Give one practical reason to benchmark with multiple prompt lengths.'
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
id: 'chat-short-3',
|
|
36
|
+
prompt: 'In under 20 words, define throughput for text generation.'
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
throughput: [
|
|
40
|
+
{
|
|
41
|
+
id: 'long-explain',
|
|
42
|
+
prompt: 'Write six concise bullets explaining the tradeoff between latency and throughput in local LLM inference.'
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
id: 'long-transform',
|
|
46
|
+
prompt: 'Rewrite this note as a polished release note with a title and five bullets: fm-bench now measures TTFT, end-to-end latency, output tokens per second, failures, and repeatability.'
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
id: 'long-plan',
|
|
50
|
+
prompt: 'Create a compact test plan for benchmarking a local foundation model across short, medium, and long prompts.'
|
|
23
51
|
}
|
|
24
52
|
],
|
|
25
53
|
stress: [
|
|
26
54
|
{
|
|
27
|
-
id: '
|
|
28
|
-
prompt: '
|
|
55
|
+
id: 'interactive-short',
|
|
56
|
+
prompt: 'Answer in one sentence: why should an on-device model benchmark report p95 latency?'
|
|
29
57
|
},
|
|
30
58
|
{
|
|
31
59
|
id: 'explain-latency',
|
package/src/report.js
CHANGED
|
@@ -13,16 +13,25 @@ export function toCsv(rows) {
|
|
|
13
13
|
export function flattenResults(results) {
|
|
14
14
|
return results.map((result) => ({
|
|
15
15
|
model: result.model,
|
|
16
|
+
concurrency: result.concurrency ?? '',
|
|
16
17
|
prompt_id: result.promptId,
|
|
17
18
|
run: result.run,
|
|
18
19
|
ok: result.ok,
|
|
19
20
|
duration_ms: round(result.durationMs),
|
|
21
|
+
ttft_ms: round(result.firstTokenMs),
|
|
22
|
+
generation_ms: round(result.generationMs),
|
|
23
|
+
tpot_ms: round(result.tpotMs),
|
|
20
24
|
prompt_tokens: result.promptTokens ?? '',
|
|
21
25
|
output_tokens: result.outputTokens ?? '',
|
|
22
26
|
chars: result.chars,
|
|
23
27
|
words: result.words,
|
|
24
28
|
tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
|
|
29
|
+
decode_tokens_per_second: result.decodeTokensPerSecond == null ? '' : round(result.decodeTokensPerSecond),
|
|
25
30
|
chars_per_second: round(result.charsPerSecond),
|
|
31
|
+
streamed: result.streamed,
|
|
32
|
+
stdout_chunks: result.stdoutChunks,
|
|
33
|
+
output_hash: result.outputHash || '',
|
|
34
|
+
good: result.good == null ? '' : result.good,
|
|
26
35
|
error: result.error || ''
|
|
27
36
|
}));
|
|
28
37
|
}
|
package/src/stats.js
CHANGED
|
@@ -6,19 +6,39 @@ export function summarizeNumbers(values) {
|
|
|
6
6
|
min: null,
|
|
7
7
|
max: null,
|
|
8
8
|
avg: null,
|
|
9
|
+
sum: 0,
|
|
10
|
+
stddev: null,
|
|
11
|
+
cv: null,
|
|
12
|
+
ci95Low: null,
|
|
13
|
+
ci95High: null,
|
|
9
14
|
p50: null,
|
|
10
|
-
|
|
15
|
+
p90: null,
|
|
16
|
+
p95: null,
|
|
17
|
+
p99: null
|
|
11
18
|
};
|
|
12
19
|
}
|
|
13
20
|
|
|
14
21
|
const total = clean.reduce((sum, value) => sum + value, 0);
|
|
22
|
+
const avg = total / clean.length;
|
|
23
|
+
const variance = clean.length > 1
|
|
24
|
+
? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
|
|
25
|
+
: 0;
|
|
26
|
+
const stddev = Math.sqrt(variance);
|
|
27
|
+
const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
|
|
15
28
|
return {
|
|
16
29
|
count: clean.length,
|
|
17
30
|
min: clean[0],
|
|
18
31
|
max: clean[clean.length - 1],
|
|
19
|
-
avg
|
|
32
|
+
avg,
|
|
33
|
+
sum: total,
|
|
34
|
+
stddev,
|
|
35
|
+
cv: avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
36
|
+
ci95Low: avg - margin,
|
|
37
|
+
ci95High: avg + margin,
|
|
20
38
|
p50: percentile(clean, 50),
|
|
21
|
-
|
|
39
|
+
p90: percentile(clean, 90),
|
|
40
|
+
p95: percentile(clean, 95),
|
|
41
|
+
p99: percentile(clean, 99)
|
|
22
42
|
};
|
|
23
43
|
}
|
|
24
44
|
|
|
@@ -34,52 +54,152 @@ export function percentile(sortedValues, percentileValue) {
|
|
|
34
54
|
return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
|
|
35
55
|
}
|
|
36
56
|
|
|
37
|
-
export function summarizeByModel(results, modelStatuses = []) {
|
|
57
|
+
export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
38
58
|
const byModel = new Map();
|
|
59
|
+
const concurrencies = options.concurrencies?.length ? options.concurrencies : [undefined];
|
|
39
60
|
|
|
40
|
-
for (const
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
61
|
+
for (const concurrency of concurrencies) {
|
|
62
|
+
for (const status of modelStatuses) {
|
|
63
|
+
const key = summaryKey(status.name, concurrency);
|
|
64
|
+
byModel.set(key, {
|
|
65
|
+
model: status.name,
|
|
66
|
+
concurrency,
|
|
67
|
+
description: status.description,
|
|
68
|
+
available: status.available,
|
|
69
|
+
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
70
|
+
results: []
|
|
71
|
+
});
|
|
72
|
+
}
|
|
48
73
|
}
|
|
49
74
|
|
|
50
75
|
for (const result of results) {
|
|
51
|
-
|
|
52
|
-
|
|
76
|
+
const key = summaryKey(result.model, result.concurrency);
|
|
77
|
+
if (!byModel.has(key)) {
|
|
78
|
+
byModel.set(key, {
|
|
53
79
|
model: result.model,
|
|
80
|
+
concurrency: result.concurrency,
|
|
54
81
|
description: '',
|
|
55
82
|
available: true,
|
|
56
83
|
skippedReason: '',
|
|
57
84
|
results: []
|
|
58
85
|
});
|
|
59
86
|
}
|
|
60
|
-
byModel.get(
|
|
87
|
+
byModel.get(key).results.push(result);
|
|
61
88
|
}
|
|
62
89
|
|
|
63
90
|
return [...byModel.values()].map((entry) => {
|
|
64
91
|
const successes = entry.results.filter((result) => result.ok);
|
|
65
92
|
const failures = entry.results.filter((result) => !result.ok);
|
|
93
|
+
const goodResults = successes.filter((result) => result.good === true);
|
|
94
|
+
const goodMeasured = successes.filter((result) => result.good != null);
|
|
66
95
|
const latency = summarizeNumbers(successes.map((result) => result.durationMs));
|
|
96
|
+
const ttft = summarizeNumbers(successes.map((result) => result.firstTokenMs).filter((value) => value != null));
|
|
97
|
+
const generation = summarizeNumbers(successes.map((result) => result.generationMs).filter((value) => value != null));
|
|
98
|
+
const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
|
|
99
|
+
const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
|
|
67
100
|
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
68
101
|
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
69
102
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
103
|
+
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
104
|
+
const windowMs = modelWindowMs(successes);
|
|
105
|
+
const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
|
|
106
|
+
const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
|
|
107
|
+
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
70
108
|
|
|
71
109
|
return {
|
|
72
110
|
model: entry.model,
|
|
111
|
+
concurrency: entry.concurrency,
|
|
73
112
|
description: entry.description,
|
|
74
113
|
available: entry.available,
|
|
75
114
|
skippedReason: entry.skippedReason,
|
|
76
115
|
attempted: entry.results.length,
|
|
77
116
|
successes: successes.length,
|
|
78
117
|
failures: failures.length,
|
|
118
|
+
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
119
|
+
goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
|
|
120
|
+
rps,
|
|
121
|
+
goodputRps,
|
|
122
|
+
outputTokenThroughput,
|
|
123
|
+
repeatability: summarizeRepeatability(successes),
|
|
79
124
|
latency,
|
|
125
|
+
ttft,
|
|
126
|
+
generation,
|
|
127
|
+
tpot,
|
|
128
|
+
promptTokens,
|
|
80
129
|
outputTokens,
|
|
81
130
|
charsPerSecond,
|
|
82
|
-
tokensPerSecond
|
|
131
|
+
tokensPerSecond,
|
|
132
|
+
decodeTokensPerSecond
|
|
83
133
|
};
|
|
84
134
|
});
|
|
85
135
|
}
|
|
136
|
+
|
|
137
|
+
function summaryKey(model, concurrency) {
|
|
138
|
+
return `${model}::${concurrency ?? 'default'}`;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function tCritical95(n) {
|
|
142
|
+
const df = Math.max(1, n - 1);
|
|
143
|
+
const table = {
|
|
144
|
+
1: 12.706,
|
|
145
|
+
2: 4.303,
|
|
146
|
+
3: 3.182,
|
|
147
|
+
4: 2.776,
|
|
148
|
+
5: 2.571,
|
|
149
|
+
6: 2.447,
|
|
150
|
+
7: 2.365,
|
|
151
|
+
8: 2.306,
|
|
152
|
+
9: 2.262,
|
|
153
|
+
10: 2.228,
|
|
154
|
+
11: 2.201,
|
|
155
|
+
12: 2.179,
|
|
156
|
+
13: 2.16,
|
|
157
|
+
14: 2.145,
|
|
158
|
+
15: 2.131,
|
|
159
|
+
16: 2.12,
|
|
160
|
+
17: 2.11,
|
|
161
|
+
18: 2.101,
|
|
162
|
+
19: 2.093,
|
|
163
|
+
20: 2.086,
|
|
164
|
+
21: 2.08,
|
|
165
|
+
22: 2.074,
|
|
166
|
+
23: 2.069,
|
|
167
|
+
24: 2.064,
|
|
168
|
+
25: 2.06,
|
|
169
|
+
26: 2.056,
|
|
170
|
+
27: 2.052,
|
|
171
|
+
28: 2.048,
|
|
172
|
+
29: 2.045,
|
|
173
|
+
30: 2.042
|
|
174
|
+
};
|
|
175
|
+
if (df <= 30) return table[df];
|
|
176
|
+
if (df <= 60) return 2;
|
|
177
|
+
return 1.96;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
function modelWindowMs(results) {
|
|
181
|
+
const starts = results.map((result) => result.startOffsetMs).filter((value) => Number.isFinite(value));
|
|
182
|
+
const ends = results.map((result) => result.endOffsetMs).filter((value) => Number.isFinite(value));
|
|
183
|
+
if (starts.length === 0 || ends.length === 0) return null;
|
|
184
|
+
return Math.max(...ends) - Math.min(...starts);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
function summarizeRepeatability(results) {
|
|
188
|
+
const byPrompt = new Map();
|
|
189
|
+
for (const result of results) {
|
|
190
|
+
if (!result.outputHash) continue;
|
|
191
|
+
if (!byPrompt.has(result.promptId)) byPrompt.set(result.promptId, []);
|
|
192
|
+
byPrompt.get(result.promptId).push(result.outputHash);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
const scores = [];
|
|
196
|
+
for (const hashes of byPrompt.values()) {
|
|
197
|
+
if (hashes.length < 2) continue;
|
|
198
|
+
const counts = new Map();
|
|
199
|
+
for (const hash of hashes) counts.set(hash, (counts.get(hash) || 0) + 1);
|
|
200
|
+
scores.push(Math.max(...counts.values()) / hashes.length);
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
if (scores.length === 0) return null;
|
|
204
|
+
return scores.reduce((sum, score) => sum + score, 0) / scores.length;
|
|
205
|
+
}
|
package/src/table.js
CHANGED
|
@@ -1,19 +1,61 @@
|
|
|
1
|
-
export function renderTable(headers, rows) {
|
|
2
|
-
const
|
|
1
|
+
export function renderTable(headers, rows, options = {}) {
|
|
2
|
+
const ascii = Boolean(options.ascii);
|
|
3
|
+
const maxCellWidth = options.maxCellWidth || 60;
|
|
4
|
+
const stringRows = rows.map((row) => row.map((cell) => truncate(formatCell(cell), maxCellWidth)));
|
|
3
5
|
const widths = headers.map((header, index) => {
|
|
4
|
-
const values = [header, ...stringRows.map((row) => row[index] ?? '')];
|
|
6
|
+
const values = [truncate(header, maxCellWidth), ...stringRows.map((row) => row[index] ?? '')];
|
|
5
7
|
return Math.max(...values.map(visibleLength));
|
|
6
8
|
});
|
|
9
|
+
const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
|
|
7
10
|
|
|
8
|
-
const
|
|
9
|
-
const
|
|
10
|
-
const
|
|
11
|
+
const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
|
|
12
|
+
const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
|
|
13
|
+
const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
|
|
14
|
+
const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true);
|
|
15
|
+
const bodyLines = stringRows.map((row) => rowLine(row, widths, style));
|
|
11
16
|
|
|
12
|
-
return [
|
|
17
|
+
return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function renderBenchmarkReport(payload, options = {}) {
|
|
21
|
+
const width = terminalWidth(options);
|
|
22
|
+
const mode = options.compact || width < 88
|
|
23
|
+
? 'compact'
|
|
24
|
+
: width < 140
|
|
25
|
+
? 'medium'
|
|
26
|
+
: 'wide';
|
|
27
|
+
const lines = [];
|
|
28
|
+
const elapsedMs = Date.parse(payload.finishedAt) - Date.parse(payload.startedAt);
|
|
29
|
+
const skipped = payload.summary.filter((item) => !item.available).length;
|
|
30
|
+
const measured = payload.summary.reduce((sum, item) => sum + item.successes, 0);
|
|
31
|
+
const failed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
|
|
32
|
+
const concurrencies = payload.options.sweepConcurrency?.length
|
|
33
|
+
? payload.options.sweepConcurrency.join(',')
|
|
34
|
+
: String(payload.options.concurrency);
|
|
35
|
+
const slo = formatSlo(payload.options.slo);
|
|
36
|
+
|
|
37
|
+
const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
|
|
38
|
+
const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
|
|
39
|
+
lines.push(truncate(title, width));
|
|
40
|
+
lines.push(truncate(meta, width));
|
|
41
|
+
lines.push('');
|
|
42
|
+
|
|
43
|
+
if (mode === 'compact') {
|
|
44
|
+
lines.push(renderCompactSummary(payload.summary, { ...options, width }));
|
|
45
|
+
} else {
|
|
46
|
+
lines.push(renderSummaryTable(payload.summary, { ...options, mode, width }));
|
|
47
|
+
lines.push('');
|
|
48
|
+
lines.push(renderDetailTable(payload.summary, { ...options, mode, width }));
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
lines.push('');
|
|
52
|
+
lines.push(compactLegend(width));
|
|
53
|
+
|
|
54
|
+
return lines.join('\n');
|
|
13
55
|
}
|
|
14
56
|
|
|
15
57
|
export function formatMs(value) {
|
|
16
|
-
if (value == null) return '-';
|
|
58
|
+
if (value == null || !Number.isFinite(value)) return '-';
|
|
17
59
|
if (value >= 1000) return `${(value / 1000).toFixed(2)}s`;
|
|
18
60
|
return `${Math.round(value)}ms`;
|
|
19
61
|
}
|
|
@@ -24,48 +66,215 @@ export function formatNumber(value, digits = 1) {
|
|
|
24
66
|
return value.toFixed(digits);
|
|
25
67
|
}
|
|
26
68
|
|
|
27
|
-
export function
|
|
69
|
+
export function formatPercent(value, digits = 0) {
|
|
70
|
+
if (value == null || !Number.isFinite(value)) return '-';
|
|
71
|
+
return `${(value * 100).toFixed(digits)}%`;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export function renderSummaryTable(summary, options = {}) {
|
|
75
|
+
const mode = options.mode || 'wide';
|
|
76
|
+
const hasGoodput = summary.some((item) => item.goodputRate != null);
|
|
28
77
|
const rows = summary.map((item) => {
|
|
29
78
|
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
30
|
-
|
|
79
|
+
const base = [
|
|
80
|
+
formatConcurrency(item.concurrency),
|
|
31
81
|
item.model,
|
|
32
82
|
status,
|
|
33
|
-
item.attempted
|
|
34
|
-
item.
|
|
35
|
-
item.
|
|
83
|
+
item.attempted ? `${item.successes}/${item.attempted}` : '-',
|
|
84
|
+
formatPercent(item.successRate),
|
|
85
|
+
formatPercent(item.goodputRate),
|
|
86
|
+
formatMs(item.ttft.p50),
|
|
87
|
+
formatMs(item.ttft.p95),
|
|
36
88
|
formatMs(item.latency.p50),
|
|
37
89
|
formatMs(item.latency.p95),
|
|
38
|
-
formatMs(item.latency.avg),
|
|
39
90
|
formatNumber(item.tokensPerSecond.avg),
|
|
40
|
-
formatNumber(item.
|
|
41
|
-
formatNumber(item.
|
|
91
|
+
formatNumber(item.outputTokenThroughput),
|
|
92
|
+
formatNumber(item.rps),
|
|
93
|
+
formatPercent(item.latency.cv),
|
|
42
94
|
item.available ? '' : compactReason(item.skippedReason)
|
|
43
95
|
];
|
|
96
|
+
|
|
97
|
+
if (mode === 'medium') {
|
|
98
|
+
const medium = [
|
|
99
|
+
base[0],
|
|
100
|
+
base[1],
|
|
101
|
+
base[2],
|
|
102
|
+
base[3]
|
|
103
|
+
];
|
|
104
|
+
if (hasGoodput) medium.push(base[5]);
|
|
105
|
+
medium.push(base[6], base[8], base[9], base[10], base[11], base[13], base[14]);
|
|
106
|
+
return medium;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const wide = [base[0], base[1], base[2], base[3], base[4]];
|
|
110
|
+
if (hasGoodput) wide.push(base[5]);
|
|
111
|
+
wide.push(
|
|
112
|
+
base[6],
|
|
113
|
+
base[7],
|
|
114
|
+
base[8],
|
|
115
|
+
base[9],
|
|
116
|
+
formatMs(item.tpot.p50),
|
|
117
|
+
formatMs(item.tpot.p95),
|
|
118
|
+
base[10],
|
|
119
|
+
base[11],
|
|
120
|
+
base[12],
|
|
121
|
+
base[13],
|
|
122
|
+
base[14]
|
|
123
|
+
);
|
|
124
|
+
return wide;
|
|
44
125
|
});
|
|
45
126
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
'
|
|
50
|
-
'
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
127
|
+
const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
|
|
128
|
+
const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
|
|
129
|
+
const headers = mode === 'medium'
|
|
130
|
+
? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
|
|
131
|
+
: (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
|
|
132
|
+
|
|
133
|
+
return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
export function renderDetailTable(summary, options = {}) {
|
|
137
|
+
const mode = options.mode || 'wide';
|
|
138
|
+
const rows = summary.map((item) => {
|
|
139
|
+
const base = [
|
|
140
|
+
formatConcurrency(item.concurrency),
|
|
141
|
+
item.model,
|
|
142
|
+
formatNumber(item.promptTokens.avg, 0),
|
|
143
|
+
formatNumber(item.outputTokens.avg, 0),
|
|
144
|
+
formatNumber(item.decodeTokensPerSecond.avg),
|
|
145
|
+
formatMs(item.latency.p99),
|
|
146
|
+
formatRangeMs(item.latency.ci95Low, item.latency.ci95High),
|
|
147
|
+
formatPercent(item.repeatability),
|
|
148
|
+
item.description || '-'
|
|
149
|
+
];
|
|
150
|
+
|
|
151
|
+
if (mode === 'medium') {
|
|
152
|
+
return base.slice(0, 8);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
return base;
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
const headers = mode === 'medium'
|
|
159
|
+
? ['c', 'model', 'in avg', 'out avg', 'decode t/s', 'e2e p99', 'e2e 95% ci', 'repeat']
|
|
160
|
+
: [
|
|
161
|
+
'c',
|
|
162
|
+
'model',
|
|
163
|
+
'in tok avg',
|
|
164
|
+
'out tok avg',
|
|
165
|
+
'decode tok/s',
|
|
166
|
+
'e2e p99',
|
|
167
|
+
'e2e 95% ci',
|
|
168
|
+
'repeat',
|
|
169
|
+
'description'
|
|
170
|
+
];
|
|
171
|
+
|
|
172
|
+
return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52 });
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
export function renderCompactSummary(summary, options = {}) {
|
|
176
|
+
const width = options.width || 80;
|
|
177
|
+
const separator = options.ascii ? '-' : '─';
|
|
178
|
+
const lines = [];
|
|
179
|
+
|
|
180
|
+
for (const item of summary) {
|
|
181
|
+
const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
|
|
182
|
+
const title = `${item.model} c${formatConcurrency(item.concurrency)} ${status} ${item.attempted ? `${item.successes}/${item.attempted}` : '-'}`;
|
|
183
|
+
lines.push(truncate(title, width));
|
|
184
|
+
|
|
185
|
+
if (item.available) {
|
|
186
|
+
lines.push(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width));
|
|
187
|
+
const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
|
|
188
|
+
lines.push(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width));
|
|
189
|
+
lines.push(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width));
|
|
190
|
+
} else {
|
|
191
|
+
lines.push(truncate(` ${compactReason(item.skippedReason)}`, width));
|
|
192
|
+
}
|
|
193
|
+
lines.push(separator.repeat(Math.min(width, 72)));
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
if (lines.at(-1)?.startsWith(separator)) lines.pop();
|
|
197
|
+
return lines.join('\n');
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
export function renderModelsTable(models, options = {}) {
|
|
201
|
+
const width = terminalWidth(options);
|
|
202
|
+
const compact = options.compact || width < 88;
|
|
203
|
+
if (compact) {
|
|
204
|
+
return models.map((model) => `${model.name} ${model.available ? 'yes' : 'no'} ${compactReason(model.reason || model.description || '-')}`).join('\n');
|
|
205
|
+
}
|
|
206
|
+
|
|
63
207
|
return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
|
|
64
208
|
model.name,
|
|
65
209
|
model.available ? 'yes' : 'no',
|
|
66
210
|
model.description || '-',
|
|
67
211
|
compactReason(model.quota || model.reason || '-')
|
|
68
|
-
]));
|
|
212
|
+
]), options);
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function formatRangeMs(low, high) {
|
|
216
|
+
if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
|
|
217
|
+
return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
function formatConcurrency(value) {
|
|
221
|
+
return value == null ? '1' : String(value);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function terminalWidth(options = {}) {
|
|
225
|
+
return options.width || process.stdout.columns || 120;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
function compactLegend(width) {
|
|
229
|
+
const text = 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence. CV = lower is steadier.';
|
|
230
|
+
return truncate(text, width);
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function formatSlo(slo = {}) {
|
|
234
|
+
const parts = [];
|
|
235
|
+
if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
|
|
236
|
+
if (slo.e2eMs) parts.push(`E2E<=${formatMs(slo.e2eMs)}`);
|
|
237
|
+
if (slo.tpotMs) parts.push(`TPOT<=${formatMs(slo.tpotMs)}`);
|
|
238
|
+
return parts.length ? `SLO ${parts.join(',')}` : '';
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
const ASCII_TABLE = {
|
|
242
|
+
topLeft: '+',
|
|
243
|
+
topJoin: '+',
|
|
244
|
+
topRight: '+',
|
|
245
|
+
midLeft: '+',
|
|
246
|
+
midJoin: '+',
|
|
247
|
+
midRight: '+',
|
|
248
|
+
bottomLeft: '+',
|
|
249
|
+
bottomJoin: '+',
|
|
250
|
+
bottomRight: '+',
|
|
251
|
+
horizontal: '-',
|
|
252
|
+
vertical: '|'
|
|
253
|
+
};
|
|
254
|
+
|
|
255
|
+
const UNICODE_TABLE = {
|
|
256
|
+
topLeft: '┌',
|
|
257
|
+
topJoin: '┬',
|
|
258
|
+
topRight: '┐',
|
|
259
|
+
midLeft: '├',
|
|
260
|
+
midJoin: '┼',
|
|
261
|
+
midRight: '┤',
|
|
262
|
+
bottomLeft: '└',
|
|
263
|
+
bottomJoin: '┴',
|
|
264
|
+
bottomRight: '┘',
|
|
265
|
+
horizontal: '─',
|
|
266
|
+
vertical: '│'
|
|
267
|
+
};
|
|
268
|
+
|
|
269
|
+
function rule(left, join, right, horizontal, widths) {
|
|
270
|
+
return `${left}${widths.map((width) => horizontal.repeat(width + 2)).join(join)}${right}`;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
function rowLine(row, widths, style, header = false) {
|
|
274
|
+
return `${style.vertical}${row.map((cell, index) => {
|
|
275
|
+
const value = header ? String(cell).toUpperCase() : formatCell(cell);
|
|
276
|
+
return ` ${pad(value, widths[index], !header && isNumericCell(value))} `;
|
|
277
|
+
}).join(style.vertical)}${style.vertical}`;
|
|
69
278
|
}
|
|
70
279
|
|
|
71
280
|
function formatCell(value) {
|
|
@@ -80,7 +289,7 @@ function pad(value, width, left = false) {
|
|
|
80
289
|
}
|
|
81
290
|
|
|
82
291
|
function isNumericCell(value) {
|
|
83
|
-
return /^-?$|^[\d,.]+(?:ms|s)?$/.test(value);
|
|
292
|
+
return /^-?$|^[\d,.]+(?:ms|s|%)?$/.test(value);
|
|
84
293
|
}
|
|
85
294
|
|
|
86
295
|
function visibleLength(value) {
|
|
@@ -92,3 +301,10 @@ function compactReason(value) {
|
|
|
92
301
|
if (clean.length <= 58) return clean;
|
|
93
302
|
return `${clean.slice(0, 55)}...`;
|
|
94
303
|
}
|
|
304
|
+
|
|
305
|
+
function truncate(value, width) {
|
|
306
|
+
const text = String(value ?? '');
|
|
307
|
+
if (visibleLength(text) <= width) return text;
|
|
308
|
+
if (width <= 1) return '…';
|
|
309
|
+
return `${text.slice(0, width - 1)}…`;
|
|
310
|
+
}
|