fm-bench 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -71,6 +71,8 @@ fm-bench doctor [options]
71
71
  fm-bench --models system,pcc --runs 3 --profile stress
72
72
  fm-bench --models system --runs 5 --profile interactive
73
73
  fm-bench --models system --runs 3 --profile throughput --warmup 1
74
+ fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
75
+ fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
74
76
  fm-bench --prompt "Reply with exactly: ok" --runs 5
75
77
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
76
78
  fm-bench --format csv --out reports/bench.csv
@@ -82,7 +84,9 @@ Useful flags:
82
84
  - `--runs <n>`: measured runs per prompt/model.
83
85
  - `--warmup <n>`: warmup runs per model before measurement.
84
86
  - `--concurrency <n>`: parallel `fm` processes.
87
+ - `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
85
88
  - `--timeout-ms <n>`: timeout per `fm` call.
89
+ - `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
86
90
  - `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
87
91
  - `--prompt <text>`: custom prompt, repeatable.
88
92
  - `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
@@ -91,6 +95,8 @@ Useful flags:
91
95
  - `--capture-output`: include raw model output in JSON reports.
92
96
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
93
97
  - `--ascii`: use plain ASCII table borders.
98
+ - `--compact`: force the narrow terminal layout.
99
+ - `--width <n>`: render as if the terminal has `n` columns.
94
100
  - `--out <file>`: save a report.
95
101
 
96
102
  ## Prompt Files
@@ -123,6 +129,8 @@ Plain text files are split on blank lines.
123
129
  - output tokens per second per request.
124
130
  - total output token throughput across the measured window.
125
131
  - requests per second across the measured window.
132
+ - goodput percentage and goodput RPS when SLO flags are set.
133
+ - coefficient of variation (CV) and confidence interval context for stability.
126
134
  - prompt and output token counts.
127
135
  - p50, p95, and p99 tail latency views.
128
136
  - repeatability across repeated runs of the same prompt.
@@ -133,6 +141,8 @@ Token counts come from `fm token-count --quiet`. If `fm` cannot count a response
133
141
 
134
142
  Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
135
143
 
144
+ Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
145
+
136
146
  See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
137
147
 
138
148
  ## Requirements
@@ -9,8 +9,9 @@ The metric set follows common LLM inference benchmark practice:
9
9
  - Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
10
10
  - NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
11
11
  - NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
12
- - vLLM benchmark tooling reports end-to-end latency and configurable percentiles, and can save JSON results: <https://docs.vllm.ai/en/latest/benchmarking/cli/>
12
+ - vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
13
13
  - MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
14
+ - MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
14
15
 
15
16
  ## Metrics
16
17
 
@@ -22,10 +23,21 @@ The metric set follows common LLM inference benchmark practice:
22
23
  - `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
23
24
  - `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
24
25
  - `RPS`: successful requests divided by that model's measured wall-clock window.
26
+ - `goodput`: successful requests that also satisfy all provided SLO thresholds.
25
27
  - `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
28
+ - `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
29
+ - `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
30
+
31
+ ## Operating Points
32
+
33
+ Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
34
+
35
+ `fm-bench` does not interpolate between operating points. It reports only what was actually measured.
26
36
 
27
37
  ## Caveats
28
38
 
29
39
  `fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
30
40
 
31
41
  Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
42
+
43
+ For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.2.0",
3
+ "version": "0.3.0",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/bench.js CHANGED
@@ -44,15 +44,69 @@ export async function runBenchmark(options = {}) {
44
44
  const runnableModels = modelStatuses.filter((model) => model.available);
45
45
  const environment = await collectEnvironment(inspection.fmBin);
46
46
  const promptTokenCounts = new Map();
47
+ const concurrencies = normalizeConcurrencySweep(options);
47
48
 
48
49
  for (const prompt of prompts) {
49
50
  const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
50
51
  promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
51
52
  }
52
53
 
54
+ const results = [];
55
+ const scenarios = [];
56
+ for (const concurrency of concurrencies) {
57
+ const scenario = await runScenario({
58
+ fmBin: inspection.fmBin,
59
+ prompts,
60
+ runnableModels,
61
+ modelStatuses,
62
+ promptTokenCounts,
63
+ options,
64
+ concurrency
65
+ });
66
+ scenarios.push(scenario);
67
+ results.push(...scenario.results);
68
+ }
69
+
70
+ results.sort((a, b) => a.model.localeCompare(b.model)
71
+ || (a.concurrency ?? 0) - (b.concurrency ?? 0)
72
+ || a.promptId.localeCompare(b.promptId)
73
+ || a.run - b.run);
74
+
75
+ const summary = summarizeByModel(results, modelStatuses, { concurrencies });
76
+ return {
77
+ tool: 'fm-bench',
78
+ version: options.version,
79
+ startedAt,
80
+ finishedAt: new Date().toISOString(),
81
+ options: publicOptions(options),
82
+ environment,
83
+ prompts: prompts.map((prompt) => ({
84
+ id: prompt.id,
85
+ prompt: prompt.prompt,
86
+ promptTokens: promptTokenCounts.get(prompt.id)
87
+ })),
88
+ models: modelStatuses,
89
+ scenarios,
90
+ summary,
91
+ results
92
+ };
93
+ }
94
+
95
+ async function runScenario(context) {
96
+ const {
97
+ fmBin,
98
+ prompts,
99
+ runnableModels,
100
+ modelStatuses,
101
+ promptTokenCounts,
102
+ options,
103
+ concurrency
104
+ } = context;
105
+ const startedAt = new Date().toISOString();
106
+
53
107
  for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
54
108
  for (const model of runnableModels) {
55
- await respond(inspection.fmBin, model.name, prompts[0].prompt, {
109
+ await respond(fmBin, model.name, prompts[0].prompt, {
56
110
  ...options,
57
111
  stream: false
58
112
  });
@@ -64,14 +118,14 @@ export async function runBenchmark(options = {}) {
64
118
  for (const model of runnableModels) {
65
119
  for (const prompt of prompts) {
66
120
  for (let run = 1; run <= options.runs; run += 1) {
67
- jobs.push({ model, prompt, run });
121
+ jobs.push({ model, prompt, run, concurrency });
68
122
  }
69
123
  }
70
124
  }
71
125
 
72
126
  const results = [];
73
- await runLimited(jobs, Math.max(1, options.concurrency), async (job) => {
74
- const result = await runSingleBenchmark(inspection.fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
127
+ await runLimited(jobs, concurrency, async (job) => {
128
+ const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
75
129
  results.push(result);
76
130
  if (!result.ok && options.failFast) {
77
131
  const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
@@ -84,21 +138,11 @@ export async function runBenchmark(options = {}) {
84
138
  || a.promptId.localeCompare(b.promptId)
85
139
  || a.run - b.run);
86
140
 
87
- const summary = summarizeByModel(results, modelStatuses);
88
141
  return {
89
- tool: 'fm-bench',
90
- version: options.version,
142
+ concurrency,
91
143
  startedAt,
92
144
  finishedAt: new Date().toISOString(),
93
- options: publicOptions(options),
94
- environment,
95
- prompts: prompts.map((prompt) => ({
96
- id: prompt.id,
97
- prompt: prompt.prompt,
98
- promptTokens: promptTokenCounts.get(prompt.id)
99
- })),
100
- models: modelStatuses,
101
- summary,
145
+ summary: summarizeByModel(results, modelStatuses, { concurrencies: [concurrency] }),
102
146
  results
103
147
  };
104
148
  }
@@ -130,6 +174,7 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
130
174
 
131
175
  return {
132
176
  model: job.model.name,
177
+ concurrency: job.concurrency,
133
178
  promptId: job.prompt.id,
134
179
  run: job.run,
135
180
  ok: response.ok,
@@ -149,6 +194,11 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
149
194
  streamed: response.streamed,
150
195
  stdoutChunks: response.stdoutChunks,
151
196
  outputHash: response.ok ? hashOutput(response.output) : null,
197
+ good: response.ok ? evaluateSlo({
198
+ firstTokenMs,
199
+ durationMs: response.durationMs,
200
+ tpotMs
201
+ }, options) : false,
152
202
  output: options.captureOutput ? response.output : undefined,
153
203
  error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
154
204
  };
@@ -175,23 +225,51 @@ function normalizeModelSelection(models) {
175
225
  }
176
226
 
177
227
  function publicOptions(options) {
228
+ const concurrencies = normalizeConcurrencySweep(options);
178
229
  return {
179
230
  models: normalizeModelSelection(options.models),
180
231
  runs: options.runs,
181
232
  warmup: options.warmup,
182
233
  concurrency: options.concurrency,
234
+ sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
183
235
  timeoutMs: options.timeoutMs,
184
236
  profile: options.profile,
185
237
  promptCount: options.promptCount,
186
238
  greedy: options.greedy,
187
239
  stream: options.stream,
240
+ slo: {
241
+ ttftMs: options.sloTtftMs || null,
242
+ e2eMs: options.sloE2eMs || null,
243
+ tpotMs: options.sloTpotMs || null
244
+ },
188
245
  instructions: options.instructions ? '[set]' : ''
189
246
  };
190
247
  }
191
248
 
249
+ function normalizeConcurrencySweep(options) {
250
+ if (options.sweepConcurrency?.length) {
251
+ return [...new Set(options.sweepConcurrency)]
252
+ .filter((value) => Number.isInteger(value) && value > 0)
253
+ .sort((a, b) => a - b);
254
+ }
255
+ return [Math.max(1, options.concurrency || 1)];
256
+ }
257
+
192
258
  function hashOutput(output) {
193
259
  return crypto.createHash('sha256')
194
260
  .update(output.replace(/\s+/g, ' ').trim())
195
261
  .digest('hex')
196
262
  .slice(0, 16);
197
263
  }
264
+
265
+ function evaluateSlo(metrics, options) {
266
+ const thresholds = [
267
+ ['firstTokenMs', options.sloTtftMs],
268
+ ['durationMs', options.sloE2eMs],
269
+ ['tpotMs', options.sloTpotMs]
270
+ ].filter(([, threshold]) => Number.isFinite(threshold));
271
+
272
+ if (thresholds.length === 0) return null;
273
+
274
+ return thresholds.every(([field, threshold]) => metrics[field] != null && metrics[field] <= threshold);
275
+ }
package/src/cli.js CHANGED
@@ -31,7 +31,7 @@ export async function runCli(argv = process.argv.slice(2)) {
31
31
  if (parsed.format === 'json') {
32
32
  console.log(JSON.stringify(inspection.models, null, 2));
33
33
  } else {
34
- console.log(renderModelsTable(inspection.models, { ascii: parsed.ascii }));
34
+ console.log(renderModelsTable(inspection.models, renderOptions(parsed)));
35
35
  }
36
36
  return;
37
37
  }
@@ -46,7 +46,7 @@ export async function runCli(argv = process.argv.slice(2)) {
46
46
  } else if (parsed.format === 'csv') {
47
47
  console.log(toCsv(flattenResults(payload.results)));
48
48
  } else {
49
- console.log(renderBenchmarkReport(payload, { ascii: parsed.ascii }));
49
+ console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
50
50
  if (parsed.verbose) {
51
51
  console.log();
52
52
  console.log(toCsv(flattenResults(payload.results)));
@@ -70,16 +70,22 @@ export function parseArgs(argv) {
70
70
  runs: 1,
71
71
  warmup: 0,
72
72
  concurrency: 1,
73
+ sweepConcurrency: [],
73
74
  timeoutMs: 60_000,
74
75
  profile: 'standard',
75
76
  greedy: true,
76
77
  stream: true,
78
+ sloTtftMs: null,
79
+ sloE2eMs: null,
80
+ sloTpotMs: null,
77
81
  format: 'table',
78
82
  captureOutput: false,
79
83
  availableOnly: false,
80
84
  failFast: false,
81
85
  verbose: false,
82
- ascii: false
86
+ ascii: false,
87
+ compact: false,
88
+ width: null
83
89
  };
84
90
 
85
91
  const args = [...argv];
@@ -118,10 +124,25 @@ export function parseArgs(argv) {
118
124
  case '--concurrency':
119
125
  options.concurrency = parsePositiveInt(requireValue(arg, args), arg);
120
126
  break;
127
+ case '--sweep-concurrency':
128
+ options.sweepConcurrency = parsePositiveIntList(requireValue(arg, args), arg);
129
+ if (options.sweepConcurrency.length > 0) {
130
+ options.concurrency = options.sweepConcurrency[0];
131
+ }
132
+ break;
121
133
  case '--timeout':
122
134
  case '--timeout-ms':
123
135
  options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
124
136
  break;
137
+ case '--slo-ttft-ms':
138
+ options.sloTtftMs = parsePositiveInt(requireValue(arg, args), arg);
139
+ break;
140
+ case '--slo-e2e-ms':
141
+ options.sloE2eMs = parsePositiveInt(requireValue(arg, args), arg);
142
+ break;
143
+ case '--slo-tpot-ms':
144
+ options.sloTpotMs = parsePositiveInt(requireValue(arg, args), arg);
145
+ break;
125
146
  case '-p':
126
147
  case '--prompt':
127
148
  options.prompts.push(requireValue(arg, args));
@@ -172,6 +193,12 @@ export function parseArgs(argv) {
172
193
  case '--ascii':
173
194
  options.ascii = true;
174
195
  break;
196
+ case '--compact':
197
+ options.compact = true;
198
+ break;
199
+ case '--width':
200
+ options.width = parsePositiveInt(requireValue(arg, args), arg);
201
+ break;
175
202
  case '-o':
176
203
  case '--out':
177
204
  options.out = requireValue(arg, args);
@@ -251,6 +278,24 @@ function parseNonNegativeInt(value, option) {
251
278
  return parsed;
252
279
  }
253
280
 
281
+ function parsePositiveIntList(value, option) {
282
+ const parsed = String(value)
283
+ .split(',')
284
+ .map((item) => item.trim())
285
+ .filter(Boolean)
286
+ .map((item) => parsePositiveInt(item, option));
287
+ if (parsed.length === 0) throw new Error(`${option} requires at least one positive integer`);
288
+ return parsed;
289
+ }
290
+
291
+ function renderOptions(parsed) {
292
+ return {
293
+ ascii: parsed.ascii,
294
+ compact: parsed.compact,
295
+ width: parsed.width
296
+ };
297
+ }
298
+
254
299
  function helpText() {
255
300
  return `fm-bench ${packageJson.version}
256
301
 
@@ -266,7 +311,12 @@ Run options:
266
311
  -r, --runs <n> Runs per prompt/model (default: 1)
267
312
  --warmup <n> Warmup runs per model before measurement
268
313
  -c, --concurrency <n> Parallel fm processes (default: 1)
314
+ --sweep-concurrency <list>
315
+ Run separate operating points, e.g. 1,2,4
269
316
  --timeout-ms <n> Timeout per fm call in ms (default: 60000)
317
+ --slo-ttft-ms <n> Count request as good only if TTFT is <= n
318
+ --slo-e2e-ms <n> Count request as good only if E2E latency is <= n
319
+ --slo-tpot-ms <n> Count request as good only if TPOT is <= n
270
320
  -p, --prompt <text> Prompt to benchmark; repeatable
271
321
  --prompt-file <file> .json, .jsonl, or blank-line separated text prompts
272
322
  --profile <name> quick, standard, interactive, throughput, or stress
@@ -286,6 +336,8 @@ Output:
286
336
  --json Alias for --format json
287
337
  --csv Alias for --format csv
288
338
  --ascii Use plain ASCII tables instead of Unicode
339
+ --compact Force compact terminal layout
340
+ --width <n> Render for a specific terminal width
289
341
  -o, --out <file> Save JSON or CSV report based on file extension
290
342
  -v, --verbose Include per-run CSV after the summary table
291
343
 
package/src/report.js CHANGED
@@ -13,6 +13,7 @@ export function toCsv(rows) {
13
13
  export function flattenResults(results) {
14
14
  return results.map((result) => ({
15
15
  model: result.model,
16
+ concurrency: result.concurrency ?? '',
16
17
  prompt_id: result.promptId,
17
18
  run: result.run,
18
19
  ok: result.ok,
@@ -30,6 +31,7 @@ export function flattenResults(results) {
30
31
  streamed: result.streamed,
31
32
  stdout_chunks: result.stdoutChunks,
32
33
  output_hash: result.outputHash || '',
34
+ good: result.good == null ? '' : result.good,
33
35
  error: result.error || ''
34
36
  }));
35
37
  }
package/src/stats.js CHANGED
@@ -7,6 +7,10 @@ export function summarizeNumbers(values) {
7
7
  max: null,
8
8
  avg: null,
9
9
  sum: 0,
10
+ stddev: null,
11
+ cv: null,
12
+ ci95Low: null,
13
+ ci95High: null,
10
14
  p50: null,
11
15
  p90: null,
12
16
  p95: null,
@@ -15,12 +19,22 @@ export function summarizeNumbers(values) {
15
19
  }
16
20
 
17
21
  const total = clean.reduce((sum, value) => sum + value, 0);
22
+ const avg = total / clean.length;
23
+ const variance = clean.length > 1
24
+ ? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
25
+ : 0;
26
+ const stddev = Math.sqrt(variance);
27
+ const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
18
28
  return {
19
29
  count: clean.length,
20
30
  min: clean[0],
21
31
  max: clean[clean.length - 1],
22
- avg: total / clean.length,
32
+ avg,
23
33
  sum: total,
34
+ stddev,
35
+ cv: avg !== 0 ? stddev / Math.abs(avg) : null,
36
+ ci95Low: avg - margin,
37
+ ci95High: avg + margin,
24
38
  p50: percentile(clean, 50),
25
39
  p90: percentile(clean, 90),
26
40
  p95: percentile(clean, 95),
@@ -40,35 +54,44 @@ export function percentile(sortedValues, percentileValue) {
40
54
  return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
41
55
  }
42
56
 
43
- export function summarizeByModel(results, modelStatuses = []) {
57
+ export function summarizeByModel(results, modelStatuses = [], options = {}) {
44
58
  const byModel = new Map();
59
+ const concurrencies = options.concurrencies?.length ? options.concurrencies : [undefined];
45
60
 
46
- for (const status of modelStatuses) {
47
- byModel.set(status.name, {
48
- model: status.name,
49
- description: status.description,
50
- available: status.available,
51
- skippedReason: status.available ? '' : status.reason || 'Unavailable',
52
- results: []
53
- });
61
+ for (const concurrency of concurrencies) {
62
+ for (const status of modelStatuses) {
63
+ const key = summaryKey(status.name, concurrency);
64
+ byModel.set(key, {
65
+ model: status.name,
66
+ concurrency,
67
+ description: status.description,
68
+ available: status.available,
69
+ skippedReason: status.available ? '' : status.reason || 'Unavailable',
70
+ results: []
71
+ });
72
+ }
54
73
  }
55
74
 
56
75
  for (const result of results) {
57
- if (!byModel.has(result.model)) {
58
- byModel.set(result.model, {
76
+ const key = summaryKey(result.model, result.concurrency);
77
+ if (!byModel.has(key)) {
78
+ byModel.set(key, {
59
79
  model: result.model,
80
+ concurrency: result.concurrency,
60
81
  description: '',
61
82
  available: true,
62
83
  skippedReason: '',
63
84
  results: []
64
85
  });
65
86
  }
66
- byModel.get(result.model).results.push(result);
87
+ byModel.get(key).results.push(result);
67
88
  }
68
89
 
69
90
  return [...byModel.values()].map((entry) => {
70
91
  const successes = entry.results.filter((result) => result.ok);
71
92
  const failures = entry.results.filter((result) => !result.ok);
93
+ const goodResults = successes.filter((result) => result.good === true);
94
+ const goodMeasured = successes.filter((result) => result.good != null);
72
95
  const latency = summarizeNumbers(successes.map((result) => result.durationMs));
73
96
  const ttft = summarizeNumbers(successes.map((result) => result.firstTokenMs).filter((value) => value != null));
74
97
  const generation = summarizeNumbers(successes.map((result) => result.generationMs).filter((value) => value != null));
@@ -80,10 +103,12 @@ export function summarizeByModel(results, modelStatuses = []) {
80
103
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
81
104
  const windowMs = modelWindowMs(successes);
82
105
  const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
106
+ const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
83
107
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
84
108
 
85
109
  return {
86
110
  model: entry.model,
111
+ concurrency: entry.concurrency,
87
112
  description: entry.description,
88
113
  available: entry.available,
89
114
  skippedReason: entry.skippedReason,
@@ -91,7 +116,9 @@ export function summarizeByModel(results, modelStatuses = []) {
91
116
  successes: successes.length,
92
117
  failures: failures.length,
93
118
  successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
119
+ goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
94
120
  rps,
121
+ goodputRps,
95
122
  outputTokenThroughput,
96
123
  repeatability: summarizeRepeatability(successes),
97
124
  latency,
@@ -107,6 +134,49 @@ export function summarizeByModel(results, modelStatuses = []) {
107
134
  });
108
135
  }
109
136
 
137
+ function summaryKey(model, concurrency) {
138
+ return `${model}::${concurrency ?? 'default'}`;
139
+ }
140
+
141
+ function tCritical95(n) {
142
+ const df = Math.max(1, n - 1);
143
+ const table = {
144
+ 1: 12.706,
145
+ 2: 4.303,
146
+ 3: 3.182,
147
+ 4: 2.776,
148
+ 5: 2.571,
149
+ 6: 2.447,
150
+ 7: 2.365,
151
+ 8: 2.306,
152
+ 9: 2.262,
153
+ 10: 2.228,
154
+ 11: 2.201,
155
+ 12: 2.179,
156
+ 13: 2.16,
157
+ 14: 2.145,
158
+ 15: 2.131,
159
+ 16: 2.12,
160
+ 17: 2.11,
161
+ 18: 2.101,
162
+ 19: 2.093,
163
+ 20: 2.086,
164
+ 21: 2.08,
165
+ 22: 2.074,
166
+ 23: 2.069,
167
+ 24: 2.064,
168
+ 25: 2.06,
169
+ 26: 2.056,
170
+ 27: 2.052,
171
+ 28: 2.048,
172
+ 29: 2.045,
173
+ 30: 2.042
174
+ };
175
+ if (df <= 30) return table[df];
176
+ if (df <= 60) return 2;
177
+ return 1.96;
178
+ }
179
+
110
180
  function modelWindowMs(results) {
111
181
  const starts = results.map((result) => result.startOffsetMs).filter((value) => Number.isFinite(value));
112
182
  const ends = results.map((result) => result.endOffsetMs).filter((value) => Number.isFinite(value));
package/src/table.js CHANGED
@@ -1,8 +1,9 @@
1
1
  export function renderTable(headers, rows, options = {}) {
2
2
  const ascii = Boolean(options.ascii);
3
- const stringRows = rows.map((row) => row.map(formatCell));
3
+ const maxCellWidth = options.maxCellWidth || 60;
4
+ const stringRows = rows.map((row) => row.map((cell) => truncate(formatCell(cell), maxCellWidth)));
4
5
  const widths = headers.map((header, index) => {
5
- const values = [header, ...stringRows.map((row) => row[index] ?? '')];
6
+ const values = [truncate(header, maxCellWidth), ...stringRows.map((row) => row[index] ?? '')];
6
7
  return Math.max(...values.map(visibleLength));
7
8
  });
8
9
  const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
@@ -10,27 +11,45 @@ export function renderTable(headers, rows, options = {}) {
10
11
  const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
11
12
  const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
12
13
  const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
13
- const headerLine = rowLine(headers, widths, style, true);
14
+ const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true);
14
15
  const bodyLines = stringRows.map((row) => rowLine(row, widths, style));
15
16
 
16
17
  return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
17
18
  }
18
19
 
19
20
  export function renderBenchmarkReport(payload, options = {}) {
21
+ const width = terminalWidth(options);
22
+ const mode = options.compact || width < 88
23
+ ? 'compact'
24
+ : width < 140
25
+ ? 'medium'
26
+ : 'wide';
20
27
  const lines = [];
21
28
  const elapsedMs = Date.parse(payload.finishedAt) - Date.parse(payload.startedAt);
22
29
  const skipped = payload.summary.filter((item) => !item.available).length;
23
30
  const measured = payload.summary.reduce((sum, item) => sum + item.successes, 0);
24
31
  const failed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
32
+ const concurrencies = payload.options.sweepConcurrency?.length
33
+ ? payload.options.sweepConcurrency.join(',')
34
+ : String(payload.options.concurrency);
35
+ const slo = formatSlo(payload.options.slo);
25
36
 
26
- lines.push(`fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`);
27
- lines.push(`prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${payload.options.concurrency} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped models ${skipped} | elapsed ${formatMs(elapsedMs)}`);
37
+ const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
38
+ const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
39
+ lines.push(truncate(title, width));
40
+ lines.push(truncate(meta, width));
28
41
  lines.push('');
29
- lines.push(renderSummaryTable(payload.summary, options));
30
- lines.push('');
31
- lines.push(renderDetailTable(payload.summary, options));
42
+
43
+ if (mode === 'compact') {
44
+ lines.push(renderCompactSummary(payload.summary, { ...options, width }));
45
+ } else {
46
+ lines.push(renderSummaryTable(payload.summary, { ...options, mode, width }));
47
+ lines.push('');
48
+ lines.push(renderDetailTable(payload.summary, { ...options, mode, width }));
49
+ }
50
+
32
51
  lines.push('');
33
- lines.push('TTFT = time to first streamed output, E2E = full response latency, TPOT = decode time per output token.');
52
+ lines.push(compactLegend(width));
34
53
 
35
54
  return lines.join('\n');
36
55
  }
@@ -53,69 +72,138 @@ export function formatPercent(value, digits = 0) {
53
72
  }
54
73
 
55
74
  export function renderSummaryTable(summary, options = {}) {
75
+ const mode = options.mode || 'wide';
76
+ const hasGoodput = summary.some((item) => item.goodputRate != null);
56
77
  const rows = summary.map((item) => {
57
78
  const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
58
- return [
79
+ const base = [
80
+ formatConcurrency(item.concurrency),
59
81
  item.model,
60
82
  status,
61
- item.attempted || '-',
62
- item.successes || '-',
83
+ item.attempted ? `${item.successes}/${item.attempted}` : '-',
63
84
  formatPercent(item.successRate),
85
+ formatPercent(item.goodputRate),
64
86
  formatMs(item.ttft.p50),
65
87
  formatMs(item.ttft.p95),
66
88
  formatMs(item.latency.p50),
67
89
  formatMs(item.latency.p95),
68
- formatMs(item.tpot.p50),
69
90
  formatNumber(item.tokensPerSecond.avg),
91
+ formatNumber(item.outputTokenThroughput),
70
92
  formatNumber(item.rps),
93
+ formatPercent(item.latency.cv),
71
94
  item.available ? '' : compactReason(item.skippedReason)
72
95
  ];
96
+
97
+ if (mode === 'medium') {
98
+ const medium = [
99
+ base[0],
100
+ base[1],
101
+ base[2],
102
+ base[3]
103
+ ];
104
+ if (hasGoodput) medium.push(base[5]);
105
+ medium.push(base[6], base[8], base[9], base[10], base[11], base[13], base[14]);
106
+ return medium;
107
+ }
108
+
109
+ const wide = [base[0], base[1], base[2], base[3], base[4]];
110
+ if (hasGoodput) wide.push(base[5]);
111
+ wide.push(
112
+ base[6],
113
+ base[7],
114
+ base[8],
115
+ base[9],
116
+ formatMs(item.tpot.p50),
117
+ formatMs(item.tpot.p95),
118
+ base[10],
119
+ base[11],
120
+ base[12],
121
+ base[13],
122
+ base[14]
123
+ );
124
+ return wide;
73
125
  });
74
126
 
75
- return renderTable([
76
- 'model',
77
- 'status',
78
- 'runs',
79
- 'ok',
80
- 'success',
81
- 'ttft p50',
82
- 'ttft p95',
83
- 'e2e p50',
84
- 'e2e p95',
85
- 'tpot p50',
86
- 'tok/s',
87
- 'rps',
88
- 'note'
89
- ], rows, options);
127
+ const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
128
+ const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
129
+ const headers = mode === 'medium'
130
+ ? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
131
+ : (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
132
+
133
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
90
134
  }
91
135
 
92
136
  export function renderDetailTable(summary, options = {}) {
93
- const rows = summary.map((item) => [
94
- item.model,
95
- formatNumber(item.promptTokens.avg, 0),
96
- formatNumber(item.outputTokens.avg, 0),
97
- formatNumber(item.outputTokenThroughput),
98
- formatNumber(item.decodeTokensPerSecond.avg),
99
- formatMs(item.latency.p99),
100
- formatMs(item.tpot.p95),
101
- formatPercent(item.repeatability),
102
- item.description || '-'
103
- ]);
104
-
105
- return renderTable([
106
- 'model',
107
- 'in tok avg',
108
- 'out tok avg',
109
- 'total tok/s',
110
- 'decode tok/s',
111
- 'e2e p99',
112
- 'tpot p95',
113
- 'repeat',
114
- 'description'
115
- ], rows, options);
137
+ const mode = options.mode || 'wide';
138
+ const rows = summary.map((item) => {
139
+ const base = [
140
+ formatConcurrency(item.concurrency),
141
+ item.model,
142
+ formatNumber(item.promptTokens.avg, 0),
143
+ formatNumber(item.outputTokens.avg, 0),
144
+ formatNumber(item.decodeTokensPerSecond.avg),
145
+ formatMs(item.latency.p99),
146
+ formatRangeMs(item.latency.ci95Low, item.latency.ci95High),
147
+ formatPercent(item.repeatability),
148
+ item.description || '-'
149
+ ];
150
+
151
+ if (mode === 'medium') {
152
+ return base.slice(0, 8);
153
+ }
154
+
155
+ return base;
156
+ });
157
+
158
+ const headers = mode === 'medium'
159
+ ? ['c', 'model', 'in avg', 'out avg', 'decode t/s', 'e2e p99', 'e2e 95% ci', 'repeat']
160
+ : [
161
+ 'c',
162
+ 'model',
163
+ 'in tok avg',
164
+ 'out tok avg',
165
+ 'decode tok/s',
166
+ 'e2e p99',
167
+ 'e2e 95% ci',
168
+ 'repeat',
169
+ 'description'
170
+ ];
171
+
172
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52 });
173
+ }
174
+
175
+ export function renderCompactSummary(summary, options = {}) {
176
+ const width = options.width || 80;
177
+ const separator = options.ascii ? '-' : '─';
178
+ const lines = [];
179
+
180
+ for (const item of summary) {
181
+ const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
182
+ const title = `${item.model} c${formatConcurrency(item.concurrency)} ${status} ${item.attempted ? `${item.successes}/${item.attempted}` : '-'}`;
183
+ lines.push(truncate(title, width));
184
+
185
+ if (item.available) {
186
+ lines.push(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width));
187
+ const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
188
+ lines.push(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width));
189
+ lines.push(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width));
190
+ } else {
191
+ lines.push(truncate(` ${compactReason(item.skippedReason)}`, width));
192
+ }
193
+ lines.push(separator.repeat(Math.min(width, 72)));
194
+ }
195
+
196
+ if (lines.at(-1)?.startsWith(separator)) lines.pop();
197
+ return lines.join('\n');
116
198
  }
117
199
 
118
200
  export function renderModelsTable(models, options = {}) {
201
+ const width = terminalWidth(options);
202
+ const compact = options.compact || width < 88;
203
+ if (compact) {
204
+ return models.map((model) => `${model.name} ${model.available ? 'yes' : 'no'} ${compactReason(model.reason || model.description || '-')}`).join('\n');
205
+ }
206
+
119
207
  return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
120
208
  model.name,
121
209
  model.available ? 'yes' : 'no',
@@ -124,6 +212,32 @@ export function renderModelsTable(models, options = {}) {
124
212
  ]), options);
125
213
  }
126
214
 
215
+ function formatRangeMs(low, high) {
216
+ if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
217
+ return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
218
+ }
219
+
220
+ function formatConcurrency(value) {
221
+ return value == null ? '1' : String(value);
222
+ }
223
+
224
+ function terminalWidth(options = {}) {
225
+ return options.width || process.stdout.columns || 120;
226
+ }
227
+
228
+ function compactLegend(width) {
229
+ const text = 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence. CV = lower is steadier.';
230
+ return truncate(text, width);
231
+ }
232
+
233
+ function formatSlo(slo = {}) {
234
+ const parts = [];
235
+ if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
236
+ if (slo.e2eMs) parts.push(`E2E<=${formatMs(slo.e2eMs)}`);
237
+ if (slo.tpotMs) parts.push(`TPOT<=${formatMs(slo.tpotMs)}`);
238
+ return parts.length ? `SLO ${parts.join(',')}` : '';
239
+ }
240
+
127
241
  const ASCII_TABLE = {
128
242
  topLeft: '+',
129
243
  topJoin: '+',
@@ -187,3 +301,10 @@ function compactReason(value) {
187
301
  if (clean.length <= 58) return clean;
188
302
  return `${clean.slice(0, 55)}...`;
189
303
  }
304
+
305
+ function truncate(value, width) {
306
+ const text = String(value ?? '');
307
+ if (visibleLength(text) <= width) return text;
308
+ if (width <= 1) return '…';
309
+ return `${text.slice(0, width - 1)}…`;
310
+ }