fm-bench 0.2.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -71,6 +71,8 @@ fm-bench doctor [options]
71
71
  fm-bench --models system,pcc --runs 3 --profile stress
72
72
  fm-bench --models system --runs 5 --profile interactive
73
73
  fm-bench --models system --runs 3 --profile throughput --warmup 1
74
+ fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
75
+ fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
74
76
  fm-bench --prompt "Reply with exactly: ok" --runs 5
75
77
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
76
78
  fm-bench --format csv --out reports/bench.csv
@@ -82,7 +84,9 @@ Useful flags:
82
84
  - `--runs <n>`: measured runs per prompt/model.
83
85
  - `--warmup <n>`: warmup runs per model before measurement.
84
86
  - `--concurrency <n>`: parallel `fm` processes.
87
+ - `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
85
88
  - `--timeout-ms <n>`: timeout per `fm` call.
89
+ - `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
86
90
  - `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
87
91
  - `--prompt <text>`: custom prompt, repeatable.
88
92
  - `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
@@ -91,6 +95,9 @@ Useful flags:
91
95
  - `--capture-output`: include raw model output in JSON reports.
92
96
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
93
97
  - `--ascii`: use plain ASCII table borders.
98
+ - `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
99
+ - `--compact`: force the narrow terminal layout.
100
+ - `--width <n>`: render as if the terminal has `n` columns.
94
101
  - `--out <file>`: save a report.
95
102
 
96
103
  ## Prompt Files
@@ -123,6 +130,8 @@ Plain text files are split on blank lines.
123
130
  - output tokens per second per request.
124
131
  - total output token throughput across the measured window.
125
132
  - requests per second across the measured window.
133
+ - goodput percentage and goodput RPS when SLO flags are set.
134
+ - coefficient of variation (CV) and confidence interval context for stability.
126
135
  - prompt and output token counts.
127
136
  - p50, p95, and p99 tail latency views.
128
137
  - repeatability across repeated runs of the same prompt.
@@ -133,6 +142,20 @@ Token counts come from `fm token-count --quiet`. If `fm` cannot count a response
133
142
 
134
143
  Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
135
144
 
145
+ Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
146
+
147
+ ## Terminal Colors
148
+
149
+ Table output uses semantic ANSI color on interactive terminals:
150
+
151
+ - green: passing, steadier, or better than the current comparison set.
152
+ - yellow: marginal, partial, or near a budget.
153
+ - red: failing a budget, unstable, or slower/lower than peers.
154
+
155
+ Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
156
+
157
+ Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
158
+
136
159
  See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
137
160
 
138
161
  ## Requirements
@@ -9,8 +9,9 @@ The metric set follows common LLM inference benchmark practice:
9
9
  - Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
10
10
  - NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
11
11
  - NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
12
- - vLLM benchmark tooling reports end-to-end latency and configurable percentiles, and can save JSON results: <https://docs.vllm.ai/en/latest/benchmarking/cli/>
12
+ - vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
13
13
  - MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
14
+ - MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
14
15
 
15
16
  ## Metrics
16
17
 
@@ -22,10 +23,21 @@ The metric set follows common LLM inference benchmark practice:
22
23
  - `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
23
24
  - `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
24
25
  - `RPS`: successful requests divided by that model's measured wall-clock window.
26
+ - `goodput`: successful requests that also satisfy all provided SLO thresholds.
25
27
  - `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
28
+ - `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
29
+ - `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
30
+
31
+ ## Operating Points
32
+
33
+ Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
34
+
35
+ `fm-bench` does not interpolate between operating points. It reports only what was actually measured.
26
36
 
27
37
  ## Caveats
28
38
 
29
39
  `fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
30
40
 
31
41
  Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
42
+
43
+ For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.2.0",
3
+ "version": "0.3.1",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/bench.js CHANGED
@@ -44,15 +44,69 @@ export async function runBenchmark(options = {}) {
44
44
  const runnableModels = modelStatuses.filter((model) => model.available);
45
45
  const environment = await collectEnvironment(inspection.fmBin);
46
46
  const promptTokenCounts = new Map();
47
+ const concurrencies = normalizeConcurrencySweep(options);
47
48
 
48
49
  for (const prompt of prompts) {
49
50
  const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
50
51
  promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
51
52
  }
52
53
 
54
+ const results = [];
55
+ const scenarios = [];
56
+ for (const concurrency of concurrencies) {
57
+ const scenario = await runScenario({
58
+ fmBin: inspection.fmBin,
59
+ prompts,
60
+ runnableModels,
61
+ modelStatuses,
62
+ promptTokenCounts,
63
+ options,
64
+ concurrency
65
+ });
66
+ scenarios.push(scenario);
67
+ results.push(...scenario.results);
68
+ }
69
+
70
+ results.sort((a, b) => a.model.localeCompare(b.model)
71
+ || (a.concurrency ?? 0) - (b.concurrency ?? 0)
72
+ || a.promptId.localeCompare(b.promptId)
73
+ || a.run - b.run);
74
+
75
+ const summary = summarizeByModel(results, modelStatuses, { concurrencies });
76
+ return {
77
+ tool: 'fm-bench',
78
+ version: options.version,
79
+ startedAt,
80
+ finishedAt: new Date().toISOString(),
81
+ options: publicOptions(options),
82
+ environment,
83
+ prompts: prompts.map((prompt) => ({
84
+ id: prompt.id,
85
+ prompt: prompt.prompt,
86
+ promptTokens: promptTokenCounts.get(prompt.id)
87
+ })),
88
+ models: modelStatuses,
89
+ scenarios,
90
+ summary,
91
+ results
92
+ };
93
+ }
94
+
95
+ async function runScenario(context) {
96
+ const {
97
+ fmBin,
98
+ prompts,
99
+ runnableModels,
100
+ modelStatuses,
101
+ promptTokenCounts,
102
+ options,
103
+ concurrency
104
+ } = context;
105
+ const startedAt = new Date().toISOString();
106
+
53
107
  for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
54
108
  for (const model of runnableModels) {
55
- await respond(inspection.fmBin, model.name, prompts[0].prompt, {
109
+ await respond(fmBin, model.name, prompts[0].prompt, {
56
110
  ...options,
57
111
  stream: false
58
112
  });
@@ -64,14 +118,14 @@ export async function runBenchmark(options = {}) {
64
118
  for (const model of runnableModels) {
65
119
  for (const prompt of prompts) {
66
120
  for (let run = 1; run <= options.runs; run += 1) {
67
- jobs.push({ model, prompt, run });
121
+ jobs.push({ model, prompt, run, concurrency });
68
122
  }
69
123
  }
70
124
  }
71
125
 
72
126
  const results = [];
73
- await runLimited(jobs, Math.max(1, options.concurrency), async (job) => {
74
- const result = await runSingleBenchmark(inspection.fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
127
+ await runLimited(jobs, concurrency, async (job) => {
128
+ const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
75
129
  results.push(result);
76
130
  if (!result.ok && options.failFast) {
77
131
  const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
@@ -84,21 +138,11 @@ export async function runBenchmark(options = {}) {
84
138
  || a.promptId.localeCompare(b.promptId)
85
139
  || a.run - b.run);
86
140
 
87
- const summary = summarizeByModel(results, modelStatuses);
88
141
  return {
89
- tool: 'fm-bench',
90
- version: options.version,
142
+ concurrency,
91
143
  startedAt,
92
144
  finishedAt: new Date().toISOString(),
93
- options: publicOptions(options),
94
- environment,
95
- prompts: prompts.map((prompt) => ({
96
- id: prompt.id,
97
- prompt: prompt.prompt,
98
- promptTokens: promptTokenCounts.get(prompt.id)
99
- })),
100
- models: modelStatuses,
101
- summary,
145
+ summary: summarizeByModel(results, modelStatuses, { concurrencies: [concurrency] }),
102
146
  results
103
147
  };
104
148
  }
@@ -130,6 +174,7 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
130
174
 
131
175
  return {
132
176
  model: job.model.name,
177
+ concurrency: job.concurrency,
133
178
  promptId: job.prompt.id,
134
179
  run: job.run,
135
180
  ok: response.ok,
@@ -149,6 +194,11 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
149
194
  streamed: response.streamed,
150
195
  stdoutChunks: response.stdoutChunks,
151
196
  outputHash: response.ok ? hashOutput(response.output) : null,
197
+ good: response.ok ? evaluateSlo({
198
+ firstTokenMs,
199
+ durationMs: response.durationMs,
200
+ tpotMs
201
+ }, options) : false,
152
202
  output: options.captureOutput ? response.output : undefined,
153
203
  error: response.ok ? '' : response.stderr || `fm exited with code ${response.code ?? response.signal}`
154
204
  };
@@ -175,23 +225,51 @@ function normalizeModelSelection(models) {
175
225
  }
176
226
 
177
227
  function publicOptions(options) {
228
+ const concurrencies = normalizeConcurrencySweep(options);
178
229
  return {
179
230
  models: normalizeModelSelection(options.models),
180
231
  runs: options.runs,
181
232
  warmup: options.warmup,
182
233
  concurrency: options.concurrency,
234
+ sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
183
235
  timeoutMs: options.timeoutMs,
184
236
  profile: options.profile,
185
237
  promptCount: options.promptCount,
186
238
  greedy: options.greedy,
187
239
  stream: options.stream,
240
+ slo: {
241
+ ttftMs: options.sloTtftMs || null,
242
+ e2eMs: options.sloE2eMs || null,
243
+ tpotMs: options.sloTpotMs || null
244
+ },
188
245
  instructions: options.instructions ? '[set]' : ''
189
246
  };
190
247
  }
191
248
 
249
+ function normalizeConcurrencySweep(options) {
250
+ if (options.sweepConcurrency?.length) {
251
+ return [...new Set(options.sweepConcurrency)]
252
+ .filter((value) => Number.isInteger(value) && value > 0)
253
+ .sort((a, b) => a - b);
254
+ }
255
+ return [Math.max(1, options.concurrency || 1)];
256
+ }
257
+
192
258
  function hashOutput(output) {
193
259
  return crypto.createHash('sha256')
194
260
  .update(output.replace(/\s+/g, ' ').trim())
195
261
  .digest('hex')
196
262
  .slice(0, 16);
197
263
  }
264
+
265
+ function evaluateSlo(metrics, options) {
266
+ const thresholds = [
267
+ ['firstTokenMs', options.sloTtftMs],
268
+ ['durationMs', options.sloE2eMs],
269
+ ['tpotMs', options.sloTpotMs]
270
+ ].filter(([, threshold]) => Number.isFinite(threshold));
271
+
272
+ if (thresholds.length === 0) return null;
273
+
274
+ return thresholds.every(([field, threshold]) => metrics[field] != null && metrics[field] <= threshold);
275
+ }
package/src/cli.js CHANGED
@@ -31,7 +31,7 @@ export async function runCli(argv = process.argv.slice(2)) {
31
31
  if (parsed.format === 'json') {
32
32
  console.log(JSON.stringify(inspection.models, null, 2));
33
33
  } else {
34
- console.log(renderModelsTable(inspection.models, { ascii: parsed.ascii }));
34
+ console.log(renderModelsTable(inspection.models, renderOptions(parsed)));
35
35
  }
36
36
  return;
37
37
  }
@@ -46,7 +46,7 @@ export async function runCli(argv = process.argv.slice(2)) {
46
46
  } else if (parsed.format === 'csv') {
47
47
  console.log(toCsv(flattenResults(payload.results)));
48
48
  } else {
49
- console.log(renderBenchmarkReport(payload, { ascii: parsed.ascii }));
49
+ console.log(renderBenchmarkReport(payload, renderOptions(parsed)));
50
50
  if (parsed.verbose) {
51
51
  console.log();
52
52
  console.log(toCsv(flattenResults(payload.results)));
@@ -70,16 +70,23 @@ export function parseArgs(argv) {
70
70
  runs: 1,
71
71
  warmup: 0,
72
72
  concurrency: 1,
73
+ sweepConcurrency: [],
73
74
  timeoutMs: 60_000,
74
75
  profile: 'standard',
75
76
  greedy: true,
76
77
  stream: true,
78
+ sloTtftMs: null,
79
+ sloE2eMs: null,
80
+ sloTpotMs: null,
77
81
  format: 'table',
78
82
  captureOutput: false,
79
83
  availableOnly: false,
80
84
  failFast: false,
81
85
  verbose: false,
82
- ascii: false
86
+ ascii: false,
87
+ color: 'auto',
88
+ compact: false,
89
+ width: null
83
90
  };
84
91
 
85
92
  const args = [...argv];
@@ -118,10 +125,25 @@ export function parseArgs(argv) {
118
125
  case '--concurrency':
119
126
  options.concurrency = parsePositiveInt(requireValue(arg, args), arg);
120
127
  break;
128
+ case '--sweep-concurrency':
129
+ options.sweepConcurrency = parsePositiveIntList(requireValue(arg, args), arg);
130
+ if (options.sweepConcurrency.length > 0) {
131
+ options.concurrency = options.sweepConcurrency[0];
132
+ }
133
+ break;
121
134
  case '--timeout':
122
135
  case '--timeout-ms':
123
136
  options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
124
137
  break;
138
+ case '--slo-ttft-ms':
139
+ options.sloTtftMs = parsePositiveInt(requireValue(arg, args), arg);
140
+ break;
141
+ case '--slo-e2e-ms':
142
+ options.sloE2eMs = parsePositiveInt(requireValue(arg, args), arg);
143
+ break;
144
+ case '--slo-tpot-ms':
145
+ options.sloTpotMs = parsePositiveInt(requireValue(arg, args), arg);
146
+ break;
125
147
  case '-p':
126
148
  case '--prompt':
127
149
  options.prompts.push(requireValue(arg, args));
@@ -172,6 +194,18 @@ export function parseArgs(argv) {
172
194
  case '--ascii':
173
195
  options.ascii = true;
174
196
  break;
197
+ case '--color':
198
+ options.color = 'always';
199
+ break;
200
+ case '--no-color':
201
+ options.color = 'never';
202
+ break;
203
+ case '--compact':
204
+ options.compact = true;
205
+ break;
206
+ case '--width':
207
+ options.width = parsePositiveInt(requireValue(arg, args), arg);
208
+ break;
175
209
  case '-o':
176
210
  case '--out':
177
211
  options.out = requireValue(arg, args);
@@ -251,6 +285,33 @@ function parseNonNegativeInt(value, option) {
251
285
  return parsed;
252
286
  }
253
287
 
288
+ function parsePositiveIntList(value, option) {
289
+ const parsed = String(value)
290
+ .split(',')
291
+ .map((item) => item.trim())
292
+ .filter(Boolean)
293
+ .map((item) => parsePositiveInt(item, option));
294
+ if (parsed.length === 0) throw new Error(`${option} requires at least one positive integer`);
295
+ return parsed;
296
+ }
297
+
298
+ function renderOptions(parsed) {
299
+ return {
300
+ ascii: parsed.ascii,
301
+ color: resolveColor(parsed.color),
302
+ compact: parsed.compact,
303
+ width: parsed.width
304
+ };
305
+ }
306
+
307
+ function resolveColor(value) {
308
+ if (value === 'always') return true;
309
+ if (value === 'never') return false;
310
+ if (process.env.NO_COLOR) return false;
311
+ if (process.env.FORCE_COLOR && process.env.FORCE_COLOR !== '0') return true;
312
+ return Boolean(process.stdout.isTTY);
313
+ }
314
+
254
315
  function helpText() {
255
316
  return `fm-bench ${packageJson.version}
256
317
 
@@ -266,7 +327,12 @@ Run options:
266
327
  -r, --runs <n> Runs per prompt/model (default: 1)
267
328
  --warmup <n> Warmup runs per model before measurement
268
329
  -c, --concurrency <n> Parallel fm processes (default: 1)
330
+ --sweep-concurrency <list>
331
+ Run separate operating points, e.g. 1,2,4
269
332
  --timeout-ms <n> Timeout per fm call in ms (default: 60000)
333
+ --slo-ttft-ms <n> Count request as good only if TTFT is <= n
334
+ --slo-e2e-ms <n> Count request as good only if E2E latency is <= n
335
+ --slo-tpot-ms <n> Count request as good only if TPOT is <= n
270
336
  -p, --prompt <text> Prompt to benchmark; repeatable
271
337
  --prompt-file <file> .json, .jsonl, or blank-line separated text prompts
272
338
  --profile <name> quick, standard, interactive, throughput, or stress
@@ -286,6 +352,10 @@ Output:
286
352
  --json Alias for --format json
287
353
  --csv Alias for --format csv
288
354
  --ascii Use plain ASCII tables instead of Unicode
355
+ --color Force ANSI colors in table output
356
+ --no-color Disable ANSI colors in table output
357
+ --compact Force compact terminal layout
358
+ --width <n> Render for a specific terminal width
289
359
  -o, --out <file> Save JSON or CSV report based on file extension
290
360
  -v, --verbose Include per-run CSV after the summary table
291
361
 
package/src/report.js CHANGED
@@ -13,6 +13,7 @@ export function toCsv(rows) {
13
13
  export function flattenResults(results) {
14
14
  return results.map((result) => ({
15
15
  model: result.model,
16
+ concurrency: result.concurrency ?? '',
16
17
  prompt_id: result.promptId,
17
18
  run: result.run,
18
19
  ok: result.ok,
@@ -30,6 +31,7 @@ export function flattenResults(results) {
30
31
  streamed: result.streamed,
31
32
  stdout_chunks: result.stdoutChunks,
32
33
  output_hash: result.outputHash || '',
34
+ good: result.good == null ? '' : result.good,
33
35
  error: result.error || ''
34
36
  }));
35
37
  }
package/src/stats.js CHANGED
@@ -7,6 +7,10 @@ export function summarizeNumbers(values) {
7
7
  max: null,
8
8
  avg: null,
9
9
  sum: 0,
10
+ stddev: null,
11
+ cv: null,
12
+ ci95Low: null,
13
+ ci95High: null,
10
14
  p50: null,
11
15
  p90: null,
12
16
  p95: null,
@@ -15,12 +19,22 @@ export function summarizeNumbers(values) {
15
19
  }
16
20
 
17
21
  const total = clean.reduce((sum, value) => sum + value, 0);
22
+ const avg = total / clean.length;
23
+ const variance = clean.length > 1
24
+ ? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
25
+ : 0;
26
+ const stddev = Math.sqrt(variance);
27
+ const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
18
28
  return {
19
29
  count: clean.length,
20
30
  min: clean[0],
21
31
  max: clean[clean.length - 1],
22
- avg: total / clean.length,
32
+ avg,
23
33
  sum: total,
34
+ stddev,
35
+ cv: avg !== 0 ? stddev / Math.abs(avg) : null,
36
+ ci95Low: avg - margin,
37
+ ci95High: avg + margin,
24
38
  p50: percentile(clean, 50),
25
39
  p90: percentile(clean, 90),
26
40
  p95: percentile(clean, 95),
@@ -40,35 +54,44 @@ export function percentile(sortedValues, percentileValue) {
40
54
  return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
41
55
  }
42
56
 
43
- export function summarizeByModel(results, modelStatuses = []) {
57
+ export function summarizeByModel(results, modelStatuses = [], options = {}) {
44
58
  const byModel = new Map();
59
+ const concurrencies = options.concurrencies?.length ? options.concurrencies : [undefined];
45
60
 
46
- for (const status of modelStatuses) {
47
- byModel.set(status.name, {
48
- model: status.name,
49
- description: status.description,
50
- available: status.available,
51
- skippedReason: status.available ? '' : status.reason || 'Unavailable',
52
- results: []
53
- });
61
+ for (const concurrency of concurrencies) {
62
+ for (const status of modelStatuses) {
63
+ const key = summaryKey(status.name, concurrency);
64
+ byModel.set(key, {
65
+ model: status.name,
66
+ concurrency,
67
+ description: status.description,
68
+ available: status.available,
69
+ skippedReason: status.available ? '' : status.reason || 'Unavailable',
70
+ results: []
71
+ });
72
+ }
54
73
  }
55
74
 
56
75
  for (const result of results) {
57
- if (!byModel.has(result.model)) {
58
- byModel.set(result.model, {
76
+ const key = summaryKey(result.model, result.concurrency);
77
+ if (!byModel.has(key)) {
78
+ byModel.set(key, {
59
79
  model: result.model,
80
+ concurrency: result.concurrency,
60
81
  description: '',
61
82
  available: true,
62
83
  skippedReason: '',
63
84
  results: []
64
85
  });
65
86
  }
66
- byModel.get(result.model).results.push(result);
87
+ byModel.get(key).results.push(result);
67
88
  }
68
89
 
69
90
  return [...byModel.values()].map((entry) => {
70
91
  const successes = entry.results.filter((result) => result.ok);
71
92
  const failures = entry.results.filter((result) => !result.ok);
93
+ const goodResults = successes.filter((result) => result.good === true);
94
+ const goodMeasured = successes.filter((result) => result.good != null);
72
95
  const latency = summarizeNumbers(successes.map((result) => result.durationMs));
73
96
  const ttft = summarizeNumbers(successes.map((result) => result.firstTokenMs).filter((value) => value != null));
74
97
  const generation = summarizeNumbers(successes.map((result) => result.generationMs).filter((value) => value != null));
@@ -80,10 +103,12 @@ export function summarizeByModel(results, modelStatuses = []) {
80
103
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
81
104
  const windowMs = modelWindowMs(successes);
82
105
  const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
106
+ const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
83
107
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
84
108
 
85
109
  return {
86
110
  model: entry.model,
111
+ concurrency: entry.concurrency,
87
112
  description: entry.description,
88
113
  available: entry.available,
89
114
  skippedReason: entry.skippedReason,
@@ -91,7 +116,9 @@ export function summarizeByModel(results, modelStatuses = []) {
91
116
  successes: successes.length,
92
117
  failures: failures.length,
93
118
  successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
119
+ goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
94
120
  rps,
121
+ goodputRps,
95
122
  outputTokenThroughput,
96
123
  repeatability: summarizeRepeatability(successes),
97
124
  latency,
@@ -107,6 +134,49 @@ export function summarizeByModel(results, modelStatuses = []) {
107
134
  });
108
135
  }
109
136
 
137
+ function summaryKey(model, concurrency) {
138
+ return `${model}::${concurrency ?? 'default'}`;
139
+ }
140
+
141
+ function tCritical95(n) {
142
+ const df = Math.max(1, n - 1);
143
+ const table = {
144
+ 1: 12.706,
145
+ 2: 4.303,
146
+ 3: 3.182,
147
+ 4: 2.776,
148
+ 5: 2.571,
149
+ 6: 2.447,
150
+ 7: 2.365,
151
+ 8: 2.306,
152
+ 9: 2.262,
153
+ 10: 2.228,
154
+ 11: 2.201,
155
+ 12: 2.179,
156
+ 13: 2.16,
157
+ 14: 2.145,
158
+ 15: 2.131,
159
+ 16: 2.12,
160
+ 17: 2.11,
161
+ 18: 2.101,
162
+ 19: 2.093,
163
+ 20: 2.086,
164
+ 21: 2.08,
165
+ 22: 2.074,
166
+ 23: 2.069,
167
+ 24: 2.064,
168
+ 25: 2.06,
169
+ 26: 2.056,
170
+ 27: 2.052,
171
+ 28: 2.048,
172
+ 29: 2.045,
173
+ 30: 2.042
174
+ };
175
+ if (df <= 30) return table[df];
176
+ if (df <= 60) return 2;
177
+ return 1.96;
178
+ }
179
+
110
180
  function modelWindowMs(results) {
111
181
  const starts = results.map((result) => result.startOffsetMs).filter((value) => Number.isFinite(value));
112
182
  const ends = results.map((result) => result.endOffsetMs).filter((value) => Number.isFinite(value));
package/src/table.js CHANGED
@@ -1,8 +1,17 @@
1
+ import { stripAnsi } from './ansi.js';
2
+
1
3
  export function renderTable(headers, rows, options = {}) {
2
4
  const ascii = Boolean(options.ascii);
3
- const stringRows = rows.map((row) => row.map(formatCell));
5
+ const maxCellWidth = options.maxCellWidth || 60;
6
+ const normalizedRows = rows.map((row) => row.map((cell) => {
7
+ const normalized = normalizeCell(cell);
8
+ return {
9
+ ...normalized,
10
+ text: truncate(normalized.text, maxCellWidth)
11
+ };
12
+ }));
4
13
  const widths = headers.map((header, index) => {
5
- const values = [header, ...stringRows.map((row) => row[index] ?? '')];
14
+ const values = [truncate(header, maxCellWidth), ...normalizedRows.map((row) => row[index]?.text ?? '')];
6
15
  return Math.max(...values.map(visibleLength));
7
16
  });
8
17
  const style = ascii ? ASCII_TABLE : UNICODE_TABLE;
@@ -10,27 +19,45 @@ export function renderTable(headers, rows, options = {}) {
10
19
  const top = rule(style.topLeft, style.topJoin, style.topRight, style.horizontal, widths);
11
20
  const middle = rule(style.midLeft, style.midJoin, style.midRight, style.horizontal, widths);
12
21
  const bottom = rule(style.bottomLeft, style.bottomJoin, style.bottomRight, style.horizontal, widths);
13
- const headerLine = rowLine(headers, widths, style, true);
14
- const bodyLines = stringRows.map((row) => rowLine(row, widths, style));
22
+ const headerLine = rowLine(headers.map((header) => truncate(header, maxCellWidth)), widths, style, true, options);
23
+ const bodyLines = normalizedRows.map((row) => rowLine(row, widths, style, false, options));
15
24
 
16
25
  return [top, headerLine, middle, ...bodyLines, bottom].join('\n');
17
26
  }
18
27
 
19
28
  export function renderBenchmarkReport(payload, options = {}) {
29
+ const width = terminalWidth(options);
30
+ const mode = options.compact || width < 88
31
+ ? 'compact'
32
+ : width < 140
33
+ ? 'medium'
34
+ : 'wide';
20
35
  const lines = [];
21
36
  const elapsedMs = Date.parse(payload.finishedAt) - Date.parse(payload.startedAt);
22
37
  const skipped = payload.summary.filter((item) => !item.available).length;
23
38
  const measured = payload.summary.reduce((sum, item) => sum + item.successes, 0);
24
39
  const failed = payload.summary.reduce((sum, item) => sum + item.failures, 0);
40
+ const concurrencies = payload.options.sweepConcurrency?.length
41
+ ? payload.options.sweepConcurrency.join(',')
42
+ : String(payload.options.concurrency);
43
+ const slo = formatSlo(payload.options.slo);
25
44
 
26
- lines.push(`fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`);
27
- lines.push(`prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${payload.options.concurrency} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped models ${skipped} | elapsed ${formatMs(elapsedMs)}`);
28
- lines.push('');
29
- lines.push(renderSummaryTable(payload.summary, options));
45
+ const title = `fm-bench ${payload.version} | ${payload.environment.platform}/${payload.environment.arch} | ${payload.environment.fmBin}`;
46
+ const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
47
+ lines.push(truncate(title, width));
48
+ lines.push(truncate(meta, width));
30
49
  lines.push('');
31
- lines.push(renderDetailTable(payload.summary, options));
50
+
51
+ if (mode === 'compact') {
52
+ lines.push(renderCompactSummary(payload.summary, { ...options, width, slo: payload.options.slo }));
53
+ } else {
54
+ lines.push(renderSummaryTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
55
+ lines.push('');
56
+ lines.push(renderDetailTable(payload.summary, { ...options, mode, width, slo: payload.options.slo }));
57
+ }
58
+
32
59
  lines.push('');
33
- lines.push('TTFT = time to first streamed output, E2E = full response latency, TPOT = decode time per output token.');
60
+ lines.push(compactLegend(width));
34
61
 
35
62
  return lines.join('\n');
36
63
  }
@@ -53,77 +80,181 @@ export function formatPercent(value, digits = 0) {
53
80
  }
54
81
 
55
82
  export function renderSummaryTable(summary, options = {}) {
83
+ const mode = options.mode || 'wide';
84
+ const hasGoodput = summary.some((item) => item.goodputRate != null);
85
+ const ranks = rankSummary(summary);
56
86
  const rows = summary.map((item) => {
57
87
  const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
58
- return [
59
- item.model,
60
- status,
61
- item.attempted || '-',
62
- item.successes || '-',
63
- formatPercent(item.successRate),
64
- formatMs(item.ttft.p50),
65
- formatMs(item.ttft.p95),
66
- formatMs(item.latency.p50),
67
- formatMs(item.latency.p95),
68
- formatMs(item.tpot.p50),
69
- formatNumber(item.tokensPerSecond.avg),
70
- formatNumber(item.rps),
71
- item.available ? '' : compactReason(item.skippedReason)
88
+ const tones = metricTones(item, ranks, options.slo);
89
+ const base = [
90
+ cell(formatConcurrency(item.concurrency)),
91
+ cell(item.model, item.available ? null : 'muted'),
92
+ cell(status, statusTone(status)),
93
+ cell(item.attempted ? `${item.successes}/${item.attempted}` : '-', item.failures > 0 ? 'yellow' : item.available ? 'green' : 'muted'),
94
+ cell(formatPercent(item.successRate), percentTone(item.successRate, 0.95, 1)),
95
+ cell(formatPercent(item.goodputRate), percentTone(item.goodputRate, 0.8, 1)),
96
+ cell(formatMs(item.ttft.p50), tones.ttft),
97
+ cell(formatMs(item.ttft.p95), tones.ttftP95),
98
+ cell(formatMs(item.latency.p50), tones.e2e),
99
+ cell(formatMs(item.latency.p95), tones.e2eP95),
100
+ cell(formatNumber(item.tokensPerSecond.avg), tones.userTps),
101
+ cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
102
+ cell(formatNumber(item.rps), tones.rps),
103
+ cell(formatPercent(item.latency.cv), cvTone(item.latency.cv)),
104
+ cell(item.available ? '' : compactReason(item.skippedReason), item.available ? null : 'yellow')
72
105
  ];
106
+
107
+ if (mode === 'medium') {
108
+ const medium = [
109
+ base[0],
110
+ base[1],
111
+ base[2],
112
+ base[3]
113
+ ];
114
+ if (hasGoodput) medium.push(base[5]);
115
+ medium.push(base[6], base[8], base[9], base[10], base[11], base[13], base[14]);
116
+ return medium;
117
+ }
118
+
119
+ const wide = [base[0], base[1], base[2], base[3], base[4]];
120
+ if (hasGoodput) wide.push(base[5]);
121
+ wide.push(
122
+ base[6],
123
+ base[7],
124
+ base[8],
125
+ base[9],
126
+ cell(formatMs(item.tpot.p50), tones.tpot),
127
+ cell(formatMs(item.tpot.p95), tones.tpotP95),
128
+ base[10],
129
+ base[11],
130
+ base[12],
131
+ base[13],
132
+ base[14]
133
+ );
134
+ return wide;
73
135
  });
74
136
 
75
- return renderTable([
76
- 'model',
77
- 'status',
78
- 'runs',
79
- 'ok',
80
- 'success',
81
- 'ttft p50',
82
- 'ttft p95',
83
- 'e2e p50',
84
- 'e2e p95',
85
- 'tpot p50',
86
- 'tok/s',
87
- 'rps',
88
- 'note'
89
- ], rows, options);
137
+ const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
138
+ const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
139
+ const headers = mode === 'medium'
140
+ ? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
141
+ : (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
142
+
143
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
90
144
  }
91
145
 
92
146
  export function renderDetailTable(summary, options = {}) {
93
- const rows = summary.map((item) => [
94
- item.model,
95
- formatNumber(item.promptTokens.avg, 0),
96
- formatNumber(item.outputTokens.avg, 0),
97
- formatNumber(item.outputTokenThroughput),
98
- formatNumber(item.decodeTokensPerSecond.avg),
99
- formatMs(item.latency.p99),
100
- formatMs(item.tpot.p95),
101
- formatPercent(item.repeatability),
102
- item.description || '-'
103
- ]);
104
-
105
- return renderTable([
106
- 'model',
107
- 'in tok avg',
108
- 'out tok avg',
109
- 'total tok/s',
110
- 'decode tok/s',
111
- 'e2e p99',
112
- 'tpot p95',
113
- 'repeat',
114
- 'description'
115
- ], rows, options);
147
+ const mode = options.mode || 'wide';
148
+ const ranks = rankSummary(summary);
149
+ const rows = summary.map((item) => {
150
+ const tones = metricTones(item, ranks, options.slo);
151
+ const base = [
152
+ cell(formatConcurrency(item.concurrency)),
153
+ cell(item.model, item.available ? null : 'muted'),
154
+ cell(formatNumber(item.promptTokens.avg, 0)),
155
+ cell(formatNumber(item.outputTokens.avg, 0)),
156
+ cell(formatNumber(item.decodeTokensPerSecond.avg), tones.decodeTps),
157
+ cell(formatMs(item.latency.p99), tones.e2eP99),
158
+ cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
159
+ cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
160
+ cell(item.description || '-', 'muted')
161
+ ];
162
+
163
+ if (mode === 'medium') {
164
+ return base.slice(0, 8);
165
+ }
166
+
167
+ return base;
168
+ });
169
+
170
+ const headers = mode === 'medium'
171
+ ? ['c', 'model', 'in avg', 'out avg', 'decode t/s', 'e2e p99', 'e2e 95% ci', 'repeat']
172
+ : [
173
+ 'c',
174
+ 'model',
175
+ 'in tok avg',
176
+ 'out tok avg',
177
+ 'decode tok/s',
178
+ 'e2e p99',
179
+ 'e2e 95% ci',
180
+ 'repeat',
181
+ 'description'
182
+ ];
183
+
184
+ return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 34 : 52 });
185
+ }
186
+
187
+ export function renderCompactSummary(summary, options = {}) {
188
+ const width = options.width || 80;
189
+ const separator = options.ascii ? '-' : '─';
190
+ const lines = [];
191
+ const ranks = rankSummary(summary);
192
+
193
+ for (const item of summary) {
194
+ const status = item.available ? (item.failures > 0 ? 'partial' : 'ok') : 'skipped';
195
+ const title = `${item.model} c${formatConcurrency(item.concurrency)} ${status} ${item.attempted ? `${item.successes}/${item.attempted}` : '-'}`;
196
+ lines.push(toneText(truncate(title, width), statusTone(status), options));
197
+
198
+ if (item.available) {
199
+ const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
200
+ const tones = metricTones(item, ranks, options.slo);
201
+ lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
202
+ lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
203
+ lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width), worstTone(tones.tpot, percentTone(item.repeatability, 0.5, 0.9)), options));
204
+ } else {
205
+ lines.push(toneText(truncate(` ${compactReason(item.skippedReason)}`, width), 'yellow', options));
206
+ }
207
+ lines.push(separator.repeat(Math.min(width, 72)));
208
+ }
209
+
210
+ if (lines.at(-1)?.startsWith(separator)) lines.pop();
211
+ return lines.join('\n');
116
212
  }
117
213
 
118
214
  export function renderModelsTable(models, options = {}) {
215
+ const width = terminalWidth(options);
216
+ const compact = options.compact || width < 88;
217
+ if (compact) {
218
+ return models.map((model) => {
219
+ const status = model.available ? 'yes' : 'no';
220
+ return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(model.reason || model.description || '-')}`;
221
+ }).join('\n');
222
+ }
223
+
119
224
  return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
120
- model.name,
121
- model.available ? 'yes' : 'no',
225
+ cell(model.name),
226
+ cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
122
227
  model.description || '-',
123
228
  compactReason(model.quota || model.reason || '-')
124
229
  ]), options);
125
230
  }
126
231
 
232
+ function formatRangeMs(low, high) {
233
+ if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
234
+ return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
235
+ }
236
+
237
+ function formatConcurrency(value) {
238
+ return value == null ? '1' : String(value);
239
+ }
240
+
241
+ function terminalWidth(options = {}) {
242
+ return options.width || process.stdout.columns || 120;
243
+ }
244
+
245
+ function compactLegend(width) {
246
+ const text = 'TTFT = first streamed output. E2E = full response. TPOT = post-first-token decode cadence. CV = lower is steadier.';
247
+ return truncate(text, width);
248
+ }
249
+
250
+ function formatSlo(slo = {}) {
251
+ const parts = [];
252
+ if (slo.ttftMs) parts.push(`TTFT<=${formatMs(slo.ttftMs)}`);
253
+ if (slo.e2eMs) parts.push(`E2E<=${formatMs(slo.e2eMs)}`);
254
+ if (slo.tpotMs) parts.push(`TPOT<=${formatMs(slo.tpotMs)}`);
255
+ return parts.length ? `SLO ${parts.join(',')}` : '';
256
+ }
257
+
127
258
  const ASCII_TABLE = {
128
259
  topLeft: '+',
129
260
  topJoin: '+',
@@ -156,13 +287,32 @@ function rule(left, join, right, horizontal, widths) {
156
287
  return `${left}${widths.map((width) => horizontal.repeat(width + 2)).join(join)}${right}`;
157
288
  }
158
289
 
159
- function rowLine(row, widths, style, header = false) {
290
+ function rowLine(row, widths, style, header = false, options = {}) {
160
291
  return `${style.vertical}${row.map((cell, index) => {
161
- const value = header ? String(cell).toUpperCase() : formatCell(cell);
162
- return ` ${pad(value, widths[index], !header && isNumericCell(value))} `;
292
+ const normalized = header ? { text: String(cell).toUpperCase(), tone: 'header' } : normalizeCell(cell);
293
+ const value = normalized.text;
294
+ const padded = pad(value, widths[index], !header && isNumericCell(value));
295
+ return ` ${toneText(padded, normalized.tone, options)} `;
163
296
  }).join(style.vertical)}${style.vertical}`;
164
297
  }
165
298
 
299
+ function cell(text, tone = null) {
300
+ return { text, tone };
301
+ }
302
+
303
+ function normalizeCell(value) {
304
+ if (value && typeof value === 'object' && Object.hasOwn(value, 'text')) {
305
+ return {
306
+ text: formatCell(value.text),
307
+ tone: value.tone || null
308
+ };
309
+ }
310
+ return {
311
+ text: formatCell(value),
312
+ tone: null
313
+ };
314
+ }
315
+
166
316
  function formatCell(value) {
167
317
  if (value == null) return '';
168
318
  return String(value);
@@ -179,7 +329,7 @@ function isNumericCell(value) {
179
329
  }
180
330
 
181
331
  function visibleLength(value) {
182
- return String(value).length;
332
+ return stripAnsi(value).length;
183
333
  }
184
334
 
185
335
  function compactReason(value) {
@@ -187,3 +337,119 @@ function compactReason(value) {
187
337
  if (clean.length <= 58) return clean;
188
338
  return `${clean.slice(0, 55)}...`;
189
339
  }
340
+
341
+ function truncate(value, width) {
342
+ const text = String(value ?? '');
343
+ if (visibleLength(text) <= width) return text;
344
+ if (width <= 1) return '…';
345
+ return `${text.slice(0, width - 1)}…`;
346
+ }
347
+
348
+ const TONES = {
349
+ header: ['\x1b[1m', '\x1b[0m'],
350
+ green: ['\x1b[32m', '\x1b[0m'],
351
+ yellow: ['\x1b[33m', '\x1b[0m'],
352
+ red: ['\x1b[31m', '\x1b[0m'],
353
+ muted: ['\x1b[2m', '\x1b[0m']
354
+ };
355
+
356
+ function toneText(text, tone, options = {}) {
357
+ if (!options.color || !tone || !TONES[tone]) return text;
358
+ const [open, close] = TONES[tone];
359
+ return `${open}${text}${close}`;
360
+ }
361
+
362
+ function statusTone(status) {
363
+ if (status === 'ok') return 'green';
364
+ if (status === 'partial') return 'yellow';
365
+ if (status === 'skipped') return 'yellow';
366
+ return 'red';
367
+ }
368
+
369
+ function percentTone(value, yellowAt, greenAt) {
370
+ if (value == null || !Number.isFinite(value)) return null;
371
+ if (value >= greenAt) return 'green';
372
+ if (value >= yellowAt) return 'yellow';
373
+ return 'red';
374
+ }
375
+
376
+ function cvTone(value) {
377
+ if (value == null || !Number.isFinite(value)) return null;
378
+ if (value <= 0.1) return 'green';
379
+ if (value <= 0.25) return 'yellow';
380
+ return 'red';
381
+ }
382
+
383
+ function thresholdTone(value, threshold) {
384
+ if (value == null || !Number.isFinite(value) || !Number.isFinite(threshold)) return null;
385
+ if (value <= threshold * 0.8) return 'green';
386
+ if (value <= threshold) return 'yellow';
387
+ return 'red';
388
+ }
389
+
390
+ function rankSummary(summary) {
391
+ return {
392
+ ttft: collectMetric(summary, (item) => item.ttft?.p50),
393
+ ttftP95: collectMetric(summary, (item) => item.ttft?.p95),
394
+ e2e: collectMetric(summary, (item) => item.latency?.p50),
395
+ e2eP95: collectMetric(summary, (item) => item.latency?.p95),
396
+ e2eP99: collectMetric(summary, (item) => item.latency?.p99),
397
+ tpot: collectMetric(summary, (item) => item.tpot?.p50),
398
+ tpotP95: collectMetric(summary, (item) => item.tpot?.p95),
399
+ userTps: collectMetric(summary, (item) => item.tokensPerSecond?.avg),
400
+ systemTps: collectMetric(summary, (item) => item.outputTokenThroughput),
401
+ rps: collectMetric(summary, (item) => item.rps),
402
+ decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg)
403
+ };
404
+ }
405
+
406
+ function collectMetric(summary, accessor) {
407
+ return summary.map(accessor).filter((value) => Number.isFinite(value));
408
+ }
409
+
410
+ function rankTone(value, values) {
411
+ if (value == null || !Number.isFinite(value) || values.length === 0) return null;
412
+ if (values.length === 1) return 'green';
413
+ const min = Math.min(...values);
414
+ const max = Math.max(...values);
415
+ if (max === min) return 'green';
416
+ const position = (value - min) / (max - min);
417
+ if (position >= 0.67) return 'green';
418
+ if (position >= 0.34) return 'yellow';
419
+ return 'red';
420
+ }
421
+
422
+ function rankToneLower(value, values) {
423
+ if (value == null || !Number.isFinite(value) || values.length === 0) return null;
424
+ if (values.length === 1) return 'green';
425
+ const min = Math.min(...values);
426
+ const max = Math.max(...values);
427
+ if (max === min) return 'green';
428
+ const position = (max - value) / (max - min);
429
+ if (position >= 0.67) return 'green';
430
+ if (position >= 0.34) return 'yellow';
431
+ return 'red';
432
+ }
433
+
434
+ function metricTones(item, ranks, slo = {}) {
435
+ return {
436
+ ttft: thresholdTone(item.ttft?.p50, slo.ttftMs) || rankToneLower(item.ttft?.p50, ranks.ttft),
437
+ ttftP95: thresholdTone(item.ttft?.p95, slo.ttftMs) || rankToneLower(item.ttft?.p95, ranks.ttftP95),
438
+ e2e: thresholdTone(item.latency?.p50, slo.e2eMs) || rankToneLower(item.latency?.p50, ranks.e2e),
439
+ e2eP95: thresholdTone(item.latency?.p95, slo.e2eMs) || rankToneLower(item.latency?.p95, ranks.e2eP95),
440
+ e2eP99: thresholdTone(item.latency?.p99, slo.e2eMs) || rankToneLower(item.latency?.p99, ranks.e2eP99),
441
+ tpot: thresholdTone(item.tpot?.p50, slo.tpotMs) || rankToneLower(item.tpot?.p50, ranks.tpot),
442
+ tpotP95: thresholdTone(item.tpot?.p95, slo.tpotMs) || rankToneLower(item.tpot?.p95, ranks.tpotP95),
443
+ userTps: rankTone(item.tokensPerSecond?.avg, ranks.userTps),
444
+ systemTps: rankTone(item.outputTokenThroughput, ranks.systemTps),
445
+ rps: rankTone(item.rps, ranks.rps),
446
+ decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps)
447
+ };
448
+ }
449
+
450
+ function worstTone(...tones) {
451
+ if (tones.includes('red')) return 'red';
452
+ if (tones.includes('yellow')) return 'yellow';
453
+ if (tones.includes('green')) return 'green';
454
+ return null;
455
+ }