fm-bench 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  `fm-bench` is a dynamic benchmark CLI for Apple's `fm` command on macOS 27 and newer.
4
4
 
5
- It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, and prints a terminal table with latency and throughput stats.
5
+ It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, shows live progress while it works, and prints terminal tables with latency, throughput, stability, goodput, and streaming-quality stats.
6
6
 
7
7
  Apple introduced the preinstalled `fm` command for macOS 27 as part of the Foundation Models tooling. `fm-bench` intentionally shells out to the system `fm` binary instead of linking private APIs, so it can adapt as Apple adds models or changes availability.
8
8
 
@@ -35,20 +35,15 @@ fm-bench
35
35
  Example output:
36
36
 
37
37
  ```text
38
- fm-bench 0.2.0 | darwin/arm64 | fm
39
- prompts 3 | runs 1 | concurrency 1 | stream on | measured 3 | failed 0 | skipped models 0 | elapsed 11.36s
40
-
41
- ┌────────┬────────┬──────┬────┬─────────┬──────────┬──────────┬─────────┬─────────┬──────────┬───────┬─────┬──────┐
42
- │ MODEL │ STATUS │ RUNS │ OK │ SUCCESS │ TTFT P50 │ TTFT P95 │ E2E P50 │ E2E P95 │ TPOT P50 │ TOK/S │ RPS │ NOTE │
43
- ├────────┼────────┼──────┼────┼─────────┼──────────┼──────────┼─────────┼─────────┼──────────┼───────┼─────┼──────┤
44
- │ system │ ok │ 3 │ 3 │ 100% │ 409ms │ 486ms │ 2.29s │ 5.97s │ 14ms │ 58.5 │ 0.3 │ │
45
- └────────┴────────┴──────┴────┴─────────┴──────────┴──────────┴─────────┴─────────┴──────────┴───────┴─────┴──────┘
46
-
47
- ┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
48
- │ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
49
- ├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
50
- │ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
51
- └────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
38
+ fm-bench 0.4.0 | darwin/arm64 | fm
39
+ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
40
+
41
+ ┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
42
+ │ C │ MODEL │ STATUS │ OK/RUNS │ SUCC │ GOOD │ GOOD RPS │ TTFT │ E2E P95 │ SYS │ CV │
43
+ ├───┼────────┼────────┼─────────┼──────┼──────┼──────────┼──────┼──────────┼─────┼─────┤
44
+ │ 1 │ system │ ok │ 15/15 │ 100% │ 93% │ 0.4 │ 318ms│ 3.20s │ 42 │ 12% │
45
+ │ 2 │ system │ ok │ 15/15 │ 100% │ 80% │ 0.7 │ 501ms│ 4.40s │ 68 │ 21% │
46
+ └───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
52
47
  ```
53
48
 
54
49
  ## Commands
@@ -72,6 +67,7 @@ fm-bench --models system,pcc --runs 3 --profile stress
72
67
  fm-bench --models system --runs 5 --profile interactive
73
68
  fm-bench --models system --runs 3 --profile throughput --warmup 1
74
69
  fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
70
+ fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5 --ramp-up-ms 2000
75
71
  fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
76
72
  fm-bench --prompt "Reply with exactly: ok" --runs 5
77
73
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
@@ -85,9 +81,11 @@ Useful flags:
85
81
  - `--warmup <n>`: warmup runs per model before measurement.
86
82
  - `--concurrency <n>`: parallel `fm` processes.
87
83
  - `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
84
+ - `--request-rate <rps>`: pace request starts at a target requests-per-second rate.
85
+ - `--ramp-up-ms <n>`: gradually ramp request pacing over `n` milliseconds.
88
86
  - `--timeout-ms <n>`: timeout per `fm` call.
89
87
  - `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
90
- - `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
88
+ - `--profile quick|standard|interactive|throughput|client|stress`: built-in prompt suite.
91
89
  - `--prompt <text>`: custom prompt, repeatable.
92
90
  - `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
93
91
  - `--instructions <text>`: passed to `fm respond`.
@@ -96,6 +94,7 @@ Useful flags:
96
94
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
97
95
  - `--ascii`: use plain ASCII table borders.
98
96
  - `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
97
+ - `--progress`, `--no-progress`: force or disable the live progress status line on stderr.
99
98
  - `--compact`: force the narrow terminal layout.
100
99
  - `--width <n>`: render as if the terminal has `n` columns.
101
100
  - `--out <file>`: save a report.
@@ -127,8 +126,11 @@ Plain text files are split on blank lines.
127
126
  - TTFT, or time to first streamed output.
128
127
  - E2E latency, or full response wall-clock latency.
129
128
  - TPOT, or decode time per output token after the first output token.
129
+ - second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
130
+ - prefill tokens per second, or prompt tokens divided by TTFT.
130
131
  - output tokens per second per request.
131
132
  - total output token throughput across the measured window.
133
+ - total token throughput, including prompt and output tokens.
132
134
  - requests per second across the measured window.
133
135
  - goodput percentage and goodput RPS when SLO flags are set.
134
136
  - coefficient of variation (CV) and confidence interval context for stability.
@@ -140,10 +142,16 @@ Plain text files are split on blank lines.
140
142
 
141
143
  Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
142
144
 
143
- Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
145
+ Measured runs stream by default so `fm-bench` can capture TTFT and streaming smoothness. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT, TPOT, second-chunk, and chunk-gap fields that depend on streaming will be blank.
144
146
 
145
147
  Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
146
148
 
149
+ ## Live Progress
150
+
151
+ Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
152
+
153
+ Progress is automatic for table output on TTYs. Use `--progress` to force it or `--no-progress` to keep the terminal completely quiet until the report is ready.
154
+
147
155
  ## Terminal Colors
148
156
 
149
157
  Table output uses semantic ANSI color on interactive terminals:
@@ -9,6 +9,7 @@ The metric set follows common LLM inference benchmark practice:
9
9
  - Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
10
10
  - NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
11
11
  - NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
12
+ - NVIDIA AIPerf documents time to second token, inter-token latency, inter-chunk latency, per-user output throughput, and prefill throughput: <https://docs.nvidia.com/aiperf/reference/ai-perf-metrics-reference>
12
13
  - vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
13
14
  - MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
14
15
  - MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
@@ -19,11 +20,16 @@ The metric set follows common LLM inference benchmark practice:
19
20
  - `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
20
21
  - `generation_ms`: `E2E - TTFT`.
21
22
  - `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
23
+ - `second_chunk_ms`: time between the first and second streamed stdout chunks. This is a terminal-side proxy for time-to-second-token style startup smoothness.
24
+ - `chunk_gap`: the distribution of time between consecutive streamed stdout chunks. It is useful for spotting streaming jitter, but it is chunk-based rather than token-based because the `fm` CLI writes stdout chunks, not token timestamp events.
25
+ - `prefill_tokens_per_second`: input prompt tokens divided by TTFT seconds. This estimates prompt-processing speed for streaming runs.
22
26
  - `tokens_per_second`: output tokens divided by E2E seconds for one request.
23
27
  - `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
24
28
  - `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
29
+ - `total token throughput`: successful prompt and output tokens divided by that model's measured wall-clock window.
25
30
  - `RPS`: successful requests divided by that model's measured wall-clock window.
26
31
  - `goodput`: successful requests that also satisfy all provided SLO thresholds.
32
+ - `goodput RPS`: SLO-passing requests divided by that model's measured wall-clock window. If SLOs are set and no requests pass, this is reported as zero.
27
33
  - `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
28
34
  - `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
29
35
  - `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
@@ -32,12 +38,20 @@ The metric set follows common LLM inference benchmark practice:
32
38
 
33
39
  Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
34
40
 
41
+ Use `--request-rate <rps>` to pace request starts independently of concurrency. Concurrency limits how many `fm respond` processes can be active at once; request rate controls how quickly new work is admitted. Use `--ramp-up-ms` to avoid instantly shocking a model or quota path when you start a higher-rate run.
42
+
35
43
  `fm-bench` does not interpolate between operating points. It reports only what was actually measured.
36
44
 
45
+ ## Prompt Profiles
46
+
47
+ The `client` profile is a pragmatic local-machine mix inspired by MLPerf Client's emphasis on multiple task categories and prompt/response lengths. It includes short chat, content generation, structured extraction, light summarization, and code analysis prompts. It is not a formal MLPerf submission suite; it is a convenient built-in workload for comparing your own Mac, OS build, and `fm` models over time.
48
+
37
49
  ## Caveats
38
50
 
39
51
  `fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
40
52
 
41
53
  Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
42
54
 
43
- For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
55
+ Stream smoothness metrics use stdout chunk arrival times. A chunk can contain more than one token, and terminal or pipe buffering can affect chunk boundaries. Treat `second_chunk_ms` and `chunk_gap` as user-visible streaming diagnostics, not raw decoder telemetry.
56
+
57
+ For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput or client profiles, compare models at the same concurrency operating points, set SLOs that match your real UX budget, and save JSON reports for later analysis.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.3.1",
3
+ "version": "0.4.0",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/bench.js CHANGED
@@ -36,7 +36,9 @@ export async function inspectModels(options = {}) {
36
36
 
37
37
  export async function runBenchmark(options = {}) {
38
38
  const startedAt = new Date().toISOString();
39
+ notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
39
40
  const prompts = await loadPrompts(options);
41
+ notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
40
42
  const inspection = await inspectModels(options);
41
43
  const modelStatuses = options.availableOnly
42
44
  ? inspection.models.filter((model) => model.available)
@@ -45,15 +47,36 @@ export async function runBenchmark(options = {}) {
45
47
  const environment = await collectEnvironment(inspection.fmBin);
46
48
  const promptTokenCounts = new Map();
47
49
  const concurrencies = normalizeConcurrencySweep(options);
50
+ const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
48
51
 
52
+ notify(options, {
53
+ type: 'tokens:start',
54
+ total: prompts.length,
55
+ message: 'counting prompt tokens'
56
+ });
49
57
  for (const prompt of prompts) {
50
58
  const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
51
59
  promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
60
+ notify(options, {
61
+ type: 'tokens:progress',
62
+ completed: promptTokenCounts.size,
63
+ total: prompts.length,
64
+ promptId: prompt.id
65
+ });
52
66
  }
53
67
 
54
68
  const results = [];
55
69
  const scenarios = [];
56
- for (const concurrency of concurrencies) {
70
+ let completedRuns = 0;
71
+ let failedRuns = 0;
72
+ notify(options, {
73
+ type: 'benchmark:start',
74
+ total: totalRuns,
75
+ modelCount: runnableModels.length,
76
+ promptCount: prompts.length,
77
+ scenarioCount: concurrencies.length
78
+ });
79
+ for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
57
80
  const scenario = await runScenario({
58
81
  fmBin: inspection.fmBin,
59
82
  prompts,
@@ -61,7 +84,26 @@ export async function runBenchmark(options = {}) {
61
84
  modelStatuses,
62
85
  promptTokenCounts,
63
86
  options,
64
- concurrency
87
+ concurrency,
88
+ scenarioIndex: scenarioIndex + 1,
89
+ scenarioCount: concurrencies.length,
90
+ onMeasuredResult: (result) => {
91
+ completedRuns += 1;
92
+ if (!result.ok) failedRuns += 1;
93
+ notify(options, {
94
+ type: 'benchmark:progress',
95
+ completed: completedRuns,
96
+ failed: failedRuns,
97
+ total: totalRuns,
98
+ concurrency,
99
+ model: result.model,
100
+ promptId: result.promptId,
101
+ run: result.run,
102
+ ok: result.ok,
103
+ durationMs: result.durationMs,
104
+ firstTokenMs: result.firstTokenMs
105
+ });
106
+ }
65
107
  });
66
108
  scenarios.push(scenario);
67
109
  results.push(...scenario.results);
@@ -73,7 +115,7 @@ export async function runBenchmark(options = {}) {
73
115
  || a.run - b.run);
74
116
 
75
117
  const summary = summarizeByModel(results, modelStatuses, { concurrencies });
76
- return {
118
+ const payload = {
77
119
  tool: 'fm-bench',
78
120
  version: options.version,
79
121
  startedAt,
@@ -90,6 +132,13 @@ export async function runBenchmark(options = {}) {
90
132
  summary,
91
133
  results
92
134
  };
135
+ notify(options, {
136
+ type: 'benchmark:complete',
137
+ completed: completedRuns,
138
+ failed: failedRuns,
139
+ total: totalRuns
140
+ });
141
+ return payload;
93
142
  }
94
143
 
95
144
  async function runScenario(context) {
@@ -100,16 +149,40 @@ async function runScenario(context) {
100
149
  modelStatuses,
101
150
  promptTokenCounts,
102
151
  options,
103
- concurrency
152
+ concurrency,
153
+ scenarioIndex,
154
+ scenarioCount,
155
+ onMeasuredResult
104
156
  } = context;
105
157
  const startedAt = new Date().toISOString();
106
158
 
159
+ const warmupTotal = options.warmup * runnableModels.length;
160
+ if (warmupTotal > 0) {
161
+ notify(options, {
162
+ type: 'warmup:start',
163
+ concurrency,
164
+ scenarioIndex,
165
+ scenarioCount,
166
+ total: warmupTotal
167
+ });
168
+ }
169
+ let warmupCompleted = 0;
107
170
  for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
108
171
  for (const model of runnableModels) {
109
172
  await respond(fmBin, model.name, prompts[0].prompt, {
110
173
  ...options,
111
174
  stream: false
112
175
  });
176
+ warmupCompleted += 1;
177
+ notify(options, {
178
+ type: 'warmup:progress',
179
+ concurrency,
180
+ scenarioIndex,
181
+ scenarioCount,
182
+ completed: warmupCompleted,
183
+ total: warmupTotal,
184
+ model: model.name
185
+ });
113
186
  }
114
187
  }
115
188
 
@@ -124,15 +197,23 @@ async function runScenario(context) {
124
197
  }
125
198
 
126
199
  const results = [];
200
+ notify(options, {
201
+ type: 'scenario:start',
202
+ concurrency,
203
+ scenarioIndex,
204
+ scenarioCount,
205
+ total: jobs.length
206
+ });
127
207
  await runLimited(jobs, concurrency, async (job) => {
128
208
  const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
129
209
  results.push(result);
210
+ if (onMeasuredResult) onMeasuredResult(result);
130
211
  if (!result.ok && options.failFast) {
131
212
  const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
132
213
  error.exitCode = 1;
133
214
  throw error;
134
215
  }
135
- });
216
+ }, options);
136
217
 
137
218
  results.sort((a, b) => a.model.localeCompare(b.model)
138
219
  || a.promptId.localeCompare(b.promptId)
@@ -171,6 +252,12 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
171
252
  : null;
172
253
  const chars = response.output.length;
173
254
  const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
255
+ const promptTokens = promptTokenCounts.get(job.prompt.id);
256
+ const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
257
+ ? promptTokens / (firstTokenMs / 1000)
258
+ : null;
259
+ const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
260
+ const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
174
261
 
175
262
  return {
176
263
  model: job.model.name,
@@ -182,17 +269,22 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
182
269
  firstTokenMs,
183
270
  generationMs,
184
271
  tpotMs,
185
- promptTokens: promptTokenCounts.get(job.prompt.id),
272
+ promptTokens,
186
273
  outputTokens: countedOutputTokens,
187
274
  chars,
188
275
  words,
189
276
  tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
190
277
  decodeTokensPerSecond,
278
+ prefillTokensPerSecond,
191
279
  charsPerSecond: seconds > 0 ? chars / seconds : 0,
192
280
  startOffsetMs,
193
281
  endOffsetMs,
194
282
  streamed: response.streamed,
195
283
  stdoutChunks: response.stdoutChunks,
284
+ secondChunkMs,
285
+ chunkGapsMs,
286
+ chunkGapAvgMs: average(chunkGapsMs),
287
+ chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
196
288
  outputHash: response.ok ? hashOutput(response.output) : null,
197
289
  good: response.ok ? evaluateSlo({
198
290
  firstTokenMs,
@@ -204,12 +296,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
204
296
  };
205
297
  }
206
298
 
207
- async function runLimited(items, concurrency, worker) {
299
+ async function runLimited(items, concurrency, worker, options = {}) {
208
300
  let nextIndex = 0;
301
+ const waitForSlot = createPacer(options.requestRate, options.rampUpMs);
209
302
  const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
210
303
  while (nextIndex < items.length) {
211
304
  const index = nextIndex;
212
305
  nextIndex += 1;
306
+ await waitForSlot(index);
213
307
  await worker(items[index]);
214
308
  }
215
309
  });
@@ -233,6 +327,8 @@ function publicOptions(options) {
233
327
  concurrency: options.concurrency,
234
328
  sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
235
329
  timeoutMs: options.timeoutMs,
330
+ requestRate: options.requestRate || null,
331
+ rampUpMs: options.rampUpMs || null,
236
332
  profile: options.profile,
237
333
  promptCount: options.promptCount,
238
334
  greedy: options.greedy,
@@ -262,6 +358,53 @@ function hashOutput(output) {
262
358
  .slice(0, 16);
263
359
  }
264
360
 
361
+ function chunkGaps(times = []) {
362
+ const gaps = [];
363
+ for (let index = 1; index < times.length; index += 1) {
364
+ gaps.push(Math.max(0, times[index] - times[index - 1]));
365
+ }
366
+ return gaps;
367
+ }
368
+
369
+ function average(values) {
370
+ const clean = values.filter((value) => Number.isFinite(value));
371
+ if (clean.length === 0) return null;
372
+ return clean.reduce((sum, value) => sum + value, 0) / clean.length;
373
+ }
374
+
375
+ function createPacer(requestRate, rampUpMs = 0) {
376
+ if (!Number.isFinite(requestRate) || requestRate <= 0) {
377
+ return async () => {};
378
+ }
379
+
380
+ const startedAt = process.hrtime.bigint();
381
+ const offsets = [];
382
+ const steadyIntervalMs = 1000 / requestRate;
383
+ const warmIntervalMs = steadyIntervalMs * 4;
384
+
385
+ return async (index) => {
386
+ while (offsets.length <= index) {
387
+ const previousOffset = offsets.length === 0 ? 0 : offsets[offsets.length - 1];
388
+ const fraction = rampUpMs > 0 ? Math.min(1, previousOffset / rampUpMs) : 1;
389
+ const interval = warmIntervalMs + ((steadyIntervalMs - warmIntervalMs) * fraction);
390
+ offsets.push(offsets.length === 0 ? 0 : previousOffset + interval);
391
+ }
392
+
393
+ const targetMs = offsets[index];
394
+ const elapsedMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
395
+ const waitMs = targetMs - elapsedMs;
396
+ if (waitMs > 0) {
397
+ await new Promise((resolve) => setTimeout(resolve, waitMs));
398
+ }
399
+ };
400
+ }
401
+
402
+ function notify(options, event) {
403
+ if (typeof options.onProgress === 'function') {
404
+ options.onProgress(event);
405
+ }
406
+ }
407
+
265
408
  function evaluateSlo(metrics, options) {
266
409
  const thresholds = [
267
410
  ['firstTokenMs', options.sloTtftMs],
package/src/cli.js CHANGED
@@ -2,6 +2,7 @@ import fs from 'node:fs/promises';
2
2
  import { createRequire } from 'node:module';
3
3
  import { inspectModels, runBenchmark } from './bench.js';
4
4
  import { runProcess } from './process.js';
5
+ import { createProgress } from './progress.js';
5
6
  import { flattenResults, toCsv, writeReport } from './report.js';
6
7
  import { renderBenchmarkReport, renderModelsTable } from './table.js';
7
8
 
@@ -36,10 +37,21 @@ export async function runCli(argv = process.argv.slice(2)) {
36
37
  return;
37
38
  }
38
39
 
39
- const payload = await runBenchmark({
40
- ...parsed,
41
- version: packageJson.version
40
+ const progress = createProgress({
41
+ ...renderOptions(parsed),
42
+ enabled: resolveProgress(parsed),
43
+ stream: process.stderr
42
44
  });
45
+ let payload;
46
+ try {
47
+ payload = await runBenchmark({
48
+ ...parsed,
49
+ version: packageJson.version,
50
+ onProgress: (event) => progress.update(event)
51
+ });
52
+ } finally {
53
+ progress.stop();
54
+ }
43
55
 
44
56
  if (parsed.format === 'json') {
45
57
  console.log(JSON.stringify(payload, null, 2));
@@ -71,6 +83,8 @@ export function parseArgs(argv) {
71
83
  warmup: 0,
72
84
  concurrency: 1,
73
85
  sweepConcurrency: [],
86
+ requestRate: null,
87
+ rampUpMs: 0,
74
88
  timeoutMs: 60_000,
75
89
  profile: 'standard',
76
90
  greedy: true,
@@ -85,6 +99,7 @@ export function parseArgs(argv) {
85
99
  verbose: false,
86
100
  ascii: false,
87
101
  color: 'auto',
102
+ progress: 'auto',
88
103
  compact: false,
89
104
  width: null
90
105
  };
@@ -131,6 +146,12 @@ export function parseArgs(argv) {
131
146
  options.concurrency = options.sweepConcurrency[0];
132
147
  }
133
148
  break;
149
+ case '--request-rate':
150
+ options.requestRate = parsePositiveNumber(requireValue(arg, args), arg);
151
+ break;
152
+ case '--ramp-up-ms':
153
+ options.rampUpMs = parseNonNegativeInt(requireValue(arg, args), arg);
154
+ break;
134
155
  case '--timeout':
135
156
  case '--timeout-ms':
136
157
  options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
@@ -200,6 +221,12 @@ export function parseArgs(argv) {
200
221
  case '--no-color':
201
222
  options.color = 'never';
202
223
  break;
224
+ case '--progress':
225
+ options.progress = 'always';
226
+ break;
227
+ case '--no-progress':
228
+ options.progress = 'never';
229
+ break;
203
230
  case '--compact':
204
231
  options.compact = true;
205
232
  break;
@@ -279,6 +306,12 @@ function parsePositiveInt(value, option) {
279
306
  return parsed;
280
307
  }
281
308
 
309
+ function parsePositiveNumber(value, option) {
310
+ const parsed = Number.parseFloat(value);
311
+ if (!Number.isFinite(parsed) || parsed <= 0) throw new Error(`${option} must be a positive number`);
312
+ return parsed;
313
+ }
314
+
282
315
  function parseNonNegativeInt(value, option) {
283
316
  const parsed = Number.parseInt(value, 10);
284
317
  if (!Number.isInteger(parsed) || parsed < 0) throw new Error(`${option} must be a non-negative integer`);
@@ -312,6 +345,12 @@ function resolveColor(value) {
312
345
  return Boolean(process.stdout.isTTY);
313
346
  }
314
347
 
348
+ function resolveProgress(parsed) {
349
+ if (parsed.progress === 'always') return true;
350
+ if (parsed.progress === 'never') return false;
351
+ return parsed.format === 'table' ? 'auto' : false;
352
+ }
353
+
315
354
  function helpText() {
316
355
  return `fm-bench ${packageJson.version}
317
356
 
@@ -329,13 +368,15 @@ Run options:
329
368
  -c, --concurrency <n> Parallel fm processes (default: 1)
330
369
  --sweep-concurrency <list>
331
370
  Run separate operating points, e.g. 1,2,4
371
+ --request-rate <rps> Pace request starts at a target requests/sec
372
+ --ramp-up-ms <n> Gradually ramp request pacing over n ms
332
373
  --timeout-ms <n> Timeout per fm call in ms (default: 60000)
333
374
  --slo-ttft-ms <n> Count request as good only if TTFT is <= n
334
375
  --slo-e2e-ms <n> Count request as good only if E2E latency is <= n
335
376
  --slo-tpot-ms <n> Count request as good only if TPOT is <= n
336
377
  -p, --prompt <text> Prompt to benchmark; repeatable
337
378
  --prompt-file <file> .json, .jsonl, or blank-line separated text prompts
338
- --profile <name> quick, standard, interactive, throughput, or stress
379
+ --profile <name> quick, standard, interactive, throughput, client, or stress
339
380
  -i, --instructions <text> Instructions passed to fm respond
340
381
  --use-case <case> Pass a system model use case through to fm
341
382
  --guardrails <level> Pass a system model guardrail level through to fm
@@ -354,6 +395,8 @@ Output:
354
395
  --ascii Use plain ASCII tables instead of Unicode
355
396
  --color Force ANSI colors in table output
356
397
  --no-color Disable ANSI colors in table output
398
+ --progress Force live progress on stderr
399
+ --no-progress Disable live progress on stderr
357
400
  --compact Force compact terminal layout
358
401
  --width <n> Render for a specific terminal width
359
402
  -o, --out <file> Save JSON or CSV report based on file extension
@@ -367,6 +410,7 @@ Environment:
367
410
  Examples:
368
411
  fm-bench
369
412
  fm-bench --models system,pcc --runs 3 --profile stress
413
+ fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
370
414
  fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
371
415
  fm-bench models
372
416
  fm-bench doctor
package/src/fm.js CHANGED
@@ -176,7 +176,8 @@ export async function respond(fmBin, model, prompt, options = {}) {
176
176
  durationMs: result.durationMs,
177
177
  firstOutputMs: streamed ? result.firstStdoutMs : null,
178
178
  streamed,
179
- stdoutChunks: result.stdoutChunks
179
+ stdoutChunks: result.stdoutChunks,
180
+ stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
180
181
  };
181
182
  }
182
183
 
package/src/process.js CHANGED
@@ -20,6 +20,7 @@ export function runProcess(command, args = [], options = {}) {
20
20
  let stderr = '';
21
21
  let stdoutChunks = 0;
22
22
  let stderrChunks = 0;
23
+ const stdoutChunkTimesMs = [];
23
24
  let firstStdoutMs = null;
24
25
  let firstStderrMs = null;
25
26
  let timedOut = false;
@@ -38,10 +39,12 @@ export function runProcess(command, args = [], options = {}) {
38
39
  child.stdout.setEncoding('utf8');
39
40
  child.stderr.setEncoding('utf8');
40
41
  child.stdout.on('data', (chunk) => {
42
+ const chunkAtMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
41
43
  stdoutChunks += 1;
42
44
  if (firstStdoutMs == null && chunk.length > 0) {
43
- firstStdoutMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
45
+ firstStdoutMs = chunkAtMs;
44
46
  }
47
+ if (chunk.length > 0) stdoutChunkTimesMs.push(chunkAtMs);
45
48
  stdout += chunk;
46
49
  });
47
50
  child.stderr.on('data', (chunk) => {
@@ -65,6 +68,7 @@ export function runProcess(command, args = [], options = {}) {
65
68
  stderr: stderr || error.message,
66
69
  stdoutChunks,
67
70
  stderrChunks,
71
+ stdoutChunkTimesMs,
68
72
  firstStdoutMs,
69
73
  firstStderrMs,
70
74
  error,
@@ -86,6 +90,7 @@ export function runProcess(command, args = [], options = {}) {
86
90
  stderr,
87
91
  stdoutChunks,
88
92
  stderrChunks,
93
+ stdoutChunkTimesMs,
89
94
  firstStdoutMs,
90
95
  firstStderrMs,
91
96
  timedOut,
@@ -0,0 +1,190 @@
1
+ import { formatMs } from './table.js';
2
+
3
+ const UNICODE_FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'];
4
+ const ASCII_FRAMES = ['-', '\\', '|', '/'];
5
+
6
+ const TONES = {
7
+ green: ['\x1b[32m', '\x1b[0m'],
8
+ yellow: ['\x1b[33m', '\x1b[0m'],
9
+ red: ['\x1b[31m', '\x1b[0m'],
10
+ dim: ['\x1b[2m', '\x1b[0m']
11
+ };
12
+
13
+ export function createProgress(options = {}) {
14
+ const stream = options.stream || process.stderr;
15
+ const enabled = options.enabled === true || (options.enabled === 'auto' && Boolean(stream.isTTY));
16
+ if (!enabled) return noopProgress();
17
+ return new StatusLine({ ...options, stream });
18
+ }
19
+
20
+ function noopProgress() {
21
+ return {
22
+ update() {},
23
+ stop() {}
24
+ };
25
+ }
26
+
27
+ class StatusLine {
28
+ constructor(options = {}) {
29
+ this.stream = options.stream || process.stderr;
30
+ this.color = Boolean(options.color);
31
+ this.ascii = Boolean(options.ascii);
32
+ this.frames = this.ascii ? ASCII_FRAMES : UNICODE_FRAMES;
33
+ this.frameIndex = 0;
34
+ this.startedAt = Date.now();
35
+ this.state = {
36
+ phase: 'starting',
37
+ message: 'starting',
38
+ completed: 0,
39
+ failed: 0,
40
+ total: null
41
+ };
42
+ this.timer = setInterval(() => this.render(), 90);
43
+ this.timer.unref?.();
44
+ this.render();
45
+ }
46
+
47
+ update(event = {}) {
48
+ if (event.type === 'phase') {
49
+ this.state.phase = event.phase || 'working';
50
+ this.state.message = event.message || this.state.message;
51
+ } else if (event.type === 'tokens:start') {
52
+ this.state.phase = 'tokens';
53
+ this.state.message = event.message || 'counting prompt tokens';
54
+ this.state.completed = 0;
55
+ this.state.total = event.total;
56
+ } else if (event.type === 'tokens:progress') {
57
+ this.state.phase = 'tokens';
58
+ this.state.message = `counted ${event.promptId || 'prompt'}`;
59
+ this.state.completed = event.completed;
60
+ this.state.total = event.total;
61
+ } else if (event.type === 'benchmark:start') {
62
+ this.state.phase = 'benchmark';
63
+ this.state.message = `running ${event.modelCount} model(s), ${event.promptCount} prompt(s)`;
64
+ this.state.completed = 0;
65
+ this.state.failed = 0;
66
+ this.state.total = event.total;
67
+ } else if (event.type === 'warmup:start') {
68
+ this.state.phase = 'warmup';
69
+ this.state.message = `warming c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
70
+ this.state.completed = 0;
71
+ this.state.total = event.total;
72
+ } else if (event.type === 'warmup:progress') {
73
+ this.state.phase = 'warmup';
74
+ this.state.message = `warmed ${event.model || 'model'} at c${event.concurrency}`;
75
+ this.state.completed = event.completed;
76
+ this.state.total = event.total;
77
+ } else if (event.type === 'scenario:start') {
78
+ this.state.phase = 'benchmark';
79
+ this.state.message = `measuring c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
80
+ this.state.total = event.total ?? this.state.total;
81
+ } else if (event.type === 'benchmark:progress') {
82
+ this.state.phase = 'benchmark';
83
+ this.state.message = `${event.model} ${event.promptId} run ${event.run} ${event.ok ? 'ok' : 'failed'} (${formatMs(event.durationMs)})`;
84
+ this.state.completed = event.completed;
85
+ this.state.failed = event.failed;
86
+ this.state.total = event.total;
87
+ } else if (event.type === 'benchmark:complete') {
88
+ this.state.phase = 'complete';
89
+ this.state.message = `complete ${event.completed}/${event.total}`;
90
+ this.state.completed = event.completed;
91
+ this.state.failed = event.failed;
92
+ this.state.total = event.total;
93
+ }
94
+ this.render();
95
+ }
96
+
97
+ stop(finalMessage = '') {
98
+ if (this.timer) clearInterval(this.timer);
99
+ this.timer = null;
100
+ if (this.stream.clearLine && this.stream.cursorTo) {
101
+ this.clearLine();
102
+ } else {
103
+ this.stream.write('\n');
104
+ }
105
+ if (finalMessage) this.stream.write(`${finalMessage}\n`);
106
+ }
107
+
108
+ render() {
109
+ const width = this.stream.columns || process.stderr.columns || 100;
110
+ const elapsed = Date.now() - this.startedAt;
111
+ const frame = this.frames[this.frameIndex % this.frames.length];
112
+ this.frameIndex += 1;
113
+ const progress = progressText(this.state);
114
+ const eta = etaText(this.state, elapsed);
115
+ const failures = this.state.failed > 0 ? this.tone(`fail ${this.state.failed}`, 'red') : '';
116
+ const parts = [
117
+ this.tone(frame, 'green'),
118
+ 'fm-bench',
119
+ this.state.phase,
120
+ progress,
121
+ eta,
122
+ failures,
123
+ this.tone(this.state.message, 'dim')
124
+ ].filter(Boolean);
125
+ this.writeLine(truncateVisible(parts.join(' '), Math.max(20, width - 1)));
126
+ }
127
+
128
+ writeLine(line) {
129
+ if (this.stream.clearLine && this.stream.cursorTo) {
130
+ this.stream.clearLine(0);
131
+ this.stream.cursorTo(0);
132
+ this.stream.write(line);
133
+ } else {
134
+ this.stream.write(`\r${line}`);
135
+ }
136
+ }
137
+
138
+ clearLine() {
139
+ if (this.stream.clearLine && this.stream.cursorTo) {
140
+ this.stream.clearLine(0);
141
+ this.stream.cursorTo(0);
142
+ } else {
143
+ this.stream.write('\r');
144
+ }
145
+ }
146
+
147
+ tone(text, tone) {
148
+ if (!this.color || !TONES[tone]) return text;
149
+ const [open, close] = TONES[tone];
150
+ return `${open}${text}${close}`;
151
+ }
152
+ }
153
+
154
+ function progressText(state) {
155
+ if (!Number.isFinite(state.total) || state.total <= 0) return '';
156
+ const completed = Math.min(state.completed || 0, state.total);
157
+ const percent = Math.round((completed / state.total) * 100);
158
+ return `${completed}/${state.total} ${percent}%`;
159
+ }
160
+
161
+ function etaText(state, elapsedMs) {
162
+ if (!Number.isFinite(state.total) || state.total <= 0 || !Number.isFinite(state.completed) || state.completed <= 0) {
163
+ return '';
164
+ }
165
+ const remaining = Math.max(0, state.total - state.completed);
166
+ if (remaining === 0) return `elapsed ${formatMs(elapsedMs)}`;
167
+ const perItemMs = elapsedMs / state.completed;
168
+ return `eta ${formatMs(perItemMs * remaining)}`;
169
+ }
170
+
171
+ function truncateVisible(value, width) {
172
+ const text = String(value);
173
+ const clean = text.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, '');
174
+ if (clean.length <= width) return text;
175
+ let visible = 0;
176
+ let output = '';
177
+ for (let index = 0; index < text.length && visible < width - 1; index += 1) {
178
+ if (text[index] === '\x1b') {
179
+ const match = text.slice(index).match(/^\u001b\[[0-?]*[ -/]*[@-~]/);
180
+ if (match) {
181
+ output += match[0];
182
+ index += match[0].length - 1;
183
+ continue;
184
+ }
185
+ }
186
+ output += text[index];
187
+ visible += 1;
188
+ }
189
+ return `${output}…`;
190
+ }
package/src/prompts.js CHANGED
@@ -50,6 +50,28 @@ const PROFILES = {
50
50
  prompt: 'Create a compact test plan for benchmarking a local foundation model across short, medium, and long prompts.'
51
51
  }
52
52
  ],
53
+ client: [
54
+ {
55
+ id: 'short-chat',
56
+ prompt: 'In one sentence, explain why time to first token matters for an interactive assistant.'
57
+ },
58
+ {
59
+ id: 'content-generation',
60
+ prompt: 'Write a practical 180-word product update for developers explaining a new terminal benchmark feature. Keep it specific and avoid marketing fluff.'
61
+ },
62
+ {
63
+ id: 'structured-extraction',
64
+ prompt: 'Return compact valid JSON with keys "risk", "owner", "deadline", and "next_step" from this note: The benchmark release is blocked by flaky p95 latency on the PCC model. Maya owns the investigation and needs a fix before Friday.'
65
+ },
66
+ {
67
+ id: 'summarization-light',
68
+ prompt: 'Summarize this in three bullets: A serious local LLM benchmark should separate time to first token from total latency, report tail percentiles, include prompt and output token counts, measure throughput at multiple concurrency operating points, and preserve the raw prompt suite so future runs are comparable.'
69
+ },
70
+ {
71
+ id: 'code-analysis',
72
+ prompt: 'Review this JavaScript function for one correctness issue and one readability improvement: function p(v){let s=0;for(let i=0;i<=v.length;i++)s+=v[i];return s/v.length}'
73
+ }
74
+ ],
53
75
  stress: [
54
76
  {
55
77
  id: 'interactive-short',
package/src/report.js CHANGED
@@ -27,9 +27,13 @@ export function flattenResults(results) {
27
27
  words: result.words,
28
28
  tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
29
29
  decode_tokens_per_second: result.decodeTokensPerSecond == null ? '' : round(result.decodeTokensPerSecond),
30
+ prefill_tokens_per_second: result.prefillTokensPerSecond == null ? '' : round(result.prefillTokensPerSecond),
30
31
  chars_per_second: round(result.charsPerSecond),
31
32
  streamed: result.streamed,
32
33
  stdout_chunks: result.stdoutChunks,
34
+ second_chunk_ms: round(result.secondChunkMs),
35
+ chunk_gap_avg_ms: round(result.chunkGapAvgMs),
36
+ chunk_gap_max_ms: round(result.chunkGapMaxMs),
33
37
  output_hash: result.outputHash || '',
34
38
  good: result.good == null ? '' : result.good,
35
39
  error: result.error || ''
package/src/stats.js CHANGED
@@ -101,10 +101,15 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
101
101
  const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
102
102
  const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
103
103
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
104
+ const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
105
+ const secondChunk = summarizeNumbers(successes.map((result) => result.secondChunkMs).filter((value) => value != null));
106
+ const chunkGap = summarizeNumbers(successes.flatMap((result) => result.chunkGapsMs || []));
104
107
  const windowMs = modelWindowMs(successes);
105
108
  const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
106
- const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
109
+ const goodputRps = goodMeasured.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
107
110
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
111
+ const totalTokens = promptTokens.sum + outputTokens.sum;
112
+ const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
108
113
 
109
114
  return {
110
115
  model: entry.model,
@@ -120,6 +125,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
120
125
  rps,
121
126
  goodputRps,
122
127
  outputTokenThroughput,
128
+ totalTokenThroughput,
123
129
  repeatability: summarizeRepeatability(successes),
124
130
  latency,
125
131
  ttft,
@@ -129,7 +135,10 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
129
135
  outputTokens,
130
136
  charsPerSecond,
131
137
  tokensPerSecond,
132
- decodeTokensPerSecond
138
+ decodeTokensPerSecond,
139
+ prefillTokensPerSecond,
140
+ secondChunk,
141
+ chunkGap
133
142
  };
134
143
  });
135
144
  }
package/src/table.js CHANGED
@@ -93,6 +93,7 @@ export function renderSummaryTable(summary, options = {}) {
93
93
  cell(item.attempted ? `${item.successes}/${item.attempted}` : '-', item.failures > 0 ? 'yellow' : item.available ? 'green' : 'muted'),
94
94
  cell(formatPercent(item.successRate), percentTone(item.successRate, 0.95, 1)),
95
95
  cell(formatPercent(item.goodputRate), percentTone(item.goodputRate, 0.8, 1)),
96
+ cell(formatNumber(item.goodputRps), tones.goodputRps),
96
97
  cell(formatMs(item.ttft.p50), tones.ttft),
97
98
  cell(formatMs(item.ttft.p95), tones.ttftP95),
98
99
  cell(formatMs(item.latency.p50), tones.e2e),
@@ -111,34 +112,34 @@ export function renderSummaryTable(summary, options = {}) {
111
112
  base[2],
112
113
  base[3]
113
114
  ];
114
- if (hasGoodput) medium.push(base[5]);
115
- medium.push(base[6], base[8], base[9], base[10], base[11], base[13], base[14]);
115
+ if (hasGoodput) medium.push(base[5], base[6]);
116
+ medium.push(base[7], base[9], base[10], base[11], base[12], base[14], base[15]);
116
117
  return medium;
117
118
  }
118
119
 
119
120
  const wide = [base[0], base[1], base[2], base[3], base[4]];
120
- if (hasGoodput) wide.push(base[5]);
121
+ if (hasGoodput) wide.push(base[5], base[6]);
121
122
  wide.push(
122
- base[6],
123
123
  base[7],
124
124
  base[8],
125
125
  base[9],
126
+ base[10],
126
127
  cell(formatMs(item.tpot.p50), tones.tpot),
127
128
  cell(formatMs(item.tpot.p95), tones.tpotP95),
128
- base[10],
129
129
  base[11],
130
130
  base[12],
131
131
  base[13],
132
- base[14]
132
+ base[14],
133
+ base[15]
133
134
  );
134
135
  return wide;
135
136
  });
136
137
 
137
- const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
138
- const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
138
+ const mediumHeaders = ['c', 'model', 'status', 'ok', 'good', 'good rps', 'ttft', 'e2e', 'e2e p95', 'user/s', 'sys/s', 'cv', 'note'];
139
+ const wideHeaders = ['c', 'model', 'status', 'ok/runs', 'succ', 'good', 'good rps', 'ttft', 'ttft p95', 'e2e', 'e2e p95', 'tpot', 'tpot p95', 'user t/s', 'sys t/s', 'rps', 'cv', 'note'];
139
140
  const headers = mode === 'medium'
140
- ? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good'))
141
- : (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good'));
141
+ ? (hasGoodput ? mediumHeaders : mediumHeaders.filter((header) => header !== 'good' && header !== 'good rps'))
142
+ : (hasGoodput ? wideHeaders : wideHeaders.filter((header) => header !== 'good' && header !== 'good rps'));
142
143
 
143
144
  return renderTable(headers, rows, { ...options, maxCellWidth: mode === 'medium' ? 24 : 52 });
144
145
  }
@@ -153,7 +154,10 @@ export function renderDetailTable(summary, options = {}) {
153
154
  cell(item.model, item.available ? null : 'muted'),
154
155
  cell(formatNumber(item.promptTokens.avg, 0)),
155
156
  cell(formatNumber(item.outputTokens.avg, 0)),
156
- cell(formatNumber(item.decodeTokensPerSecond.avg), tones.decodeTps),
157
+ cell(formatNumber(item.prefillTokensPerSecond?.avg), tones.prefillTps),
158
+ cell(formatNumber(item.decodeTokensPerSecond?.avg), tones.decodeTps),
159
+ cell(formatMs(item.secondChunk?.p50), tones.secondChunk),
160
+ cell(formatMs(item.chunkGap?.p95), tones.chunkGapP95),
157
161
  cell(formatMs(item.latency.p99), tones.e2eP99),
158
162
  cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
159
163
  cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
@@ -161,20 +165,23 @@ export function renderDetailTable(summary, options = {}) {
161
165
  ];
162
166
 
163
167
  if (mode === 'medium') {
164
- return base.slice(0, 8);
168
+ return [base[0], base[1], base[2], base[3], base[4], base[5], base[7], base[8], base[10]];
165
169
  }
166
170
 
167
171
  return base;
168
172
  });
169
173
 
170
174
  const headers = mode === 'medium'
171
- ? ['c', 'model', 'in avg', 'out avg', 'decode t/s', 'e2e p99', 'e2e 95% ci', 'repeat']
175
+ ? ['c', 'model', 'in avg', 'out avg', 'prefill/s', 'decode/s', 'chunk p95', 'e2e p99', 'repeat']
172
176
  : [
173
177
  'c',
174
178
  'model',
175
179
  'in tok avg',
176
180
  'out tok avg',
181
+ 'prefill tok/s',
177
182
  'decode tok/s',
183
+ '2nd chunk',
184
+ 'chunk p95',
178
185
  'e2e p99',
179
186
  'e2e 95% ci',
180
187
  'repeat',
@@ -200,7 +207,8 @@ export function renderCompactSummary(summary, options = {}) {
200
207
  const tones = metricTones(item, ranks, options.slo);
201
208
  lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
202
209
  lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
203
- lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | TPOT ${formatMs(item.tpot.p50)} | repeat ${formatPercent(item.repeatability)}`, width), worstTone(tones.tpot, percentTone(item.repeatability, 0.5, 0.9)), options));
210
+ lines.push(toneText(truncate(` prefill ${formatNumber(item.prefillTokensPerSecond?.avg)} tok/s | TPOT ${formatMs(item.tpot.p50)} | chunk p95 ${formatMs(item.chunkGap?.p95)}`, width), worstTone(tones.prefillTps, tones.tpot, tones.chunkGapP95), options));
211
+ lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | repeat ${formatPercent(item.repeatability)}`, width), percentTone(item.repeatability, 0.5, 0.9), options));
204
212
  } else {
205
213
  lines.push(toneText(truncate(` ${compactReason(item.skippedReason)}`, width), 'yellow', options));
206
214
  }
@@ -396,10 +404,15 @@ function rankSummary(summary) {
396
404
  e2eP99: collectMetric(summary, (item) => item.latency?.p99),
397
405
  tpot: collectMetric(summary, (item) => item.tpot?.p50),
398
406
  tpotP95: collectMetric(summary, (item) => item.tpot?.p95),
407
+ secondChunk: collectMetric(summary, (item) => item.secondChunk?.p50),
408
+ chunkGapP95: collectMetric(summary, (item) => item.chunkGap?.p95),
399
409
  userTps: collectMetric(summary, (item) => item.tokensPerSecond?.avg),
400
410
  systemTps: collectMetric(summary, (item) => item.outputTokenThroughput),
411
+ totalTps: collectMetric(summary, (item) => item.totalTokenThroughput),
401
412
  rps: collectMetric(summary, (item) => item.rps),
402
- decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg)
413
+ goodputRps: collectMetric(summary, (item) => item.goodputRps),
414
+ decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg),
415
+ prefillTps: collectMetric(summary, (item) => item.prefillTokensPerSecond?.avg)
403
416
  };
404
417
  }
405
418
 
@@ -440,10 +453,15 @@ function metricTones(item, ranks, slo = {}) {
440
453
  e2eP99: thresholdTone(item.latency?.p99, slo.e2eMs) || rankToneLower(item.latency?.p99, ranks.e2eP99),
441
454
  tpot: thresholdTone(item.tpot?.p50, slo.tpotMs) || rankToneLower(item.tpot?.p50, ranks.tpot),
442
455
  tpotP95: thresholdTone(item.tpot?.p95, slo.tpotMs) || rankToneLower(item.tpot?.p95, ranks.tpotP95),
456
+ secondChunk: rankToneLower(item.secondChunk?.p50, ranks.secondChunk),
457
+ chunkGapP95: rankToneLower(item.chunkGap?.p95, ranks.chunkGapP95),
443
458
  userTps: rankTone(item.tokensPerSecond?.avg, ranks.userTps),
444
459
  systemTps: rankTone(item.outputTokenThroughput, ranks.systemTps),
460
+ totalTps: rankTone(item.totalTokenThroughput, ranks.totalTps),
445
461
  rps: rankTone(item.rps, ranks.rps),
446
- decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps)
462
+ goodputRps: goodputRpsTone(item, ranks.goodputRps),
463
+ decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps),
464
+ prefillTps: rankTone(item.prefillTokensPerSecond?.avg, ranks.prefillTps)
447
465
  };
448
466
  }
449
467
 
@@ -453,3 +471,9 @@ function worstTone(...tones) {
453
471
  if (tones.includes('green')) return 'green';
454
472
  return null;
455
473
  }
474
+
475
+ function goodputRpsTone(item, values) {
476
+ const passTone = percentTone(item.goodputRate, 0.8, 1);
477
+ if (passTone && passTone !== 'green') return passTone;
478
+ return rankTone(item.goodputRps, values) || passTone;
479
+ }