fm-bench 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  `fm-bench` is a dynamic benchmark CLI for Apple's `fm` command on macOS 27 and newer.
4
4
 
5
- It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, and prints a terminal table with latency and throughput stats.
5
+ It discovers the models reported by `fm --help`, checks availability with `fm available`, runs repeatable prompt suites through `fm respond`, counts tokens with `fm token-count`, shows live progress while it works, and prints terminal tables with latency, throughput, stability, goodput, and streaming-quality stats.
6
6
 
7
7
  Apple introduced the preinstalled `fm` command for macOS 27 as part of the Foundation Models tooling. `fm-bench` intentionally shells out to the system `fm` binary instead of linking private APIs, so it can adapt as Apple adds models or changes availability.
8
8
 
@@ -35,20 +35,15 @@ fm-bench
35
35
  Example output:
36
36
 
37
37
  ```text
38
- fm-bench 0.2.0 | darwin/arm64 | fm
39
- prompts 3 | runs 1 | concurrency 1 | stream on | measured 3 | failed 0 | skipped models 0 | elapsed 11.36s
40
-
41
- ┌────────┬────────┬──────┬────┬─────────┬──────────┬──────────┬─────────┬─────────┬──────────┬───────┬─────┬──────┐
42
- │ MODEL │ STATUS │ RUNS │ OK │ SUCCESS │ TTFT P50 │ TTFT P95 │ E2E P50 │ E2E P95 │ TPOT P50 │ TOK/S │ RPS │ NOTE │
43
- ├────────┼────────┼──────┼────┼─────────┼──────────┼──────────┼─────────┼─────────┼──────────┼───────┼─────┼──────┤
44
- │ system │ ok │ 3 │ 3 │ 100% │ 409ms │ 486ms │ 2.29s │ 5.97s │ 14ms │ 58.5 │ 0.3 │ │
45
- └────────┴────────┴──────┴────┴─────────┴──────────┴──────────┴─────────┴─────────┴──────────┴───────┴─────┴──────┘
46
-
47
- ┌────────┬────────────┬─────────────┬─────────────┬──────────────┬─────────┬──────────┬────────┬──────────────────────────────────┐
48
- │ MODEL │ IN TOK AVG │ OUT TOK AVG │ TOTAL TOK/S │ DECODE TOK/S │ E2E P99 │ TPOT P95 │ REPEAT │ DESCRIPTION │
49
- ├────────┼────────────┼─────────────┼─────────────┼──────────────┼─────────┼──────────┼────────┼──────────────────────────────────┤
50
- │ system │ 35 │ 196 │ 59.5 │ 75.0 │ 6.30s │ 15ms │ - │ On-device Apple Foundation Model │
51
- └────────┴────────────┴─────────────┴─────────────┴──────────────┴─────────┴──────────┴────────┴──────────────────────────────────┘
38
+ fm-bench 0.4.0 | darwin/arm64 | fm
39
+ prompts 5 | runs 3 | concurrency 1,2 | stream on | measured 30 | failed 0 | skipped 0 | elapsed 42.10s | SLO TTFT<=750ms,E2E<=4.00s
40
+
41
+ ┌───┬────────┬────────┬─────────┬──────┬──────┬──────────┬──────┬──────────┬─────┬─────┐
42
+ │ C │ MODEL │ STATUS │ OK/RUNS │ SUCC │ GOOD │ GOOD RPS │ TTFT │ E2E P95 │ SYS │ CV │
43
+ ├───┼────────┼────────┼─────────┼──────┼──────┼──────────┼──────┼──────────┼─────┼─────┤
44
+ │ 1 │ system │ ok │ 15/15 │ 100% │ 93% │ 0.4 │ 318ms│ 3.20s │ 42 │ 12% │
45
+ │ 2 │ system │ ok │ 15/15 │ 100% │ 80% │ 0.7 │ 501ms│ 4.40s │ 68 │ 21% │
46
+ └───┴────────┴────────┴─────────┴──────┴──────┴──────────┴──────┴──────────┴─────┴─────┘
52
47
  ```
53
48
 
54
49
  ## Commands
@@ -72,6 +67,7 @@ fm-bench --models system,pcc --runs 3 --profile stress
72
67
  fm-bench --models system --runs 5 --profile interactive
73
68
  fm-bench --models system --runs 3 --profile throughput --warmup 1
74
69
  fm-bench --models system --profile interactive --sweep-concurrency 1,2,4
70
+ fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5 --ramp-up-ms 2000
75
71
  fm-bench --models system --runs 5 --slo-ttft-ms 750 --slo-e2e-ms 4000
76
72
  fm-bench --prompt "Reply with exactly: ok" --runs 5
77
73
  fm-bench --prompt-file prompts.json --format json --out reports/bench.json
@@ -85,9 +81,11 @@ Useful flags:
85
81
  - `--warmup <n>`: warmup runs per model before measurement.
86
82
  - `--concurrency <n>`: parallel `fm` processes.
87
83
  - `--sweep-concurrency <list>`: run separate measured operating points, such as `1,2,4`.
84
+ - `--request-rate <rps>`: pace request starts at a target requests-per-second rate.
85
+ - `--ramp-up-ms <n>`: gradually ramp request pacing over `n` milliseconds.
88
86
  - `--timeout-ms <n>`: timeout per `fm` call.
89
87
  - `--slo-ttft-ms <n>`, `--slo-e2e-ms <n>`, `--slo-tpot-ms <n>`: count goodput against latency budgets.
90
- - `--profile quick|standard|interactive|throughput|stress`: built-in prompt suite.
88
+ - `--profile quick|standard|interactive|throughput|client|stress`: built-in prompt suite.
91
89
  - `--prompt <text>`: custom prompt, repeatable.
92
90
  - `--prompt-file <file>`: JSON, JSONL, or blank-line separated text prompts.
93
91
  - `--instructions <text>`: passed to `fm respond`.
@@ -95,6 +93,8 @@ Useful flags:
95
93
  - `--capture-output`: include raw model output in JSON reports.
96
94
  - `--json`, `--csv`, `--format table|json|csv`: choose output format.
97
95
  - `--ascii`: use plain ASCII table borders.
96
+ - `--color`, `--no-color`: force or disable semantic ANSI colors. Colors are automatic on TTYs.
97
+ - `--progress`, `--no-progress`: force or disable the live progress status line on stderr.
98
98
  - `--compact`: force the narrow terminal layout.
99
99
  - `--width <n>`: render as if the terminal has `n` columns.
100
100
  - `--out <file>`: save a report.
@@ -126,8 +126,11 @@ Plain text files are split on blank lines.
126
126
  - TTFT, or time to first streamed output.
127
127
  - E2E latency, or full response wall-clock latency.
128
128
  - TPOT, or decode time per output token after the first output token.
129
+ - second-chunk delay and chunk-gap p95 as terminal-side streaming smoothness signals.
130
+ - prefill tokens per second, or prompt tokens divided by TTFT.
129
131
  - output tokens per second per request.
130
132
  - total output token throughput across the measured window.
133
+ - total token throughput, including prompt and output tokens.
131
134
  - requests per second across the measured window.
132
135
  - goodput percentage and goodput RPS when SLO flags are set.
133
136
  - coefficient of variation (CV) and confidence interval context for stability.
@@ -139,10 +142,28 @@ Plain text files are split on blank lines.
139
142
 
140
143
  Token counts come from `fm token-count --quiet`. If `fm` cannot count a response, token fields are left blank while character throughput is still reported.
141
144
 
142
- Measured runs stream by default so `fm-bench` can capture TTFT. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT and TPOT fields that depend on streaming will be blank.
145
+ Measured runs stream by default so `fm-bench` can capture TTFT and streaming smoothness. Use `--no-stream` if you need buffered `fm respond` behavior; TTFT, TPOT, second-chunk, and chunk-gap fields that depend on streaming will be blank.
143
146
 
144
147
  Terminal output is responsive. Wide terminals show full scoreboard and detail tables, medium terminals show a tighter operating-point table, and narrow terminals switch to compact model cards. Use `--width` to preview a layout and `--ascii` for log systems that do not render Unicode borders well.
145
148
 
149
+ ## Live Progress
150
+
151
+ Interactive terminal runs show a single-line status indicator on stderr while prompts are loaded, models are inspected, tokens are counted, warmups run, and benchmark jobs complete. The final report still prints to stdout, so `--json`, `--csv`, and `--out` remain automation-friendly.
152
+
153
+ Progress is automatic for table output on TTYs. Use `--progress` to force it or `--no-progress` to keep the terminal completely quiet until the report is ready.
154
+
155
+ ## Terminal Colors
156
+
157
+ Table output uses semantic ANSI color on interactive terminals:
158
+
159
+ - green: passing, steadier, or better than the current comparison set.
160
+ - yellow: marginal, partial, or near a budget.
161
+ - red: failing a budget, unstable, or slower/lower than peers.
162
+
163
+ Success rate, goodput, repeatability, and CV use fixed benchmark thresholds. Throughput columns use relative ranking within the current run because “good” depends on the machine, model, prompt mix, and concurrency. TTFT, E2E, and TPOT use SLO thresholds when you pass `--slo-ttft-ms`, `--slo-e2e-ms`, or `--slo-tpot-ms`; otherwise they use lower-is-better relative ranking across the models and operating points in the report.
164
+
165
+ Use `--color` to force ANSI colors in captured logs, or `--no-color` for plain output. `NO_COLOR=1` disables automatic color and `FORCE_COLOR=1` enables it.
166
+
146
167
  See [docs/methodology.md](docs/methodology.md) for the benchmark methodology and source references.
147
168
 
148
169
  ## Requirements
@@ -9,6 +9,7 @@ The metric set follows common LLM inference benchmark practice:
9
9
  - Apple introduces the macOS 27 `fm` command as a preinstalled way to use Foundation Models from the terminal and scripts: <https://developer.apple.com/videos/play/wwdc2026/334/>
10
10
  - NVIDIA NIM benchmarking defines TTFT, end-to-end latency, inter-token latency / TPOT, tokens per second, and requests per second: <https://docs.nvidia.com/nim/benchmarking/llm/latest/metrics.html>
11
11
  - NVIDIA GenAI-Perf reports TTFT, inter-token latency, request latency, sequence lengths, output token throughput, and JSON/CSV artifacts: <https://docs.nvidia.com/deeplearning/triton-inference-server/user-guide/docs/perf_analyzer/genai-perf/README.html>
12
+ - NVIDIA AIPerf documents time to second token, inter-token latency, inter-chunk latency, per-user output throughput, and prefill throughput: <https://docs.nvidia.com/aiperf/reference/ai-perf-metrics-reference>
12
13
  - vLLM benchmark tooling reports TTFT, TPOT, ITL, E2E percentiles and SLO-oriented goodput: <https://docs.vllm.ai/en/stable/cli/bench/serve/>
13
14
  - MLCommons describes varying concurrency and reporting verified operating points for TTFT, throughput, interactivity, and response latency rather than interpolated performance: <https://mlcommons.org/2026/03/mlperf-endpoints-gen-ai-benchmarking/>
14
15
  - MLPerf Client emphasizes local client workloads with multiple task types and varying prompt/response lengths: <https://mlcommons.org/benchmarks/client/>
@@ -19,11 +20,16 @@ The metric set follows common LLM inference benchmark practice:
19
20
  - `E2E latency`: time from starting `fm respond` until the process exits and the full response is captured.
20
21
  - `generation_ms`: `E2E - TTFT`.
21
22
  - `TPOT`: `(E2E - TTFT) / (output_tokens - 1)`. The first output token is excluded so TPOT focuses on decode cadence.
23
+ - `second_chunk_ms`: time between the first and second streamed stdout chunks. This is a terminal-side proxy for time-to-second-token style startup smoothness.
24
+ - `chunk_gap`: the distribution of time between consecutive streamed stdout chunks. It is useful for spotting streaming jitter, but it is chunk-based rather than token-based because the `fm` CLI writes stdout chunks, not token timestamp events.
25
+ - `prefill_tokens_per_second`: input prompt tokens divided by TTFT seconds. This estimates prompt-processing speed for streaming runs.
22
26
  - `tokens_per_second`: output tokens divided by E2E seconds for one request.
23
27
  - `decode_tokens_per_second`: output tokens after the first token divided by generation seconds.
24
28
  - `total output token throughput`: all successful output tokens for a model divided by that model's measured wall-clock window.
29
+ - `total token throughput`: successful prompt and output tokens divided by that model's measured wall-clock window.
25
30
  - `RPS`: successful requests divided by that model's measured wall-clock window.
26
31
  - `goodput`: successful requests that also satisfy all provided SLO thresholds.
32
+ - `goodput RPS`: SLO-passing requests divided by that model's measured wall-clock window. If SLOs are set and no requests pass, this is reported as zero.
27
33
  - `repeatability`: for repeated runs of the same prompt, the average share of runs that produced the most common normalized output hash.
28
34
  - `CV`: coefficient of variation, or sample standard deviation divided by the mean. Lower values indicate steadier latency for that metric.
29
35
  - `95% CI`: a t-distribution confidence interval around the sample mean. Treat it as useful context, not proof, especially with very small sample sizes.
@@ -32,12 +38,20 @@ The metric set follows common LLM inference benchmark practice:
32
38
 
33
39
  Use `--sweep-concurrency 1,2,4` to measure separate concurrency operating points. This follows the same idea as MLCommons endpoint reporting: a single peak number hides the tradeoff between system throughput and per-user responsiveness.
34
40
 
41
+ Use `--request-rate <rps>` to pace request starts independently of concurrency. Concurrency limits how many `fm respond` processes can be active at once; request rate controls how quickly new work is admitted. Use `--ramp-up-ms` to avoid instantly shocking a model or quota path when you start a higher-rate run.
42
+
35
43
  `fm-bench` does not interpolate between operating points. It reports only what was actually measured.
36
44
 
45
+ ## Prompt Profiles
46
+
47
+ The `client` profile is a pragmatic local-machine mix inspired by MLPerf Client's emphasis on multiple task categories and prompt/response lengths. It includes short chat, content generation, structured extraction, light summarization, and code analysis prompts. It is not a formal MLPerf submission suite; it is a convenient built-in workload for comparing your own Mac, OS build, and `fm` models over time.
48
+
37
49
  ## Caveats
38
50
 
39
51
  `fm-bench` uses `fm token-count --quiet` as the source of token counts, so token values follow Apple's local tokenizer behavior. It does not judge semantic quality unless you provide your own prompt suite and inspect captured outputs with `--capture-output`.
40
52
 
41
53
  Client-side measurements include process startup, local queueing, model prefill, streaming, detokenization, and terminal pipe overhead. That is intentional for a command-line benchmark, but it is not the same as an internal model-kernel benchmark.
42
54
 
43
- For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput profiles, and compare models at the same concurrency operating points.
55
+ Stream smoothness metrics use stdout chunk arrival times. A chunk can contain more than one token, and terminal or pipe buffering can affect chunk boundaries. Treat `second_chunk_ms` and `chunk_gap` as user-visible streaming diagnostics, not raw decoder telemetry.
56
+
57
+ For serious comparisons, prefer at least three runs per prompt, include warmups, benchmark both interactive and throughput or client profiles, compare models at the same concurrency operating points, set SLOs that match your real UX budget, and save JSON reports for later analysis.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fm-bench",
3
- "version": "0.3.0",
3
+ "version": "0.4.0",
4
4
  "description": "Dynamic benchmark CLI for Apple's fm command on macOS 27+.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/bench.js CHANGED
@@ -36,7 +36,9 @@ export async function inspectModels(options = {}) {
36
36
 
37
37
  export async function runBenchmark(options = {}) {
38
38
  const startedAt = new Date().toISOString();
39
+ notify(options, { type: 'phase', phase: 'prompts', message: 'loading prompts' });
39
40
  const prompts = await loadPrompts(options);
41
+ notify(options, { type: 'phase', phase: 'models', message: 'discovering models' });
40
42
  const inspection = await inspectModels(options);
41
43
  const modelStatuses = options.availableOnly
42
44
  ? inspection.models.filter((model) => model.available)
@@ -45,15 +47,36 @@ export async function runBenchmark(options = {}) {
45
47
  const environment = await collectEnvironment(inspection.fmBin);
46
48
  const promptTokenCounts = new Map();
47
49
  const concurrencies = normalizeConcurrencySweep(options);
50
+ const totalRuns = concurrencies.length * runnableModels.length * prompts.length * options.runs;
48
51
 
52
+ notify(options, {
53
+ type: 'tokens:start',
54
+ total: prompts.length,
55
+ message: 'counting prompt tokens'
56
+ });
49
57
  for (const prompt of prompts) {
50
58
  const counted = await countTokens(inspection.fmBin, prompt.prompt, options);
51
59
  promptTokenCounts.set(prompt.id, counted.ok ? counted.count : null);
60
+ notify(options, {
61
+ type: 'tokens:progress',
62
+ completed: promptTokenCounts.size,
63
+ total: prompts.length,
64
+ promptId: prompt.id
65
+ });
52
66
  }
53
67
 
54
68
  const results = [];
55
69
  const scenarios = [];
56
- for (const concurrency of concurrencies) {
70
+ let completedRuns = 0;
71
+ let failedRuns = 0;
72
+ notify(options, {
73
+ type: 'benchmark:start',
74
+ total: totalRuns,
75
+ modelCount: runnableModels.length,
76
+ promptCount: prompts.length,
77
+ scenarioCount: concurrencies.length
78
+ });
79
+ for (const [scenarioIndex, concurrency] of concurrencies.entries()) {
57
80
  const scenario = await runScenario({
58
81
  fmBin: inspection.fmBin,
59
82
  prompts,
@@ -61,7 +84,26 @@ export async function runBenchmark(options = {}) {
61
84
  modelStatuses,
62
85
  promptTokenCounts,
63
86
  options,
64
- concurrency
87
+ concurrency,
88
+ scenarioIndex: scenarioIndex + 1,
89
+ scenarioCount: concurrencies.length,
90
+ onMeasuredResult: (result) => {
91
+ completedRuns += 1;
92
+ if (!result.ok) failedRuns += 1;
93
+ notify(options, {
94
+ type: 'benchmark:progress',
95
+ completed: completedRuns,
96
+ failed: failedRuns,
97
+ total: totalRuns,
98
+ concurrency,
99
+ model: result.model,
100
+ promptId: result.promptId,
101
+ run: result.run,
102
+ ok: result.ok,
103
+ durationMs: result.durationMs,
104
+ firstTokenMs: result.firstTokenMs
105
+ });
106
+ }
65
107
  });
66
108
  scenarios.push(scenario);
67
109
  results.push(...scenario.results);
@@ -73,7 +115,7 @@ export async function runBenchmark(options = {}) {
73
115
  || a.run - b.run);
74
116
 
75
117
  const summary = summarizeByModel(results, modelStatuses, { concurrencies });
76
- return {
118
+ const payload = {
77
119
  tool: 'fm-bench',
78
120
  version: options.version,
79
121
  startedAt,
@@ -90,6 +132,13 @@ export async function runBenchmark(options = {}) {
90
132
  summary,
91
133
  results
92
134
  };
135
+ notify(options, {
136
+ type: 'benchmark:complete',
137
+ completed: completedRuns,
138
+ failed: failedRuns,
139
+ total: totalRuns
140
+ });
141
+ return payload;
93
142
  }
94
143
 
95
144
  async function runScenario(context) {
@@ -100,16 +149,40 @@ async function runScenario(context) {
100
149
  modelStatuses,
101
150
  promptTokenCounts,
102
151
  options,
103
- concurrency
152
+ concurrency,
153
+ scenarioIndex,
154
+ scenarioCount,
155
+ onMeasuredResult
104
156
  } = context;
105
157
  const startedAt = new Date().toISOString();
106
158
 
159
+ const warmupTotal = options.warmup * runnableModels.length;
160
+ if (warmupTotal > 0) {
161
+ notify(options, {
162
+ type: 'warmup:start',
163
+ concurrency,
164
+ scenarioIndex,
165
+ scenarioCount,
166
+ total: warmupTotal
167
+ });
168
+ }
169
+ let warmupCompleted = 0;
107
170
  for (let warmupIndex = 0; warmupIndex < options.warmup; warmupIndex += 1) {
108
171
  for (const model of runnableModels) {
109
172
  await respond(fmBin, model.name, prompts[0].prompt, {
110
173
  ...options,
111
174
  stream: false
112
175
  });
176
+ warmupCompleted += 1;
177
+ notify(options, {
178
+ type: 'warmup:progress',
179
+ concurrency,
180
+ scenarioIndex,
181
+ scenarioCount,
182
+ completed: warmupCompleted,
183
+ total: warmupTotal,
184
+ model: model.name
185
+ });
113
186
  }
114
187
  }
115
188
 
@@ -124,15 +197,23 @@ async function runScenario(context) {
124
197
  }
125
198
 
126
199
  const results = [];
200
+ notify(options, {
201
+ type: 'scenario:start',
202
+ concurrency,
203
+ scenarioIndex,
204
+ scenarioCount,
205
+ total: jobs.length
206
+ });
127
207
  await runLimited(jobs, concurrency, async (job) => {
128
208
  const result = await runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchmarkStartedAt);
129
209
  results.push(result);
210
+ if (onMeasuredResult) onMeasuredResult(result);
130
211
  if (!result.ok && options.failFast) {
131
212
  const error = new Error(result.error || `Benchmark failed for ${job.model.name}`);
132
213
  error.exitCode = 1;
133
214
  throw error;
134
215
  }
135
- });
216
+ }, options);
136
217
 
137
218
  results.sort((a, b) => a.model.localeCompare(b.model)
138
219
  || a.promptId.localeCompare(b.promptId)
@@ -171,6 +252,12 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
171
252
  : null;
172
253
  const chars = response.output.length;
173
254
  const words = response.output.trim() ? response.output.trim().split(/\s+/).length : 0;
255
+ const promptTokens = promptTokenCounts.get(job.prompt.id);
256
+ const prefillTokensPerSecond = promptTokens != null && firstTokenMs > 0
257
+ ? promptTokens / (firstTokenMs / 1000)
258
+ : null;
259
+ const chunkGapsMs = chunkGaps(response.stdoutChunkTimesMs);
260
+ const secondChunkMs = chunkGapsMs.length > 0 ? chunkGapsMs[0] : null;
174
261
 
175
262
  return {
176
263
  model: job.model.name,
@@ -182,17 +269,22 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
182
269
  firstTokenMs,
183
270
  generationMs,
184
271
  tpotMs,
185
- promptTokens: promptTokenCounts.get(job.prompt.id),
272
+ promptTokens,
186
273
  outputTokens: countedOutputTokens,
187
274
  chars,
188
275
  words,
189
276
  tokensPerSecond: countedOutputTokens != null && seconds > 0 ? countedOutputTokens / seconds : null,
190
277
  decodeTokensPerSecond,
278
+ prefillTokensPerSecond,
191
279
  charsPerSecond: seconds > 0 ? chars / seconds : 0,
192
280
  startOffsetMs,
193
281
  endOffsetMs,
194
282
  streamed: response.streamed,
195
283
  stdoutChunks: response.stdoutChunks,
284
+ secondChunkMs,
285
+ chunkGapsMs,
286
+ chunkGapAvgMs: average(chunkGapsMs),
287
+ chunkGapMaxMs: chunkGapsMs.length > 0 ? Math.max(...chunkGapsMs) : null,
196
288
  outputHash: response.ok ? hashOutput(response.output) : null,
197
289
  good: response.ok ? evaluateSlo({
198
290
  firstTokenMs,
@@ -204,12 +296,14 @@ async function runSingleBenchmark(fmBin, job, promptTokenCounts, options, benchm
204
296
  };
205
297
  }
206
298
 
207
- async function runLimited(items, concurrency, worker) {
299
+ async function runLimited(items, concurrency, worker, options = {}) {
208
300
  let nextIndex = 0;
301
+ const waitForSlot = createPacer(options.requestRate, options.rampUpMs);
209
302
  const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
210
303
  while (nextIndex < items.length) {
211
304
  const index = nextIndex;
212
305
  nextIndex += 1;
306
+ await waitForSlot(index);
213
307
  await worker(items[index]);
214
308
  }
215
309
  });
@@ -233,6 +327,8 @@ function publicOptions(options) {
233
327
  concurrency: options.concurrency,
234
328
  sweepConcurrency: concurrencies.length > 1 ? concurrencies : [],
235
329
  timeoutMs: options.timeoutMs,
330
+ requestRate: options.requestRate || null,
331
+ rampUpMs: options.rampUpMs || null,
236
332
  profile: options.profile,
237
333
  promptCount: options.promptCount,
238
334
  greedy: options.greedy,
@@ -262,6 +358,53 @@ function hashOutput(output) {
262
358
  .slice(0, 16);
263
359
  }
264
360
 
361
+ function chunkGaps(times = []) {
362
+ const gaps = [];
363
+ for (let index = 1; index < times.length; index += 1) {
364
+ gaps.push(Math.max(0, times[index] - times[index - 1]));
365
+ }
366
+ return gaps;
367
+ }
368
+
369
+ function average(values) {
370
+ const clean = values.filter((value) => Number.isFinite(value));
371
+ if (clean.length === 0) return null;
372
+ return clean.reduce((sum, value) => sum + value, 0) / clean.length;
373
+ }
374
+
375
+ function createPacer(requestRate, rampUpMs = 0) {
376
+ if (!Number.isFinite(requestRate) || requestRate <= 0) {
377
+ return async () => {};
378
+ }
379
+
380
+ const startedAt = process.hrtime.bigint();
381
+ const offsets = [];
382
+ const steadyIntervalMs = 1000 / requestRate;
383
+ const warmIntervalMs = steadyIntervalMs * 4;
384
+
385
+ return async (index) => {
386
+ while (offsets.length <= index) {
387
+ const previousOffset = offsets.length === 0 ? 0 : offsets[offsets.length - 1];
388
+ const fraction = rampUpMs > 0 ? Math.min(1, previousOffset / rampUpMs) : 1;
389
+ const interval = warmIntervalMs + ((steadyIntervalMs - warmIntervalMs) * fraction);
390
+ offsets.push(offsets.length === 0 ? 0 : previousOffset + interval);
391
+ }
392
+
393
+ const targetMs = offsets[index];
394
+ const elapsedMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
395
+ const waitMs = targetMs - elapsedMs;
396
+ if (waitMs > 0) {
397
+ await new Promise((resolve) => setTimeout(resolve, waitMs));
398
+ }
399
+ };
400
+ }
401
+
402
+ function notify(options, event) {
403
+ if (typeof options.onProgress === 'function') {
404
+ options.onProgress(event);
405
+ }
406
+ }
407
+
265
408
  function evaluateSlo(metrics, options) {
266
409
  const thresholds = [
267
410
  ['firstTokenMs', options.sloTtftMs],
package/src/cli.js CHANGED
@@ -2,6 +2,7 @@ import fs from 'node:fs/promises';
2
2
  import { createRequire } from 'node:module';
3
3
  import { inspectModels, runBenchmark } from './bench.js';
4
4
  import { runProcess } from './process.js';
5
+ import { createProgress } from './progress.js';
5
6
  import { flattenResults, toCsv, writeReport } from './report.js';
6
7
  import { renderBenchmarkReport, renderModelsTable } from './table.js';
7
8
 
@@ -36,10 +37,21 @@ export async function runCli(argv = process.argv.slice(2)) {
36
37
  return;
37
38
  }
38
39
 
39
- const payload = await runBenchmark({
40
- ...parsed,
41
- version: packageJson.version
40
+ const progress = createProgress({
41
+ ...renderOptions(parsed),
42
+ enabled: resolveProgress(parsed),
43
+ stream: process.stderr
42
44
  });
45
+ let payload;
46
+ try {
47
+ payload = await runBenchmark({
48
+ ...parsed,
49
+ version: packageJson.version,
50
+ onProgress: (event) => progress.update(event)
51
+ });
52
+ } finally {
53
+ progress.stop();
54
+ }
43
55
 
44
56
  if (parsed.format === 'json') {
45
57
  console.log(JSON.stringify(payload, null, 2));
@@ -71,6 +83,8 @@ export function parseArgs(argv) {
71
83
  warmup: 0,
72
84
  concurrency: 1,
73
85
  sweepConcurrency: [],
86
+ requestRate: null,
87
+ rampUpMs: 0,
74
88
  timeoutMs: 60_000,
75
89
  profile: 'standard',
76
90
  greedy: true,
@@ -84,6 +98,8 @@ export function parseArgs(argv) {
84
98
  failFast: false,
85
99
  verbose: false,
86
100
  ascii: false,
101
+ color: 'auto',
102
+ progress: 'auto',
87
103
  compact: false,
88
104
  width: null
89
105
  };
@@ -130,6 +146,12 @@ export function parseArgs(argv) {
130
146
  options.concurrency = options.sweepConcurrency[0];
131
147
  }
132
148
  break;
149
+ case '--request-rate':
150
+ options.requestRate = parsePositiveNumber(requireValue(arg, args), arg);
151
+ break;
152
+ case '--ramp-up-ms':
153
+ options.rampUpMs = parseNonNegativeInt(requireValue(arg, args), arg);
154
+ break;
133
155
  case '--timeout':
134
156
  case '--timeout-ms':
135
157
  options.timeoutMs = parsePositiveInt(requireValue(arg, args), arg);
@@ -193,6 +215,18 @@ export function parseArgs(argv) {
193
215
  case '--ascii':
194
216
  options.ascii = true;
195
217
  break;
218
+ case '--color':
219
+ options.color = 'always';
220
+ break;
221
+ case '--no-color':
222
+ options.color = 'never';
223
+ break;
224
+ case '--progress':
225
+ options.progress = 'always';
226
+ break;
227
+ case '--no-progress':
228
+ options.progress = 'never';
229
+ break;
196
230
  case '--compact':
197
231
  options.compact = true;
198
232
  break;
@@ -272,6 +306,12 @@ function parsePositiveInt(value, option) {
272
306
  return parsed;
273
307
  }
274
308
 
309
+ function parsePositiveNumber(value, option) {
310
+ const parsed = Number.parseFloat(value);
311
+ if (!Number.isFinite(parsed) || parsed <= 0) throw new Error(`${option} must be a positive number`);
312
+ return parsed;
313
+ }
314
+
275
315
  function parseNonNegativeInt(value, option) {
276
316
  const parsed = Number.parseInt(value, 10);
277
317
  if (!Number.isInteger(parsed) || parsed < 0) throw new Error(`${option} must be a non-negative integer`);
@@ -291,11 +331,26 @@ function parsePositiveIntList(value, option) {
291
331
  function renderOptions(parsed) {
292
332
  return {
293
333
  ascii: parsed.ascii,
334
+ color: resolveColor(parsed.color),
294
335
  compact: parsed.compact,
295
336
  width: parsed.width
296
337
  };
297
338
  }
298
339
 
340
+ function resolveColor(value) {
341
+ if (value === 'always') return true;
342
+ if (value === 'never') return false;
343
+ if (process.env.NO_COLOR) return false;
344
+ if (process.env.FORCE_COLOR && process.env.FORCE_COLOR !== '0') return true;
345
+ return Boolean(process.stdout.isTTY);
346
+ }
347
+
348
+ function resolveProgress(parsed) {
349
+ if (parsed.progress === 'always') return true;
350
+ if (parsed.progress === 'never') return false;
351
+ return parsed.format === 'table' ? 'auto' : false;
352
+ }
353
+
299
354
  function helpText() {
300
355
  return `fm-bench ${packageJson.version}
301
356
 
@@ -313,13 +368,15 @@ Run options:
313
368
  -c, --concurrency <n> Parallel fm processes (default: 1)
314
369
  --sweep-concurrency <list>
315
370
  Run separate operating points, e.g. 1,2,4
371
+ --request-rate <rps> Pace request starts at a target requests/sec
372
+ --ramp-up-ms <n> Gradually ramp request pacing over n ms
316
373
  --timeout-ms <n> Timeout per fm call in ms (default: 60000)
317
374
  --slo-ttft-ms <n> Count request as good only if TTFT is <= n
318
375
  --slo-e2e-ms <n> Count request as good only if E2E latency is <= n
319
376
  --slo-tpot-ms <n> Count request as good only if TPOT is <= n
320
377
  -p, --prompt <text> Prompt to benchmark; repeatable
321
378
  --prompt-file <file> .json, .jsonl, or blank-line separated text prompts
322
- --profile <name> quick, standard, interactive, throughput, or stress
379
+ --profile <name> quick, standard, interactive, throughput, client, or stress
323
380
  -i, --instructions <text> Instructions passed to fm respond
324
381
  --use-case <case> Pass a system model use case through to fm
325
382
  --guardrails <level> Pass a system model guardrail level through to fm
@@ -336,6 +393,10 @@ Output:
336
393
  --json Alias for --format json
337
394
  --csv Alias for --format csv
338
395
  --ascii Use plain ASCII tables instead of Unicode
396
+ --color Force ANSI colors in table output
397
+ --no-color Disable ANSI colors in table output
398
+ --progress Force live progress on stderr
399
+ --no-progress Disable live progress on stderr
339
400
  --compact Force compact terminal layout
340
401
  --width <n> Render for a specific terminal width
341
402
  -o, --out <file> Save JSON or CSV report based on file extension
@@ -349,6 +410,7 @@ Environment:
349
410
  Examples:
350
411
  fm-bench
351
412
  fm-bench --models system,pcc --runs 3 --profile stress
413
+ fm-bench --profile client --sweep-concurrency 1,2 --request-rate 0.5
352
414
  fm-bench --prompt "Reply with exactly: ok" --json --out bench.json
353
415
  fm-bench models
354
416
  fm-bench doctor
package/src/fm.js CHANGED
@@ -176,7 +176,8 @@ export async function respond(fmBin, model, prompt, options = {}) {
176
176
  durationMs: result.durationMs,
177
177
  firstOutputMs: streamed ? result.firstStdoutMs : null,
178
178
  streamed,
179
- stdoutChunks: result.stdoutChunks
179
+ stdoutChunks: result.stdoutChunks,
180
+ stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
180
181
  };
181
182
  }
182
183
 
package/src/process.js CHANGED
@@ -20,6 +20,7 @@ export function runProcess(command, args = [], options = {}) {
20
20
  let stderr = '';
21
21
  let stdoutChunks = 0;
22
22
  let stderrChunks = 0;
23
+ const stdoutChunkTimesMs = [];
23
24
  let firstStdoutMs = null;
24
25
  let firstStderrMs = null;
25
26
  let timedOut = false;
@@ -38,10 +39,12 @@ export function runProcess(command, args = [], options = {}) {
38
39
  child.stdout.setEncoding('utf8');
39
40
  child.stderr.setEncoding('utf8');
40
41
  child.stdout.on('data', (chunk) => {
42
+ const chunkAtMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
41
43
  stdoutChunks += 1;
42
44
  if (firstStdoutMs == null && chunk.length > 0) {
43
- firstStdoutMs = Number(process.hrtime.bigint() - startedAt) / 1e6;
45
+ firstStdoutMs = chunkAtMs;
44
46
  }
47
+ if (chunk.length > 0) stdoutChunkTimesMs.push(chunkAtMs);
45
48
  stdout += chunk;
46
49
  });
47
50
  child.stderr.on('data', (chunk) => {
@@ -65,6 +68,7 @@ export function runProcess(command, args = [], options = {}) {
65
68
  stderr: stderr || error.message,
66
69
  stdoutChunks,
67
70
  stderrChunks,
71
+ stdoutChunkTimesMs,
68
72
  firstStdoutMs,
69
73
  firstStderrMs,
70
74
  error,
@@ -86,6 +90,7 @@ export function runProcess(command, args = [], options = {}) {
86
90
  stderr,
87
91
  stdoutChunks,
88
92
  stderrChunks,
93
+ stdoutChunkTimesMs,
89
94
  firstStdoutMs,
90
95
  firstStderrMs,
91
96
  timedOut,