fm-bench 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/metrics.js ADDED
@@ -0,0 +1,212 @@
1
+ // Metric provenance catalog.
2
+ //
3
+ // fm-bench measures `fm` from the outside, so not every named metric can be
4
+ // observed directly. Each metric declares how it is obtained:
5
+ //
6
+ // measured — observed directly (process timings, exit codes, token counts)
7
+ // proxy — observed at a coarser granularity than the ideal metric
8
+ // derived — computed from other measured values
9
+ // controlled — an input setting, not a measurement
10
+ //
11
+ // A metric whose requirement is missing from the detected `fm` build is
12
+ // reported as unavailable with a reason instead of a blank or invented value.
13
+
14
+ const DEFINITIONS = [
15
+ {
16
+ key: 'ttft',
17
+ label: 'TTFT',
18
+ kind: 'proxy',
19
+ source: 'arrival time of the first streamed stdout chunk',
20
+ requires: ['streaming'],
21
+ reason: 'requires an fm build whose output can be streamed'
22
+ },
23
+ {
24
+ key: 'e2eLatency',
25
+ label: 'E2E latency',
26
+ kind: 'measured',
27
+ source: 'wall clock from spawning fm until it exits',
28
+ requires: []
29
+ },
30
+ {
31
+ key: 'generationMs',
32
+ label: 'generation time',
33
+ kind: 'derived',
34
+ source: 'E2E latency minus TTFT',
35
+ requires: ['streaming'],
36
+ reason: 'requires at least two streamed output chunks to separate prefill from decode'
37
+ },
38
+ {
39
+ key: 'tpot',
40
+ label: 'TPOT',
41
+ kind: 'derived',
42
+ source: '(E2E - TTFT) / (output tokens - 1)',
43
+ requires: ['streaming', 'tokenCounting'],
44
+ reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
45
+ },
46
+ {
47
+ key: 'promptTokens',
48
+ label: 'prompt tokens',
49
+ kind: 'measured',
50
+ source: 'fm count-tokens on the prompt',
51
+ requires: ['tokenCounting'],
52
+ reason: 'this fm build exposes no token-counting command'
53
+ },
54
+ {
55
+ key: 'outputTokens',
56
+ label: 'output tokens',
57
+ kind: 'measured',
58
+ source: 'fm count-tokens on the captured output',
59
+ requires: ['tokenCounting'],
60
+ reason: 'this fm build exposes no token-counting command'
61
+ },
62
+ {
63
+ key: 'tokensPerSecond',
64
+ label: 'per-request output tokens/s',
65
+ kind: 'derived',
66
+ source: 'output tokens / E2E seconds',
67
+ requires: ['tokenCounting'],
68
+ reason: 'requires a token-counting fm command'
69
+ },
70
+ {
71
+ key: 'decodeTokensPerSecond',
72
+ label: 'decode tokens/s',
73
+ kind: 'derived',
74
+ source: '(output tokens - 1) / generation seconds',
75
+ requires: ['streaming', 'tokenCounting'],
76
+ reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
77
+ },
78
+ {
79
+ key: 'prefillTokensPerSecond',
80
+ label: 'prefill tokens/s',
81
+ kind: 'proxy',
82
+ source: 'prompt tokens / TTFT seconds',
83
+ requires: ['streaming', 'tokenCounting'],
84
+ reason: 'requires streaming plus a token-counting fm command'
85
+ },
86
+ {
87
+ key: 'outputTokenThroughput',
88
+ label: 'aggregate output token throughput',
89
+ kind: 'derived',
90
+ source: 'successful output tokens / measured wall-clock window',
91
+ requires: ['tokenCounting'],
92
+ reason: 'requires a token-counting fm command'
93
+ },
94
+ {
95
+ key: 'rps',
96
+ label: 'request throughput',
97
+ kind: 'measured',
98
+ source: 'successful requests / measured wall-clock window',
99
+ requires: []
100
+ },
101
+ {
102
+ key: 'chunkGaps',
103
+ label: 'chunk gaps and second-chunk delay',
104
+ kind: 'proxy',
105
+ source: 'gaps between consecutive streamed stdout chunks',
106
+ requires: ['streaming'],
107
+ reason: 'requires an fm build whose output can be streamed'
108
+ },
109
+ {
110
+ key: 'percentiles',
111
+ label: 'percentiles',
112
+ kind: 'derived',
113
+ source: 'percentile interpolation over successful samples',
114
+ requires: []
115
+ },
116
+ {
117
+ key: 'successRate',
118
+ label: 'success rate',
119
+ kind: 'measured',
120
+ source: 'successful runs / attempted runs',
121
+ requires: []
122
+ },
123
+ {
124
+ key: 'goodput',
125
+ label: 'goodput',
126
+ kind: 'derived',
127
+ source: 'successful runs meeting every configured SLO / runs with an SLO verdict',
128
+ requires: ['slo'],
129
+ reason: 'set --slo-ttft-ms, --slo-e2e-ms, or --slo-tpot-ms to enable goodput'
130
+ },
131
+ {
132
+ key: 'repeatability',
133
+ label: 'repeatability',
134
+ kind: 'derived',
135
+ source: 'most common normalized output hash share across repeated runs',
136
+ requires: []
137
+ },
138
+ {
139
+ key: 'variability',
140
+ label: 'CV and 95% confidence interval',
141
+ kind: 'derived',
142
+ source: 'sample standard deviation over successful samples',
143
+ requires: [],
144
+ reason: 'needs at least two successful samples'
145
+ },
146
+ {
147
+ key: 'quota',
148
+ label: 'quota',
149
+ kind: 'measured',
150
+ source: 'fm quota-usage',
151
+ requires: ['quota'],
152
+ reason: 'this fm build exposes no quota command'
153
+ }
154
+ ];
155
+
156
+ function requirementsMet(requires = [], context = {}) {
157
+ return requires.every((name) => {
158
+ if (name === 'streaming') return Boolean(context.streaming);
159
+ if (name === 'tokenCounting') return Boolean(context.tokenCounting);
160
+ if (name === 'quota') return Boolean(context.quota);
161
+ if (name === 'slo') return Boolean(context.slo);
162
+ return true;
163
+ });
164
+ }
165
+
166
+ /**
167
+ * Describe which metrics the detected `fm` build can support for a run.
168
+ *
169
+ * @param {{ features?: Record<string, any> }} [capabilities]
170
+ * @param {{ stream?: boolean, slo?: boolean }} [options]
171
+ */
172
+ export function metricAvailability(capabilities = {}, options = {}) {
173
+ const features = capabilities.features ?? {};
174
+ const context = {
175
+ streaming: options.stream !== false && features.streaming !== false,
176
+ tokenCounting: features.tokenCounting !== false,
177
+ quota: features.quota === true,
178
+ slo: options.slo === true
179
+ };
180
+
181
+ const metrics = {};
182
+ for (const definition of DEFINITIONS) {
183
+ const available = requirementsMet(definition.requires, context);
184
+ metrics[definition.key] = {
185
+ label: definition.label,
186
+ kind: definition.kind,
187
+ source: definition.source,
188
+ available,
189
+ unavailableReason: available ? '' : (definition.reason ?? 'unavailable in this environment')
190
+ };
191
+ }
192
+ return metrics;
193
+ }
194
+
195
+ export function metricDefinition(key) {
196
+ const definition = DEFINITIONS.find((item) => item.key === key);
197
+ return definition ? { ...definition, requires: [...definition.requires] } : null;
198
+ }
199
+
200
+ export function metricDefinitions() {
201
+ return DEFINITIONS.map((definition) => ({ ...definition, requires: [...definition.requires] }));
202
+ }
203
+
204
+ /** Short human line for CLI output, e.g. "token counting: yes, quota: no". */
205
+ export function formatCapabilitySummary(capabilities = {}) {
206
+ const features = capabilities.features ?? {};
207
+ return [
208
+ `token counting ${features.tokenCounting ? 'yes' : 'no'}`,
209
+ `streaming ${features.streaming ? 'yes' : 'no'}`,
210
+ `quota ${features.quota ? 'yes' : 'no'}`
211
+ ].join(', ');
212
+ }
package/src/process.js CHANGED
@@ -1,5 +1,31 @@
1
1
  import { spawn } from 'node:child_process';
2
2
 
3
+ // Children currently alive. The CLI registers signal handlers that terminate
4
+ // these so Ctrl+C cannot leave `fm` processes behind.
5
+ const activeChildren = new Set();
6
+
7
+ export function activeChildCount() {
8
+ return activeChildren.size;
9
+ }
10
+
11
+ /**
12
+ * Terminate every running child process. Returns how many were signalled.
13
+ * @param {NodeJS.Signals} signal
14
+ */
15
+ export function killActiveChildren(signal = 'SIGTERM') {
16
+ let killed = 0;
17
+ for (const child of activeChildren) {
18
+ if (child.exitCode != null || child.signalCode != null) continue;
19
+ try {
20
+ child.kill(signal);
21
+ killed += 1;
22
+ } catch {
23
+ // Process already gone.
24
+ }
25
+ }
26
+ return killed;
27
+ }
28
+
3
29
  export function runProcess(command, args = [], options = {}) {
4
30
  const {
5
31
  input,
@@ -15,6 +41,7 @@ export function runProcess(command, args = [], options = {}) {
15
41
  env,
16
42
  stdio: ['pipe', 'pipe', 'pipe']
17
43
  });
44
+ activeChildren.add(child);
18
45
 
19
46
  let stdout = '';
20
47
  let stderr = '';
@@ -55,11 +82,16 @@ export function runProcess(command, args = [], options = {}) {
55
82
  stderr += chunk;
56
83
  });
57
84
 
58
- child.on('error', (error) => {
85
+ const finish = (result) => {
86
+ if (settled) return;
59
87
  settled = true;
88
+ activeChildren.delete(child);
60
89
  if (timer) clearTimeout(timer);
61
- const endedAt = process.hrtime.bigint();
62
- resolve({
90
+ resolve(result);
91
+ };
92
+
93
+ child.on('error', (error) => {
94
+ finish({
63
95
  command,
64
96
  args,
65
97
  code: null,
@@ -73,15 +105,12 @@ export function runProcess(command, args = [], options = {}) {
73
105
  firstStderrMs,
74
106
  error,
75
107
  timedOut,
76
- durationMs: Number(endedAt - startedAt) / 1e6
108
+ durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
77
109
  });
78
110
  });
79
111
 
80
112
  child.on('close', (code, signal) => {
81
- settled = true;
82
- if (timer) clearTimeout(timer);
83
- const endedAt = process.hrtime.bigint();
84
- resolve({
113
+ finish({
85
114
  command,
86
115
  args,
87
116
  code,
@@ -93,8 +122,9 @@ export function runProcess(command, args = [], options = {}) {
93
122
  stdoutChunkTimesMs,
94
123
  firstStdoutMs,
95
124
  firstStderrMs,
125
+ error: null,
96
126
  timedOut,
97
- durationMs: Number(endedAt - startedAt) / 1e6
127
+ durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
98
128
  });
99
129
  });
100
130
 
package/src/prompts.js CHANGED
@@ -195,16 +195,28 @@ export async function loadPrompts(options = {}) {
195
195
 
196
196
  async function loadPromptFile(filePath) {
197
197
  const absolutePath = path.resolve(filePath);
198
- const content = await fs.readFile(absolutePath, 'utf8');
198
+ let content;
199
+ try {
200
+ content = await fs.readFile(absolutePath, 'utf8');
201
+ } catch (error) {
202
+ throw new Error(error.code === 'ENOENT'
203
+ ? `Prompt file not found: ${absolutePath}`
204
+ : `Cannot read prompt file ${absolutePath}: ${error.message}`);
205
+ }
199
206
  const trimmed = content.trim();
200
207
 
201
208
  if (!trimmed) return [];
202
209
 
203
210
  if (absolutePath.endsWith('.json')) {
204
- const parsed = JSON.parse(trimmed);
211
+ let parsed;
212
+ try {
213
+ parsed = JSON.parse(trimmed);
214
+ } catch (error) {
215
+ throw new Error(`Cannot parse ${absolutePath} as JSON: ${error.message}`);
216
+ }
205
217
  const items = Array.isArray(parsed) ? parsed : parsed.prompts;
206
218
  if (!Array.isArray(items)) {
207
- throw new Error('Prompt JSON must be an array or an object with a prompts array');
219
+ throw new Error(`${absolutePath} must be a JSON array of prompts or an object with a prompts array`);
208
220
  }
209
221
  return items.map((item, index) => normalizePromptItem(item, index));
210
222
  }
@@ -212,7 +224,13 @@ async function loadPromptFile(filePath) {
212
224
  if (absolutePath.endsWith('.jsonl')) {
213
225
  return trimmed.split(/\r?\n/)
214
226
  .filter(Boolean)
215
- .map((line, index) => normalizePromptItem(JSON.parse(line), index));
227
+ .map((line, index) => {
228
+ try {
229
+ return normalizePromptItem(JSON.parse(line), index);
230
+ } catch (error) {
231
+ throw new Error(`Cannot parse line ${index + 1} of ${absolutePath} as JSON: ${error.message}`);
232
+ }
233
+ });
216
234
  }
217
235
 
218
236
  return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({
package/src/report.js CHANGED
@@ -16,6 +16,7 @@ export function flattenResults(results) {
16
16
  concurrency: result.concurrency ?? '',
17
17
  prompt_id: result.promptId,
18
18
  run: result.run,
19
+ attempts: result.attempts ?? 1,
19
20
  ok: result.ok,
20
21
  duration_ms: round(result.durationMs),
21
22
  ttft_ms: round(result.firstTokenMs),
@@ -58,9 +59,19 @@ export async function writeReport(filePath, payload, format) {
58
59
 
59
60
  function csvEscape(value) {
60
61
  const text = String(value ?? '');
61
- if (/[",\n\r]/.test(text)) {
62
- return `"${text.replaceAll('"', '""')}"`;
62
+ const safe = guardFormula(text);
63
+ if (/[",\n\r]/.test(safe)) {
64
+ return `"${safe.replaceAll('"', '""')}"`;
63
65
  }
66
+ return safe;
67
+ }
68
+
69
+ // Spreadsheet programs execute cells that start with =, +, @, or a non-numeric
70
+ // leading -. Prompt text and captured output are untrusted input, so prefix
71
+ // those cells with a single quote to keep them literal text.
72
+ function guardFormula(text) {
73
+ if (/^[=+@\t\r]/.test(text)) return `'${text}`;
74
+ if (text.startsWith('-') && !/^-?\d+(\.\d+)?$/.test(text)) return `'${text}`;
64
75
  return text;
65
76
  }
66
77
 
package/src/schema.js CHANGED
@@ -126,10 +126,26 @@ export function validateReport(value) {
126
126
  for (const key of REQUIRED_TOP_LEVEL) {
127
127
  if (!(key in report)) errors.push(`missing required field: ${key}`);
128
128
  }
129
- if (!Array.isArray(report.summary)) errors.push('summary must be an array');
129
+ if (!Array.isArray(report.summary)) {
130
+ errors.push('summary must be an array');
131
+ } else {
132
+ for (const [index, item] of report.summary.entries()) {
133
+ if (!item || typeof item !== 'object') {
134
+ errors.push(`summary[${index}] must be an object`);
135
+ continue;
136
+ }
137
+ const row = /** @type {Record<string, unknown>} */ (item);
138
+ if (typeof row.model !== 'string' || row.model === '') {
139
+ errors.push(`summary[${index}].model must be a non-empty string`);
140
+ }
141
+ }
142
+ }
130
143
  if (report.schemaVersion != null && report.schemaVersion !== REPORT_SCHEMA_VERSION) {
131
144
  errors.push(`unsupported schemaVersion: ${report.schemaVersion} (expected ${REPORT_SCHEMA_VERSION})`);
132
145
  }
146
+ if (report.metrics != null && (typeof report.metrics !== 'object' || Array.isArray(report.metrics))) {
147
+ errors.push('metrics must be an object when present');
148
+ }
133
149
  if (errors.length > 0) return { ok: false, errors };
134
150
  return { ok: true, report };
135
151
  }
package/src/stats.js CHANGED
@@ -1,3 +1,12 @@
1
+ /**
2
+ * Sample statistics for one metric.
3
+ *
4
+ * Spread metrics (`stddev`, `cv`, `ci95*`) need at least two samples. With a
5
+ * single sample they are `null` rather than `0`, because a "0% variation" or a
6
+ * zero-width confidence interval is invented precision, not a measurement.
7
+ *
8
+ * @param {number[]} values
9
+ */
1
10
  export function summarizeNumbers(values) {
2
11
  const clean = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
3
12
  if (clean.length === 0) {
@@ -20,11 +29,12 @@ export function summarizeNumbers(values) {
20
29
 
21
30
  const total = clean.reduce((sum, value) => sum + value, 0);
22
31
  const avg = total / clean.length;
23
- const variance = clean.length > 1
32
+ const hasSpread = clean.length > 1;
33
+ const variance = hasSpread
24
34
  ? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
25
- : 0;
26
- const stddev = Math.sqrt(variance);
27
- const margin = clean.length > 1 ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : 0;
35
+ : null;
36
+ const stddev = variance == null ? null : Math.sqrt(variance);
37
+ const margin = hasSpread ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : null;
28
38
  return {
29
39
  count: clean.length,
30
40
  min: clean[0],
@@ -32,9 +42,9 @@ export function summarizeNumbers(values) {
32
42
  avg,
33
43
  sum: total,
34
44
  stddev,
35
- cv: avg !== 0 ? stddev / Math.abs(avg) : null,
36
- ci95Low: avg - margin,
37
- ci95High: avg + margin,
45
+ cv: hasSpread && avg !== 0 ? stddev / Math.abs(avg) : null,
46
+ ci95Low: hasSpread ? avg - margin : null,
47
+ ci95High: hasSpread ? avg + margin : null,
38
48
  p50: percentile(clean, 50),
39
49
  p90: percentile(clean, 90),
40
50
  p95: percentile(clean, 95),
@@ -42,16 +52,25 @@ export function summarizeNumbers(values) {
42
52
  };
43
53
  }
44
54
 
45
- export function percentile(sortedValues, percentileValue) {
46
- if (sortedValues.length === 0) return null;
47
- if (sortedValues.length === 1) return sortedValues[0];
55
+ /**
56
+ * Percentile with linear interpolation between closest ranks (the same method
57
+ * as Excel's PERCENTILE.INC / NumPy's default). Accepts unsorted input so
58
+ * callers cannot silently get a wrong answer from an unsorted array.
59
+ *
60
+ * @param {number[]} values
61
+ * @param {number} percentileValue 0-100
62
+ */
63
+ export function percentile(values, percentileValue) {
64
+ const sorted = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
65
+ if (sorted.length === 0) return null;
66
+ if (sorted.length === 1) return sorted[0];
48
67
 
49
- const rank = (percentileValue / 100) * (sortedValues.length - 1);
68
+ const rank = (percentileValue / 100) * (sorted.length - 1);
50
69
  const low = Math.floor(rank);
51
70
  const high = Math.ceil(rank);
52
- if (low === high) return sortedValues[low];
71
+ if (low === high) return sorted[low];
53
72
  const weight = rank - low;
54
- return sortedValues[low] * (1 - weight) + sortedValues[high] * weight;
73
+ return sorted[low] * (1 - weight) + sorted[high] * weight;
55
74
  }
56
75
 
57
76
  export function summarizeByModel(results, modelStatuses = [], options = {}) {
@@ -66,6 +85,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
66
85
  concurrency,
67
86
  description: status.description,
68
87
  available: status.available,
88
+ unsupported: Boolean(status.unsupported),
69
89
  skippedReason: status.available ? '' : status.reason || 'Unavailable',
70
90
  results: []
71
91
  });
@@ -80,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
80
100
  concurrency: result.concurrency,
81
101
  description: '',
82
102
  available: true,
103
+ unsupported: false,
83
104
  skippedReason: '',
84
105
  results: []
85
106
  });
@@ -98,7 +119,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
98
119
  const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
99
120
  const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
100
121
  const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
101
- const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
122
+ const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
102
123
  const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
103
124
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
104
125
  const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
@@ -110,14 +131,18 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
110
131
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
111
132
  const totalTokens = promptTokens.sum + outputTokens.sum;
112
133
  const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
134
+ const attempts = entry.results.reduce((sum, result) => sum + (result.attempts ?? 1), 0);
113
135
 
114
136
  return {
115
137
  model: entry.model,
116
138
  concurrency: entry.concurrency,
117
139
  description: entry.description,
118
140
  available: entry.available,
141
+ unsupported: entry.unsupported,
119
142
  skippedReason: entry.skippedReason,
120
143
  attempted: entry.results.length,
144
+ attempts,
145
+ retried: entry.results.length > 0 ? Math.max(0, attempts - entry.results.length) : 0,
121
146
  successes: successes.length,
122
147
  failures: failures.length,
123
148
  successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
@@ -147,6 +172,10 @@ function summaryKey(model, concurrency) {
147
172
  return `${model}::${concurrency ?? 'default'}`;
148
173
  }
149
174
 
175
+ // Two-sided 95% t critical values. Exact table entries up to 30 degrees of
176
+ // freedom, then the standard 2.0 / 1.96 approximations for larger samples.
177
+ // fm-bench uses this for a mean confidence interval, which is context for
178
+ // small samples rather than a hypothesis test.
150
179
  function tCritical95(n) {
151
180
  const df = Math.max(1, n - 1);
152
181
  const table = {