fm-bench 0.6.3 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/fm.js CHANGED
@@ -1,60 +1,16 @@
1
- import crypto from 'node:crypto';
2
1
  import os from 'node:os';
3
2
  import { stripAnsi } from './ansi.js';
3
+ import { detectFmCapabilities } from './capabilities.js';
4
+ import { firstLine, isUnsupportedModelError, parseAvailabilityOutput } from './fm-help.js';
4
5
  import { runProcess } from './process.js';
5
6
  import { parseBatteryOutput, parseThermalOutput } from './system.js';
6
7
 
7
- const DEFAULT_MODELS = [
8
- { name: 'system', description: 'On-device Apple Foundation Model' },
9
- { name: 'pcc', description: 'Apple Foundation Model on Private Cloud Compute' }
10
- ];
8
+ export { parseModelsFromHelp, parseAvailabilityOutput } from './fm-help.js';
11
9
 
12
10
  export function fmBinaryFromOptions(options = {}) {
13
11
  return options.fmBin || process.env.FM_BIN || 'fm';
14
12
  }
15
13
 
16
- export function parseModelsFromHelp(helpText) {
17
- const clean = stripAnsi(helpText);
18
- const lines = clean.split(/\r?\n/);
19
- const models = new Map();
20
- let inModels = false;
21
-
22
- for (const line of lines) {
23
- if (/^\s*MODELS\s*$/.test(line)) {
24
- inModels = true;
25
- continue;
26
- }
27
-
28
- if (inModels && /^\s*[A-Z][A-Z -]+\s*$/.test(line) && !/^\s*MODELS\s*$/.test(line)) {
29
- inModels = false;
30
- }
31
-
32
- if (inModels) {
33
- const match = line.match(/^\s*([A-Za-z0-9._:-]+)\s{2,}(.+?)\s*$/);
34
- if (match) {
35
- models.set(match[1], {
36
- name: match[1],
37
- description: match[2].replace(/\s*\(default\)\s*$/, '').trim()
38
- });
39
- }
40
- }
41
-
42
- const optionMatch = /--model\b/.test(line)
43
- ? line.match(/\bmodel\b.*?\(([^)]+)\)/i)
44
- : null;
45
- if (optionMatch) {
46
- for (const raw of optionMatch[1].split(',')) {
47
- const name = raw.trim();
48
- if (/^[A-Za-z0-9._:-]+$/.test(name) && !models.has(name)) {
49
- models.set(name, { name, description: '' });
50
- }
51
- }
52
- }
53
- }
54
-
55
- return [...models.values()];
56
- }
57
-
58
14
  export async function getFmHelp(fmBin, timeoutMs = 10_000) {
59
15
  const result = await runProcess(fmBin, ['--help'], { timeoutMs });
60
16
  if (result.error) {
@@ -69,40 +25,42 @@ export async function getFmHelp(fmBin, timeoutMs = 10_000) {
69
25
  };
70
26
  }
71
27
 
28
+ /**
29
+ * Discover models, preferring an already-detected capability probe so a run
30
+ * does not spawn `fm --help` more than once.
31
+ * @param {Record<string, any>} options
32
+ */
72
33
  export async function discoverModels(options = {}) {
73
34
  const fmBin = fmBinaryFromOptions(options);
74
- const help = await getFmHelp(fmBin, options.timeoutMs ?? 10_000);
75
- let models = parseModelsFromHelp(help.text);
76
-
77
- if (models.length === 0 && /Apple Foundation Models CLI/i.test(stripAnsi(help.text))) {
78
- models = DEFAULT_MODELS;
79
- }
80
-
35
+ const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
81
36
  return {
82
37
  fmBin,
83
- models,
84
- help: stripAnsi(help.text)
85
- };
86
- }
87
-
88
- export function parseAvailabilityOutput(model, output, code) {
89
- const clean = stripAnsi(output).trim();
90
- const lower = clean.toLowerCase();
91
- const modelLower = model.toLowerCase();
92
- const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b/.test(lower);
93
- const hasAvailable = new RegExp(`\\b${escapeRegExp(modelLower)}\\b[\\s\\S]{0,80}\\bavailable\\b|\\bavailable\\b[\\s\\S]{0,80}\\b${escapeRegExp(modelLower)}\\b`).test(lower)
94
- || lower.includes(`${modelLower} model available`)
95
- || lower.includes(`${titleCase(modelLower)} model available`.toLowerCase());
96
-
97
- return {
98
- model,
99
- available: code === 0 && hasAvailable && !hasError,
100
- raw: clean,
101
- reason: hasError ? clean : ''
38
+ models: capabilities.models,
39
+ help: capabilities.help,
40
+ capabilities
102
41
  };
103
42
  }
104
43
 
44
+ /**
45
+ * Check one model against the detected `fm` build.
46
+ *
47
+ * Models the build does not expose are reported as unsupported without
48
+ * spawning `fm` at all, so a raw argument-error blob from `fm` can never end
49
+ * up in a report or in `fm-bench models` output.
50
+ */
105
51
  export async function checkModelAvailability(fmBin, model, options = {}) {
52
+ const capabilities = options.capabilities;
53
+ const known = capabilities?.models?.map((entry) => entry.name) ?? [];
54
+ if (capabilities && known.length > 0 && !known.includes(model)) {
55
+ return {
56
+ model,
57
+ available: false,
58
+ unsupported: true,
59
+ raw: '',
60
+ reason: `not supported by this fm build (supported: ${known.join(', ')})`
61
+ };
62
+ }
63
+
106
64
  const result = await runProcess(fmBin, ['available', '--model', model], {
107
65
  timeoutMs: options.timeoutMs ?? 15_000
108
66
  });
@@ -111,53 +69,99 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
111
69
  if (result.error) {
112
70
  parsed.available = false;
113
71
  parsed.reason = result.stderr || result.error.message;
72
+ return parsed;
73
+ }
74
+ if (isUnsupportedModelError(output)) {
75
+ parsed.available = false;
76
+ parsed.unsupported = true;
77
+ const supported = known.length > 0 ? ` (supported: ${known.join(', ')})` : '';
78
+ parsed.reason = `not supported by this fm build${supported}`;
114
79
  }
115
80
  return parsed;
116
81
  }
117
82
 
83
+ /**
84
+ * Query quota information when the build exposes it.
85
+ * @returns {Promise<{ model: string, supported: boolean, ok: boolean, raw: string, reason: string }>}
86
+ */
118
87
  export async function getQuotaUsage(fmBin, model, options = {}) {
88
+ const features = options.capabilities?.features;
89
+ if (features && !features.quota) {
90
+ return {
91
+ model,
92
+ supported: false,
93
+ ok: false,
94
+ raw: '',
95
+ reason: 'unavailable: this fm build exposes no quota command'
96
+ };
97
+ }
98
+
119
99
  const result = await runProcess(fmBin, ['quota-usage', '--model', model], {
120
100
  timeoutMs: options.timeoutMs ?? 15_000
121
101
  });
122
102
  const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
123
103
  return {
124
104
  model,
105
+ supported: true,
125
106
  ok: result.code === 0,
126
107
  raw: output,
127
- unavailable: /\bunavailable\b|\bnot available\b|\berror:/i.test(output)
108
+ reason: result.code === 0 ? '' : firstLine(output)
128
109
  };
129
110
  }
130
111
 
112
+ /**
113
+ * Count tokens with whichever token-counting command this `fm` build exposes.
114
+ * Returns `ok: false` with a reason when the build cannot count tokens; it
115
+ * never invents a count.
116
+ */
131
117
  export async function countTokens(fmBin, text, options = {}) {
132
- const result = await runProcess(fmBin, ['token-count', '--quiet'], {
118
+ const command = options.capabilities?.features?.tokenCountCommand ?? 'count-tokens';
119
+ const supported = options.capabilities?.features?.tokenCounting ?? true;
120
+ if (!supported || !command) {
121
+ return {
122
+ ok: false,
123
+ count: null,
124
+ unsupported: true,
125
+ raw: '',
126
+ reason: 'token counting is unavailable in this fm build'
127
+ };
128
+ }
129
+
130
+ const result = await runProcess(fmBin, [command, '--quiet'], {
133
131
  input: text,
134
132
  timeoutMs: options.timeoutMs ?? 15_000
135
133
  });
136
134
  const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
137
- const match = output.match(/-?\d+/);
138
- if (result.code !== 0 || !match) {
135
+ const match = output.match(/\d+/);
136
+ if (result.error || result.code !== 0 || !match) {
139
137
  return {
140
138
  ok: false,
141
139
  count: null,
142
- raw: output
140
+ raw: output,
141
+ reason: result.error?.message || firstLine(output) || `fm ${command} exited with code ${result.code}`
143
142
  };
144
143
  }
145
144
  return {
146
145
  ok: true,
147
146
  count: Number.parseInt(match[0], 10),
148
- raw: output
147
+ raw: output,
148
+ reason: ''
149
149
  };
150
150
  }
151
151
 
152
152
  export async function respond(fmBin, model, prompt, options = {}) {
153
- const args = ['respond', '--model', model];
154
- const streamed = options.stream !== false;
155
-
156
- if (!streamed) args.push('--no-stream');
157
- if (options.greedy) args.push('--greedy');
158
- if (options.instructions) args.push('--instructions', options.instructions);
159
- if (options.useCase) args.push('--use-case', options.useCase);
160
- if (options.guardrails) args.push('--guardrails', options.guardrails);
153
+ const features = options.capabilities?.features;
154
+ const streamControl = features ? features.streaming : true;
155
+ const modelSelection = features ? features.modelSelection : true;
156
+ const streamed = streamControl && options.stream !== false;
157
+
158
+ const args = ['respond'];
159
+ if (modelSelection) args.push('--model', model);
160
+ if (streamControl && !streamed) args.push('--no-stream');
161
+ if (options.greedy && (features?.greedy ?? true)) args.push('--greedy');
162
+ if (options.instructions && (features?.instructions ?? true)) args.push('--instructions', options.instructions);
163
+ if (options.useCase && (features?.useCase ?? true)) args.push('--use-case', options.useCase);
164
+ if (options.guardrails && (features?.guardrails ?? true)) args.push('--guardrails', options.guardrails);
161
165
 
162
166
  const result = await runProcess(fmBin, args, {
163
167
  input: prompt,
@@ -166,8 +170,10 @@ export async function respond(fmBin, model, prompt, options = {}) {
166
170
 
167
171
  const output = stripAnsi(result.stdout).trim();
168
172
  const errorText = stripAnsi(result.stderr).trim();
173
+ const failed = result.code !== 0 || result.timedOut;
174
+
169
175
  return {
170
- ok: result.code === 0 && !result.timedOut,
176
+ ok: !failed,
171
177
  model,
172
178
  prompt,
173
179
  output,
@@ -179,11 +185,17 @@ export async function respond(fmBin, model, prompt, options = {}) {
179
185
  firstOutputMs: streamed ? result.firstStdoutMs : null,
180
186
  streamed,
181
187
  stdoutChunks: result.stdoutChunks,
182
- stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
188
+ stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : [],
189
+ error: failed
190
+ ? (result.timedOut
191
+ ? `timed out after ${options.timeoutMs ?? 60_000}ms`
192
+ : firstLine(errorText) || `fm exited with code ${result.code ?? result.signal}`)
193
+ : ''
183
194
  };
184
195
  }
185
196
 
186
- export async function collectEnvironment(fmBin) {
197
+ export async function collectEnvironment(fmBin, options = {}) {
198
+ const capabilities = options.capabilities;
187
199
  const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
188
200
  const macOS = stripAnsi(swVers.stdout).trim() || null;
189
201
 
@@ -196,15 +208,15 @@ export async function collectEnvironment(fmBin) {
196
208
  const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
197
209
  const battery = parseBatteryOutput(`${batteryResult.stdout || ''}${batteryResult.stderr || ''}`);
198
210
 
199
- let fmHelpDigest = null;
200
- try {
201
- const help = await getFmHelp(fmBin, 10_000);
202
- const text = stripAnsi(help.text).trim();
203
- if (text) {
204
- fmHelpDigest = crypto.createHash('sha256').update(text).digest('hex').slice(0, 16);
211
+ let fmHelpDigest = capabilities?.digest ?? null;
212
+ if (fmHelpDigest == null) {
213
+ try {
214
+ const help = await getFmHelp(fmBin, 10_000);
215
+ const detected = await detectFmCapabilities(fmBin, { help: { text: help.text } });
216
+ fmHelpDigest = detected.digest;
217
+ } catch {
218
+ fmHelpDigest = null;
205
219
  }
206
- } catch {
207
- fmHelpDigest = null;
208
220
  }
209
221
 
210
222
  const memRaw = (memBytes.stdout || '').trim();
@@ -237,11 +249,3 @@ export async function collectEnvironment(fmBin) {
237
249
  : null
238
250
  };
239
251
  }
240
-
241
- function escapeRegExp(value) {
242
- return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
243
- }
244
-
245
- function titleCase(value) {
246
- return value.slice(0, 1).toUpperCase() + value.slice(1);
247
- }
package/src/history.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import fs from 'node:fs/promises';
2
2
  import path from 'node:path';
3
+ import { stripAnsi } from './ansi.js';
3
4
  import { formatMs, formatNumber, formatPercent } from './table.js';
4
5
 
5
6
  export async function loadHistory(dir) {
@@ -100,7 +101,7 @@ export function renderHistoryReport(reports, options = {}) {
100
101
  }
101
102
 
102
103
  function fit(text, width) {
103
- const str = String(text ?? '');
104
+ const str = stripAnsi(String(text ?? '')).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
104
105
  if (str.length <= width) return str + ' '.repeat(width - str.length);
105
106
  return `${str.slice(0, width - 1)}…`;
106
107
  }
package/src/macos.js CHANGED
@@ -57,7 +57,8 @@ export function evaluateMacosSupport(platform, parsed) {
57
57
  export function formatMacosRequirementError({ reason, latestSupported }) {
58
58
  return [
59
59
  `unsupported macOS: ${reason}`,
60
- `Latest supported: ${latestSupported} (fm is not available on older macOS releases).`
60
+ `Latest supported: ${latestSupported} (fm is not available on older macOS releases).`,
61
+ 'Pass --fm-bin <path> (or set FM_BIN) to benchmark an fm binary you provide on this host.'
61
62
  ].join('\n');
62
63
  }
63
64
 
package/src/metrics.js ADDED
@@ -0,0 +1,212 @@
1
+ // Metric provenance catalog.
2
+ //
3
+ // fm-bench measures `fm` from the outside, so not every named metric can be
4
+ // observed directly. Each metric declares how it is obtained:
5
+ //
6
+ // measured — observed directly (process timings, exit codes, token counts)
7
+ // proxy — observed at a coarser granularity than the ideal metric
8
+ // derived — computed from other measured values
9
+ // controlled — an input setting, not a measurement
10
+ //
11
+ // A metric whose requirement is missing from the detected `fm` build is
12
+ // reported as unavailable with a reason instead of a blank or invented value.
13
+
14
+ const DEFINITIONS = [
15
+ {
16
+ key: 'ttft',
17
+ label: 'TTFT',
18
+ kind: 'proxy',
19
+ source: 'arrival time of the first streamed stdout chunk',
20
+ requires: ['streaming'],
21
+ reason: 'requires an fm build whose output can be streamed'
22
+ },
23
+ {
24
+ key: 'e2eLatency',
25
+ label: 'E2E latency',
26
+ kind: 'measured',
27
+ source: 'wall clock from spawning fm until it exits',
28
+ requires: []
29
+ },
30
+ {
31
+ key: 'generationMs',
32
+ label: 'generation time',
33
+ kind: 'derived',
34
+ source: 'E2E latency minus TTFT',
35
+ requires: ['streaming'],
36
+ reason: 'requires at least two streamed output chunks to separate prefill from decode'
37
+ },
38
+ {
39
+ key: 'tpot',
40
+ label: 'TPOT',
41
+ kind: 'derived',
42
+ source: '(E2E - TTFT) / (output tokens - 1)',
43
+ requires: ['streaming', 'tokenCounting'],
44
+ reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
45
+ },
46
+ {
47
+ key: 'promptTokens',
48
+ label: 'prompt tokens',
49
+ kind: 'measured',
50
+ source: 'fm count-tokens on the prompt',
51
+ requires: ['tokenCounting'],
52
+ reason: 'this fm build exposes no token-counting command'
53
+ },
54
+ {
55
+ key: 'outputTokens',
56
+ label: 'output tokens',
57
+ kind: 'measured',
58
+ source: 'fm count-tokens on the captured output',
59
+ requires: ['tokenCounting'],
60
+ reason: 'this fm build exposes no token-counting command'
61
+ },
62
+ {
63
+ key: 'tokensPerSecond',
64
+ label: 'per-request output tokens/s',
65
+ kind: 'derived',
66
+ source: 'output tokens / E2E seconds',
67
+ requires: ['tokenCounting'],
68
+ reason: 'requires a token-counting fm command'
69
+ },
70
+ {
71
+ key: 'decodeTokensPerSecond',
72
+ label: 'decode tokens/s',
73
+ kind: 'derived',
74
+ source: '(output tokens - 1) / generation seconds',
75
+ requires: ['streaming', 'tokenCounting'],
76
+ reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
77
+ },
78
+ {
79
+ key: 'prefillTokensPerSecond',
80
+ label: 'prefill tokens/s',
81
+ kind: 'proxy',
82
+ source: 'prompt tokens / TTFT seconds',
83
+ requires: ['streaming', 'tokenCounting'],
84
+ reason: 'requires streaming plus a token-counting fm command'
85
+ },
86
+ {
87
+ key: 'outputTokenThroughput',
88
+ label: 'aggregate output token throughput',
89
+ kind: 'derived',
90
+ source: 'successful output tokens / measured wall-clock window',
91
+ requires: ['tokenCounting'],
92
+ reason: 'requires a token-counting fm command'
93
+ },
94
+ {
95
+ key: 'rps',
96
+ label: 'request throughput',
97
+ kind: 'measured',
98
+ source: 'successful requests / measured wall-clock window',
99
+ requires: []
100
+ },
101
+ {
102
+ key: 'chunkGaps',
103
+ label: 'chunk gaps and second-chunk delay',
104
+ kind: 'proxy',
105
+ source: 'gaps between consecutive streamed stdout chunks',
106
+ requires: ['streaming'],
107
+ reason: 'requires an fm build whose output can be streamed'
108
+ },
109
+ {
110
+ key: 'percentiles',
111
+ label: 'percentiles',
112
+ kind: 'derived',
113
+ source: 'percentile interpolation over successful samples',
114
+ requires: []
115
+ },
116
+ {
117
+ key: 'successRate',
118
+ label: 'success rate',
119
+ kind: 'measured',
120
+ source: 'successful runs / attempted runs',
121
+ requires: []
122
+ },
123
+ {
124
+ key: 'goodput',
125
+ label: 'goodput',
126
+ kind: 'derived',
127
+ source: 'successful runs meeting every configured SLO / runs with an SLO verdict',
128
+ requires: ['slo'],
129
+ reason: 'set --slo-ttft-ms, --slo-e2e-ms, or --slo-tpot-ms to enable goodput'
130
+ },
131
+ {
132
+ key: 'repeatability',
133
+ label: 'repeatability',
134
+ kind: 'derived',
135
+ source: 'most common normalized output hash share across repeated runs',
136
+ requires: []
137
+ },
138
+ {
139
+ key: 'variability',
140
+ label: 'CV and 95% confidence interval',
141
+ kind: 'derived',
142
+ source: 'sample standard deviation over successful samples',
143
+ requires: [],
144
+ reason: 'needs at least two successful samples'
145
+ },
146
+ {
147
+ key: 'quota',
148
+ label: 'quota',
149
+ kind: 'measured',
150
+ source: 'fm quota-usage',
151
+ requires: ['quota'],
152
+ reason: 'this fm build exposes no quota command'
153
+ }
154
+ ];
155
+
156
+ function requirementsMet(requires = [], context = {}) {
157
+ return requires.every((name) => {
158
+ if (name === 'streaming') return Boolean(context.streaming);
159
+ if (name === 'tokenCounting') return Boolean(context.tokenCounting);
160
+ if (name === 'quota') return Boolean(context.quota);
161
+ if (name === 'slo') return Boolean(context.slo);
162
+ return true;
163
+ });
164
+ }
165
+
166
+ /**
167
+ * Describe which metrics the detected `fm` build can support for a run.
168
+ *
169
+ * @param {{ features?: Record<string, any> }} [capabilities]
170
+ * @param {{ stream?: boolean, slo?: boolean }} [options]
171
+ */
172
+ export function metricAvailability(capabilities = {}, options = {}) {
173
+ const features = capabilities.features ?? {};
174
+ const context = {
175
+ streaming: options.stream !== false && features.streaming !== false,
176
+ tokenCounting: features.tokenCounting !== false,
177
+ quota: features.quota === true,
178
+ slo: options.slo === true
179
+ };
180
+
181
+ const metrics = {};
182
+ for (const definition of DEFINITIONS) {
183
+ const available = requirementsMet(definition.requires, context);
184
+ metrics[definition.key] = {
185
+ label: definition.label,
186
+ kind: definition.kind,
187
+ source: definition.source,
188
+ available,
189
+ unavailableReason: available ? '' : (definition.reason ?? 'unavailable in this environment')
190
+ };
191
+ }
192
+ return metrics;
193
+ }
194
+
195
+ export function metricDefinition(key) {
196
+ const definition = DEFINITIONS.find((item) => item.key === key);
197
+ return definition ? { ...definition, requires: [...definition.requires] } : null;
198
+ }
199
+
200
+ export function metricDefinitions() {
201
+ return DEFINITIONS.map((definition) => ({ ...definition, requires: [...definition.requires] }));
202
+ }
203
+
204
+ /** Short human line for CLI output, e.g. "token counting: yes, quota: no". */
205
+ export function formatCapabilitySummary(capabilities = {}) {
206
+ const features = capabilities.features ?? {};
207
+ return [
208
+ `token counting ${features.tokenCounting ? 'yes' : 'no'}`,
209
+ `streaming ${features.streaming ? 'yes' : 'no'}`,
210
+ `quota ${features.quota ? 'yes' : 'no'}`
211
+ ].join(', ');
212
+ }
package/src/process.js CHANGED
@@ -1,5 +1,31 @@
1
1
  import { spawn } from 'node:child_process';
2
2
 
3
+ // Children currently alive. The CLI registers signal handlers that terminate
4
+ // these so Ctrl+C cannot leave `fm` processes behind.
5
+ const activeChildren = new Set();
6
+
7
+ export function activeChildCount() {
8
+ return activeChildren.size;
9
+ }
10
+
11
+ /**
12
+ * Terminate every running child process. Returns how many were signalled.
13
+ * @param {NodeJS.Signals} signal
14
+ */
15
+ export function killActiveChildren(signal = 'SIGTERM') {
16
+ let killed = 0;
17
+ for (const child of activeChildren) {
18
+ if (child.exitCode != null || child.signalCode != null) continue;
19
+ try {
20
+ child.kill(signal);
21
+ killed += 1;
22
+ } catch {
23
+ // Process already gone.
24
+ }
25
+ }
26
+ return killed;
27
+ }
28
+
3
29
  export function runProcess(command, args = [], options = {}) {
4
30
  const {
5
31
  input,
@@ -15,6 +41,7 @@ export function runProcess(command, args = [], options = {}) {
15
41
  env,
16
42
  stdio: ['pipe', 'pipe', 'pipe']
17
43
  });
44
+ activeChildren.add(child);
18
45
 
19
46
  let stdout = '';
20
47
  let stderr = '';
@@ -55,11 +82,16 @@ export function runProcess(command, args = [], options = {}) {
55
82
  stderr += chunk;
56
83
  });
57
84
 
58
- child.on('error', (error) => {
85
+ const finish = (result) => {
86
+ if (settled) return;
59
87
  settled = true;
88
+ activeChildren.delete(child);
60
89
  if (timer) clearTimeout(timer);
61
- const endedAt = process.hrtime.bigint();
62
- resolve({
90
+ resolve(result);
91
+ };
92
+
93
+ child.on('error', (error) => {
94
+ finish({
63
95
  command,
64
96
  args,
65
97
  code: null,
@@ -73,15 +105,12 @@ export function runProcess(command, args = [], options = {}) {
73
105
  firstStderrMs,
74
106
  error,
75
107
  timedOut,
76
- durationMs: Number(endedAt - startedAt) / 1e6
108
+ durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
77
109
  });
78
110
  });
79
111
 
80
112
  child.on('close', (code, signal) => {
81
- settled = true;
82
- if (timer) clearTimeout(timer);
83
- const endedAt = process.hrtime.bigint();
84
- resolve({
113
+ finish({
85
114
  command,
86
115
  args,
87
116
  code,
@@ -93,8 +122,9 @@ export function runProcess(command, args = [], options = {}) {
93
122
  stdoutChunkTimesMs,
94
123
  firstStdoutMs,
95
124
  firstStderrMs,
125
+ error: null,
96
126
  timedOut,
97
- durationMs: Number(endedAt - startedAt) / 1e6
127
+ durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
98
128
  });
99
129
  });
100
130
 
package/src/prompts.js CHANGED
@@ -195,16 +195,28 @@ export async function loadPrompts(options = {}) {
195
195
 
196
196
  async function loadPromptFile(filePath) {
197
197
  const absolutePath = path.resolve(filePath);
198
- const content = await fs.readFile(absolutePath, 'utf8');
198
+ let content;
199
+ try {
200
+ content = await fs.readFile(absolutePath, 'utf8');
201
+ } catch (error) {
202
+ throw new Error(error.code === 'ENOENT'
203
+ ? `Prompt file not found: ${absolutePath}`
204
+ : `Cannot read prompt file ${absolutePath}: ${error.message}`);
205
+ }
199
206
  const trimmed = content.trim();
200
207
 
201
208
  if (!trimmed) return [];
202
209
 
203
210
  if (absolutePath.endsWith('.json')) {
204
- const parsed = JSON.parse(trimmed);
211
+ let parsed;
212
+ try {
213
+ parsed = JSON.parse(trimmed);
214
+ } catch (error) {
215
+ throw new Error(`Cannot parse ${absolutePath} as JSON: ${error.message}`);
216
+ }
205
217
  const items = Array.isArray(parsed) ? parsed : parsed.prompts;
206
218
  if (!Array.isArray(items)) {
207
- throw new Error('Prompt JSON must be an array or an object with a prompts array');
219
+ throw new Error(`${absolutePath} must be a JSON array of prompts or an object with a prompts array`);
208
220
  }
209
221
  return items.map((item, index) => normalizePromptItem(item, index));
210
222
  }
@@ -212,7 +224,13 @@ async function loadPromptFile(filePath) {
212
224
  if (absolutePath.endsWith('.jsonl')) {
213
225
  return trimmed.split(/\r?\n/)
214
226
  .filter(Boolean)
215
- .map((line, index) => normalizePromptItem(JSON.parse(line), index));
227
+ .map((line, index) => {
228
+ try {
229
+ return normalizePromptItem(JSON.parse(line), index);
230
+ } catch (error) {
231
+ throw new Error(`Cannot parse line ${index + 1} of ${absolutePath} as JSON: ${error.message}`);
232
+ }
233
+ });
216
234
  }
217
235
 
218
236
  return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({