fm-bench 0.6.3 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +33 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +19 -2
- package/package.json +6 -5
- package/src/bench.js +129 -37
- package/src/capabilities.js +188 -0
- package/src/cli.js +153 -68
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +100 -112
- package/src/history.js +2 -1
- package/src/macos.js +2 -1
- package/src/metrics.js +203 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/src/fm.js
CHANGED
|
@@ -1,60 +1,16 @@
|
|
|
1
|
-
import crypto from 'node:crypto';
|
|
2
1
|
import os from 'node:os';
|
|
3
2
|
import { stripAnsi } from './ansi.js';
|
|
3
|
+
import { detectFmCapabilities } from './capabilities.js';
|
|
4
|
+
import { firstLine, isUnsupportedModelError, parseAvailabilityOutput } from './fm-help.js';
|
|
4
5
|
import { runProcess } from './process.js';
|
|
5
6
|
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
6
7
|
|
|
7
|
-
|
|
8
|
-
{ name: 'system', description: 'On-device Apple Foundation Model' },
|
|
9
|
-
{ name: 'pcc', description: 'Apple Foundation Model on Private Cloud Compute' }
|
|
10
|
-
];
|
|
8
|
+
export { parseModelsFromHelp, parseAvailabilityOutput } from './fm-help.js';
|
|
11
9
|
|
|
12
10
|
export function fmBinaryFromOptions(options = {}) {
|
|
13
11
|
return options.fmBin || process.env.FM_BIN || 'fm';
|
|
14
12
|
}
|
|
15
13
|
|
|
16
|
-
export function parseModelsFromHelp(helpText) {
|
|
17
|
-
const clean = stripAnsi(helpText);
|
|
18
|
-
const lines = clean.split(/\r?\n/);
|
|
19
|
-
const models = new Map();
|
|
20
|
-
let inModels = false;
|
|
21
|
-
|
|
22
|
-
for (const line of lines) {
|
|
23
|
-
if (/^\s*MODELS\s*$/.test(line)) {
|
|
24
|
-
inModels = true;
|
|
25
|
-
continue;
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
if (inModels && /^\s*[A-Z][A-Z -]+\s*$/.test(line) && !/^\s*MODELS\s*$/.test(line)) {
|
|
29
|
-
inModels = false;
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
if (inModels) {
|
|
33
|
-
const match = line.match(/^\s*([A-Za-z0-9._:-]+)\s{2,}(.+?)\s*$/);
|
|
34
|
-
if (match) {
|
|
35
|
-
models.set(match[1], {
|
|
36
|
-
name: match[1],
|
|
37
|
-
description: match[2].replace(/\s*\(default\)\s*$/, '').trim()
|
|
38
|
-
});
|
|
39
|
-
}
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
const optionMatch = /--model\b/.test(line)
|
|
43
|
-
? line.match(/\bmodel\b.*?\(([^)]+)\)/i)
|
|
44
|
-
: null;
|
|
45
|
-
if (optionMatch) {
|
|
46
|
-
for (const raw of optionMatch[1].split(',')) {
|
|
47
|
-
const name = raw.trim();
|
|
48
|
-
if (/^[A-Za-z0-9._:-]+$/.test(name) && !models.has(name)) {
|
|
49
|
-
models.set(name, { name, description: '' });
|
|
50
|
-
}
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
return [...models.values()];
|
|
56
|
-
}
|
|
57
|
-
|
|
58
14
|
export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
59
15
|
const result = await runProcess(fmBin, ['--help'], { timeoutMs });
|
|
60
16
|
if (result.error) {
|
|
@@ -69,40 +25,26 @@ export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
|
69
25
|
};
|
|
70
26
|
}
|
|
71
27
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
28
|
+
/**
|
|
29
|
+
* Check one model against the detected `fm` build.
|
|
30
|
+
*
|
|
31
|
+
* Models the build does not expose are reported as unsupported without
|
|
32
|
+
* spawning `fm` at all, so a raw argument-error blob from `fm` can never end
|
|
33
|
+
* up in a report or in `fm-bench models` output.
|
|
34
|
+
*/
|
|
35
|
+
export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
36
|
+
const capabilities = options.capabilities;
|
|
37
|
+
const known = capabilities?.models?.map((entry) => entry.name) ?? [];
|
|
38
|
+
if (capabilities && known.length > 0 && !known.includes(model)) {
|
|
39
|
+
return {
|
|
40
|
+
model,
|
|
41
|
+
available: false,
|
|
42
|
+
unsupported: true,
|
|
43
|
+
raw: '',
|
|
44
|
+
reason: `not supported by this fm build (supported: ${known.join(', ')})`
|
|
45
|
+
};
|
|
79
46
|
}
|
|
80
47
|
|
|
81
|
-
return {
|
|
82
|
-
fmBin,
|
|
83
|
-
models,
|
|
84
|
-
help: stripAnsi(help.text)
|
|
85
|
-
};
|
|
86
|
-
}
|
|
87
|
-
|
|
88
|
-
export function parseAvailabilityOutput(model, output, code) {
|
|
89
|
-
const clean = stripAnsi(output).trim();
|
|
90
|
-
const lower = clean.toLowerCase();
|
|
91
|
-
const modelLower = model.toLowerCase();
|
|
92
|
-
const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b/.test(lower);
|
|
93
|
-
const hasAvailable = new RegExp(`\\b${escapeRegExp(modelLower)}\\b[\\s\\S]{0,80}\\bavailable\\b|\\bavailable\\b[\\s\\S]{0,80}\\b${escapeRegExp(modelLower)}\\b`).test(lower)
|
|
94
|
-
|| lower.includes(`${modelLower} model available`)
|
|
95
|
-
|| lower.includes(`${titleCase(modelLower)} model available`.toLowerCase());
|
|
96
|
-
|
|
97
|
-
return {
|
|
98
|
-
model,
|
|
99
|
-
available: code === 0 && hasAvailable && !hasError,
|
|
100
|
-
raw: clean,
|
|
101
|
-
reason: hasError ? clean : ''
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
|
-
|
|
105
|
-
export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
106
48
|
const result = await runProcess(fmBin, ['available', '--model', model], {
|
|
107
49
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
108
50
|
});
|
|
@@ -111,53 +53,99 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
|
111
53
|
if (result.error) {
|
|
112
54
|
parsed.available = false;
|
|
113
55
|
parsed.reason = result.stderr || result.error.message;
|
|
56
|
+
return parsed;
|
|
57
|
+
}
|
|
58
|
+
if (isUnsupportedModelError(output)) {
|
|
59
|
+
parsed.available = false;
|
|
60
|
+
parsed.unsupported = true;
|
|
61
|
+
const supported = known.length > 0 ? ` (supported: ${known.join(', ')})` : '';
|
|
62
|
+
parsed.reason = `not supported by this fm build${supported}`;
|
|
114
63
|
}
|
|
115
64
|
return parsed;
|
|
116
65
|
}
|
|
117
66
|
|
|
67
|
+
/**
|
|
68
|
+
* Query quota information when the build exposes it.
|
|
69
|
+
* @returns {Promise<{ model: string, supported: boolean, ok: boolean, raw: string, reason: string }>}
|
|
70
|
+
*/
|
|
118
71
|
export async function getQuotaUsage(fmBin, model, options = {}) {
|
|
72
|
+
const features = options.capabilities?.features;
|
|
73
|
+
if (features && !features.quota) {
|
|
74
|
+
return {
|
|
75
|
+
model,
|
|
76
|
+
supported: false,
|
|
77
|
+
ok: false,
|
|
78
|
+
raw: '',
|
|
79
|
+
reason: 'unavailable: this fm build exposes no quota command'
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
|
|
119
83
|
const result = await runProcess(fmBin, ['quota-usage', '--model', model], {
|
|
120
84
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
121
85
|
});
|
|
122
86
|
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
123
87
|
return {
|
|
124
88
|
model,
|
|
89
|
+
supported: true,
|
|
125
90
|
ok: result.code === 0,
|
|
126
91
|
raw: output,
|
|
127
|
-
|
|
92
|
+
reason: result.code === 0 ? '' : firstLine(output)
|
|
128
93
|
};
|
|
129
94
|
}
|
|
130
95
|
|
|
96
|
+
/**
|
|
97
|
+
* Count tokens with whichever token-counting command this `fm` build exposes.
|
|
98
|
+
* Returns `ok: false` with a reason when the build cannot count tokens; it
|
|
99
|
+
* never invents a count.
|
|
100
|
+
*/
|
|
131
101
|
export async function countTokens(fmBin, text, options = {}) {
|
|
132
|
-
const
|
|
102
|
+
const command = options.capabilities?.features?.tokenCountCommand ?? 'count-tokens';
|
|
103
|
+
const supported = options.capabilities?.features?.tokenCounting ?? true;
|
|
104
|
+
if (!supported || !command) {
|
|
105
|
+
return {
|
|
106
|
+
ok: false,
|
|
107
|
+
count: null,
|
|
108
|
+
unsupported: true,
|
|
109
|
+
raw: '',
|
|
110
|
+
reason: 'token counting is unavailable in this fm build'
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const result = await runProcess(fmBin, [command, '--quiet'], {
|
|
133
115
|
input: text,
|
|
134
116
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
135
117
|
});
|
|
136
118
|
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
137
|
-
const match = output.match(
|
|
138
|
-
if (result.code !== 0 || !match) {
|
|
119
|
+
const match = output.match(/\d+/);
|
|
120
|
+
if (result.error || result.code !== 0 || !match) {
|
|
139
121
|
return {
|
|
140
122
|
ok: false,
|
|
141
123
|
count: null,
|
|
142
|
-
raw: output
|
|
124
|
+
raw: output,
|
|
125
|
+
reason: result.error?.message || firstLine(output) || `fm ${command} exited with code ${result.code}`
|
|
143
126
|
};
|
|
144
127
|
}
|
|
145
128
|
return {
|
|
146
129
|
ok: true,
|
|
147
130
|
count: Number.parseInt(match[0], 10),
|
|
148
|
-
raw: output
|
|
131
|
+
raw: output,
|
|
132
|
+
reason: ''
|
|
149
133
|
};
|
|
150
134
|
}
|
|
151
135
|
|
|
152
136
|
export async function respond(fmBin, model, prompt, options = {}) {
|
|
153
|
-
const
|
|
154
|
-
const
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
if (
|
|
160
|
-
if (
|
|
137
|
+
const features = options.capabilities?.features;
|
|
138
|
+
const streamControl = features ? features.streaming : true;
|
|
139
|
+
const modelSelection = features ? features.modelSelection : true;
|
|
140
|
+
const streamed = streamControl && options.stream !== false;
|
|
141
|
+
|
|
142
|
+
const args = ['respond'];
|
|
143
|
+
if (modelSelection) args.push('--model', model);
|
|
144
|
+
if (streamControl && !streamed) args.push('--no-stream');
|
|
145
|
+
if (options.greedy && (features?.greedy ?? true)) args.push('--greedy');
|
|
146
|
+
if (options.instructions && (features?.instructions ?? true)) args.push('--instructions', options.instructions);
|
|
147
|
+
if (options.useCase && (features?.useCase ?? true)) args.push('--use-case', options.useCase);
|
|
148
|
+
if (options.guardrails && (features?.guardrails ?? true)) args.push('--guardrails', options.guardrails);
|
|
161
149
|
|
|
162
150
|
const result = await runProcess(fmBin, args, {
|
|
163
151
|
input: prompt,
|
|
@@ -166,8 +154,10 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
166
154
|
|
|
167
155
|
const output = stripAnsi(result.stdout).trim();
|
|
168
156
|
const errorText = stripAnsi(result.stderr).trim();
|
|
157
|
+
const failed = result.code !== 0 || result.timedOut;
|
|
158
|
+
|
|
169
159
|
return {
|
|
170
|
-
ok:
|
|
160
|
+
ok: !failed,
|
|
171
161
|
model,
|
|
172
162
|
prompt,
|
|
173
163
|
output,
|
|
@@ -179,11 +169,17 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
179
169
|
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
180
170
|
streamed,
|
|
181
171
|
stdoutChunks: result.stdoutChunks,
|
|
182
|
-
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
|
|
172
|
+
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : [],
|
|
173
|
+
error: failed
|
|
174
|
+
? (result.timedOut
|
|
175
|
+
? `timed out after ${options.timeoutMs ?? 60_000}ms`
|
|
176
|
+
: firstLine(errorText) || `fm exited with code ${result.code ?? result.signal}`)
|
|
177
|
+
: ''
|
|
183
178
|
};
|
|
184
179
|
}
|
|
185
180
|
|
|
186
|
-
export async function collectEnvironment(fmBin) {
|
|
181
|
+
export async function collectEnvironment(fmBin, options = {}) {
|
|
182
|
+
const capabilities = options.capabilities;
|
|
187
183
|
const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
|
|
188
184
|
const macOS = stripAnsi(swVers.stdout).trim() || null;
|
|
189
185
|
|
|
@@ -196,15 +192,15 @@ export async function collectEnvironment(fmBin) {
|
|
|
196
192
|
const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
|
|
197
193
|
const battery = parseBatteryOutput(`${batteryResult.stdout || ''}${batteryResult.stderr || ''}`);
|
|
198
194
|
|
|
199
|
-
let fmHelpDigest = null;
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
fmHelpDigest =
|
|
195
|
+
let fmHelpDigest = capabilities?.digest ?? null;
|
|
196
|
+
if (fmHelpDigest == null) {
|
|
197
|
+
try {
|
|
198
|
+
const help = await getFmHelp(fmBin, 10_000);
|
|
199
|
+
const detected = await detectFmCapabilities(fmBin, { help: { text: help.text } });
|
|
200
|
+
fmHelpDigest = detected.digest;
|
|
201
|
+
} catch {
|
|
202
|
+
fmHelpDigest = null;
|
|
205
203
|
}
|
|
206
|
-
} catch {
|
|
207
|
-
fmHelpDigest = null;
|
|
208
204
|
}
|
|
209
205
|
|
|
210
206
|
const memRaw = (memBytes.stdout || '').trim();
|
|
@@ -237,11 +233,3 @@ export async function collectEnvironment(fmBin) {
|
|
|
237
233
|
: null
|
|
238
234
|
};
|
|
239
235
|
}
|
|
240
|
-
|
|
241
|
-
function escapeRegExp(value) {
|
|
242
|
-
return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
function titleCase(value) {
|
|
246
|
-
return value.slice(0, 1).toUpperCase() + value.slice(1);
|
|
247
|
-
}
|
package/src/history.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import fs from 'node:fs/promises';
|
|
2
2
|
import path from 'node:path';
|
|
3
|
+
import { stripAnsi } from './ansi.js';
|
|
3
4
|
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
4
5
|
|
|
5
6
|
export async function loadHistory(dir) {
|
|
@@ -100,7 +101,7 @@ export function renderHistoryReport(reports, options = {}) {
|
|
|
100
101
|
}
|
|
101
102
|
|
|
102
103
|
function fit(text, width) {
|
|
103
|
-
const str = String(text ?? '');
|
|
104
|
+
const str = stripAnsi(String(text ?? '')).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
|
|
104
105
|
if (str.length <= width) return str + ' '.repeat(width - str.length);
|
|
105
106
|
return `${str.slice(0, width - 1)}…`;
|
|
106
107
|
}
|
package/src/macos.js
CHANGED
|
@@ -57,7 +57,8 @@ export function evaluateMacosSupport(platform, parsed) {
|
|
|
57
57
|
export function formatMacosRequirementError({ reason, latestSupported }) {
|
|
58
58
|
return [
|
|
59
59
|
`unsupported macOS: ${reason}`,
|
|
60
|
-
`Latest supported: ${latestSupported} (fm is not available on older macOS releases)
|
|
60
|
+
`Latest supported: ${latestSupported} (fm is not available on older macOS releases).`,
|
|
61
|
+
'Pass --fm-bin <path> (or set FM_BIN) to benchmark an fm binary you provide on this host.'
|
|
61
62
|
].join('\n');
|
|
62
63
|
}
|
|
63
64
|
|
package/src/metrics.js
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
// Metric provenance catalog.
|
|
2
|
+
//
|
|
3
|
+
// fm-bench measures `fm` from the outside, so not every named metric can be
|
|
4
|
+
// observed directly. Each metric declares how it is obtained:
|
|
5
|
+
//
|
|
6
|
+
// measured — observed directly (process timings, exit codes, token counts)
|
|
7
|
+
// proxy — observed at a coarser granularity than the ideal metric
|
|
8
|
+
// derived — computed from other measured values
|
|
9
|
+
// controlled — an input setting, not a measurement
|
|
10
|
+
//
|
|
11
|
+
// A metric whose requirement is missing from the detected `fm` build is
|
|
12
|
+
// reported as unavailable with a reason instead of a blank or invented value.
|
|
13
|
+
|
|
14
|
+
const DEFINITIONS = [
|
|
15
|
+
{
|
|
16
|
+
key: 'ttft',
|
|
17
|
+
label: 'TTFT',
|
|
18
|
+
kind: 'proxy',
|
|
19
|
+
source: 'arrival time of the first streamed stdout chunk',
|
|
20
|
+
requires: ['streaming'],
|
|
21
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
key: 'e2eLatency',
|
|
25
|
+
label: 'E2E latency',
|
|
26
|
+
kind: 'measured',
|
|
27
|
+
source: 'wall clock from spawning fm until it exits',
|
|
28
|
+
requires: []
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
key: 'generationMs',
|
|
32
|
+
label: 'generation time',
|
|
33
|
+
kind: 'derived',
|
|
34
|
+
source: 'E2E latency minus TTFT',
|
|
35
|
+
requires: ['streaming'],
|
|
36
|
+
reason: 'requires at least two streamed output chunks to separate prefill from decode'
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
key: 'tpot',
|
|
40
|
+
label: 'TPOT',
|
|
41
|
+
kind: 'derived',
|
|
42
|
+
source: '(E2E - TTFT) / (output tokens - 1)',
|
|
43
|
+
requires: ['streaming', 'tokenCounting'],
|
|
44
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
key: 'promptTokens',
|
|
48
|
+
label: 'prompt tokens',
|
|
49
|
+
kind: 'measured',
|
|
50
|
+
source: 'fm count-tokens on the prompt',
|
|
51
|
+
requires: ['tokenCounting'],
|
|
52
|
+
reason: 'this fm build exposes no token-counting command'
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
key: 'outputTokens',
|
|
56
|
+
label: 'output tokens',
|
|
57
|
+
kind: 'measured',
|
|
58
|
+
source: 'fm count-tokens on the captured output',
|
|
59
|
+
requires: ['tokenCounting'],
|
|
60
|
+
reason: 'this fm build exposes no token-counting command'
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
key: 'tokensPerSecond',
|
|
64
|
+
label: 'per-request output tokens/s',
|
|
65
|
+
kind: 'derived',
|
|
66
|
+
source: 'output tokens / E2E seconds',
|
|
67
|
+
requires: ['tokenCounting'],
|
|
68
|
+
reason: 'requires a token-counting fm command'
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
key: 'decodeTokensPerSecond',
|
|
72
|
+
label: 'decode tokens/s',
|
|
73
|
+
kind: 'derived',
|
|
74
|
+
source: '(output tokens - 1) / generation seconds',
|
|
75
|
+
requires: ['streaming', 'tokenCounting'],
|
|
76
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
key: 'prefillTokensPerSecond',
|
|
80
|
+
label: 'prefill tokens/s',
|
|
81
|
+
kind: 'proxy',
|
|
82
|
+
source: 'prompt tokens / TTFT seconds',
|
|
83
|
+
requires: ['streaming', 'tokenCounting'],
|
|
84
|
+
reason: 'requires streaming plus a token-counting fm command'
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
key: 'outputTokenThroughput',
|
|
88
|
+
label: 'aggregate output token throughput',
|
|
89
|
+
kind: 'derived',
|
|
90
|
+
source: 'successful output tokens / measured wall-clock window',
|
|
91
|
+
requires: ['tokenCounting'],
|
|
92
|
+
reason: 'requires a token-counting fm command'
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
key: 'rps',
|
|
96
|
+
label: 'request throughput',
|
|
97
|
+
kind: 'measured',
|
|
98
|
+
source: 'successful requests / measured wall-clock window',
|
|
99
|
+
requires: []
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
key: 'chunkGaps',
|
|
103
|
+
label: 'chunk gaps and second-chunk delay',
|
|
104
|
+
kind: 'proxy',
|
|
105
|
+
source: 'gaps between consecutive streamed stdout chunks',
|
|
106
|
+
requires: ['streaming'],
|
|
107
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
key: 'percentiles',
|
|
111
|
+
label: 'percentiles',
|
|
112
|
+
kind: 'derived',
|
|
113
|
+
source: 'percentile interpolation over successful samples',
|
|
114
|
+
requires: []
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
key: 'successRate',
|
|
118
|
+
label: 'success rate',
|
|
119
|
+
kind: 'measured',
|
|
120
|
+
source: 'successful runs / attempted runs',
|
|
121
|
+
requires: []
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
key: 'goodput',
|
|
125
|
+
label: 'goodput',
|
|
126
|
+
kind: 'derived',
|
|
127
|
+
source: 'successful runs meeting every configured SLO / runs with an SLO verdict',
|
|
128
|
+
requires: ['slo'],
|
|
129
|
+
reason: 'set --slo-ttft-ms, --slo-e2e-ms, or --slo-tpot-ms to enable goodput'
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
key: 'repeatability',
|
|
133
|
+
label: 'repeatability',
|
|
134
|
+
kind: 'derived',
|
|
135
|
+
source: 'most common normalized output hash share across repeated runs',
|
|
136
|
+
requires: []
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
key: 'variability',
|
|
140
|
+
label: 'CV and 95% confidence interval',
|
|
141
|
+
kind: 'derived',
|
|
142
|
+
source: 'sample standard deviation over successful samples',
|
|
143
|
+
requires: [],
|
|
144
|
+
reason: 'needs at least two successful samples'
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
key: 'quota',
|
|
148
|
+
label: 'quota',
|
|
149
|
+
kind: 'measured',
|
|
150
|
+
source: 'fm quota-usage',
|
|
151
|
+
requires: ['quota'],
|
|
152
|
+
reason: 'this fm build exposes no quota command'
|
|
153
|
+
}
|
|
154
|
+
];
|
|
155
|
+
|
|
156
|
+
function requirementsMet(requires = [], context = {}) {
|
|
157
|
+
return requires.every((name) => {
|
|
158
|
+
if (name === 'streaming') return Boolean(context.streaming);
|
|
159
|
+
if (name === 'tokenCounting') return Boolean(context.tokenCounting);
|
|
160
|
+
if (name === 'quota') return Boolean(context.quota);
|
|
161
|
+
if (name === 'slo') return Boolean(context.slo);
|
|
162
|
+
return true;
|
|
163
|
+
});
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Describe which metrics the detected `fm` build can support for a run.
|
|
168
|
+
*
|
|
169
|
+
* @param {{ features?: Record<string, any> }} [capabilities]
|
|
170
|
+
* @param {{ stream?: boolean, slo?: boolean }} [options]
|
|
171
|
+
*/
|
|
172
|
+
export function metricAvailability(capabilities = {}, options = {}) {
|
|
173
|
+
const features = capabilities.features ?? {};
|
|
174
|
+
const context = {
|
|
175
|
+
streaming: options.stream !== false && features.streaming !== false,
|
|
176
|
+
tokenCounting: features.tokenCounting !== false,
|
|
177
|
+
quota: features.quota === true,
|
|
178
|
+
slo: options.slo === true
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
const metrics = {};
|
|
182
|
+
for (const definition of DEFINITIONS) {
|
|
183
|
+
const available = requirementsMet(definition.requires, context);
|
|
184
|
+
metrics[definition.key] = {
|
|
185
|
+
label: definition.label,
|
|
186
|
+
kind: definition.kind,
|
|
187
|
+
source: definition.source,
|
|
188
|
+
available,
|
|
189
|
+
unavailableReason: available ? '' : (definition.reason ?? 'unavailable in this environment')
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
return metrics;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Short human line for CLI output, e.g. "token counting: yes, quota: no". */
|
|
196
|
+
export function formatCapabilitySummary(capabilities = {}) {
|
|
197
|
+
const features = capabilities.features ?? {};
|
|
198
|
+
return [
|
|
199
|
+
`token counting ${features.tokenCounting ? 'yes' : 'no'}`,
|
|
200
|
+
`streaming ${features.streaming ? 'yes' : 'no'}`,
|
|
201
|
+
`quota ${features.quota ? 'yes' : 'no'}`
|
|
202
|
+
].join(', ');
|
|
203
|
+
}
|
package/src/process.js
CHANGED
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
import { spawn } from 'node:child_process';
|
|
2
2
|
|
|
3
|
+
// Children currently alive. The CLI registers signal handlers that terminate
|
|
4
|
+
// these so Ctrl+C cannot leave `fm` processes behind.
|
|
5
|
+
const activeChildren = new Set();
|
|
6
|
+
|
|
7
|
+
export function activeChildCount() {
|
|
8
|
+
return activeChildren.size;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Terminate every running child process. Returns how many were signalled.
|
|
13
|
+
* @param {NodeJS.Signals} signal
|
|
14
|
+
*/
|
|
15
|
+
export function killActiveChildren(signal = 'SIGTERM') {
|
|
16
|
+
let killed = 0;
|
|
17
|
+
for (const child of activeChildren) {
|
|
18
|
+
if (child.exitCode != null || child.signalCode != null) continue;
|
|
19
|
+
try {
|
|
20
|
+
child.kill(signal);
|
|
21
|
+
killed += 1;
|
|
22
|
+
} catch {
|
|
23
|
+
// Process already gone.
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
return killed;
|
|
27
|
+
}
|
|
28
|
+
|
|
3
29
|
export function runProcess(command, args = [], options = {}) {
|
|
4
30
|
const {
|
|
5
31
|
input,
|
|
@@ -15,6 +41,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
15
41
|
env,
|
|
16
42
|
stdio: ['pipe', 'pipe', 'pipe']
|
|
17
43
|
});
|
|
44
|
+
activeChildren.add(child);
|
|
18
45
|
|
|
19
46
|
let stdout = '';
|
|
20
47
|
let stderr = '';
|
|
@@ -55,11 +82,16 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
55
82
|
stderr += chunk;
|
|
56
83
|
});
|
|
57
84
|
|
|
58
|
-
|
|
85
|
+
const finish = (result) => {
|
|
86
|
+
if (settled) return;
|
|
59
87
|
settled = true;
|
|
88
|
+
activeChildren.delete(child);
|
|
60
89
|
if (timer) clearTimeout(timer);
|
|
61
|
-
|
|
62
|
-
|
|
90
|
+
resolve(result);
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
child.on('error', (error) => {
|
|
94
|
+
finish({
|
|
63
95
|
command,
|
|
64
96
|
args,
|
|
65
97
|
code: null,
|
|
@@ -73,15 +105,12 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
73
105
|
firstStderrMs,
|
|
74
106
|
error,
|
|
75
107
|
timedOut,
|
|
76
|
-
durationMs: Number(
|
|
108
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
77
109
|
});
|
|
78
110
|
});
|
|
79
111
|
|
|
80
112
|
child.on('close', (code, signal) => {
|
|
81
|
-
|
|
82
|
-
if (timer) clearTimeout(timer);
|
|
83
|
-
const endedAt = process.hrtime.bigint();
|
|
84
|
-
resolve({
|
|
113
|
+
finish({
|
|
85
114
|
command,
|
|
86
115
|
args,
|
|
87
116
|
code,
|
|
@@ -93,8 +122,9 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
93
122
|
stdoutChunkTimesMs,
|
|
94
123
|
firstStdoutMs,
|
|
95
124
|
firstStderrMs,
|
|
125
|
+
error: null,
|
|
96
126
|
timedOut,
|
|
97
|
-
durationMs: Number(
|
|
127
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
98
128
|
});
|
|
99
129
|
});
|
|
100
130
|
|
package/src/prompts.js
CHANGED
|
@@ -195,16 +195,28 @@ export async function loadPrompts(options = {}) {
|
|
|
195
195
|
|
|
196
196
|
async function loadPromptFile(filePath) {
|
|
197
197
|
const absolutePath = path.resolve(filePath);
|
|
198
|
-
|
|
198
|
+
let content;
|
|
199
|
+
try {
|
|
200
|
+
content = await fs.readFile(absolutePath, 'utf8');
|
|
201
|
+
} catch (error) {
|
|
202
|
+
throw new Error(error.code === 'ENOENT'
|
|
203
|
+
? `Prompt file not found: ${absolutePath}`
|
|
204
|
+
: `Cannot read prompt file ${absolutePath}: ${error.message}`);
|
|
205
|
+
}
|
|
199
206
|
const trimmed = content.trim();
|
|
200
207
|
|
|
201
208
|
if (!trimmed) return [];
|
|
202
209
|
|
|
203
210
|
if (absolutePath.endsWith('.json')) {
|
|
204
|
-
|
|
211
|
+
let parsed;
|
|
212
|
+
try {
|
|
213
|
+
parsed = JSON.parse(trimmed);
|
|
214
|
+
} catch (error) {
|
|
215
|
+
throw new Error(`Cannot parse ${absolutePath} as JSON: ${error.message}`);
|
|
216
|
+
}
|
|
205
217
|
const items = Array.isArray(parsed) ? parsed : parsed.prompts;
|
|
206
218
|
if (!Array.isArray(items)) {
|
|
207
|
-
throw new Error(
|
|
219
|
+
throw new Error(`${absolutePath} must be a JSON array of prompts or an object with a prompts array`);
|
|
208
220
|
}
|
|
209
221
|
return items.map((item, index) => normalizePromptItem(item, index));
|
|
210
222
|
}
|
|
@@ -212,7 +224,13 @@ async function loadPromptFile(filePath) {
|
|
|
212
224
|
if (absolutePath.endsWith('.jsonl')) {
|
|
213
225
|
return trimmed.split(/\r?\n/)
|
|
214
226
|
.filter(Boolean)
|
|
215
|
-
.map((line, index) =>
|
|
227
|
+
.map((line, index) => {
|
|
228
|
+
try {
|
|
229
|
+
return normalizePromptItem(JSON.parse(line), index);
|
|
230
|
+
} catch (error) {
|
|
231
|
+
throw new Error(`Cannot parse line ${index + 1} of ${absolutePath} as JSON: ${error.message}`);
|
|
232
|
+
}
|
|
233
|
+
});
|
|
216
234
|
}
|
|
217
235
|
|
|
218
236
|
return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({
|