fm-bench 0.6.3 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +33 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +19 -2
- package/package.json +5 -4
- package/src/bench.js +129 -37
- package/src/capabilities.js +196 -0
- package/src/cli.js +153 -68
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +110 -106
- package/src/history.js +2 -1
- package/src/macos.js +2 -1
- package/src/metrics.js +212 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/src/fm.js
CHANGED
|
@@ -1,60 +1,16 @@
|
|
|
1
|
-
import crypto from 'node:crypto';
|
|
2
1
|
import os from 'node:os';
|
|
3
2
|
import { stripAnsi } from './ansi.js';
|
|
3
|
+
import { detectFmCapabilities } from './capabilities.js';
|
|
4
|
+
import { firstLine, isUnsupportedModelError, parseAvailabilityOutput } from './fm-help.js';
|
|
4
5
|
import { runProcess } from './process.js';
|
|
5
6
|
import { parseBatteryOutput, parseThermalOutput } from './system.js';
|
|
6
7
|
|
|
7
|
-
|
|
8
|
-
{ name: 'system', description: 'On-device Apple Foundation Model' },
|
|
9
|
-
{ name: 'pcc', description: 'Apple Foundation Model on Private Cloud Compute' }
|
|
10
|
-
];
|
|
8
|
+
export { parseModelsFromHelp, parseAvailabilityOutput } from './fm-help.js';
|
|
11
9
|
|
|
12
10
|
export function fmBinaryFromOptions(options = {}) {
|
|
13
11
|
return options.fmBin || process.env.FM_BIN || 'fm';
|
|
14
12
|
}
|
|
15
13
|
|
|
16
|
-
export function parseModelsFromHelp(helpText) {
|
|
17
|
-
const clean = stripAnsi(helpText);
|
|
18
|
-
const lines = clean.split(/\r?\n/);
|
|
19
|
-
const models = new Map();
|
|
20
|
-
let inModels = false;
|
|
21
|
-
|
|
22
|
-
for (const line of lines) {
|
|
23
|
-
if (/^\s*MODELS\s*$/.test(line)) {
|
|
24
|
-
inModels = true;
|
|
25
|
-
continue;
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
if (inModels && /^\s*[A-Z][A-Z -]+\s*$/.test(line) && !/^\s*MODELS\s*$/.test(line)) {
|
|
29
|
-
inModels = false;
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
if (inModels) {
|
|
33
|
-
const match = line.match(/^\s*([A-Za-z0-9._:-]+)\s{2,}(.+?)\s*$/);
|
|
34
|
-
if (match) {
|
|
35
|
-
models.set(match[1], {
|
|
36
|
-
name: match[1],
|
|
37
|
-
description: match[2].replace(/\s*\(default\)\s*$/, '').trim()
|
|
38
|
-
});
|
|
39
|
-
}
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
const optionMatch = /--model\b/.test(line)
|
|
43
|
-
? line.match(/\bmodel\b.*?\(([^)]+)\)/i)
|
|
44
|
-
: null;
|
|
45
|
-
if (optionMatch) {
|
|
46
|
-
for (const raw of optionMatch[1].split(',')) {
|
|
47
|
-
const name = raw.trim();
|
|
48
|
-
if (/^[A-Za-z0-9._:-]+$/.test(name) && !models.has(name)) {
|
|
49
|
-
models.set(name, { name, description: '' });
|
|
50
|
-
}
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
return [...models.values()];
|
|
56
|
-
}
|
|
57
|
-
|
|
58
14
|
export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
59
15
|
const result = await runProcess(fmBin, ['--help'], { timeoutMs });
|
|
60
16
|
if (result.error) {
|
|
@@ -69,40 +25,42 @@ export async function getFmHelp(fmBin, timeoutMs = 10_000) {
|
|
|
69
25
|
};
|
|
70
26
|
}
|
|
71
27
|
|
|
28
|
+
/**
|
|
29
|
+
* Discover models, preferring an already-detected capability probe so a run
|
|
30
|
+
* does not spawn `fm --help` more than once.
|
|
31
|
+
* @param {Record<string, any>} options
|
|
32
|
+
*/
|
|
72
33
|
export async function discoverModels(options = {}) {
|
|
73
34
|
const fmBin = fmBinaryFromOptions(options);
|
|
74
|
-
const
|
|
75
|
-
let models = parseModelsFromHelp(help.text);
|
|
76
|
-
|
|
77
|
-
if (models.length === 0 && /Apple Foundation Models CLI/i.test(stripAnsi(help.text))) {
|
|
78
|
-
models = DEFAULT_MODELS;
|
|
79
|
-
}
|
|
80
|
-
|
|
35
|
+
const capabilities = options.capabilities ?? await detectFmCapabilities(fmBin, options);
|
|
81
36
|
return {
|
|
82
37
|
fmBin,
|
|
83
|
-
models,
|
|
84
|
-
help:
|
|
85
|
-
|
|
86
|
-
}
|
|
87
|
-
|
|
88
|
-
export function parseAvailabilityOutput(model, output, code) {
|
|
89
|
-
const clean = stripAnsi(output).trim();
|
|
90
|
-
const lower = clean.toLowerCase();
|
|
91
|
-
const modelLower = model.toLowerCase();
|
|
92
|
-
const hasError = /\berror:|\bunavailable\b|\bnot available\b|\bnot supported\b/.test(lower);
|
|
93
|
-
const hasAvailable = new RegExp(`\\b${escapeRegExp(modelLower)}\\b[\\s\\S]{0,80}\\bavailable\\b|\\bavailable\\b[\\s\\S]{0,80}\\b${escapeRegExp(modelLower)}\\b`).test(lower)
|
|
94
|
-
|| lower.includes(`${modelLower} model available`)
|
|
95
|
-
|| lower.includes(`${titleCase(modelLower)} model available`.toLowerCase());
|
|
96
|
-
|
|
97
|
-
return {
|
|
98
|
-
model,
|
|
99
|
-
available: code === 0 && hasAvailable && !hasError,
|
|
100
|
-
raw: clean,
|
|
101
|
-
reason: hasError ? clean : ''
|
|
38
|
+
models: capabilities.models,
|
|
39
|
+
help: capabilities.help,
|
|
40
|
+
capabilities
|
|
102
41
|
};
|
|
103
42
|
}
|
|
104
43
|
|
|
44
|
+
/**
|
|
45
|
+
* Check one model against the detected `fm` build.
|
|
46
|
+
*
|
|
47
|
+
* Models the build does not expose are reported as unsupported without
|
|
48
|
+
* spawning `fm` at all, so a raw argument-error blob from `fm` can never end
|
|
49
|
+
* up in a report or in `fm-bench models` output.
|
|
50
|
+
*/
|
|
105
51
|
export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
52
|
+
const capabilities = options.capabilities;
|
|
53
|
+
const known = capabilities?.models?.map((entry) => entry.name) ?? [];
|
|
54
|
+
if (capabilities && known.length > 0 && !known.includes(model)) {
|
|
55
|
+
return {
|
|
56
|
+
model,
|
|
57
|
+
available: false,
|
|
58
|
+
unsupported: true,
|
|
59
|
+
raw: '',
|
|
60
|
+
reason: `not supported by this fm build (supported: ${known.join(', ')})`
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
106
64
|
const result = await runProcess(fmBin, ['available', '--model', model], {
|
|
107
65
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
108
66
|
});
|
|
@@ -111,53 +69,99 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
|
|
|
111
69
|
if (result.error) {
|
|
112
70
|
parsed.available = false;
|
|
113
71
|
parsed.reason = result.stderr || result.error.message;
|
|
72
|
+
return parsed;
|
|
73
|
+
}
|
|
74
|
+
if (isUnsupportedModelError(output)) {
|
|
75
|
+
parsed.available = false;
|
|
76
|
+
parsed.unsupported = true;
|
|
77
|
+
const supported = known.length > 0 ? ` (supported: ${known.join(', ')})` : '';
|
|
78
|
+
parsed.reason = `not supported by this fm build${supported}`;
|
|
114
79
|
}
|
|
115
80
|
return parsed;
|
|
116
81
|
}
|
|
117
82
|
|
|
83
|
+
/**
|
|
84
|
+
* Query quota information when the build exposes it.
|
|
85
|
+
* @returns {Promise<{ model: string, supported: boolean, ok: boolean, raw: string, reason: string }>}
|
|
86
|
+
*/
|
|
118
87
|
export async function getQuotaUsage(fmBin, model, options = {}) {
|
|
88
|
+
const features = options.capabilities?.features;
|
|
89
|
+
if (features && !features.quota) {
|
|
90
|
+
return {
|
|
91
|
+
model,
|
|
92
|
+
supported: false,
|
|
93
|
+
ok: false,
|
|
94
|
+
raw: '',
|
|
95
|
+
reason: 'unavailable: this fm build exposes no quota command'
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
|
|
119
99
|
const result = await runProcess(fmBin, ['quota-usage', '--model', model], {
|
|
120
100
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
121
101
|
});
|
|
122
102
|
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
123
103
|
return {
|
|
124
104
|
model,
|
|
105
|
+
supported: true,
|
|
125
106
|
ok: result.code === 0,
|
|
126
107
|
raw: output,
|
|
127
|
-
|
|
108
|
+
reason: result.code === 0 ? '' : firstLine(output)
|
|
128
109
|
};
|
|
129
110
|
}
|
|
130
111
|
|
|
112
|
+
/**
|
|
113
|
+
* Count tokens with whichever token-counting command this `fm` build exposes.
|
|
114
|
+
* Returns `ok: false` with a reason when the build cannot count tokens; it
|
|
115
|
+
* never invents a count.
|
|
116
|
+
*/
|
|
131
117
|
export async function countTokens(fmBin, text, options = {}) {
|
|
132
|
-
const
|
|
118
|
+
const command = options.capabilities?.features?.tokenCountCommand ?? 'count-tokens';
|
|
119
|
+
const supported = options.capabilities?.features?.tokenCounting ?? true;
|
|
120
|
+
if (!supported || !command) {
|
|
121
|
+
return {
|
|
122
|
+
ok: false,
|
|
123
|
+
count: null,
|
|
124
|
+
unsupported: true,
|
|
125
|
+
raw: '',
|
|
126
|
+
reason: 'token counting is unavailable in this fm build'
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const result = await runProcess(fmBin, [command, '--quiet'], {
|
|
133
131
|
input: text,
|
|
134
132
|
timeoutMs: options.timeoutMs ?? 15_000
|
|
135
133
|
});
|
|
136
134
|
const output = stripAnsi(`${result.stdout}${result.stderr}`).trim();
|
|
137
|
-
const match = output.match(
|
|
138
|
-
if (result.code !== 0 || !match) {
|
|
135
|
+
const match = output.match(/\d+/);
|
|
136
|
+
if (result.error || result.code !== 0 || !match) {
|
|
139
137
|
return {
|
|
140
138
|
ok: false,
|
|
141
139
|
count: null,
|
|
142
|
-
raw: output
|
|
140
|
+
raw: output,
|
|
141
|
+
reason: result.error?.message || firstLine(output) || `fm ${command} exited with code ${result.code}`
|
|
143
142
|
};
|
|
144
143
|
}
|
|
145
144
|
return {
|
|
146
145
|
ok: true,
|
|
147
146
|
count: Number.parseInt(match[0], 10),
|
|
148
|
-
raw: output
|
|
147
|
+
raw: output,
|
|
148
|
+
reason: ''
|
|
149
149
|
};
|
|
150
150
|
}
|
|
151
151
|
|
|
152
152
|
export async function respond(fmBin, model, prompt, options = {}) {
|
|
153
|
-
const
|
|
154
|
-
const
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
if (
|
|
160
|
-
if (
|
|
153
|
+
const features = options.capabilities?.features;
|
|
154
|
+
const streamControl = features ? features.streaming : true;
|
|
155
|
+
const modelSelection = features ? features.modelSelection : true;
|
|
156
|
+
const streamed = streamControl && options.stream !== false;
|
|
157
|
+
|
|
158
|
+
const args = ['respond'];
|
|
159
|
+
if (modelSelection) args.push('--model', model);
|
|
160
|
+
if (streamControl && !streamed) args.push('--no-stream');
|
|
161
|
+
if (options.greedy && (features?.greedy ?? true)) args.push('--greedy');
|
|
162
|
+
if (options.instructions && (features?.instructions ?? true)) args.push('--instructions', options.instructions);
|
|
163
|
+
if (options.useCase && (features?.useCase ?? true)) args.push('--use-case', options.useCase);
|
|
164
|
+
if (options.guardrails && (features?.guardrails ?? true)) args.push('--guardrails', options.guardrails);
|
|
161
165
|
|
|
162
166
|
const result = await runProcess(fmBin, args, {
|
|
163
167
|
input: prompt,
|
|
@@ -166,8 +170,10 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
166
170
|
|
|
167
171
|
const output = stripAnsi(result.stdout).trim();
|
|
168
172
|
const errorText = stripAnsi(result.stderr).trim();
|
|
173
|
+
const failed = result.code !== 0 || result.timedOut;
|
|
174
|
+
|
|
169
175
|
return {
|
|
170
|
-
ok:
|
|
176
|
+
ok: !failed,
|
|
171
177
|
model,
|
|
172
178
|
prompt,
|
|
173
179
|
output,
|
|
@@ -179,11 +185,17 @@ export async function respond(fmBin, model, prompt, options = {}) {
|
|
|
179
185
|
firstOutputMs: streamed ? result.firstStdoutMs : null,
|
|
180
186
|
streamed,
|
|
181
187
|
stdoutChunks: result.stdoutChunks,
|
|
182
|
-
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : []
|
|
188
|
+
stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : [],
|
|
189
|
+
error: failed
|
|
190
|
+
? (result.timedOut
|
|
191
|
+
? `timed out after ${options.timeoutMs ?? 60_000}ms`
|
|
192
|
+
: firstLine(errorText) || `fm exited with code ${result.code ?? result.signal}`)
|
|
193
|
+
: ''
|
|
183
194
|
};
|
|
184
195
|
}
|
|
185
196
|
|
|
186
|
-
export async function collectEnvironment(fmBin) {
|
|
197
|
+
export async function collectEnvironment(fmBin, options = {}) {
|
|
198
|
+
const capabilities = options.capabilities;
|
|
187
199
|
const swVers = await runProcess('sw_vers', [], { timeoutMs: 5_000 });
|
|
188
200
|
const macOS = stripAnsi(swVers.stdout).trim() || null;
|
|
189
201
|
|
|
@@ -196,15 +208,15 @@ export async function collectEnvironment(fmBin) {
|
|
|
196
208
|
const batteryResult = await runProcess('pmset', ['-g', 'batt'], { timeoutMs: 5_000 });
|
|
197
209
|
const battery = parseBatteryOutput(`${batteryResult.stdout || ''}${batteryResult.stderr || ''}`);
|
|
198
210
|
|
|
199
|
-
let fmHelpDigest = null;
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
fmHelpDigest =
|
|
211
|
+
let fmHelpDigest = capabilities?.digest ?? null;
|
|
212
|
+
if (fmHelpDigest == null) {
|
|
213
|
+
try {
|
|
214
|
+
const help = await getFmHelp(fmBin, 10_000);
|
|
215
|
+
const detected = await detectFmCapabilities(fmBin, { help: { text: help.text } });
|
|
216
|
+
fmHelpDigest = detected.digest;
|
|
217
|
+
} catch {
|
|
218
|
+
fmHelpDigest = null;
|
|
205
219
|
}
|
|
206
|
-
} catch {
|
|
207
|
-
fmHelpDigest = null;
|
|
208
220
|
}
|
|
209
221
|
|
|
210
222
|
const memRaw = (memBytes.stdout || '').trim();
|
|
@@ -237,11 +249,3 @@ export async function collectEnvironment(fmBin) {
|
|
|
237
249
|
: null
|
|
238
250
|
};
|
|
239
251
|
}
|
|
240
|
-
|
|
241
|
-
function escapeRegExp(value) {
|
|
242
|
-
return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
function titleCase(value) {
|
|
246
|
-
return value.slice(0, 1).toUpperCase() + value.slice(1);
|
|
247
|
-
}
|
package/src/history.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import fs from 'node:fs/promises';
|
|
2
2
|
import path from 'node:path';
|
|
3
|
+
import { stripAnsi } from './ansi.js';
|
|
3
4
|
import { formatMs, formatNumber, formatPercent } from './table.js';
|
|
4
5
|
|
|
5
6
|
export async function loadHistory(dir) {
|
|
@@ -100,7 +101,7 @@ export function renderHistoryReport(reports, options = {}) {
|
|
|
100
101
|
}
|
|
101
102
|
|
|
102
103
|
function fit(text, width) {
|
|
103
|
-
const str = String(text ?? '');
|
|
104
|
+
const str = stripAnsi(String(text ?? '')).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
|
|
104
105
|
if (str.length <= width) return str + ' '.repeat(width - str.length);
|
|
105
106
|
return `${str.slice(0, width - 1)}…`;
|
|
106
107
|
}
|
package/src/macos.js
CHANGED
|
@@ -57,7 +57,8 @@ export function evaluateMacosSupport(platform, parsed) {
|
|
|
57
57
|
export function formatMacosRequirementError({ reason, latestSupported }) {
|
|
58
58
|
return [
|
|
59
59
|
`unsupported macOS: ${reason}`,
|
|
60
|
-
`Latest supported: ${latestSupported} (fm is not available on older macOS releases)
|
|
60
|
+
`Latest supported: ${latestSupported} (fm is not available on older macOS releases).`,
|
|
61
|
+
'Pass --fm-bin <path> (or set FM_BIN) to benchmark an fm binary you provide on this host.'
|
|
61
62
|
].join('\n');
|
|
62
63
|
}
|
|
63
64
|
|
package/src/metrics.js
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
// Metric provenance catalog.
|
|
2
|
+
//
|
|
3
|
+
// fm-bench measures `fm` from the outside, so not every named metric can be
|
|
4
|
+
// observed directly. Each metric declares how it is obtained:
|
|
5
|
+
//
|
|
6
|
+
// measured — observed directly (process timings, exit codes, token counts)
|
|
7
|
+
// proxy — observed at a coarser granularity than the ideal metric
|
|
8
|
+
// derived — computed from other measured values
|
|
9
|
+
// controlled — an input setting, not a measurement
|
|
10
|
+
//
|
|
11
|
+
// A metric whose requirement is missing from the detected `fm` build is
|
|
12
|
+
// reported as unavailable with a reason instead of a blank or invented value.
|
|
13
|
+
|
|
14
|
+
const DEFINITIONS = [
|
|
15
|
+
{
|
|
16
|
+
key: 'ttft',
|
|
17
|
+
label: 'TTFT',
|
|
18
|
+
kind: 'proxy',
|
|
19
|
+
source: 'arrival time of the first streamed stdout chunk',
|
|
20
|
+
requires: ['streaming'],
|
|
21
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
key: 'e2eLatency',
|
|
25
|
+
label: 'E2E latency',
|
|
26
|
+
kind: 'measured',
|
|
27
|
+
source: 'wall clock from spawning fm until it exits',
|
|
28
|
+
requires: []
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
key: 'generationMs',
|
|
32
|
+
label: 'generation time',
|
|
33
|
+
kind: 'derived',
|
|
34
|
+
source: 'E2E latency minus TTFT',
|
|
35
|
+
requires: ['streaming'],
|
|
36
|
+
reason: 'requires at least two streamed output chunks to separate prefill from decode'
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
key: 'tpot',
|
|
40
|
+
label: 'TPOT',
|
|
41
|
+
kind: 'derived',
|
|
42
|
+
source: '(E2E - TTFT) / (output tokens - 1)',
|
|
43
|
+
requires: ['streaming', 'tokenCounting'],
|
|
44
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
key: 'promptTokens',
|
|
48
|
+
label: 'prompt tokens',
|
|
49
|
+
kind: 'measured',
|
|
50
|
+
source: 'fm count-tokens on the prompt',
|
|
51
|
+
requires: ['tokenCounting'],
|
|
52
|
+
reason: 'this fm build exposes no token-counting command'
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
key: 'outputTokens',
|
|
56
|
+
label: 'output tokens',
|
|
57
|
+
kind: 'measured',
|
|
58
|
+
source: 'fm count-tokens on the captured output',
|
|
59
|
+
requires: ['tokenCounting'],
|
|
60
|
+
reason: 'this fm build exposes no token-counting command'
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
key: 'tokensPerSecond',
|
|
64
|
+
label: 'per-request output tokens/s',
|
|
65
|
+
kind: 'derived',
|
|
66
|
+
source: 'output tokens / E2E seconds',
|
|
67
|
+
requires: ['tokenCounting'],
|
|
68
|
+
reason: 'requires a token-counting fm command'
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
key: 'decodeTokensPerSecond',
|
|
72
|
+
label: 'decode tokens/s',
|
|
73
|
+
kind: 'derived',
|
|
74
|
+
source: '(output tokens - 1) / generation seconds',
|
|
75
|
+
requires: ['streaming', 'tokenCounting'],
|
|
76
|
+
reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
key: 'prefillTokensPerSecond',
|
|
80
|
+
label: 'prefill tokens/s',
|
|
81
|
+
kind: 'proxy',
|
|
82
|
+
source: 'prompt tokens / TTFT seconds',
|
|
83
|
+
requires: ['streaming', 'tokenCounting'],
|
|
84
|
+
reason: 'requires streaming plus a token-counting fm command'
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
key: 'outputTokenThroughput',
|
|
88
|
+
label: 'aggregate output token throughput',
|
|
89
|
+
kind: 'derived',
|
|
90
|
+
source: 'successful output tokens / measured wall-clock window',
|
|
91
|
+
requires: ['tokenCounting'],
|
|
92
|
+
reason: 'requires a token-counting fm command'
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
key: 'rps',
|
|
96
|
+
label: 'request throughput',
|
|
97
|
+
kind: 'measured',
|
|
98
|
+
source: 'successful requests / measured wall-clock window',
|
|
99
|
+
requires: []
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
key: 'chunkGaps',
|
|
103
|
+
label: 'chunk gaps and second-chunk delay',
|
|
104
|
+
kind: 'proxy',
|
|
105
|
+
source: 'gaps between consecutive streamed stdout chunks',
|
|
106
|
+
requires: ['streaming'],
|
|
107
|
+
reason: 'requires an fm build whose output can be streamed'
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
key: 'percentiles',
|
|
111
|
+
label: 'percentiles',
|
|
112
|
+
kind: 'derived',
|
|
113
|
+
source: 'percentile interpolation over successful samples',
|
|
114
|
+
requires: []
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
key: 'successRate',
|
|
118
|
+
label: 'success rate',
|
|
119
|
+
kind: 'measured',
|
|
120
|
+
source: 'successful runs / attempted runs',
|
|
121
|
+
requires: []
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
key: 'goodput',
|
|
125
|
+
label: 'goodput',
|
|
126
|
+
kind: 'derived',
|
|
127
|
+
source: 'successful runs meeting every configured SLO / runs with an SLO verdict',
|
|
128
|
+
requires: ['slo'],
|
|
129
|
+
reason: 'set --slo-ttft-ms, --slo-e2e-ms, or --slo-tpot-ms to enable goodput'
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
key: 'repeatability',
|
|
133
|
+
label: 'repeatability',
|
|
134
|
+
kind: 'derived',
|
|
135
|
+
source: 'most common normalized output hash share across repeated runs',
|
|
136
|
+
requires: []
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
key: 'variability',
|
|
140
|
+
label: 'CV and 95% confidence interval',
|
|
141
|
+
kind: 'derived',
|
|
142
|
+
source: 'sample standard deviation over successful samples',
|
|
143
|
+
requires: [],
|
|
144
|
+
reason: 'needs at least two successful samples'
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
key: 'quota',
|
|
148
|
+
label: 'quota',
|
|
149
|
+
kind: 'measured',
|
|
150
|
+
source: 'fm quota-usage',
|
|
151
|
+
requires: ['quota'],
|
|
152
|
+
reason: 'this fm build exposes no quota command'
|
|
153
|
+
}
|
|
154
|
+
];
|
|
155
|
+
|
|
156
|
+
function requirementsMet(requires = [], context = {}) {
|
|
157
|
+
return requires.every((name) => {
|
|
158
|
+
if (name === 'streaming') return Boolean(context.streaming);
|
|
159
|
+
if (name === 'tokenCounting') return Boolean(context.tokenCounting);
|
|
160
|
+
if (name === 'quota') return Boolean(context.quota);
|
|
161
|
+
if (name === 'slo') return Boolean(context.slo);
|
|
162
|
+
return true;
|
|
163
|
+
});
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Describe which metrics the detected `fm` build can support for a run.
|
|
168
|
+
*
|
|
169
|
+
* @param {{ features?: Record<string, any> }} [capabilities]
|
|
170
|
+
* @param {{ stream?: boolean, slo?: boolean }} [options]
|
|
171
|
+
*/
|
|
172
|
+
export function metricAvailability(capabilities = {}, options = {}) {
|
|
173
|
+
const features = capabilities.features ?? {};
|
|
174
|
+
const context = {
|
|
175
|
+
streaming: options.stream !== false && features.streaming !== false,
|
|
176
|
+
tokenCounting: features.tokenCounting !== false,
|
|
177
|
+
quota: features.quota === true,
|
|
178
|
+
slo: options.slo === true
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
const metrics = {};
|
|
182
|
+
for (const definition of DEFINITIONS) {
|
|
183
|
+
const available = requirementsMet(definition.requires, context);
|
|
184
|
+
metrics[definition.key] = {
|
|
185
|
+
label: definition.label,
|
|
186
|
+
kind: definition.kind,
|
|
187
|
+
source: definition.source,
|
|
188
|
+
available,
|
|
189
|
+
unavailableReason: available ? '' : (definition.reason ?? 'unavailable in this environment')
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
return metrics;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
export function metricDefinition(key) {
|
|
196
|
+
const definition = DEFINITIONS.find((item) => item.key === key);
|
|
197
|
+
return definition ? { ...definition, requires: [...definition.requires] } : null;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
export function metricDefinitions() {
|
|
201
|
+
return DEFINITIONS.map((definition) => ({ ...definition, requires: [...definition.requires] }));
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** Short human line for CLI output, e.g. "token counting: yes, quota: no". */
|
|
205
|
+
export function formatCapabilitySummary(capabilities = {}) {
|
|
206
|
+
const features = capabilities.features ?? {};
|
|
207
|
+
return [
|
|
208
|
+
`token counting ${features.tokenCounting ? 'yes' : 'no'}`,
|
|
209
|
+
`streaming ${features.streaming ? 'yes' : 'no'}`,
|
|
210
|
+
`quota ${features.quota ? 'yes' : 'no'}`
|
|
211
|
+
].join(', ');
|
|
212
|
+
}
|
package/src/process.js
CHANGED
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
import { spawn } from 'node:child_process';
|
|
2
2
|
|
|
3
|
+
// Children currently alive. The CLI registers signal handlers that terminate
|
|
4
|
+
// these so Ctrl+C cannot leave `fm` processes behind.
|
|
5
|
+
const activeChildren = new Set();
|
|
6
|
+
|
|
7
|
+
export function activeChildCount() {
|
|
8
|
+
return activeChildren.size;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Terminate every running child process. Returns how many were signalled.
|
|
13
|
+
* @param {NodeJS.Signals} signal
|
|
14
|
+
*/
|
|
15
|
+
export function killActiveChildren(signal = 'SIGTERM') {
|
|
16
|
+
let killed = 0;
|
|
17
|
+
for (const child of activeChildren) {
|
|
18
|
+
if (child.exitCode != null || child.signalCode != null) continue;
|
|
19
|
+
try {
|
|
20
|
+
child.kill(signal);
|
|
21
|
+
killed += 1;
|
|
22
|
+
} catch {
|
|
23
|
+
// Process already gone.
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
return killed;
|
|
27
|
+
}
|
|
28
|
+
|
|
3
29
|
export function runProcess(command, args = [], options = {}) {
|
|
4
30
|
const {
|
|
5
31
|
input,
|
|
@@ -15,6 +41,7 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
15
41
|
env,
|
|
16
42
|
stdio: ['pipe', 'pipe', 'pipe']
|
|
17
43
|
});
|
|
44
|
+
activeChildren.add(child);
|
|
18
45
|
|
|
19
46
|
let stdout = '';
|
|
20
47
|
let stderr = '';
|
|
@@ -55,11 +82,16 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
55
82
|
stderr += chunk;
|
|
56
83
|
});
|
|
57
84
|
|
|
58
|
-
|
|
85
|
+
const finish = (result) => {
|
|
86
|
+
if (settled) return;
|
|
59
87
|
settled = true;
|
|
88
|
+
activeChildren.delete(child);
|
|
60
89
|
if (timer) clearTimeout(timer);
|
|
61
|
-
|
|
62
|
-
|
|
90
|
+
resolve(result);
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
child.on('error', (error) => {
|
|
94
|
+
finish({
|
|
63
95
|
command,
|
|
64
96
|
args,
|
|
65
97
|
code: null,
|
|
@@ -73,15 +105,12 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
73
105
|
firstStderrMs,
|
|
74
106
|
error,
|
|
75
107
|
timedOut,
|
|
76
|
-
durationMs: Number(
|
|
108
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
77
109
|
});
|
|
78
110
|
});
|
|
79
111
|
|
|
80
112
|
child.on('close', (code, signal) => {
|
|
81
|
-
|
|
82
|
-
if (timer) clearTimeout(timer);
|
|
83
|
-
const endedAt = process.hrtime.bigint();
|
|
84
|
-
resolve({
|
|
113
|
+
finish({
|
|
85
114
|
command,
|
|
86
115
|
args,
|
|
87
116
|
code,
|
|
@@ -93,8 +122,9 @@ export function runProcess(command, args = [], options = {}) {
|
|
|
93
122
|
stdoutChunkTimesMs,
|
|
94
123
|
firstStdoutMs,
|
|
95
124
|
firstStderrMs,
|
|
125
|
+
error: null,
|
|
96
126
|
timedOut,
|
|
97
|
-
durationMs: Number(
|
|
127
|
+
durationMs: Number(process.hrtime.bigint() - startedAt) / 1e6
|
|
98
128
|
});
|
|
99
129
|
});
|
|
100
130
|
|
package/src/prompts.js
CHANGED
|
@@ -195,16 +195,28 @@ export async function loadPrompts(options = {}) {
|
|
|
195
195
|
|
|
196
196
|
async function loadPromptFile(filePath) {
|
|
197
197
|
const absolutePath = path.resolve(filePath);
|
|
198
|
-
|
|
198
|
+
let content;
|
|
199
|
+
try {
|
|
200
|
+
content = await fs.readFile(absolutePath, 'utf8');
|
|
201
|
+
} catch (error) {
|
|
202
|
+
throw new Error(error.code === 'ENOENT'
|
|
203
|
+
? `Prompt file not found: ${absolutePath}`
|
|
204
|
+
: `Cannot read prompt file ${absolutePath}: ${error.message}`);
|
|
205
|
+
}
|
|
199
206
|
const trimmed = content.trim();
|
|
200
207
|
|
|
201
208
|
if (!trimmed) return [];
|
|
202
209
|
|
|
203
210
|
if (absolutePath.endsWith('.json')) {
|
|
204
|
-
|
|
211
|
+
let parsed;
|
|
212
|
+
try {
|
|
213
|
+
parsed = JSON.parse(trimmed);
|
|
214
|
+
} catch (error) {
|
|
215
|
+
throw new Error(`Cannot parse ${absolutePath} as JSON: ${error.message}`);
|
|
216
|
+
}
|
|
205
217
|
const items = Array.isArray(parsed) ? parsed : parsed.prompts;
|
|
206
218
|
if (!Array.isArray(items)) {
|
|
207
|
-
throw new Error(
|
|
219
|
+
throw new Error(`${absolutePath} must be a JSON array of prompts or an object with a prompts array`);
|
|
208
220
|
}
|
|
209
221
|
return items.map((item, index) => normalizePromptItem(item, index));
|
|
210
222
|
}
|
|
@@ -212,7 +224,13 @@ async function loadPromptFile(filePath) {
|
|
|
212
224
|
if (absolutePath.endsWith('.jsonl')) {
|
|
213
225
|
return trimmed.split(/\r?\n/)
|
|
214
226
|
.filter(Boolean)
|
|
215
|
-
.map((line, index) =>
|
|
227
|
+
.map((line, index) => {
|
|
228
|
+
try {
|
|
229
|
+
return normalizePromptItem(JSON.parse(line), index);
|
|
230
|
+
} catch (error) {
|
|
231
|
+
throw new Error(`Cannot parse line ${index + 1} of ${absolutePath} as JSON: ${error.message}`);
|
|
232
|
+
}
|
|
233
|
+
});
|
|
216
234
|
}
|
|
217
235
|
|
|
218
236
|
return trimmed.split(/\n\s*\n/g).map((prompt, index) => ({
|