fm-bench 0.6.3 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -68
- package/bin/fm-bench.js +14 -0
- package/docs/compatibility.md +46 -0
- package/docs/methodology.md +46 -20
- package/docs/releasing.md +33 -8
- package/docs/report-format.md +79 -2
- package/docs/supported-platforms.md +19 -2
- package/package.json +6 -5
- package/src/bench.js +129 -37
- package/src/capabilities.js +188 -0
- package/src/cli.js +153 -68
- package/src/compare.js +11 -3
- package/src/fm-help.js +131 -0
- package/src/fm.js +100 -112
- package/src/history.js +2 -1
- package/src/macos.js +2 -1
- package/src/metrics.js +203 -0
- package/src/process.js +39 -9
- package/src/prompts.js +22 -4
- package/src/report.js +13 -2
- package/src/schema.js +17 -1
- package/src/stats.js +43 -14
- package/src/table.js +124 -50
package/src/report.js
CHANGED
|
@@ -16,6 +16,7 @@ export function flattenResults(results) {
|
|
|
16
16
|
concurrency: result.concurrency ?? '',
|
|
17
17
|
prompt_id: result.promptId,
|
|
18
18
|
run: result.run,
|
|
19
|
+
attempts: result.attempts ?? 1,
|
|
19
20
|
ok: result.ok,
|
|
20
21
|
duration_ms: round(result.durationMs),
|
|
21
22
|
ttft_ms: round(result.firstTokenMs),
|
|
@@ -58,9 +59,19 @@ export async function writeReport(filePath, payload, format) {
|
|
|
58
59
|
|
|
59
60
|
function csvEscape(value) {
|
|
60
61
|
const text = String(value ?? '');
|
|
61
|
-
|
|
62
|
-
|
|
62
|
+
const safe = guardFormula(text);
|
|
63
|
+
if (/[",\n\r]/.test(safe)) {
|
|
64
|
+
return `"${safe.replaceAll('"', '""')}"`;
|
|
63
65
|
}
|
|
66
|
+
return safe;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// Spreadsheet programs execute cells that start with =, +, @, or a non-numeric
|
|
70
|
+
// leading -. Prompt text and captured output are untrusted input, so prefix
|
|
71
|
+
// those cells with a single quote to keep them literal text.
|
|
72
|
+
function guardFormula(text) {
|
|
73
|
+
if (/^[=+@\t\r]/.test(text)) return `'${text}`;
|
|
74
|
+
if (text.startsWith('-') && !/^-?\d+(\.\d+)?$/.test(text)) return `'${text}`;
|
|
64
75
|
return text;
|
|
65
76
|
}
|
|
66
77
|
|
package/src/schema.js
CHANGED
|
@@ -126,10 +126,26 @@ export function validateReport(value) {
|
|
|
126
126
|
for (const key of REQUIRED_TOP_LEVEL) {
|
|
127
127
|
if (!(key in report)) errors.push(`missing required field: ${key}`);
|
|
128
128
|
}
|
|
129
|
-
if (!Array.isArray(report.summary))
|
|
129
|
+
if (!Array.isArray(report.summary)) {
|
|
130
|
+
errors.push('summary must be an array');
|
|
131
|
+
} else {
|
|
132
|
+
for (const [index, item] of report.summary.entries()) {
|
|
133
|
+
if (!item || typeof item !== 'object') {
|
|
134
|
+
errors.push(`summary[${index}] must be an object`);
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
const row = /** @type {Record<string, unknown>} */ (item);
|
|
138
|
+
if (typeof row.model !== 'string' || row.model === '') {
|
|
139
|
+
errors.push(`summary[${index}].model must be a non-empty string`);
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
130
143
|
if (report.schemaVersion != null && report.schemaVersion !== REPORT_SCHEMA_VERSION) {
|
|
131
144
|
errors.push(`unsupported schemaVersion: ${report.schemaVersion} (expected ${REPORT_SCHEMA_VERSION})`);
|
|
132
145
|
}
|
|
146
|
+
if (report.metrics != null && (typeof report.metrics !== 'object' || Array.isArray(report.metrics))) {
|
|
147
|
+
errors.push('metrics must be an object when present');
|
|
148
|
+
}
|
|
133
149
|
if (errors.length > 0) return { ok: false, errors };
|
|
134
150
|
return { ok: true, report };
|
|
135
151
|
}
|
package/src/stats.js
CHANGED
|
@@ -1,3 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sample statistics for one metric.
|
|
3
|
+
*
|
|
4
|
+
* Spread metrics (`stddev`, `cv`, `ci95*`) need at least two samples. With a
|
|
5
|
+
* single sample they are `null` rather than `0`, because a "0% variation" or a
|
|
6
|
+
* zero-width confidence interval is invented precision, not a measurement.
|
|
7
|
+
*
|
|
8
|
+
* @param {number[]} values
|
|
9
|
+
*/
|
|
1
10
|
export function summarizeNumbers(values) {
|
|
2
11
|
const clean = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
|
|
3
12
|
if (clean.length === 0) {
|
|
@@ -20,11 +29,12 @@ export function summarizeNumbers(values) {
|
|
|
20
29
|
|
|
21
30
|
const total = clean.reduce((sum, value) => sum + value, 0);
|
|
22
31
|
const avg = total / clean.length;
|
|
23
|
-
const
|
|
32
|
+
const hasSpread = clean.length > 1;
|
|
33
|
+
const variance = hasSpread
|
|
24
34
|
? clean.reduce((sum, value) => sum + (value - avg) ** 2, 0) / (clean.length - 1)
|
|
25
|
-
:
|
|
26
|
-
const stddev = Math.sqrt(variance);
|
|
27
|
-
const margin =
|
|
35
|
+
: null;
|
|
36
|
+
const stddev = variance == null ? null : Math.sqrt(variance);
|
|
37
|
+
const margin = hasSpread ? tCritical95(clean.length) * (stddev / Math.sqrt(clean.length)) : null;
|
|
28
38
|
return {
|
|
29
39
|
count: clean.length,
|
|
30
40
|
min: clean[0],
|
|
@@ -32,9 +42,9 @@ export function summarizeNumbers(values) {
|
|
|
32
42
|
avg,
|
|
33
43
|
sum: total,
|
|
34
44
|
stddev,
|
|
35
|
-
cv: avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
36
|
-
ci95Low: avg - margin,
|
|
37
|
-
ci95High: avg + margin,
|
|
45
|
+
cv: hasSpread && avg !== 0 ? stddev / Math.abs(avg) : null,
|
|
46
|
+
ci95Low: hasSpread ? avg - margin : null,
|
|
47
|
+
ci95High: hasSpread ? avg + margin : null,
|
|
38
48
|
p50: percentile(clean, 50),
|
|
39
49
|
p90: percentile(clean, 90),
|
|
40
50
|
p95: percentile(clean, 95),
|
|
@@ -42,16 +52,25 @@ export function summarizeNumbers(values) {
|
|
|
42
52
|
};
|
|
43
53
|
}
|
|
44
54
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
/**
|
|
56
|
+
* Percentile with linear interpolation between closest ranks (the same method
|
|
57
|
+
* as Excel's PERCENTILE.INC / NumPy's default). Accepts unsorted input so
|
|
58
|
+
* callers cannot silently get a wrong answer from an unsorted array.
|
|
59
|
+
*
|
|
60
|
+
* @param {number[]} values
|
|
61
|
+
* @param {number} percentileValue 0-100
|
|
62
|
+
*/
|
|
63
|
+
export function percentile(values, percentileValue) {
|
|
64
|
+
const sorted = values.filter((value) => Number.isFinite(value)).sort((a, b) => a - b);
|
|
65
|
+
if (sorted.length === 0) return null;
|
|
66
|
+
if (sorted.length === 1) return sorted[0];
|
|
48
67
|
|
|
49
|
-
const rank = (percentileValue / 100) * (
|
|
68
|
+
const rank = (percentileValue / 100) * (sorted.length - 1);
|
|
50
69
|
const low = Math.floor(rank);
|
|
51
70
|
const high = Math.ceil(rank);
|
|
52
|
-
if (low === high) return
|
|
71
|
+
if (low === high) return sorted[low];
|
|
53
72
|
const weight = rank - low;
|
|
54
|
-
return
|
|
73
|
+
return sorted[low] * (1 - weight) + sorted[high] * weight;
|
|
55
74
|
}
|
|
56
75
|
|
|
57
76
|
export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
@@ -66,6 +85,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
66
85
|
concurrency,
|
|
67
86
|
description: status.description,
|
|
68
87
|
available: status.available,
|
|
88
|
+
unsupported: Boolean(status.unsupported),
|
|
69
89
|
skippedReason: status.available ? '' : status.reason || 'Unavailable',
|
|
70
90
|
results: []
|
|
71
91
|
});
|
|
@@ -80,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
80
100
|
concurrency: result.concurrency,
|
|
81
101
|
description: '',
|
|
82
102
|
available: true,
|
|
103
|
+
unsupported: false,
|
|
83
104
|
skippedReason: '',
|
|
84
105
|
results: []
|
|
85
106
|
});
|
|
@@ -98,7 +119,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
98
119
|
const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
|
|
99
120
|
const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
|
|
100
121
|
const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
|
|
101
|
-
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
|
|
122
|
+
const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
|
|
102
123
|
const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
|
|
103
124
|
const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
|
|
104
125
|
const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
|
|
@@ -110,14 +131,18 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
|
|
|
110
131
|
const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
|
|
111
132
|
const totalTokens = promptTokens.sum + outputTokens.sum;
|
|
112
133
|
const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
|
|
134
|
+
const attempts = entry.results.reduce((sum, result) => sum + (result.attempts ?? 1), 0);
|
|
113
135
|
|
|
114
136
|
return {
|
|
115
137
|
model: entry.model,
|
|
116
138
|
concurrency: entry.concurrency,
|
|
117
139
|
description: entry.description,
|
|
118
140
|
available: entry.available,
|
|
141
|
+
unsupported: entry.unsupported,
|
|
119
142
|
skippedReason: entry.skippedReason,
|
|
120
143
|
attempted: entry.results.length,
|
|
144
|
+
attempts,
|
|
145
|
+
retried: entry.results.length > 0 ? Math.max(0, attempts - entry.results.length) : 0,
|
|
121
146
|
successes: successes.length,
|
|
122
147
|
failures: failures.length,
|
|
123
148
|
successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
|
|
@@ -147,6 +172,10 @@ function summaryKey(model, concurrency) {
|
|
|
147
172
|
return `${model}::${concurrency ?? 'default'}`;
|
|
148
173
|
}
|
|
149
174
|
|
|
175
|
+
// Two-sided 95% t critical values. Exact table entries up to 30 degrees of
|
|
176
|
+
// freedom, then the standard 2.0 / 1.96 approximations for larger samples.
|
|
177
|
+
// fm-bench uses this for a mean confidence interval, which is context for
|
|
178
|
+
// small samples rather than a hypothesis test.
|
|
150
179
|
function tCritical95(n) {
|
|
151
180
|
const df = Math.max(1, n - 1);
|
|
152
181
|
const table = {
|
package/src/table.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { stripAnsi } from './ansi.js';
|
|
2
|
+
import { formatCapabilitySummary } from './metrics.js';
|
|
2
3
|
|
|
3
4
|
export function renderTable(headers, rows, options = {}) {
|
|
4
5
|
const ascii = Boolean(options.ascii);
|
|
@@ -52,10 +53,19 @@ export function renderBenchmarkReport(payload, options = {}) {
|
|
|
52
53
|
const meta = `prompts ${payload.prompts.length} | runs ${payload.options.runs} | concurrency ${concurrencies} | stream ${payload.options.stream ? 'on' : 'off'} | measured ${measured} | failed ${failed} | skipped ${skipped} | elapsed ${formatMs(elapsedMs)}${slo ? ` | ${slo}` : ''}`;
|
|
53
54
|
const tags = payload.options?.tags?.length ? payload.options.tags : [];
|
|
54
55
|
const note = payload.options?.note ?? null;
|
|
55
|
-
lines.push(
|
|
56
|
-
lines.push(
|
|
57
|
-
if (tags.length > 0)
|
|
58
|
-
|
|
56
|
+
lines.push(...wrapText(title, width));
|
|
57
|
+
lines.push(...wrapText(meta, width));
|
|
58
|
+
if (tags.length > 0) {
|
|
59
|
+
for (const line of wrapText(`tags: ${tags.join(', ')}`, width)) lines.push(line);
|
|
60
|
+
}
|
|
61
|
+
if (note) {
|
|
62
|
+
for (const line of wrapText(`note: ${note}`, width)) lines.push(line);
|
|
63
|
+
}
|
|
64
|
+
for (const metricNote of unavailableMetricNotes(payload.metrics)) {
|
|
65
|
+
for (const line of wrapText(`unavailable: ${metricNote}`, width)) {
|
|
66
|
+
lines.push(line);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
59
69
|
lines.push('');
|
|
60
70
|
|
|
61
71
|
if (mode === 'compact') {
|
|
@@ -71,41 +81,43 @@ export function renderBenchmarkReport(payload, options = {}) {
|
|
|
71
81
|
|
|
72
82
|
export function legendEntries() {
|
|
73
83
|
return [
|
|
74
|
-
entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.'),
|
|
75
|
-
entry('summary', 'MODEL', 'fm model name
|
|
76
|
-
entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable.'),
|
|
77
|
-
entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.'),
|
|
78
|
-
entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.'),
|
|
79
|
-
entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set.
|
|
80
|
-
entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.'),
|
|
81
|
-
entry('summary', 'TTFT', 'Time to first streamed
|
|
82
|
-
entry('summary', 'TTFT P95', '95th percentile time to first streamed
|
|
83
|
-
entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better.'),
|
|
84
|
-
entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.'),
|
|
85
|
-
entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', '
|
|
86
|
-
entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.'),
|
|
87
|
-
entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Higher is better
|
|
88
|
-
entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better
|
|
89
|
-
entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Higher is better
|
|
90
|
-
entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.'),
|
|
91
|
-
entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', ''),
|
|
92
|
-
entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm
|
|
93
|
-
entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm
|
|
94
|
-
entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', '
|
|
95
|
-
entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', '
|
|
96
|
-
entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.'),
|
|
97
|
-
entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Lower is smoother
|
|
98
|
-
entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.'),
|
|
99
|
-
entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', '
|
|
100
|
-
entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.'),
|
|
101
|
-
entry('detail', '
|
|
102
|
-
entry('
|
|
103
|
-
entry('models', '
|
|
104
|
-
entry('
|
|
105
|
-
entry('
|
|
106
|
-
entry('
|
|
107
|
-
entry('colors', '
|
|
108
|
-
entry('colors', '
|
|
84
|
+
entry('summary', 'C', 'Concurrency operating point for this row.', 'Higher C means more parallel fm respond processes.', 'controlled'),
|
|
85
|
+
entry('summary', 'MODEL', 'fm model name reported by the detected fm build.', '', 'measured'),
|
|
86
|
+
entry('summary', 'STATUS', 'Run status for this model and operating point.', 'ok = all measured jobs passed; partial = at least one failed; skipped = unavailable or unsupported.', 'measured'),
|
|
87
|
+
entry('summary', 'OK / OK/RUNS', 'Successful measured runs over attempted measured runs.', 'Green when no failures; yellow when partial.', 'measured'),
|
|
88
|
+
entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.', 'measured'),
|
|
89
|
+
entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Runs whose SLO metric is unmeasurable count as not good.', 'derived'),
|
|
90
|
+
entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.', 'derived'),
|
|
91
|
+
entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity, not per token. Lower is better.', 'proxy'),
|
|
92
|
+
entry('summary', 'TTFT P95', '95th percentile time to the first streamed stdout chunk.', 'Lower is better.', 'proxy'),
|
|
93
|
+
entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better. This is a direct wall-clock measurement.', 'measured'),
|
|
94
|
+
entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.', 'measured'),
|
|
95
|
+
entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Derived from fm token counts and stream timings. Lower is better.', 'derived'),
|
|
96
|
+
entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.', 'derived'),
|
|
97
|
+
entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Derived from fm token counts. Higher is better.', 'derived'),
|
|
98
|
+
entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better.', 'derived'),
|
|
99
|
+
entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Measured from process timings. Higher is better.', 'measured'),
|
|
100
|
+
entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.', 'derived'),
|
|
101
|
+
entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', '', 'measured'),
|
|
102
|
+
entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
|
|
103
|
+
entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
|
|
104
|
+
entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Proxy: prefill is inferred from time to first chunk, not observed directly. Higher is better.', 'proxy'),
|
|
105
|
+
entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Derived; requires streaming and token counts. Higher is better.', 'derived'),
|
|
106
|
+
entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.', 'proxy'),
|
|
107
|
+
entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Proxy for decode smoothness at chunk granularity. Lower is smoother.', 'proxy'),
|
|
108
|
+
entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.', 'measured'),
|
|
109
|
+
entry('detail', 'E2E 95% CI', '95% confidence interval around mean E2E latency.', 'Blank with fewer than two successful samples; treat small samples carefully.', 'derived'),
|
|
110
|
+
entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.', 'derived'),
|
|
111
|
+
entry('detail', 'ATTEMPTS', 'Total fm invocations including retries.', 'Shown in reports; a retried run is still one measured result.', 'measured'),
|
|
112
|
+
entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', '', 'measured'),
|
|
113
|
+
entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', '', 'measured'),
|
|
114
|
+
entry('models', 'QUOTA', 'Quota information when the fm build exposes a quota command.', 'Column is omitted entirely when the installed fm has no quota command.', 'measured'),
|
|
115
|
+
entry('metrics', 'SOURCE', 'How a metric is obtained: measured, proxy, derived, or controlled.', 'measured = observed directly; proxy = observed at coarser granularity; derived = computed from measured values.', 'measured'),
|
|
116
|
+
entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.', 'derived'),
|
|
117
|
+
entry('colors', 'GREEN', 'Passing, steadier, faster, or better within this benchmark context.', '', 'controlled'),
|
|
118
|
+
entry('colors', 'YELLOW', 'Marginal, partial, near a budget, or middle-ranked within this benchmark context.', '', 'controlled'),
|
|
119
|
+
entry('colors', 'RED', 'Failed budget, unstable, slower, lower, or worse within this benchmark context.', '', 'controlled'),
|
|
120
|
+
entry('colors', 'MUTED', 'Unavailable model, skipped value, or descriptive text.', '', 'controlled')
|
|
109
121
|
];
|
|
110
122
|
}
|
|
111
123
|
|
|
@@ -114,6 +126,7 @@ export function renderLegend(options = {}) {
|
|
|
114
126
|
const rows = legendEntries().map((item) => [
|
|
115
127
|
item.table,
|
|
116
128
|
item.column,
|
|
129
|
+
item.kind || 'measured',
|
|
117
130
|
item.definition,
|
|
118
131
|
item.rule || '-'
|
|
119
132
|
]);
|
|
@@ -122,7 +135,7 @@ export function renderLegend(options = {}) {
|
|
|
122
135
|
return renderCompactLegend(width);
|
|
123
136
|
}
|
|
124
137
|
|
|
125
|
-
return renderWrappedTable(['table', 'column', 'definition', 'rule'], rows, legendColumnWidths(width), options);
|
|
138
|
+
return renderWrappedTable(['table', 'column', 'source', 'definition', 'rule'], rows, legendColumnWidths(width), options);
|
|
126
139
|
}
|
|
127
140
|
|
|
128
141
|
export function renderLatencyHistogram(results, options = {}) {
|
|
@@ -335,6 +348,19 @@ export function renderModelsTable(models, options = {}) {
|
|
|
335
348
|
}).join('\n');
|
|
336
349
|
}
|
|
337
350
|
|
|
351
|
+
const quotaSupported = options.capabilities
|
|
352
|
+
? Boolean(options.capabilities.features?.quota)
|
|
353
|
+
: models.some((model) => model.quotaSupported === true || (model.quota != null && model.quota !== ''));
|
|
354
|
+
|
|
355
|
+
if (!quotaSupported) {
|
|
356
|
+
return renderTable(['model', 'available', 'description', 'notes'], models.map((model) => [
|
|
357
|
+
cell(model.name),
|
|
358
|
+
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
|
|
359
|
+
model.description || '-',
|
|
360
|
+
cleanReason(model.reason || '-')
|
|
361
|
+
]), { ...options, wrapColumns: ['description', 'notes'] });
|
|
362
|
+
}
|
|
363
|
+
|
|
338
364
|
return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
|
|
339
365
|
cell(model.name),
|
|
340
366
|
cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
|
|
@@ -343,6 +369,27 @@ export function renderModelsTable(models, options = {}) {
|
|
|
343
369
|
]), { ...options, wrapColumns: ['description', 'quota'] });
|
|
344
370
|
}
|
|
345
371
|
|
|
372
|
+
/**
|
|
373
|
+
* `fm-bench models` body: fm capability summary, then the model table.
|
|
374
|
+
* Unsupported capabilities are stated up front so a missing quota column or
|
|
375
|
+
* blank metric never looks like a bug.
|
|
376
|
+
*/
|
|
377
|
+
export function renderModelsReport(models, options = {}) {
|
|
378
|
+
const capabilities = options.capabilities;
|
|
379
|
+
const lines = [];
|
|
380
|
+
if (capabilities) {
|
|
381
|
+
lines.push(`fm ${capabilities.bin}${capabilities.digest ? ` help ${capabilities.digest}` : ''}${capabilities.ok ? '' : ' (no commands detected)'}`);
|
|
382
|
+
lines.push(`commands ${capabilities.commands.join(', ') || 'none detected'}`);
|
|
383
|
+
lines.push(`features ${formatCapabilitySummary(capabilities)}`);
|
|
384
|
+
for (const warning of capabilities.warnings) {
|
|
385
|
+
lines.push(`warn ${warning}`);
|
|
386
|
+
}
|
|
387
|
+
lines.push('');
|
|
388
|
+
}
|
|
389
|
+
lines.push(renderModelsTable(models, options));
|
|
390
|
+
return lines.join('\n');
|
|
391
|
+
}
|
|
392
|
+
|
|
346
393
|
function formatRangeMs(low, high) {
|
|
347
394
|
if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
|
|
348
395
|
return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
|
|
@@ -364,15 +411,38 @@ function formatSlo(slo = {}) {
|
|
|
364
411
|
return parts.length ? `SLO ${parts.join(',')}` : '';
|
|
365
412
|
}
|
|
366
413
|
|
|
367
|
-
function entry(table, column, definition, rule) {
|
|
414
|
+
function entry(table, column, definition, rule, kind = 'measured') {
|
|
368
415
|
return {
|
|
369
416
|
table,
|
|
370
417
|
column,
|
|
371
418
|
definition,
|
|
372
|
-
rule
|
|
419
|
+
rule,
|
|
420
|
+
kind
|
|
373
421
|
};
|
|
374
422
|
}
|
|
375
423
|
|
|
424
|
+
/**
|
|
425
|
+
* Reasons why metric columns are blank. Keeps "unavailable" visible in table
|
|
426
|
+
* output instead of leaving readers to guess why a column is empty.
|
|
427
|
+
*/
|
|
428
|
+
function unavailableMetricNotes(metrics) {
|
|
429
|
+
if (!metrics) return [];
|
|
430
|
+
const notes = [];
|
|
431
|
+
const tokenMetric = metrics.promptTokens;
|
|
432
|
+
if (tokenMetric && !tokenMetric.available) {
|
|
433
|
+
notes.push(`${tokenMetric.label} unavailable — ${tokenMetric.unavailableReason}`);
|
|
434
|
+
}
|
|
435
|
+
const ttft = metrics.ttft;
|
|
436
|
+
if (ttft && !ttft.available) {
|
|
437
|
+
notes.push(`${ttft.label} unavailable — ${ttft.unavailableReason}`);
|
|
438
|
+
}
|
|
439
|
+
const quota = metrics.quota;
|
|
440
|
+
if (quota && !quota.available) {
|
|
441
|
+
notes.push(`${quota.label} unavailable — ${quota.unavailableReason}`);
|
|
442
|
+
}
|
|
443
|
+
return notes;
|
|
444
|
+
}
|
|
445
|
+
|
|
376
446
|
const ASCII_TABLE = {
|
|
377
447
|
topLeft: '+',
|
|
378
448
|
topJoin: '+',
|
|
@@ -516,7 +586,10 @@ function normalizeCell(value) {
|
|
|
516
586
|
|
|
517
587
|
function formatCell(value) {
|
|
518
588
|
if (value == null) return '';
|
|
519
|
-
|
|
589
|
+
// Terminal output carries prompt text, captured model output, and fm
|
|
590
|
+
// diagnostics. Strip ANSI escapes and control characters so they cannot
|
|
591
|
+
// corrupt the table layout or emit terminal control sequences.
|
|
592
|
+
return stripAnsi(String(value)).replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g, '');
|
|
520
593
|
}
|
|
521
594
|
|
|
522
595
|
function pad(value, width, left = false) {
|
|
@@ -546,7 +619,7 @@ function compactReason(value) {
|
|
|
546
619
|
function renderCompactLegend(width) {
|
|
547
620
|
const lines = [];
|
|
548
621
|
for (const item of legendEntries()) {
|
|
549
|
-
lines.push(...wrapText(`${item.table.toUpperCase()} ${item.column}`, width));
|
|
622
|
+
lines.push(...wrapText(`${item.table.toUpperCase()} ${item.column} (${item.kind || 'measured'})`, width));
|
|
550
623
|
lines.push(...wrapText(` Definition: ${item.definition}`, width));
|
|
551
624
|
if (item.rule) {
|
|
552
625
|
lines.push(...wrapText(` Rule: ${item.rule}`, width));
|
|
@@ -558,13 +631,14 @@ function renderCompactLegend(width) {
|
|
|
558
631
|
}
|
|
559
632
|
|
|
560
633
|
function legendColumnWidths(width) {
|
|
561
|
-
const available = Math.max(40, width -
|
|
634
|
+
const available = Math.max(40, width - 16);
|
|
562
635
|
const tableWidth = 7;
|
|
563
|
-
const columnWidth = Math.min(25, Math.max(18, Math.floor(available * 0.
|
|
564
|
-
const
|
|
565
|
-
const
|
|
566
|
-
const
|
|
567
|
-
|
|
636
|
+
const columnWidth = Math.min(25, Math.max(18, Math.floor(available * 0.22)));
|
|
637
|
+
const sourceWidth = 11;
|
|
638
|
+
const remaining = Math.max(32, available - tableWidth - columnWidth - sourceWidth);
|
|
639
|
+
const definitionWidth = Math.max(18, Math.floor(remaining * 0.58));
|
|
640
|
+
const ruleWidth = Math.max(14, remaining - definitionWidth);
|
|
641
|
+
return [tableWidth, columnWidth, sourceWidth, definitionWidth, ruleWidth];
|
|
568
642
|
}
|
|
569
643
|
|
|
570
644
|
function wrapText(value, width) {
|