fm-bench 0.7.1 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/fm.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import os from 'node:os';
2
2
  import { stripAnsi } from './ansi.js';
3
3
  import { detectFmCapabilities } from './capabilities.js';
4
- import { firstLine, isUnsupportedModelError, parseAvailabilityOutput } from './fm-help.js';
4
+ import { firstLine, isUnsupportedModelError, parseAvailabilityOutput, parseModelList, withoutWarnings } from './fm-help.js';
5
5
  import { runProcess } from './process.js';
6
6
  import { parseBatteryOutput, parseThermalOutput } from './system.js';
7
7
 
@@ -25,12 +25,33 @@ export async function getFmHelp(fmBin, timeoutMs = 10_000) {
25
25
  };
26
26
  }
27
27
 
28
+ /**
29
+ * Ask `fm models` once for every model's status. Returns `null` when the
30
+ * build has no `models` command (older builds use per-model `fm available`).
31
+ * @returns {Promise<{ entries: Map<string, { name: string, available: boolean, identity: string, reason: string }>, error: string } | null>}
32
+ */
33
+ export async function listModelStatus(fmBin, options = {}) {
34
+ if (options.capabilities?.features?.modelListCommand !== 'models') return null;
35
+ const result = await runProcess(fmBin, ['models'], {
36
+ timeoutMs: options.timeoutMs ?? 15_000
37
+ });
38
+ const output = `${result.stdout}${result.stderr}`;
39
+ const entries = new Map(parseModelList(output).map((entry) => [entry.name, entry]));
40
+ const error = result.error
41
+ ? (result.stderr || result.error.message)
42
+ : (entries.size === 0 ? firstLine(output) || `fm models exited with code ${result.code}` : '');
43
+ return { entries, error };
44
+ }
45
+
28
46
  /**
29
47
  * Check one model against the detected `fm` build.
30
48
  *
31
49
  * Models the build does not expose are reported as unsupported without
32
50
  * spawning `fm` at all, so a raw argument-error blob from `fm` can never end
33
51
  * up in a report or in `fm-bench models` output.
52
+ *
53
+ * Pass `options.modelList` (from `listModelStatus`) to reuse one `fm models`
54
+ * call across several models.
34
55
  */
35
56
  export async function checkModelAvailability(fmBin, model, options = {}) {
36
57
  const capabilities = options.capabilities;
@@ -40,11 +61,33 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
40
61
  model,
41
62
  available: false,
42
63
  unsupported: true,
64
+ identity: '',
43
65
  raw: '',
44
66
  reason: `not supported by this fm build (supported: ${known.join(', ')})`
45
67
  };
46
68
  }
47
69
 
70
+ const listCommand = capabilities ? capabilities.features?.modelListCommand : 'available';
71
+ if (listCommand === 'models') {
72
+ const list = options.modelList ?? await listModelStatus(fmBin, options);
73
+ const entry = list?.entries.get(model);
74
+ if (entry) {
75
+ return { model, available: entry.available, identity: entry.identity, raw: '', reason: entry.reason };
76
+ }
77
+ return {
78
+ model,
79
+ available: false,
80
+ identity: '',
81
+ raw: '',
82
+ reason: list?.error || 'not reported by fm models'
83
+ };
84
+ }
85
+ if (listCommand == null) {
86
+ // No availability command at all: the first fm respond call is the only
87
+ // way to find out, and a failure there is recorded per run.
88
+ return { model, available: true, identity: '', raw: '', reason: '' };
89
+ }
90
+
48
91
  const result = await runProcess(fmBin, ['available', '--model', model], {
49
92
  timeoutMs: options.timeoutMs ?? 15_000
50
93
  });
@@ -61,7 +104,27 @@ export async function checkModelAvailability(fmBin, model, options = {}) {
61
104
  const supported = known.length > 0 ? ` (supported: ${known.join(', ')})` : '';
62
105
  parsed.reason = `not supported by this fm build${supported}`;
63
106
  }
64
- return parsed;
107
+ return { identity: '', ...parsed };
108
+ }
109
+
110
+ /**
111
+ * Whether the fm Legal Notice & Terms have been accepted. `fm respond` cannot
112
+ * run until they are, so `doctor` reports it explicitly.
113
+ * @returns {Promise<{ supported: boolean, agreed: boolean|null, detail: string }>}
114
+ */
115
+ export async function getLicenseStatus(fmBin, options = {}) {
116
+ if (options.capabilities && !options.capabilities.features?.license) {
117
+ return { supported: false, agreed: null, detail: 'this fm build has no license command' };
118
+ }
119
+ const result = await runProcess(fmBin, ['license', '--status'], {
120
+ timeoutMs: options.timeoutMs ?? 10_000
121
+ });
122
+ const output = withoutWarnings(`${result.stdout}${result.stderr}`).replace(/\s+/g, ' ').trim();
123
+ if (result.error) {
124
+ return { supported: true, agreed: null, detail: result.stderr || result.error.message };
125
+ }
126
+ const agreed = result.code === 0 && /\bagreed\b/i.test(output) && !/\bnot\b|\bnever\b/i.test(output);
127
+ return { supported: true, agreed, detail: output || `fm license --status exited with code ${result.code}` };
65
128
  }
66
129
 
67
130
  /**
@@ -133,6 +196,61 @@ export async function countTokens(fmBin, text, options = {}) {
133
196
  };
134
197
  }
135
198
 
199
+ /**
200
+ * Measure the constant framing overhead `fm count-tokens` adds to every count.
201
+ *
202
+ * On macOS 27.2 `count-tokens` reports 2 for "a", 3 for "a a", and 5 for
203
+ * "a a a a": one token per word plus one framing token. Model output must not
204
+ * carry that extra token, so the overhead is derived from three counts whose
205
+ * per-word step must agree; anything inconsistent leaves counts uncorrected.
206
+ *
207
+ * @returns {Promise<{ overhead: number, calibrated: boolean }>}
208
+ */
209
+ export async function calibrateTokenCounter(fmBin, options = {}) {
210
+ const counts = [];
211
+ for (const text of ['a', 'a a', 'a a a']) {
212
+ const counted = await countTokens(fmBin, text, options);
213
+ if (!counted.ok) return { overhead: 0, calibrated: false };
214
+ counts.push(counted.count);
215
+ }
216
+ const step = counts[1] - counts[0];
217
+ const overhead = counts[0] - step;
218
+ const consistent = step >= 1 && counts[2] - counts[1] === step && overhead >= 0 && overhead <= 16;
219
+ return consistent ? { overhead, calibrated: true } : { overhead: 0, calibrated: false };
220
+ }
221
+
222
+ // Stdout chunks closer together than this are one write burst from fm, not
223
+ // separate streaming steps. On macOS 27.2 a short answer's tail arrives as
224
+ // several writes well under 1 ms apart, while real deltas are 20 ms or more
225
+ // apart; counting the burst as decode steps reported tens of thousands of
226
+ // tokens per second.
227
+ export const DELIVERY_COALESCE_MS = 5;
228
+
229
+ /**
230
+ * Group stdout chunk arrivals into deliveries: runs of chunks that each
231
+ * arrived less than `coalesceMs` after the previous one.
232
+ * @param {number[]} timesMs chunk arrival times
233
+ * @param {number[]} lengths chunk lengths in characters
234
+ * @returns {{ atMs: number, endChars: number }[]} delivery start time and the
235
+ * stdout length once the delivery is complete
236
+ */
237
+ export function groupDeliveries(timesMs = [], lengths = [], coalesceMs = DELIVERY_COALESCE_MS) {
238
+ const deliveries = [];
239
+ let chars = 0;
240
+ let previousAtMs = null;
241
+ for (const [index, atMs] of timesMs.entries()) {
242
+ chars += lengths[index] ?? 0;
243
+ const current = deliveries.at(-1);
244
+ if (current && atMs - previousAtMs < coalesceMs) {
245
+ current.endChars = chars;
246
+ } else {
247
+ deliveries.push({ atMs, endChars: chars });
248
+ }
249
+ previousAtMs = atMs;
250
+ }
251
+ return deliveries;
252
+ }
253
+
136
254
  export async function respond(fmBin, model, prompt, options = {}) {
137
255
  const features = options.capabilities?.features;
138
256
  const streamControl = features ? features.streaming : true;
@@ -155,6 +273,7 @@ export async function respond(fmBin, model, prompt, options = {}) {
155
273
  const output = stripAnsi(result.stdout).trim();
156
274
  const errorText = stripAnsi(result.stderr).trim();
157
275
  const failed = result.code !== 0 || result.timedOut;
276
+ const deliveries = streamed ? groupDeliveries(result.stdoutChunkTimesMs, result.stdoutChunkLengths) : [];
158
277
 
159
278
  return {
160
279
  ok: !failed,
@@ -167,9 +286,11 @@ export async function respond(fmBin, model, prompt, options = {}) {
167
286
  timedOut: result.timedOut,
168
287
  durationMs: result.durationMs,
169
288
  firstOutputMs: streamed ? result.firstStdoutMs : null,
289
+ firstChunkText: deliveries.length > 0 ? stripAnsi(result.stdout.slice(0, deliveries[0].endChars)) : null,
170
290
  streamed,
171
291
  stdoutChunks: result.stdoutChunks,
172
292
  stdoutChunkTimesMs: streamed ? result.stdoutChunkTimesMs : [],
293
+ deliveryTimesMs: deliveries.map((delivery) => delivery.atMs),
173
294
  error: failed
174
295
  ? (result.timedOut
175
296
  ? `timed out after ${options.timeoutMs ?? 60_000}ms`
package/src/metrics.js CHANGED
@@ -31,7 +31,7 @@ const DEFINITIONS = [
31
31
  key: 'generationMs',
32
32
  label: 'generation time',
33
33
  kind: 'derived',
34
- source: 'E2E latency minus TTFT',
34
+ source: 'last streamed stdout delivery minus the first (chunks under 5 ms apart are one delivery)',
35
35
  requires: ['streaming'],
36
36
  reason: 'requires at least two streamed output chunks to separate prefill from decode'
37
37
  },
@@ -39,9 +39,17 @@ const DEFINITIONS = [
39
39
  key: 'tpot',
40
40
  label: 'TPOT',
41
41
  kind: 'derived',
42
- source: '(E2E - TTFT) / (output tokens - 1)',
42
+ source: 'generation time / output tokens that arrived after the first chunk',
43
43
  requires: ['streaming', 'tokenCounting'],
44
- reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
44
+ reason: 'requires streaming, a token-counting fm command, and at least two tokens after the first chunk'
45
+ },
46
+ {
47
+ key: 'firstChunkTokens',
48
+ label: 'first-chunk tokens',
49
+ kind: 'measured',
50
+ source: 'fm count-tokens on the first streamed stdout delivery, minus the counter framing overhead',
51
+ requires: ['streaming', 'tokenCounting'],
52
+ reason: 'requires streaming plus a token-counting fm command'
45
53
  },
46
54
  {
47
55
  key: 'promptTokens',
@@ -55,7 +63,7 @@ const DEFINITIONS = [
55
63
  key: 'outputTokens',
56
64
  label: 'output tokens',
57
65
  kind: 'measured',
58
- source: 'fm count-tokens on the captured output',
66
+ source: 'fm count-tokens on the captured output, minus the calibrated counter framing overhead',
59
67
  requires: ['tokenCounting'],
60
68
  reason: 'this fm build exposes no token-counting command'
61
69
  },
@@ -71,9 +79,9 @@ const DEFINITIONS = [
71
79
  key: 'decodeTokensPerSecond',
72
80
  label: 'decode tokens/s',
73
81
  kind: 'derived',
74
- source: '(output tokens - 1) / generation seconds',
82
+ source: 'output tokens after the first chunk / generation seconds',
75
83
  requires: ['streaming', 'tokenCounting'],
76
- reason: 'requires streaming, a token-counting fm command, and at least three output tokens'
84
+ reason: 'requires streaming, a token-counting fm command, and at least two tokens after the first chunk'
77
85
  },
78
86
  {
79
87
  key: 'prefillTokensPerSecond',
@@ -102,7 +110,7 @@ const DEFINITIONS = [
102
110
  key: 'chunkGaps',
103
111
  label: 'chunk gaps and second-chunk delay',
104
112
  kind: 'proxy',
105
- source: 'gaps between consecutive streamed stdout chunks',
113
+ source: 'gaps between consecutive streamed stdout deliveries (chunks under 5 ms apart are one delivery)',
106
114
  requires: ['streaming'],
107
115
  reason: 'requires an fm build whose output can be streamed'
108
116
  },
package/src/process.js CHANGED
@@ -48,6 +48,7 @@ export function runProcess(command, args = [], options = {}) {
48
48
  let stdoutChunks = 0;
49
49
  let stderrChunks = 0;
50
50
  const stdoutChunkTimesMs = [];
51
+ const stdoutChunkLengths = [];
51
52
  let firstStdoutMs = null;
52
53
  let firstStderrMs = null;
53
54
  let timedOut = false;
@@ -71,7 +72,10 @@ export function runProcess(command, args = [], options = {}) {
71
72
  if (firstStdoutMs == null && chunk.length > 0) {
72
73
  firstStdoutMs = chunkAtMs;
73
74
  }
74
- if (chunk.length > 0) stdoutChunkTimesMs.push(chunkAtMs);
75
+ if (chunk.length > 0) {
76
+ stdoutChunkTimesMs.push(chunkAtMs);
77
+ stdoutChunkLengths.push(chunk.length);
78
+ }
75
79
  stdout += chunk;
76
80
  });
77
81
  child.stderr.on('data', (chunk) => {
@@ -101,6 +105,7 @@ export function runProcess(command, args = [], options = {}) {
101
105
  stdoutChunks,
102
106
  stderrChunks,
103
107
  stdoutChunkTimesMs,
108
+ stdoutChunkLengths,
104
109
  firstStdoutMs,
105
110
  firstStderrMs,
106
111
  error,
@@ -120,6 +125,7 @@ export function runProcess(command, args = [], options = {}) {
120
125
  stdoutChunks,
121
126
  stderrChunks,
122
127
  stdoutChunkTimesMs,
128
+ stdoutChunkLengths,
123
129
  firstStdoutMs,
124
130
  firstStderrMs,
125
131
  error: null,
package/src/report.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import fs from 'node:fs/promises';
2
2
  import path from 'node:path';
3
+ import { renderHtmlReport } from './export.js';
3
4
 
4
5
  export function toCsv(rows) {
5
6
  if (rows.length === 0) return '';
@@ -37,7 +38,8 @@ export function flattenResults(results) {
37
38
  chunk_gap_max_ms: round(result.chunkGapMaxMs),
38
39
  output_hash: result.outputHash || '',
39
40
  good: result.good == null ? '' : result.good,
40
- error: result.error || ''
41
+ error: result.error || '',
42
+ first_chunk_tokens: result.firstChunkTokens ?? ''
41
43
  }));
42
44
  }
43
45
 
@@ -48,7 +50,6 @@ export async function writeReport(filePath, payload, format) {
48
50
  if (format === 'csv') {
49
51
  content = toCsv(flattenResults(payload.results));
50
52
  } else if (format === 'html') {
51
- const { renderHtmlReport } = await import('./export.js');
52
53
  content = renderHtmlReport(payload);
53
54
  } else {
54
55
  content = `${JSON.stringify(payload, null, 2)}\n`;
package/src/schema.js CHANGED
@@ -62,11 +62,30 @@ export function environmentFingerprint(report) {
62
62
  macOSBuildVersion: build,
63
63
  fmBin: env.fmBin ?? null,
64
64
  fmHelpDigest: env.fmHelpDigest ?? null,
65
+ modelIdentities: modelIdentities(report),
65
66
  thermal: env.thermal ?? null,
66
67
  power: env.power ?? null
67
68
  };
68
69
  }
69
70
 
71
+ /**
72
+ * Model name → identity reported by `fm models` (for example
73
+ * `{ system: 'AFM 3 Core Advanced' }`). Empty for reports from builds or
74
+ * fm-bench versions that do not expose it.
75
+ * @param {Record<string, unknown>} report
76
+ * @returns {Record<string, string>}
77
+ */
78
+ export function modelIdentities(report) {
79
+ const identities = {};
80
+ const models = Array.isArray(report.models) ? report.models : [];
81
+ for (const model of models) {
82
+ if (model && typeof model.name === 'string' && typeof model.identity === 'string' && model.identity) {
83
+ identities[model.name] = model.identity;
84
+ }
85
+ }
86
+ return identities;
87
+ }
88
+
70
89
  /**
71
90
  * @param {Record<string, unknown>} before
72
91
  * @param {Record<string, unknown>} after
@@ -99,6 +118,12 @@ export function compareCompatibility(before, after) {
99
118
  if (bFp.macOSBuildVersion && aFp.macOSBuildVersion && bFp.macOSBuildVersion !== aFp.macOSBuildVersion) {
100
119
  warnings.push(`macOS build differs (${bFp.macOSBuildVersion} vs ${aFp.macOSBuildVersion})`);
101
120
  }
121
+ for (const [name, identity] of Object.entries(bFp.modelIdentities)) {
122
+ const other = aFp.modelIdentities[name];
123
+ if (other && other !== identity) {
124
+ warnings.push(`model ${name} differs (${identity} vs ${other})`);
125
+ }
126
+ }
102
127
 
103
128
  return {
104
129
  compatible: errors.length === 0,
package/src/stats.js CHANGED
@@ -84,6 +84,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
84
84
  model: status.name,
85
85
  concurrency,
86
86
  description: status.description,
87
+ identity: status.identity || '',
87
88
  available: status.available,
88
89
  unsupported: Boolean(status.unsupported),
89
90
  skippedReason: status.available ? '' : status.reason || 'Unavailable',
@@ -99,6 +100,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
99
100
  model: result.model,
100
101
  concurrency: result.concurrency,
101
102
  description: '',
103
+ identity: '',
102
104
  available: true,
103
105
  unsupported: false,
104
106
  skippedReason: '',
@@ -119,12 +121,19 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
119
121
  const tpot = summarizeNumbers(successes.map((result) => result.tpotMs).filter((value) => value != null));
120
122
  const promptTokens = summarizeNumbers(successes.map((result) => result.promptTokens).filter((value) => value != null));
121
123
  const outputTokens = summarizeNumbers(successes.map((result) => result.outputTokens).filter((value) => value != null));
124
+ const firstChunkTokens = summarizeNumbers(successes.map((result) => result.firstChunkTokens).filter((value) => value != null));
122
125
  const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond).filter((value) => value != null));
123
126
  const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
124
127
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
125
128
  const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
126
129
  const secondChunk = summarizeNumbers(successes.map((result) => result.secondChunkMs).filter((value) => value != null));
127
130
  const chunkGap = summarizeNumbers(successes.flatMap((result) => result.chunkGapsMs || []));
131
+ const decodeRuns = successes.filter((result) => result.decodeTokens != null && result.generationMs > 0);
132
+ const decodeMs = decodeRuns.reduce((sum, result) => sum + result.generationMs, 0);
133
+ // Token-weighted: a 3-token tail cannot outweigh a 200-token generation.
134
+ const decodeThroughput = decodeMs > 0
135
+ ? decodeRuns.reduce((sum, result) => sum + result.decodeTokens, 0) / (decodeMs / 1000)
136
+ : null;
128
137
  const windowMs = modelWindowMs(successes);
129
138
  const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
130
139
  const goodputRps = goodMeasured.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
@@ -137,6 +146,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
137
146
  model: entry.model,
138
147
  concurrency: entry.concurrency,
139
148
  description: entry.description,
149
+ identity: entry.identity,
140
150
  available: entry.available,
141
151
  unsupported: entry.unsupported,
142
152
  skippedReason: entry.skippedReason,
@@ -147,6 +157,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
147
157
  failures: failures.length,
148
158
  successRate: entry.results.length > 0 ? successes.length / entry.results.length : null,
149
159
  goodputRate: goodMeasured.length > 0 ? goodResults.length / goodMeasured.length : null,
160
+ stabilityCv: summarizeStability(successes),
150
161
  rps,
151
162
  goodputRps,
152
163
  outputTokenThroughput,
@@ -158,9 +169,11 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
158
169
  tpot,
159
170
  promptTokens,
160
171
  outputTokens,
172
+ firstChunkTokens,
161
173
  charsPerSecond,
162
174
  tokensPerSecond,
163
175
  decodeTokensPerSecond,
176
+ decodeThroughput,
164
177
  prefillTokensPerSecond,
165
178
  secondChunk,
166
179
  chunkGap
@@ -222,6 +235,29 @@ function modelWindowMs(results) {
222
235
  return Math.max(...ends) - Math.min(...starts);
223
236
  }
224
237
 
238
+ /**
239
+ * Run-to-run latency stability: the E2E coefficient of variation of each
240
+ * prompt across its repeated runs, averaged over prompts.
241
+ *
242
+ * The pooled `latency.cv` mixes every prompt, so a suite with a one-line answer and a
243
+ * long generation reports high "variation" even when every prompt is
244
+ * perfectly steady. Grouping by prompt isolates repeat noise. `null` until at
245
+ * least one prompt has two successful runs.
246
+ */
247
+ function summarizeStability(results) {
248
+ const byPrompt = new Map();
249
+ for (const result of results) {
250
+ if (!Number.isFinite(result.durationMs)) continue;
251
+ if (!byPrompt.has(result.promptId)) byPrompt.set(result.promptId, []);
252
+ byPrompt.get(result.promptId).push(result.durationMs);
253
+ }
254
+ const cvs = [...byPrompt.values()]
255
+ .map((durations) => summarizeNumbers(durations).cv)
256
+ .filter((cv) => cv != null);
257
+ if (cvs.length === 0) return null;
258
+ return cvs.reduce((sum, cv) => sum + cv, 0) / cvs.length;
259
+ }
260
+
225
261
  function summarizeRepeatability(results) {
226
262
  const byPrompt = new Map();
227
263
  for (const result of results) {
package/src/table.js CHANGED
@@ -55,6 +55,12 @@ export function renderBenchmarkReport(payload, options = {}) {
55
55
  const note = payload.options?.note ?? null;
56
56
  lines.push(...wrapText(title, width));
57
57
  lines.push(...wrapText(meta, width));
58
+ const identities = (payload.models ?? [])
59
+ .filter((model) => model.available && model.identity)
60
+ .map((model) => `${model.name} = ${model.identity}`);
61
+ if (identities.length > 0) {
62
+ lines.push(...wrapText(`models: ${identities.join(', ')}`, width));
63
+ }
58
64
  if (tags.length > 0) {
59
65
  for (const line of wrapText(`tags: ${tags.join(', ')}`, width)) lines.push(line);
60
66
  }
@@ -88,21 +94,22 @@ export function legendEntries() {
88
94
  entry('summary', 'SUCC / SUCCESS', 'Success rate: successful runs divided by attempted runs.', 'Green 100%, yellow >=95%, red <95%.', 'measured'),
89
95
  entry('summary', 'GOOD', 'Goodput rate: successful runs that also met every configured SLO.', 'Only appears when SLO flags are set. Runs whose SLO metric is unmeasurable count as not good.', 'derived'),
90
96
  entry('summary', 'GOOD RPS', 'SLO-passing requests per second during this measured window.', 'Zero is shown when SLOs are set and no request meets them.', 'derived'),
91
- entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity, not per token. Lower is better.', 'proxy'),
97
+ entry('summary', 'TTFT', 'Time from starting fm respond to the first streamed stdout chunk, p50.', 'Proxy for time to first token: measured at chunk granularity, and the first chunk can already hold many tokens (see 1ST CHUNK). Lower is better.', 'proxy'),
92
98
  entry('summary', 'TTFT P95', '95th percentile time to the first streamed stdout chunk.', 'Lower is better.', 'proxy'),
93
99
  entry('summary', 'E2E', 'End-to-end latency, p50, from starting fm respond until full response exits.', 'Lower is better. This is a direct wall-clock measurement.', 'measured'),
94
100
  entry('summary', 'E2E P95', '95th percentile end-to-end latency.', 'Lower is better; this is usually the main interactive tail-latency signal.', 'measured'),
95
- entry('summary', 'TPOT', 'Time per output token after the first output token, p50.', 'Derived from fm token counts and stream timings. Lower is better.', 'derived'),
96
- entry('summary', 'TPOT P95', '95th percentile time per output token after first token.', 'Lower is better.', 'derived'),
101
+ entry('summary', 'TPOT', 'Time per output token for tokens that arrived after the first streamed chunk, p50.', 'Derived from fm token counts and stream timings. Lower is better.', 'derived'),
102
+ entry('summary', 'TPOT P95', '95th percentile time per output token after the first streamed chunk.', 'Lower is better.', 'derived'),
97
103
  entry('summary', 'USER/S / USER T/S', 'Per-request output tokens per second.', 'Derived from fm token counts. Higher is better.', 'derived'),
98
104
  entry('summary', 'SYS/S / SYS T/S', 'Aggregate successful output-token throughput for the model row.', 'Higher is better.', 'derived'),
99
105
  entry('summary', 'RPS', 'Successful requests per second over the model row measured window.', 'Measured from process timings. Higher is better.', 'measured'),
100
- entry('summary', 'CV', 'Coefficient of variation for E2E latency: sample stddev divided by mean.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%.', 'derived'),
106
+ entry('summary', 'CV', 'Run-to-run E2E variation: coefficient of variation of each prompt across its repeated runs, averaged over prompts.', 'Lower is steadier. Green <=10%, yellow <=25%, red >25%. Blank until a prompt has two successful runs (use --runs 2 or more).', 'derived'),
101
107
  entry('summary', 'NOTE', 'Short unavailable, skipped, or error note.', '', 'measured'),
102
108
  entry('detail', 'IN AVG / IN TOK AVG', 'Average prompt/input token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
103
- entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens.', 'Blank when the fm build cannot count tokens.', 'measured'),
109
+ entry('detail', 'OUT AVG / OUT TOK AVG', 'Average output token count from fm count-tokens, minus its constant framing token.', 'Blank when the fm build cannot count tokens.', 'measured'),
110
+ entry('detail', '1ST CHUNK', 'Average output tokens carried by the first streamed stdout chunk.', 'Shows what TTFT really measures: fm streams coarse deltas (about 20 tokens in the first chunk on macOS 27.2). Wide layout only.', 'measured'),
104
111
  entry('detail', 'PREFILL/S / PREFILL TOK/S', 'Prompt tokens divided by TTFT seconds.', 'Proxy: prefill is inferred from time to first chunk, not observed directly. Higher is better.', 'proxy'),
105
- entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens after the first token divided by generation seconds.', 'Derived; requires streaming and token counts. Higher is better.', 'derived'),
112
+ entry('detail', 'DECODE/S / DECODE TOK/S', 'Output tokens that arrived after the first streamed delivery, summed over runs, divided by the summed generation time.', 'Derived and token-weighted, so long generations dominate short tails; requires streaming and token counts. Higher is better.', 'derived'),
106
113
  entry('detail', '2ND CHUNK', 'Delay between the first and second streamed stdout chunks, p50.', 'Lower is smoother startup. Chunk-based, not raw token telemetry.', 'proxy'),
107
114
  entry('detail', 'CHUNK P95', '95th percentile gap between consecutive streamed stdout chunks.', 'Proxy for decode smoothness at chunk granularity. Lower is smoother.', 'proxy'),
108
115
  entry('detail', 'E2E P99', '99th percentile end-to-end latency.', 'Lower is better; useful for worst-case UX.', 'measured'),
@@ -110,7 +117,8 @@ export function legendEntries() {
110
117
  entry('detail', 'REPEAT', 'Share of repeated runs for a prompt that produced the most common normalized output hash.', 'Green 90%+, yellow 50%+, red below 50%. Blank when there are not repeated comparable outputs.', 'derived'),
111
118
  entry('detail', 'ATTEMPTS', 'Total fm invocations including retries.', 'Shown in reports; a retried run is still one measured result.', 'measured'),
112
119
  entry('detail', 'DESCRIPTION', 'Model description discovered from fm help.', '', 'measured'),
113
- entry('models', 'AVAILABLE', 'Whether fm available reports the model as usable on this machine right now.', '', 'measured'),
120
+ entry('models', 'AVAILABLE', 'Whether fm models (fm available on older builds) reports the model as usable on this machine right now.', '', 'measured'),
121
+ entry('models', 'IDENTITY', 'Model identity reported by fm models, for example AFM 3 Core Advanced.', 'Recorded in reports; compare warns when it changes between two runs.', 'measured'),
114
122
  entry('models', 'QUOTA', 'Quota information when the fm build exposes a quota command.', 'Column is omitted entirely when the installed fm has no quota command.', 'measured'),
115
123
  entry('metrics', 'SOURCE', 'How a metric is obtained: measured, proxy, derived, or controlled.', 'measured = observed directly; proxy = observed at coarser granularity; derived = computed from measured values.', 'measured'),
116
124
  entry('compact', 'GOOD / CV / TPOT / CHUNK', 'Compact output combines the same summary and detail metrics into model cards.', 'Same definitions and color rules as table columns.', 'derived'),
@@ -220,7 +228,7 @@ export function renderSummaryTable(summary, options = {}) {
220
228
  cell(formatNumber(item.tokensPerSecond.avg), tones.userTps),
221
229
  cell(formatNumber(item.outputTokenThroughput), tones.systemTps),
222
230
  cell(formatNumber(item.rps), tones.rps),
223
- cell(formatPercent(item.latency.cv), cvTone(item.latency.cv)),
231
+ cell(formatPercent(stabilityCv(item)), cvTone(stabilityCv(item))),
224
232
  cell(item.available ? '' : cleanReason(item.skippedReason), item.available ? null : 'yellow')
225
233
  ];
226
234
 
@@ -274,20 +282,21 @@ export function renderDetailTable(summary, options = {}) {
274
282
  cell(formatNumber(item.promptTokens.avg, 0)),
275
283
  cell(formatNumber(item.outputTokens.avg, 0)),
276
284
  cell(formatNumber(item.prefillTokensPerSecond?.avg), tones.prefillTps),
277
- cell(formatNumber(item.decodeTokensPerSecond?.avg), tones.decodeTps),
285
+ cell(formatNumber(decodeRate(item)), tones.decodeTps),
278
286
  cell(formatMs(item.secondChunk?.p50), tones.secondChunk),
279
287
  cell(formatMs(item.chunkGap?.p95), tones.chunkGapP95),
280
288
  cell(formatMs(item.latency.p99), tones.e2eP99),
281
- cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(item.latency.cv)),
289
+ cell(formatRangeMs(item.latency.ci95Low, item.latency.ci95High), cvTone(stabilityCv(item))),
282
290
  cell(formatPercent(item.repeatability), percentTone(item.repeatability, 0.5, 0.9)),
283
- cell(item.description || '-', 'muted')
291
+ cell(item.description || '-', 'muted'),
292
+ cell(formatNumber(item.firstChunkTokens?.avg, 0))
284
293
  ];
285
294
 
286
295
  if (mode === 'medium') {
287
296
  return [base[0], base[1], base[2], base[3], base[4], base[5], base[7], base[8], base[10]];
288
297
  }
289
298
 
290
- return base;
299
+ return [base[0], base[1], base[2], base[3], base[12], ...base.slice(4, 12)];
291
300
  });
292
301
 
293
302
  const headers = mode === 'medium'
@@ -297,6 +306,7 @@ export function renderDetailTable(summary, options = {}) {
297
306
  'model',
298
307
  'in tok avg',
299
308
  'out tok avg',
309
+ '1st chunk',
300
310
  'prefill tok/s',
301
311
  'decode tok/s',
302
312
  '2nd chunk',
@@ -325,7 +335,7 @@ export function renderCompactSummary(summary, options = {}) {
325
335
  const goodput = item.goodputRate == null ? '' : ` | good ${formatPercent(item.goodputRate)}`;
326
336
  const tones = metricTones(item, ranks, options.slo);
327
337
  lines.push(toneText(truncate(` TTFT ${formatMs(item.ttft.p50)} p95 ${formatMs(item.ttft.p95)} | E2E ${formatMs(item.latency.p50)} p95 ${formatMs(item.latency.p95)} p99 ${formatMs(item.latency.p99)}`, width), worstTone(tones.ttft, tones.e2e, tones.e2eP95), options));
328
- lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(item.latency.cv)}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(item.latency.cv), percentTone(item.goodputRate, 0.8, 1)), options));
338
+ lines.push(toneText(truncate(` user ${formatNumber(item.tokensPerSecond.avg)} tok/s | system ${formatNumber(item.outputTokenThroughput)} tok/s | RPS ${formatNumber(item.rps)} | CV ${formatPercent(stabilityCv(item))}${goodput}`, width), worstTone(tones.userTps, tones.systemTps, cvTone(stabilityCv(item)), percentTone(item.goodputRate, 0.8, 1)), options));
329
339
  lines.push(toneText(truncate(` prefill ${formatNumber(item.prefillTokensPerSecond?.avg)} tok/s | TPOT ${formatMs(item.tpot.p50)} | chunk p95 ${formatMs(item.chunkGap?.p95)}`, width), worstTone(tones.prefillTps, tones.tpot, tones.chunkGapP95), options));
330
340
  lines.push(toneText(truncate(` in/out ${formatNumber(item.promptTokens.avg, 0)}/${formatNumber(item.outputTokens.avg, 0)} tok avg | repeat ${formatPercent(item.repeatability)}`, width), percentTone(item.repeatability, 0.5, 0.9), options));
331
341
  } else {
@@ -344,29 +354,37 @@ export function renderModelsTable(models, options = {}) {
344
354
  if (compact) {
345
355
  return models.map((model) => {
346
356
  const status = model.available ? 'yes' : 'no';
347
- return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(model.reason || model.description || '-')}`;
357
+ const detail = model.available ? (model.identity || model.description) : model.reason;
358
+ return `${model.name} ${toneText(status, model.available ? 'green' : 'yellow', options)} ${compactReason(detail || model.description || '-')}`;
348
359
  }).join('\n');
349
360
  }
350
361
 
351
362
  const quotaSupported = options.capabilities
352
363
  ? Boolean(options.capabilities.features?.quota)
353
364
  : models.some((model) => model.quotaSupported === true || (model.quota != null && model.quota !== ''));
365
+ const hasIdentity = models.some((model) => model.identity);
366
+ const base = (model) => {
367
+ const cells = [
368
+ cell(model.name),
369
+ cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow')
370
+ ];
371
+ if (hasIdentity) cells.push(model.identity || '-');
372
+ cells.push(model.description || '-');
373
+ return cells;
374
+ };
375
+ const leading = hasIdentity ? ['model', 'available', 'identity', 'description'] : ['model', 'available', 'description'];
354
376
 
355
377
  if (!quotaSupported) {
356
- return renderTable(['model', 'available', 'description', 'notes'], models.map((model) => [
357
- cell(model.name),
358
- cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
359
- model.description || '-',
378
+ return renderTable([...leading, 'notes'], models.map((model) => [
379
+ ...base(model),
360
380
  cleanReason(model.reason || '-')
361
- ]), { ...options, wrapColumns: ['description', 'notes'] });
381
+ ]), { ...options, wrapColumns: ['identity', 'description', 'notes'] });
362
382
  }
363
383
 
364
- return renderTable(['model', 'available', 'description', 'quota'], models.map((model) => [
365
- cell(model.name),
366
- cell(model.available ? 'yes' : 'no', model.available ? 'green' : 'yellow'),
367
- model.description || '-',
384
+ return renderTable([...leading, 'quota'], models.map((model) => [
385
+ ...base(model),
368
386
  cleanReason(model.quota || model.reason || '-')
369
- ]), { ...options, wrapColumns: ['description', 'quota'] });
387
+ ]), { ...options, wrapColumns: ['identity', 'description', 'quota'] });
370
388
  }
371
389
 
372
390
  /**
@@ -390,6 +408,22 @@ export function renderModelsReport(models, options = {}) {
390
408
  return lines.join('\n');
391
409
  }
392
410
 
411
+ /**
412
+ * Run-to-run CV for a summary row. Reports written before `stabilityCv`
413
+ * existed only have the suite-wide `latency.cv`.
414
+ */
415
+ export function stabilityCv(item) {
416
+ return Object.hasOwn(item, 'stabilityCv') ? item.stabilityCv : item.latency?.cv ?? null;
417
+ }
418
+
419
+ /**
420
+ * Token-weighted decode rate for a summary row. Reports written before
421
+ * `decodeThroughput` existed only have the mean of per-run rates.
422
+ */
423
+ export function decodeRate(item) {
424
+ return Object.hasOwn(item, 'decodeThroughput') ? item.decodeThroughput : item.decodeTokensPerSecond?.avg ?? null;
425
+ }
426
+
393
427
  function formatRangeMs(low, high) {
394
428
  if (low == null || high == null || !Number.isFinite(low) || !Number.isFinite(high)) return '-';
395
429
  return `${formatMs(Math.max(0, low))}..${formatMs(Math.max(0, high))}`;
@@ -739,7 +773,7 @@ function rankSummary(summary) {
739
773
  totalTps: collectMetric(summary, (item) => item.totalTokenThroughput),
740
774
  rps: collectMetric(summary, (item) => item.rps),
741
775
  goodputRps: collectMetric(summary, (item) => item.goodputRps),
742
- decodeTps: collectMetric(summary, (item) => item.decodeTokensPerSecond?.avg),
776
+ decodeTps: collectMetric(summary, decodeRate),
743
777
  prefillTps: collectMetric(summary, (item) => item.prefillTokensPerSecond?.avg)
744
778
  };
745
779
  }
@@ -788,7 +822,7 @@ function metricTones(item, ranks, slo = {}) {
788
822
  totalTps: rankTone(item.totalTokenThroughput, ranks.totalTps),
789
823
  rps: rankTone(item.rps, ranks.rps),
790
824
  goodputRps: goodputRpsTone(item, ranks.goodputRps),
791
- decodeTps: rankTone(item.decodeTokensPerSecond?.avg, ranks.decodeTps),
825
+ decodeTps: rankTone(decodeRate(item), ranks.decodeTps),
792
826
  prefillTps: rankTone(item.prefillTokensPerSecond?.avg, ranks.prefillTps)
793
827
  };
794
828
  }