fm-bench 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,190 @@
1
+ import { formatMs } from './table.js';
2
+
3
+ const UNICODE_FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'];
4
+ const ASCII_FRAMES = ['-', '\\', '|', '/'];
5
+
6
+ const TONES = {
7
+ green: ['\x1b[32m', '\x1b[0m'],
8
+ yellow: ['\x1b[33m', '\x1b[0m'],
9
+ red: ['\x1b[31m', '\x1b[0m'],
10
+ dim: ['\x1b[2m', '\x1b[0m']
11
+ };
12
+
13
+ export function createProgress(options = {}) {
14
+ const stream = options.stream || process.stderr;
15
+ const enabled = options.enabled === true || (options.enabled === 'auto' && Boolean(stream.isTTY));
16
+ if (!enabled) return noopProgress();
17
+ return new StatusLine({ ...options, stream });
18
+ }
19
+
20
+ function noopProgress() {
21
+ return {
22
+ update() {},
23
+ stop() {}
24
+ };
25
+ }
26
+
27
+ class StatusLine {
28
+ constructor(options = {}) {
29
+ this.stream = options.stream || process.stderr;
30
+ this.color = Boolean(options.color);
31
+ this.ascii = Boolean(options.ascii);
32
+ this.frames = this.ascii ? ASCII_FRAMES : UNICODE_FRAMES;
33
+ this.frameIndex = 0;
34
+ this.startedAt = Date.now();
35
+ this.state = {
36
+ phase: 'starting',
37
+ message: 'starting',
38
+ completed: 0,
39
+ failed: 0,
40
+ total: null
41
+ };
42
+ this.timer = setInterval(() => this.render(), 90);
43
+ this.timer.unref?.();
44
+ this.render();
45
+ }
46
+
47
+ update(event = {}) {
48
+ if (event.type === 'phase') {
49
+ this.state.phase = event.phase || 'working';
50
+ this.state.message = event.message || this.state.message;
51
+ } else if (event.type === 'tokens:start') {
52
+ this.state.phase = 'tokens';
53
+ this.state.message = event.message || 'counting prompt tokens';
54
+ this.state.completed = 0;
55
+ this.state.total = event.total;
56
+ } else if (event.type === 'tokens:progress') {
57
+ this.state.phase = 'tokens';
58
+ this.state.message = `counted ${event.promptId || 'prompt'}`;
59
+ this.state.completed = event.completed;
60
+ this.state.total = event.total;
61
+ } else if (event.type === 'benchmark:start') {
62
+ this.state.phase = 'benchmark';
63
+ this.state.message = `running ${event.modelCount} model(s), ${event.promptCount} prompt(s)`;
64
+ this.state.completed = 0;
65
+ this.state.failed = 0;
66
+ this.state.total = event.total;
67
+ } else if (event.type === 'warmup:start') {
68
+ this.state.phase = 'warmup';
69
+ this.state.message = `warming c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
70
+ this.state.completed = 0;
71
+ this.state.total = event.total;
72
+ } else if (event.type === 'warmup:progress') {
73
+ this.state.phase = 'warmup';
74
+ this.state.message = `warmed ${event.model || 'model'} at c${event.concurrency}`;
75
+ this.state.completed = event.completed;
76
+ this.state.total = event.total;
77
+ } else if (event.type === 'scenario:start') {
78
+ this.state.phase = 'benchmark';
79
+ this.state.message = `measuring c${event.concurrency} (${event.scenarioIndex}/${event.scenarioCount})`;
80
+ this.state.total = event.total ?? this.state.total;
81
+ } else if (event.type === 'benchmark:progress') {
82
+ this.state.phase = 'benchmark';
83
+ this.state.message = `${event.model} ${event.promptId} run ${event.run} ${event.ok ? 'ok' : 'failed'} (${formatMs(event.durationMs)})`;
84
+ this.state.completed = event.completed;
85
+ this.state.failed = event.failed;
86
+ this.state.total = event.total;
87
+ } else if (event.type === 'benchmark:complete') {
88
+ this.state.phase = 'complete';
89
+ this.state.message = `complete ${event.completed}/${event.total}`;
90
+ this.state.completed = event.completed;
91
+ this.state.failed = event.failed;
92
+ this.state.total = event.total;
93
+ }
94
+ this.render();
95
+ }
96
+
97
+ stop(finalMessage = '') {
98
+ if (this.timer) clearInterval(this.timer);
99
+ this.timer = null;
100
+ if (this.stream.clearLine && this.stream.cursorTo) {
101
+ this.clearLine();
102
+ } else {
103
+ this.stream.write('\n');
104
+ }
105
+ if (finalMessage) this.stream.write(`${finalMessage}\n`);
106
+ }
107
+
108
+ render() {
109
+ const width = this.stream.columns || process.stderr.columns || 100;
110
+ const elapsed = Date.now() - this.startedAt;
111
+ const frame = this.frames[this.frameIndex % this.frames.length];
112
+ this.frameIndex += 1;
113
+ const progress = progressText(this.state);
114
+ const eta = etaText(this.state, elapsed);
115
+ const failures = this.state.failed > 0 ? this.tone(`fail ${this.state.failed}`, 'red') : '';
116
+ const parts = [
117
+ this.tone(frame, 'green'),
118
+ 'fm-bench',
119
+ this.state.phase,
120
+ progress,
121
+ eta,
122
+ failures,
123
+ this.tone(this.state.message, 'dim')
124
+ ].filter(Boolean);
125
+ this.writeLine(truncateVisible(parts.join(' '), Math.max(20, width - 1)));
126
+ }
127
+
128
+ writeLine(line) {
129
+ if (this.stream.clearLine && this.stream.cursorTo) {
130
+ this.stream.clearLine(0);
131
+ this.stream.cursorTo(0);
132
+ this.stream.write(line);
133
+ } else {
134
+ this.stream.write(`\r${line}`);
135
+ }
136
+ }
137
+
138
+ clearLine() {
139
+ if (this.stream.clearLine && this.stream.cursorTo) {
140
+ this.stream.clearLine(0);
141
+ this.stream.cursorTo(0);
142
+ } else {
143
+ this.stream.write('\r');
144
+ }
145
+ }
146
+
147
+ tone(text, tone) {
148
+ if (!this.color || !TONES[tone]) return text;
149
+ const [open, close] = TONES[tone];
150
+ return `${open}${text}${close}`;
151
+ }
152
+ }
153
+
154
+ function progressText(state) {
155
+ if (!Number.isFinite(state.total) || state.total <= 0) return '';
156
+ const completed = Math.min(state.completed || 0, state.total);
157
+ const percent = Math.round((completed / state.total) * 100);
158
+ return `${completed}/${state.total} ${percent}%`;
159
+ }
160
+
161
+ function etaText(state, elapsedMs) {
162
+ if (!Number.isFinite(state.total) || state.total <= 0 || !Number.isFinite(state.completed) || state.completed <= 0) {
163
+ return '';
164
+ }
165
+ const remaining = Math.max(0, state.total - state.completed);
166
+ if (remaining === 0) return `elapsed ${formatMs(elapsedMs)}`;
167
+ const perItemMs = elapsedMs / state.completed;
168
+ return `eta ${formatMs(perItemMs * remaining)}`;
169
+ }
170
+
171
+ function truncateVisible(value, width) {
172
+ const text = String(value);
173
+ const clean = text.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, '');
174
+ if (clean.length <= width) return text;
175
+ let visible = 0;
176
+ let output = '';
177
+ for (let index = 0; index < text.length && visible < width - 1; index += 1) {
178
+ if (text[index] === '\x1b') {
179
+ const match = text.slice(index).match(/^\u001b\[[0-?]*[ -/]*[@-~]/);
180
+ if (match) {
181
+ output += match[0];
182
+ index += match[0].length - 1;
183
+ continue;
184
+ }
185
+ }
186
+ output += text[index];
187
+ visible += 1;
188
+ }
189
+ return `${output}…`;
190
+ }
package/src/prompts.js CHANGED
@@ -50,6 +50,28 @@ const PROFILES = {
50
50
  prompt: 'Create a compact test plan for benchmarking a local foundation model across short, medium, and long prompts.'
51
51
  }
52
52
  ],
53
+ client: [
54
+ {
55
+ id: 'short-chat',
56
+ prompt: 'In one sentence, explain why time to first token matters for an interactive assistant.'
57
+ },
58
+ {
59
+ id: 'content-generation',
60
+ prompt: 'Write a practical 180-word product update for developers explaining a new terminal benchmark feature. Keep it specific and avoid marketing fluff.'
61
+ },
62
+ {
63
+ id: 'structured-extraction',
64
+ prompt: 'Return compact valid JSON with keys "risk", "owner", "deadline", and "next_step" from this note: The benchmark release is blocked by flaky p95 latency on the PCC model. Maya owns the investigation and needs a fix before Friday.'
65
+ },
66
+ {
67
+ id: 'summarization-light',
68
+ prompt: 'Summarize this in three bullets: A serious local LLM benchmark should separate time to first token from total latency, report tail percentiles, include prompt and output token counts, measure throughput at multiple concurrency operating points, and preserve the raw prompt suite so future runs are comparable.'
69
+ },
70
+ {
71
+ id: 'code-analysis',
72
+ prompt: 'Review this JavaScript function for one correctness issue and one readability improvement: function p(v){let s=0;for(let i=0;i<=v.length;i++)s+=v[i];return s/v.length}'
73
+ }
74
+ ],
53
75
  stress: [
54
76
  {
55
77
  id: 'interactive-short',
package/src/report.js CHANGED
@@ -27,9 +27,13 @@ export function flattenResults(results) {
27
27
  words: result.words,
28
28
  tokens_per_second: result.tokensPerSecond == null ? '' : round(result.tokensPerSecond),
29
29
  decode_tokens_per_second: result.decodeTokensPerSecond == null ? '' : round(result.decodeTokensPerSecond),
30
+ prefill_tokens_per_second: result.prefillTokensPerSecond == null ? '' : round(result.prefillTokensPerSecond),
30
31
  chars_per_second: round(result.charsPerSecond),
31
32
  streamed: result.streamed,
32
33
  stdout_chunks: result.stdoutChunks,
34
+ second_chunk_ms: round(result.secondChunkMs),
35
+ chunk_gap_avg_ms: round(result.chunkGapAvgMs),
36
+ chunk_gap_max_ms: round(result.chunkGapMaxMs),
33
37
  output_hash: result.outputHash || '',
34
38
  good: result.good == null ? '' : result.good,
35
39
  error: result.error || ''
package/src/stats.js CHANGED
@@ -101,10 +101,15 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
101
101
  const charsPerSecond = summarizeNumbers(successes.map((result) => result.charsPerSecond));
102
102
  const tokensPerSecond = summarizeNumbers(successes.map((result) => result.tokensPerSecond).filter((value) => value != null));
103
103
  const decodeTokensPerSecond = summarizeNumbers(successes.map((result) => result.decodeTokensPerSecond).filter((value) => value != null));
104
+ const prefillTokensPerSecond = summarizeNumbers(successes.map((result) => result.prefillTokensPerSecond).filter((value) => value != null));
105
+ const secondChunk = summarizeNumbers(successes.map((result) => result.secondChunkMs).filter((value) => value != null));
106
+ const chunkGap = summarizeNumbers(successes.flatMap((result) => result.chunkGapsMs || []));
104
107
  const windowMs = modelWindowMs(successes);
105
108
  const rps = successes.length > 0 && windowMs > 0 ? successes.length / (windowMs / 1000) : null;
106
- const goodputRps = goodResults.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
109
+ const goodputRps = goodMeasured.length > 0 && windowMs > 0 ? goodResults.length / (windowMs / 1000) : null;
107
110
  const outputTokenThroughput = outputTokens.sum > 0 && windowMs > 0 ? outputTokens.sum / (windowMs / 1000) : null;
111
+ const totalTokens = promptTokens.sum + outputTokens.sum;
112
+ const totalTokenThroughput = totalTokens > 0 && windowMs > 0 ? totalTokens / (windowMs / 1000) : null;
108
113
 
109
114
  return {
110
115
  model: entry.model,
@@ -120,6 +125,7 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
120
125
  rps,
121
126
  goodputRps,
122
127
  outputTokenThroughput,
128
+ totalTokenThroughput,
123
129
  repeatability: summarizeRepeatability(successes),
124
130
  latency,
125
131
  ttft,
@@ -129,7 +135,10 @@ export function summarizeByModel(results, modelStatuses = [], options = {}) {
129
135
  outputTokens,
130
136
  charsPerSecond,
131
137
  tokensPerSecond,
132
- decodeTokensPerSecond
138
+ decodeTokensPerSecond,
139
+ prefillTokensPerSecond,
140
+ secondChunk,
141
+ chunkGap
133
142
  };
134
143
  });
135
144
  }