pi-voicekit 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -3
- package/extensions/voice/local.ts +37 -12
- package/extensions/voice/post-process-queue.ts +532 -0
- package/extensions/voice/post-process.ts +37 -3
- package/extensions/voice/sherpa-engine.ts +41 -17
- package/extensions/voice.ts +363 -50
- package/package.json +1 -1
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Segmented transcript post-processing: a bounded, fail-open polish queue.
|
|
3
|
+
*
|
|
4
|
+
* Pure and dependency-injected like ./post-process: no Pi types, no TUI, no filesystem
|
|
5
|
+
* and no network. The model call arrives as `call`, so ordering, the concurrency cap,
|
|
6
|
+
* the per-segment timeout, the single retry and per-segment isolation are all provable offline.
|
|
7
|
+
*
|
|
8
|
+
* The cap counts real calls, not segment promises: a slot is released when the injected
|
|
9
|
+
* transport settles, so a caller that ignores the abort signal keeps its slot after its
|
|
10
|
+
* segment timed out instead of letting a hung endpoint pile up calls past the cap.
|
|
11
|
+
*
|
|
12
|
+
* Spec: docs/superpowers/specs/2026-09-26-polish-pipeline-design.md §4.2, §4.3, §4.4
|
|
13
|
+
* (a local design record, not part of the published package)
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import {
|
|
17
|
+
polishSamplingOptions,
|
|
18
|
+
polishTranscript,
|
|
19
|
+
type AssistantLike,
|
|
20
|
+
type PolishRequest,
|
|
21
|
+
type PolishResult,
|
|
22
|
+
} from "./post-process";
|
|
23
|
+
import { assembleContext, DEFAULT_CONTEXT_LIMITS, type ContextLimits, type EntryLike } from "./post-process-context";
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Start at most this many segment calls at once. A long dictation must not open one request
|
|
27
|
+
* per segment, and wall-clock must stay close to the slowest segment rather than their sum
|
|
28
|
+
* (spec §4.2, open question 1).
|
|
29
|
+
*/
|
|
30
|
+
export const POLISH_QUEUE_CONCURRENCY = 3;
|
|
31
|
+
|
|
32
|
+
/** The extra request fields `polishSamplingOptions` decided for one segment. */
|
|
33
|
+
export interface PolishSampling {
|
|
34
|
+
samplingParams?: { reasoning_effort: string };
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* The polish request as the injected caller sees it: the prompt plus this segment's sampling
|
|
39
|
+
* decision. The caller forwards `samplingParams` to its transport options; it is absent when
|
|
40
|
+
* the model has no thinking to gate.
|
|
41
|
+
*/
|
|
42
|
+
export type QueuePolishRequest = PolishRequest & PolishSampling;
|
|
43
|
+
|
|
44
|
+
/** The model call, injected. The queue decides `samplingParams` per segment. */
|
|
45
|
+
export type PolishQueueCaller = (request: QueuePolishRequest, signal: AbortSignal) => Promise<AssistantLike>;
|
|
46
|
+
|
|
47
|
+
export interface SegmentOutcome {
|
|
48
|
+
index: number;
|
|
49
|
+
status: "applied" | "fallback";
|
|
50
|
+
/** What this segment contributes: the rewrite when applied, its own raw text otherwise. */
|
|
51
|
+
text: string;
|
|
52
|
+
reason?: string;
|
|
53
|
+
retried: boolean;
|
|
54
|
+
latencyMs: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export interface QueueResult {
|
|
58
|
+
/** The ordered join of every segment outcome (spec §4.2 stitching rule). */
|
|
59
|
+
text: string;
|
|
60
|
+
segments: SegmentOutcome[];
|
|
61
|
+
polished: number;
|
|
62
|
+
failed: number;
|
|
63
|
+
retried: number;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export interface PolishQueue {
|
|
67
|
+
/** Enqueue one recogniser segment. Empty text is skipped; work starts immediately. */
|
|
68
|
+
push(index: number, raw: string): void;
|
|
69
|
+
/** Barrier for one dictation: resolves once every segment pushed before this call settled. */
|
|
70
|
+
finish(): Promise<QueueResult>;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface PolishQueueOptions {
|
|
74
|
+
call: PolishQueueCaller;
|
|
75
|
+
/** Per-segment timeout, the same knob as the single-call path (`postProcessTimeoutMs`). */
|
|
76
|
+
timeoutMs: number;
|
|
77
|
+
/** Session context entries; attached to segment 0 only (spec §4.3). */
|
|
78
|
+
entries?: readonly EntryLike[];
|
|
79
|
+
limits?: ContextLimits;
|
|
80
|
+
/** The model whose thinking policy gates each segment; absent or non-reasoning keeps thinking on. */
|
|
81
|
+
model?: { reasoning?: boolean } | undefined | null;
|
|
82
|
+
/** Defaults to POLISH_QUEUE_CONCURRENCY; clamped to at least one. */
|
|
83
|
+
concurrency?: number;
|
|
84
|
+
/**
|
|
85
|
+
* One pass id for the whole queue. Checked before each attempt is scheduled (first try and
|
|
86
|
+
* retry), so an invalidated pass stops sending new requests instead of merely discarding
|
|
87
|
+
* the answers; `polishTranscript` checks it again before and after every call.
|
|
88
|
+
*/
|
|
89
|
+
isCurrent?: () => boolean;
|
|
90
|
+
/** Injected clock, so tests can pin latency without waiting. */
|
|
91
|
+
now?: () => number;
|
|
92
|
+
debug?: (reason: string, data?: Record<string, unknown>) => void;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
interface SegmentJob {
|
|
96
|
+
index: number;
|
|
97
|
+
raw: string;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** One real call, which owns a concurrency slot until the transport settles - timeout or not. */
|
|
101
|
+
interface LiveCall {
|
|
102
|
+
settled: boolean;
|
|
103
|
+
orphaned: boolean;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** One attempt: its result, whether a request really went out, or that no request was sent. */
|
|
107
|
+
interface AttemptRun {
|
|
108
|
+
issued: boolean;
|
|
109
|
+
denied: boolean;
|
|
110
|
+
/** The pass lost ownership before this attempt could send anything. */
|
|
111
|
+
stale: boolean;
|
|
112
|
+
result: PolishResult | null;
|
|
113
|
+
error?: unknown;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Stitch segment texts into one transcript. Two adjacent CJK characters take no separator,
|
|
118
|
+
* so Chinese stays "这是第一段这是第二段" while English keeps "first part second part"; every
|
|
119
|
+
* other boundary takes a single space, which is what zh-en dictation reads as. Empty parts
|
|
120
|
+
* are dropped so a silent segment cannot inject a stray space, and each part is trimmed at
|
|
121
|
+
* its edges only.
|
|
122
|
+
*/
|
|
123
|
+
export function joinSegments(parts: readonly string[]): string {
|
|
124
|
+
let joined = "";
|
|
125
|
+
for (const part of parts) {
|
|
126
|
+
const text = part.trim();
|
|
127
|
+
if (!text) continue;
|
|
128
|
+
if (joined && needsSeparator(joined, text)) joined += " ";
|
|
129
|
+
joined += text;
|
|
130
|
+
}
|
|
131
|
+
return joined;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function needsSeparator(left: string, right: string): boolean {
|
|
135
|
+
return !(isCjk(lastCodePoint(left)) && isCjk(firstCodePoint(right)));
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function firstCodePoint(text: string): number {
|
|
139
|
+
return text.codePointAt(0) ?? -1;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** The code point at the end of `text`, surrogate pairs included. */
|
|
143
|
+
function lastCodePoint(text: string): number {
|
|
144
|
+
const last = text.charCodeAt(text.length - 1);
|
|
145
|
+
const previous = text.charCodeAt(text.length - 2);
|
|
146
|
+
if (last >= 0xdc00 && last <= 0xdfff && previous >= 0xd800 && previous <= 0xdbff) {
|
|
147
|
+
return (previous - 0xd800) * 0x400 + (last - 0xdc00) + 0x10000;
|
|
148
|
+
}
|
|
149
|
+
return last;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* CJK for the stitching rule: ideographs and their extensions, kana, Bopomofo, CJK
|
|
154
|
+
* punctuation and fullwidth forms. Hangul is left out on purpose — Korean separates words
|
|
155
|
+
* with spaces.
|
|
156
|
+
*/
|
|
157
|
+
function isCjk(codePoint: number): boolean {
|
|
158
|
+
return (
|
|
159
|
+
(codePoint >= 0x2e80 && codePoint <= 0x2eff) || // CJK Radicals Supplement
|
|
160
|
+
(codePoint >= 0x3000 && codePoint <= 0x303f) || // CJK Symbols and Punctuation
|
|
161
|
+
(codePoint >= 0x3040 && codePoint <= 0x30ff) || // Hiragana and Katakana
|
|
162
|
+
(codePoint >= 0x3100 && codePoint <= 0x312f) || // Bopomofo
|
|
163
|
+
(codePoint >= 0x3190 && codePoint <= 0x319f) || // Kanbun
|
|
164
|
+
(codePoint >= 0x31c0 && codePoint <= 0x31ef) || // CJK Strokes
|
|
165
|
+
(codePoint >= 0x3400 && codePoint <= 0x4dbf) || // CJK Unified Ideographs Extension A
|
|
166
|
+
(codePoint >= 0x4e00 && codePoint <= 0x9fff) || // CJK Unified Ideographs
|
|
167
|
+
(codePoint >= 0xf900 && codePoint <= 0xfaff) || // CJK Compatibility Ideographs
|
|
168
|
+
(codePoint >= 0xfe30 && codePoint <= 0xfe4f) || // CJK Compatibility Forms
|
|
169
|
+
(codePoint >= 0xff01 && codePoint <= 0xff60) || // Fullwidth forms
|
|
170
|
+
(codePoint >= 0x20000 && codePoint <= 0x3ffff) // Extensions B and beyond
|
|
171
|
+
);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** Same shape as `resolveTurns` in post-process-context: floor locally, never let 0 through. */
|
|
175
|
+
function resolveConcurrency(value: number | undefined): number {
|
|
176
|
+
if (value === undefined || !Number.isFinite(value)) return POLISH_QUEUE_CONCURRENCY;
|
|
177
|
+
return Math.max(1, Math.floor(value));
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* One polish queue per dictation. Every segment runs `polishTranscript` with the same
|
|
182
|
+
* guardrails as the single-call path, so nothing here re-implements the prompt, the output
|
|
183
|
+
* validation or the fail-open fallback.
|
|
184
|
+
*/
|
|
185
|
+
export function createPolishQueue(options: PolishQueueOptions): PolishQueue {
|
|
186
|
+
const concurrency = resolveConcurrency(options.concurrency);
|
|
187
|
+
const limits = options.limits ?? DEFAULT_CONTEXT_LIMITS;
|
|
188
|
+
const clock = (): number => {
|
|
189
|
+
try {
|
|
190
|
+
return (options.now ?? Date.now)();
|
|
191
|
+
} catch {
|
|
192
|
+
// A broken injected clock must not cost the user their dictation.
|
|
193
|
+
return Date.now();
|
|
194
|
+
}
|
|
195
|
+
};
|
|
196
|
+
/** Raw text by index: the join order, and the reference turn for the next segment. */
|
|
197
|
+
const raws = new Map<number, string>();
|
|
198
|
+
const outcomes = new Map<number, SegmentOutcome>();
|
|
199
|
+
/**
|
|
200
|
+
* Real calls in flight, and how many of them outlived their segment. A slot is released when
|
|
201
|
+
* the transport settles, never when the segment deadline fires: an injected caller that
|
|
202
|
+
* ignores the abort signal keeps its slot, so a hung endpoint cannot push the live request
|
|
203
|
+
* count past `concurrency`. `running` counts segments whose outcome is not decided yet.
|
|
204
|
+
*/
|
|
205
|
+
let liveCalls = 0;
|
|
206
|
+
let orphanedCalls = 0;
|
|
207
|
+
let running = 0;
|
|
208
|
+
const slotWaiters: ((granted: boolean) => void)[] = [];
|
|
209
|
+
let finished: Promise<QueueResult> | null = null;
|
|
210
|
+
let resolveIdle: (() => void) | null = null;
|
|
211
|
+
|
|
212
|
+
function orderedIndexes(): number[] {
|
|
213
|
+
return [...raws.keys()].sort((left, right) => left - right);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
function hasFreeSlot(): boolean {
|
|
217
|
+
return liveCalls < concurrency;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* True when no slot can free on its own: every live call has already outlived its deadline,
|
|
222
|
+
* so waiting only delays the fallback. Every issued call sits inside a `polishTranscript`
|
|
223
|
+
* timeout, so a live call either settles or becomes an orphan within one deadline.
|
|
224
|
+
*/
|
|
225
|
+
function capacityStuck(): boolean {
|
|
226
|
+
return !hasFreeSlot() && orphanedCalls >= liveCalls;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/** After a slot frees or an orphan appears: grant while slots last, else deny when stuck. */
|
|
230
|
+
function wakeWaiters(): void {
|
|
231
|
+
while (slotWaiters.length > 0) {
|
|
232
|
+
if (hasFreeSlot()) {
|
|
233
|
+
liveCalls += 1;
|
|
234
|
+
slotWaiters.shift()!(true);
|
|
235
|
+
continue;
|
|
236
|
+
}
|
|
237
|
+
if (capacityStuck()) {
|
|
238
|
+
for (const waiter of slotWaiters.splice(0)) waiter(false);
|
|
239
|
+
}
|
|
240
|
+
return;
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/** Take a slot right now, so a pushed segment reaches its caller without a microtask delay. */
|
|
245
|
+
function tryTakeSlot(): boolean {
|
|
246
|
+
if (!hasFreeSlot()) return false;
|
|
247
|
+
liveCalls += 1;
|
|
248
|
+
return true;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/** Wait for a slot; false means no live call will ever free one, so the segment falls back. */
|
|
252
|
+
function waitForSlot(): Promise<boolean> {
|
|
253
|
+
if (tryTakeSlot()) return Promise.resolve(true);
|
|
254
|
+
if (capacityStuck()) return Promise.resolve(false);
|
|
255
|
+
return new Promise<boolean>((resolve) => {
|
|
256
|
+
slotWaiters.push(resolve);
|
|
257
|
+
});
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
function releaseSlot(orphaned: boolean): void {
|
|
261
|
+
if (orphaned) orphanedCalls = Math.max(0, orphanedCalls - 1);
|
|
262
|
+
liveCalls = Math.max(0, liveCalls - 1);
|
|
263
|
+
wakeWaiters();
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/** The attempt is over but its call never settled: the slot stays held until it does. */
|
|
267
|
+
function markOrphaned(call: LiveCall): void {
|
|
268
|
+
if (call.orphaned) return;
|
|
269
|
+
call.orphaned = true;
|
|
270
|
+
orphanedCalls += 1;
|
|
271
|
+
wakeWaiters();
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
function resolveIdleNow(): void {
|
|
275
|
+
if (!resolveIdle || running > 0) return;
|
|
276
|
+
const resolve = resolveIdle;
|
|
277
|
+
resolveIdle = null;
|
|
278
|
+
resolve();
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
function settle(job: SegmentJob, outcome: SegmentOutcome): void {
|
|
282
|
+
outcomes.set(job.index, outcome);
|
|
283
|
+
running -= 1;
|
|
284
|
+
resolveIdleNow();
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Spec §4.3: segment 0 carries the session context; a later segment carries the previous
|
|
289
|
+
* segment's raw text as a single reference turn instead. The reference turn goes through
|
|
290
|
+
* `assembleContext` so the same caps apply to it as to the session context.
|
|
291
|
+
*/
|
|
292
|
+
function entriesForSegment(index: number): readonly EntryLike[] {
|
|
293
|
+
if (index === 0) return options.entries ?? [];
|
|
294
|
+
const previous = raws.get(index - 1);
|
|
295
|
+
if (!previous) return [];
|
|
296
|
+
const reference = assembleContext([{ type: "message", message: { role: "user", content: previous } }], limits);
|
|
297
|
+
const text = reference.turns
|
|
298
|
+
.map((turn) => turn.text)
|
|
299
|
+
.join("\n")
|
|
300
|
+
.trim();
|
|
301
|
+
if (!text) return [];
|
|
302
|
+
return [{ type: "message", message: { role: "user", content: text } }];
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
function debugFor(index: number): ((reason: string, data?: Record<string, unknown>) => void) | undefined {
|
|
306
|
+
const debug = options.debug;
|
|
307
|
+
if (!debug) return undefined;
|
|
308
|
+
return (reason, data) => {
|
|
309
|
+
try {
|
|
310
|
+
debug(reason, { ...data, segment: index });
|
|
311
|
+
} catch {
|
|
312
|
+
// The debug hook is observational: a throw here must not break fail-open.
|
|
313
|
+
}
|
|
314
|
+
};
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
function appliedOutcome(job: SegmentJob, text: string, retried: boolean, started: number): SegmentOutcome {
|
|
318
|
+
return { index: job.index, status: "applied", text, retried, latencyMs: Math.max(0, clock() - started) };
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
function fallbackOutcome(
|
|
322
|
+
job: SegmentJob,
|
|
323
|
+
reason: string | undefined,
|
|
324
|
+
retried: boolean,
|
|
325
|
+
started: number
|
|
326
|
+
): SegmentOutcome {
|
|
327
|
+
const outcome: SegmentOutcome = {
|
|
328
|
+
index: job.index,
|
|
329
|
+
status: "fallback",
|
|
330
|
+
text: job.raw,
|
|
331
|
+
retried,
|
|
332
|
+
latencyMs: Math.max(0, clock() - started),
|
|
333
|
+
};
|
|
334
|
+
if (reason !== undefined) outcome.reason = reason;
|
|
335
|
+
return outcome;
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
function reasonOf(run: AttemptRun): string | undefined {
|
|
339
|
+
if (run.stale) return "invalidated";
|
|
340
|
+
if (run.error !== undefined) {
|
|
341
|
+
return run.error instanceof Error && run.error.message ? run.error.message : "segment-error";
|
|
342
|
+
}
|
|
343
|
+
if (run.denied) return "no-capacity";
|
|
344
|
+
return run.result?.reason;
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/** True once the pass that owns this queue has been invalidated. A throwing check keeps
|
|
348
|
+
* fail-open intact: it must never block a dictation that might still be valid. */
|
|
349
|
+
function stale(): boolean {
|
|
350
|
+
try {
|
|
351
|
+
return options.isCurrent?.() === false;
|
|
352
|
+
} catch {
|
|
353
|
+
return false;
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* Spec §4.2: one retry, and only for a call that timed out or failed at the transport. A
|
|
359
|
+
* guardrail rejection means the rewrite was wrong rather than slow, so repeating it cannot
|
|
360
|
+
* help; an attempt that never issued a request has nothing to repeat.
|
|
361
|
+
*/
|
|
362
|
+
function isRetryable(run: AttemptRun): boolean {
|
|
363
|
+
if (!run.issued || !run.result || run.result.status !== "rejected") return false;
|
|
364
|
+
return run.result.reason === "timeout" || run.result.reason === "call-failed";
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
function unexpectedOutcome(job: SegmentJob, error: unknown, latencyMs: number): SegmentOutcome {
|
|
368
|
+
return {
|
|
369
|
+
index: job.index,
|
|
370
|
+
status: "fallback",
|
|
371
|
+
text: job.raw,
|
|
372
|
+
reason: error instanceof Error && error.message ? error.message : "segment-error",
|
|
373
|
+
retried: false,
|
|
374
|
+
latencyMs,
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* One model attempt. The slot is taken before `polishTranscript` starts its deadline, so a
|
|
380
|
+
* segment that waited for capacity still gets its full timeout, and the slot goes back only
|
|
381
|
+
* once the transport settles.
|
|
382
|
+
*/
|
|
383
|
+
async function runAttempt(job: SegmentJob, forceOff: boolean): Promise<AttemptRun> {
|
|
384
|
+
// Invalidation is checked before a slot is taken: a cancelled or superseded pass must stop
|
|
385
|
+
// scheduling requests, not just ignore their answers.
|
|
386
|
+
if (stale()) return { issued: false, denied: false, stale: true, result: null };
|
|
387
|
+
if (!tryTakeSlot() && !(await waitForSlot())) return { issued: false, denied: true, stale: false, result: null };
|
|
388
|
+
// The pass may have died while this segment waited for a slot.
|
|
389
|
+
if (stale()) {
|
|
390
|
+
releaseSlot(false);
|
|
391
|
+
return { issued: false, denied: false, stale: true, result: null };
|
|
392
|
+
}
|
|
393
|
+
const call: LiveCall = { settled: false, orphaned: false };
|
|
394
|
+
let issued = false;
|
|
395
|
+
try {
|
|
396
|
+
const sampling = polishSamplingOptions(options.model, job.raw.length, forceOff);
|
|
397
|
+
const result = await polishTranscript({
|
|
398
|
+
raw: job.raw,
|
|
399
|
+
entries: entriesForSegment(job.index),
|
|
400
|
+
limits,
|
|
401
|
+
timeoutMs: options.timeoutMs,
|
|
402
|
+
timestamp: clock(),
|
|
403
|
+
isCurrent: options.isCurrent,
|
|
404
|
+
call: async (request, signal) => {
|
|
405
|
+
issued = true;
|
|
406
|
+
try {
|
|
407
|
+
return await options.call({ ...request, ...sampling }, signal);
|
|
408
|
+
} finally {
|
|
409
|
+
call.settled = true;
|
|
410
|
+
releaseSlot(call.orphaned);
|
|
411
|
+
}
|
|
412
|
+
},
|
|
413
|
+
debug: debugFor(job.index),
|
|
414
|
+
});
|
|
415
|
+
if (!call.settled) markOrphaned(call);
|
|
416
|
+
return { issued, denied: false, stale: false, result };
|
|
417
|
+
} catch (error) {
|
|
418
|
+
// `polishTranscript` normally reports through its result; this only covers a fault
|
|
419
|
+
// thrown before it could invoke the caller.
|
|
420
|
+
if (!issued) releaseSlot(false);
|
|
421
|
+
else if (!call.settled) markOrphaned(call);
|
|
422
|
+
return { issued, denied: false, stale: false, result: null, error };
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
async function runSegment(job: SegmentJob): Promise<SegmentOutcome> {
|
|
427
|
+
const started = clock();
|
|
428
|
+
try {
|
|
429
|
+
const first = await runAttempt(job, false);
|
|
430
|
+
if (first.result?.status === "applied") return appliedOutcome(job, first.result.text, false, started);
|
|
431
|
+
const firstReason = reasonOf(first);
|
|
432
|
+
if (!isRetryable(first)) return fallbackOutcome(job, firstReason, false, started);
|
|
433
|
+
// Never retry for a pass that lost ownership while the first attempt ran.
|
|
434
|
+
if (stale()) return fallbackOutcome(job, "invalidated", false, started);
|
|
435
|
+
const second = await runAttempt(job, true);
|
|
436
|
+
if (second.result?.status === "applied") return appliedOutcome(job, second.result.text, second.issued, started);
|
|
437
|
+
// A retry that never reached the transport leaves the first failure as the reason.
|
|
438
|
+
const reason = second.issued ? (reasonOf(second) ?? firstReason) : firstReason;
|
|
439
|
+
return fallbackOutcome(job, reason, second.issued, started);
|
|
440
|
+
} catch (error) {
|
|
441
|
+
// Fail-open, last resort: this segment keeps its own raw text.
|
|
442
|
+
return unexpectedOutcome(job, error, Math.max(0, clock() - started));
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
function buildResult(): QueueResult {
|
|
447
|
+
const segments: SegmentOutcome[] = [];
|
|
448
|
+
for (const index of orderedIndexes()) {
|
|
449
|
+
const outcome = outcomes.get(index);
|
|
450
|
+
if (outcome) {
|
|
451
|
+
segments.push(outcome);
|
|
452
|
+
} else {
|
|
453
|
+
// A pushed segment always settles before finish() resolves, so this only fires if
|
|
454
|
+
// an unexpected throw slipped past the per-segment catch. Keep the raw text.
|
|
455
|
+
segments.push({
|
|
456
|
+
index,
|
|
457
|
+
status: "fallback",
|
|
458
|
+
text: raws.get(index) ?? "",
|
|
459
|
+
reason: "missing-outcome",
|
|
460
|
+
retried: false,
|
|
461
|
+
latencyMs: 0,
|
|
462
|
+
});
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
return {
|
|
466
|
+
text: joinSegments(segments.map((segment) => segment.text)),
|
|
467
|
+
segments,
|
|
468
|
+
polished: segments.filter((segment) => segment.status === "applied").length,
|
|
469
|
+
failed: segments.filter((segment) => segment.status === "fallback").length,
|
|
470
|
+
retried: segments.filter((segment) => segment.retried).length,
|
|
471
|
+
};
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
/** Every segment raw, order kept: the last-resort result if the builder itself faults. */
|
|
475
|
+
function rawOnlyResult(reason: string): QueueResult {
|
|
476
|
+
const segments: SegmentOutcome[] = orderedIndexes().map((index) => ({
|
|
477
|
+
index,
|
|
478
|
+
status: "fallback" as const,
|
|
479
|
+
text: raws.get(index) ?? "",
|
|
480
|
+
reason,
|
|
481
|
+
retried: false,
|
|
482
|
+
latencyMs: 0,
|
|
483
|
+
}));
|
|
484
|
+
return {
|
|
485
|
+
text: joinSegments(segments.map((segment) => segment.text)),
|
|
486
|
+
segments,
|
|
487
|
+
polished: 0,
|
|
488
|
+
failed: segments.length,
|
|
489
|
+
retried: 0,
|
|
490
|
+
};
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
return {
|
|
494
|
+
push(index: number, raw: string): void {
|
|
495
|
+
// finish() is the barrier for one dictation: anything offered afterwards belongs to
|
|
496
|
+
// a pass that is already being written, so it can only be dropped.
|
|
497
|
+
if (finished) return;
|
|
498
|
+
// Silence contributes nothing, exactly as today's segment concatenation skips it.
|
|
499
|
+
if (!raw.trim()) return;
|
|
500
|
+
// One segment per index: a repeated push would duplicate the text in the join.
|
|
501
|
+
if (raws.has(index)) return;
|
|
502
|
+
raws.set(index, raw);
|
|
503
|
+
running += 1;
|
|
504
|
+
const job: SegmentJob = { index, raw };
|
|
505
|
+
// Every pushed segment starts here; the slot pool, not this call site, bounds requests.
|
|
506
|
+
void runSegment(job).then(
|
|
507
|
+
(outcome) => settle(job, outcome),
|
|
508
|
+
(error: unknown) => settle(job, unexpectedOutcome(job, error, 0))
|
|
509
|
+
);
|
|
510
|
+
},
|
|
511
|
+
|
|
512
|
+
finish(): Promise<QueueResult> {
|
|
513
|
+
if (!finished) {
|
|
514
|
+
finished = new Promise<QueueResult>((resolve) => {
|
|
515
|
+
// The result builder is the last place a fault could cost the dictation its
|
|
516
|
+
// text, so a throw still resolves with the ordered raw transcript.
|
|
517
|
+
resolveIdle = () => {
|
|
518
|
+
try {
|
|
519
|
+
resolve(buildResult());
|
|
520
|
+
} catch {
|
|
521
|
+
resolve(rawOnlyResult("result-error"));
|
|
522
|
+
}
|
|
523
|
+
};
|
|
524
|
+
});
|
|
525
|
+
// Nothing pushed: resolve in this tick instead of waiting for a segment that may
|
|
526
|
+
// never come.
|
|
527
|
+
resolveIdleNow();
|
|
528
|
+
}
|
|
529
|
+
return finished;
|
|
530
|
+
},
|
|
531
|
+
};
|
|
532
|
+
}
|
|
@@ -142,7 +142,7 @@ export type EditorRead = string | typeof EDITOR_READ_FAILED;
|
|
|
142
142
|
* while the pass still owns the flow AND the editor still holds the value the pass
|
|
143
143
|
* snapshotted. Used by the normal path and by the pass's own throw path.
|
|
144
144
|
*/
|
|
145
|
-
export function decideApply(input: { tokenCurrent: boolean; editorSnapshot:
|
|
145
|
+
export function decideApply(input: { tokenCurrent: boolean; editorSnapshot: EditorRead; currentEditor: EditorRead }): {
|
|
146
146
|
apply: boolean;
|
|
147
147
|
reason?: string;
|
|
148
148
|
} {
|
|
@@ -180,6 +180,12 @@ export async function polishTranscript(input: PolishInput): Promise<PolishResult
|
|
|
180
180
|
const context = assembleContext(input.entries, input.limits);
|
|
181
181
|
const shape = { contextChars: context.characters, truncatedContext: context.truncated };
|
|
182
182
|
const request = buildPolishRequest(context, input.raw, input.timestamp);
|
|
183
|
+
// Invalidation stops scheduling, not merely the write: a cancelled or superseded pass must
|
|
184
|
+
// not send transcript text to the provider at all, so the check in front of the call is as
|
|
185
|
+
// important as the one after it.
|
|
186
|
+
if (input.isCurrent && !input.isCurrent()) {
|
|
187
|
+
return { status: "skipped", text: input.raw, reason: "invalidated", ...shape };
|
|
188
|
+
}
|
|
183
189
|
const controller = new AbortController();
|
|
184
190
|
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
185
191
|
|
|
@@ -218,6 +224,17 @@ export async function polishTranscript(input: PolishInput): Promise<PolishResult
|
|
|
218
224
|
}
|
|
219
225
|
}
|
|
220
226
|
|
|
227
|
+
/**
|
|
228
|
+
* Per-segment outcome of the pipelined pass, one summary per dictation. Absent for the
|
|
229
|
+
* single-call path, where there are no recogniser segments to report.
|
|
230
|
+
*/
|
|
231
|
+
export interface PolishAuditSegments {
|
|
232
|
+
count: number;
|
|
233
|
+
polished: number;
|
|
234
|
+
failed: number;
|
|
235
|
+
retried: number;
|
|
236
|
+
}
|
|
237
|
+
|
|
221
238
|
/**
|
|
222
239
|
* One durable record of what a pass did, written into the session file by the caller.
|
|
223
240
|
*
|
|
@@ -248,6 +265,14 @@ export interface PolishAudit {
|
|
|
248
265
|
durationSec?: number;
|
|
249
266
|
/** Which recogniser produced it, so results are never pooled across backends. */
|
|
250
267
|
backend?: string;
|
|
268
|
+
/**
|
|
269
|
+
* The local model that produced the text (e.g. "parakeet-v3"), so dictations can be
|
|
270
|
+
* grouped by recogniser within the local backend. Absent for cloud backends, which
|
|
271
|
+
* carry their own server-side model choice.
|
|
272
|
+
*/
|
|
273
|
+
recognizer?: string;
|
|
274
|
+
/** Per-segment outcome when the segmented queue produced this dictation. */
|
|
275
|
+
segments?: PolishAuditSegments;
|
|
251
276
|
/** True when the pass asked the model not to think. */
|
|
252
277
|
thinkingOff?: boolean;
|
|
253
278
|
/** The output-token cap the request carried, a cap and not a spend. */
|
|
@@ -265,6 +290,8 @@ export function buildPolishAudit(input: {
|
|
|
265
290
|
thinkingOff?: boolean;
|
|
266
291
|
durationSec?: number;
|
|
267
292
|
backend?: string;
|
|
293
|
+
recognizer?: string;
|
|
294
|
+
segments?: PolishAuditSegments;
|
|
268
295
|
maxTokens?: number;
|
|
269
296
|
telemetry?: {
|
|
270
297
|
model?: string;
|
|
@@ -290,6 +317,8 @@ export function buildPolishAudit(input: {
|
|
|
290
317
|
if (input.maxTokens !== undefined) audit.maxTokens = input.maxTokens;
|
|
291
318
|
if (input.durationSec !== undefined) audit.durationSec = input.durationSec;
|
|
292
319
|
if (input.backend !== undefined) audit.backend = input.backend;
|
|
320
|
+
if (input.recognizer !== undefined) audit.recognizer = input.recognizer;
|
|
321
|
+
if (input.segments !== undefined) audit.segments = input.segments;
|
|
293
322
|
const telemetry = input.telemetry;
|
|
294
323
|
if (telemetry) {
|
|
295
324
|
if (telemetry.model !== undefined) audit.model = telemetry.model;
|
|
@@ -319,14 +348,19 @@ export function buildPolishAudit(input: {
|
|
|
319
348
|
*
|
|
320
349
|
* `samplingParams` is applied by OpenAI-compatible adapters only, and the `reasoning` gate keeps
|
|
321
350
|
* the field away from models with no thinking at all.
|
|
351
|
+
*
|
|
352
|
+
* `forceOff` is the segmented queue's retry: the attempt already burned its deadline, so it runs
|
|
353
|
+
* small and cheap regardless of length. It overrides the length gate only - a model with no
|
|
354
|
+
* thinking still gets no sampling fields at all.
|
|
322
355
|
*/
|
|
323
356
|
export const THINKING_MAX_CHARS = 200;
|
|
324
357
|
|
|
325
358
|
export function polishSamplingOptions(
|
|
326
359
|
model: { reasoning?: boolean } | undefined | null,
|
|
327
|
-
rawLength: number
|
|
360
|
+
rawLength: number,
|
|
361
|
+
forceOff = false
|
|
328
362
|
): { samplingParams?: { reasoning_effort: string } } {
|
|
329
363
|
if (!model || model.reasoning !== true) return {};
|
|
330
|
-
if (Number.isFinite(rawLength) && rawLength <= THINKING_MAX_CHARS) return {};
|
|
364
|
+
if (!forceOff && Number.isFinite(rawLength) && rawLength <= THINKING_MAX_CHARS) return {};
|
|
331
365
|
return { samplingParams: { reasoning_effort: "none" } };
|
|
332
366
|
}
|
|
@@ -432,33 +432,57 @@ export function segmentPcmForLongAudio(samples: Float32Array, sampleRate: number
|
|
|
432
432
|
return segments.length > 0 ? segments : [samples];
|
|
433
433
|
}
|
|
434
434
|
|
|
435
|
+
/**
|
|
436
|
+
* Decode VAD segments in order, calling `onSegment(text, index)` as soon as a segment's text
|
|
437
|
+
* is available and before the next segment starts decoding. Returns the non-empty trimmed
|
|
438
|
+
* texts in order — the parts `transcribeBufferSegmented` joins.
|
|
439
|
+
*
|
|
440
|
+
* The recognizer is already a parameter, so this seam can be driven with a stub in tests.
|
|
441
|
+
* `onSegment` is observational: a throwing callback is swallowed, because an observer must
|
|
442
|
+
* never cost the user their transcript.
|
|
443
|
+
*/
|
|
444
|
+
export async function decodeSegmentsInOrder(
|
|
445
|
+
recognizer: SherpaRecognizer,
|
|
446
|
+
segments: readonly Float32Array[],
|
|
447
|
+
onSegment?: (text: string, index: number) => void
|
|
448
|
+
): Promise<string[]> {
|
|
449
|
+
const parts: string[] = [];
|
|
450
|
+
for (const [index, segment] of segments.entries()) {
|
|
451
|
+
const stream = recognizer.createStream();
|
|
452
|
+
stream.acceptWaveform({ sampleRate: 16000, samples: segment });
|
|
453
|
+
await recognizer.decodeAsync(stream);
|
|
454
|
+
const r = recognizer.getResult(stream);
|
|
455
|
+
const t = (r?.text || "").trim();
|
|
456
|
+
if (onSegment) {
|
|
457
|
+
try {
|
|
458
|
+
onSegment(t, index);
|
|
459
|
+
} catch {}
|
|
460
|
+
}
|
|
461
|
+
if (t) parts.push(t);
|
|
462
|
+
}
|
|
463
|
+
return parts;
|
|
464
|
+
}
|
|
465
|
+
|
|
435
466
|
/**
|
|
436
467
|
* Transcribe PCM with automatic VAD segmentation for long recordings.
|
|
437
468
|
* Byte-identical fast path (single decode) for audio ≤ thresholdSecs.
|
|
469
|
+
*
|
|
470
|
+
* `onSegment` is called for every decoded segment — the fast path's single segment included —
|
|
471
|
+
* as soon as it decodes and before the next decode starts. Leaving it out keeps today's exact
|
|
472
|
+
* return value (docs/superpowers/specs/2026-09-26-polish-pipeline-design.md §4.1).
|
|
438
473
|
*/
|
|
439
474
|
export async function transcribeBufferSegmented(
|
|
440
475
|
pcmData: Buffer,
|
|
441
476
|
recognizer: SherpaRecognizer,
|
|
442
|
-
thresholdSecs = 10
|
|
477
|
+
thresholdSecs = 10,
|
|
478
|
+
onSegment?: (text: string, index: number) => void
|
|
443
479
|
): Promise<string> {
|
|
444
480
|
getSherpaModule();
|
|
445
481
|
const samples = pcmToFloat32(pcmData);
|
|
482
|
+
// The fast path is one segment too, so a pipelining caller sees exactly one segment and a
|
|
483
|
+
// short dictation keeps today's one-call behaviour.
|
|
446
484
|
if (samples.length / 16000 <= thresholdSecs) {
|
|
447
|
-
|
|
448
|
-
stream.acceptWaveform({ sampleRate: 16000, samples });
|
|
449
|
-
await recognizer.decodeAsync(stream);
|
|
450
|
-
const r = recognizer.getResult(stream);
|
|
451
|
-
return (r?.text || "").trim();
|
|
452
|
-
}
|
|
453
|
-
|
|
454
|
-
const parts: string[] = [];
|
|
455
|
-
for (const seg of segmentPcmForLongAudio(samples, 16000)) {
|
|
456
|
-
const stream = recognizer.createStream();
|
|
457
|
-
stream.acceptWaveform({ sampleRate: 16000, samples: seg });
|
|
458
|
-
await recognizer.decodeAsync(stream);
|
|
459
|
-
const r = recognizer.getResult(stream);
|
|
460
|
-
const t = (r?.text || "").trim();
|
|
461
|
-
if (t) parts.push(t);
|
|
485
|
+
return (await decodeSegmentsInOrder(recognizer, [samples], onSegment)).join(" ");
|
|
462
486
|
}
|
|
463
|
-
return
|
|
487
|
+
return (await decodeSegmentsInOrder(recognizer, segmentPcmForLongAudio(samples, 16000), onSegment)).join(" ");
|
|
464
488
|
}
|