@livx.cc/agentx 0.99.6 → 0.99.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -974,8 +974,10 @@ declare class DuplexAgent {
974
974
  * left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
975
975
  * Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
976
976
  private turnDispatched;
977
+ private spokeBeforeDispatch;
977
978
  private turnBriefs;
978
979
  private spokeThisTurn;
980
+ private externalSpeech;
979
981
  private heldThisTurn;
980
982
  private nudging;
981
983
  private reflexBuf;
@@ -1001,6 +1003,18 @@ declare class DuplexAgent {
1001
1003
  question: string;
1002
1004
  resolve: (answer: string) => void;
1003
1005
  }>;
1006
+ /** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
1007
+ * PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
1008
+ * confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
1009
+ * aborts it: output dropped, history rolled back, the final dispatches normally. */
1010
+ private spec?;
1011
+ /** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
1012
+ speculativeAbortedCalls: number;
1013
+ /** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
1014
+ * intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
1015
+ * writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
1016
+ * instead (the endpointed final then dispatches normally and may use the tool for real). */
1017
+ private static readonly SPEC_SAFE_TOOLS;
1004
1018
  /** Lazily resolved memory tools (async loadMemory runs in initMemory). */
1005
1019
  private memoryReady;
1006
1020
  constructor(options?: Partial<DuplexAgentOptions>);
@@ -1009,6 +1023,10 @@ declare class DuplexAgent {
1009
1023
  /** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
1010
1024
  * turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
1011
1025
  private flushHeldReflexTail;
1026
+ /** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
1027
+ * Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
1028
+ * reflexForwarded would corrupt the forward offset. */
1029
+ private scrubStageDirections;
1012
1030
  /** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
1013
1031
  private resetTurn;
1014
1032
  /** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
@@ -1016,6 +1034,11 @@ declare class DuplexAgent {
1016
1034
  * re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
1017
1035
  * still allowed (parallel independent work). During a re-ack pass, block every tool. */
1018
1036
  private dispatchGuard;
1037
+ /** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
1038
+ * micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
1039
+ * inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
1040
+ * still dead air, so silentTurn ignores external speech unless work was dispatched. */
1041
+ noteExternalSpeech(): void;
1019
1042
  /** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
1020
1043
  * (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
1021
1044
  * gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
@@ -1026,8 +1049,27 @@ declare class DuplexAgent {
1026
1049
  * line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
1027
1050
  * Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
1028
1051
  private ackIfSilent;
1029
- /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns. */
1052
+ /** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
1053
+ * hosts/tests key on it. */
1054
+ private static readonly FALLBACK_ACKS;
1055
+ private static readonly FALLBACK_RETRY;
1056
+ private lastFallback;
1057
+ /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
1058
+ * If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
1059
+ * speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
1060
+ * host NOW (this is the latency win) and streaming continues live. Any other content aborts the
1061
+ * speculation first (rolled back silently) and runs a normal turn behind it. */
1030
1062
  send(content: MessageContent): Promise<RunResult>;
1063
+ /** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
1064
+ * output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
1065
+ * voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
1066
+ * into the transcript between the speculative messages and their rollback. No-op if a speculation
1067
+ * is already in flight. */
1068
+ speculate(text: string): void;
1069
+ /** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
1070
+ * Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
1071
+ * attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
1072
+ abortSpeculation(): void;
1031
1073
  /** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
1032
1074
  cancelTask(id: string): string;
1033
1075
  /** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
@@ -1040,6 +1082,11 @@ declare class DuplexAgent {
1040
1082
  /** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
1041
1083
  private enqueue;
1042
1084
  private notify;
1085
+ /** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
1086
+ * hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
1087
+ * order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
1088
+ * which bypasses this via host.notify directly) passes through untouched. */
1089
+ private emitHost;
1043
1090
  /** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
1044
1091
  * `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
1045
1092
  * marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
@@ -1221,6 +1268,22 @@ interface AudioSink {
1221
1268
  type AuthProvider = string | (() => string | Promise<string>);
1222
1269
  declare function resolveAuth(auth: AuthProvider): Promise<string>;
1223
1270
 
1271
+ /** Injectable time source — ALL engine timing (merge windows, barge grace, overlap resume, drain
1272
+ * settles) routes through it. Defaults to real timers; the deterministic bench swaps a virtual
1273
+ * clock so timing scenarios run instantly and exactly. */
1274
+ interface EngineClock {
1275
+ now(): number;
1276
+ setTimeout(fn: () => void, ms: number): unknown;
1277
+ clearTimeout(handle: unknown): void;
1278
+ }
1279
+ /** One structured diagnostic event (fire-and-forget, never affects behavior). `t` = engine clock ms.
1280
+ * Emitted at every DECISION point; kinds + fields are the per-session forensic record
1281
+ * (CLI: `<session>.voice.jsonl`; lab: `<session>.timeline.jsonl`). See mind/11-voice.md §Diagnostics. */
1282
+ interface DiagEvent {
1283
+ t: number;
1284
+ kind: string;
1285
+ [k: string]: unknown;
1286
+ }
1224
1287
  /** Structural contracts (satisfied by SonioxSTT/CartesiaTTS or test fakes). */
1225
1288
  interface SttLike {
1226
1289
  usingAec: boolean;
@@ -1236,7 +1299,14 @@ interface SttLike {
1236
1299
  interface TtsLike {
1237
1300
  onAudio: (chunk: Uint8Array) => void;
1238
1301
  onDone: () => void;
1302
+ /** Optional word-timestamp seam for karaoke reveal (revealMode==='word'). `start[i]` = seconds
1303
+ * from turn-audio start. Set `wantTimestamps` to request them from the provider. */
1304
+ onTimestamps?: (words: string[], start: number[]) => void;
1305
+ wantTimestamps?: boolean;
1239
1306
  connect(): Promise<void> | void;
1307
+ /** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
1308
+ * so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
1309
+ warmup?(): void;
1240
1310
  newContext(): string;
1241
1311
  speak(text: string, cont: boolean): void;
1242
1312
  end(): void;
@@ -1258,6 +1328,12 @@ declare class VoiceEngineOptions {
1258
1328
  onBargeIn: (phase: 'speaking' | 'drain') => void;
1259
1329
  /** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
1260
1330
  ackPhrase: string;
1331
+ /** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
1332
+ * delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
1333
+ * per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
1334
+ adaptiveAckMs: number;
1335
+ /** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
1336
+ onAdaptiveAck: () => void;
1261
1337
  /** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
1262
1338
  * letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
1263
1339
  * ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
@@ -1303,11 +1379,58 @@ declare class VoiceEngineOptions {
1303
1379
  * speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
1304
1380
  * mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
1305
1381
  overlapEnergyHold: boolean;
1382
+ /** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
1383
+ * this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
1384
+ * start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
1385
+ * speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
1386
+ * confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
1387
+ speculativeMs: number;
1388
+ /** Minimum word count for a partial to qualify as a speculation trigger. */
1389
+ speculativeMinWords: number;
1390
+ /** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
1391
+ onSpeculate: (text: string) => void;
1392
+ /** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
1393
+ * partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
1394
+ * UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
1395
+ * quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
1396
+ * touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
1397
+ * context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
1398
+ * boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
1399
+ * 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
1400
+ * ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
1401
+ backchannelMs: number;
1402
+ /** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
1403
+ backchannelMinGapMs: number;
1404
+ /** Only multi-clause turns: the partial must carry at least this many words before a blip. */
1405
+ backchannelMinWords: number;
1406
+ /** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
1407
+ onBackchannel: (phrase: string) => void;
1408
+ /** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
1409
+ * quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
1410
+ onSpeculateAbort: () => void;
1306
1411
  /** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
1307
1412
  * tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
1308
1413
  emotions: boolean;
1309
1414
  /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
1310
1415
  showEmotions: boolean;
1416
+ /**
1417
+ * Progressive text reveal — the "karaoke" capability, opt-in.
1418
+ * 'off' — no reveal events (CLI default; the host renders text however it likes).
1419
+ * 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
1420
+ * in step with the model stream).
1421
+ * 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
1422
+ * timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
1423
+ */
1424
+ revealMode: 'off' | 'delta' | 'word';
1425
+ /** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
1426
+ * Reset to '' at the start of each spoken turn. */
1427
+ onReveal: (revealed: string) => void;
1428
+ /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
1429
+ clock: EngineClock;
1430
+ /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
1431
+ * barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
1432
+ * a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
1433
+ onDiag: (ev: DiagEvent) => void;
1311
1434
  }
1312
1435
  declare class VoiceEngine {
1313
1436
  options: VoiceEngineOptions;
@@ -1319,6 +1442,11 @@ declare class VoiceEngine {
1319
1442
  private ctxOpen;
1320
1443
  private interrupted;
1321
1444
  private spokeDeltas;
1445
+ private revealText;
1446
+ private wordStarts;
1447
+ private revealedN;
1448
+ private revealPoll;
1449
+ private clock;
1322
1450
  private drainTimer;
1323
1451
  private echoWords;
1324
1452
  private prevReply;
@@ -1328,15 +1456,36 @@ declare class VoiceEngine {
1328
1456
  private hot;
1329
1457
  private suspectUntil;
1330
1458
  private ackAt;
1459
+ private lastAck;
1460
+ private ackTimer;
1331
1461
  private bargeGraceUntil;
1332
1462
  private pendingUtt;
1463
+ private mergePath;
1464
+ private lastGraceDiag;
1333
1465
  private pendingTimer;
1466
+ private lastDispatchFlat;
1467
+ private lastDispatchWords;
1468
+ private lastDispatchAt;
1469
+ private repliedSinceDispatch;
1470
+ private static readonly DUP_FINAL_MS;
1334
1471
  private lastInterrupted;
1335
1472
  private pausedAt;
1336
1473
  private lastResumeAt;
1337
1474
  private lastOverlapPartial;
1338
1475
  private resumeTimer;
1339
1476
  private turnStartAt;
1477
+ private specPartial;
1478
+ private specTimer;
1479
+ private specText;
1480
+ private specSpent;
1481
+ private bcPartial;
1482
+ private bcTimer;
1483
+ private bcCount;
1484
+ private bcActive;
1485
+ private bcActiveTimer;
1486
+ private lastBcAt;
1487
+ private lastBcPhrase;
1488
+ private recentBc;
1340
1489
  private uttQueue;
1341
1490
  private emo;
1342
1491
  constructor(options?: Partial<VoiceEngineOptions>);
@@ -1346,6 +1495,10 @@ declare class VoiceEngine {
1346
1495
  setBargeIn(on: boolean): void;
1347
1496
  /** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
1348
1497
  setShowEmotions(on: boolean): void;
1498
+ /** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
1499
+ * it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
1500
+ private diagOn;
1501
+ protected diag(kind: string, fields?: Record<string, unknown>): void;
1349
1502
  private idleWaiters;
1350
1503
  private setState;
1351
1504
  /** Resolve when the engine is no longer speaking (immediate if already idle). */
@@ -1357,6 +1510,15 @@ declare class VoiceEngine {
1357
1510
  /** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
1358
1511
  * host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
1359
1512
  speakDelta(text: string): string;
1513
+ /** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
1514
+ * Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
1515
+ * paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
1516
+ * deterministic under the virtual clock. */
1517
+ private startWordReveal;
1518
+ /** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
1519
+ * undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
1520
+ * is what was actually spoken). */
1521
+ private stopWordReveal;
1360
1522
  /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
1361
1523
  endSpeech(): void;
1362
1524
  /** text of the reply cut by the last barge-in — consumed by the host to tell the model what
@@ -1374,6 +1536,20 @@ declare class VoiceEngine {
1374
1536
  /** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
1375
1537
  * next one from the settle completion (endSpeech), so utterances serialize without overlap. */
1376
1538
  private pumpQueue;
1539
+ /** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
1540
+ * utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
1541
+ * of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
1542
+ * many times"). All phrases are sub-second and semantically safe for their shape.
1543
+ * No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
1544
+ static readonly ACKS_NEUTRAL: string[];
1545
+ static readonly ACKS_QUESTION: string[];
1546
+ private recentAcks;
1547
+ private pickAck;
1548
+ private lastDispatchWasQuestion;
1549
+ private clearAckTimer;
1550
+ /** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
1551
+ * to be a Hold (intentionally silent; an ack would read as the start of an answer). */
1552
+ cancelPendingAck(): void;
1377
1553
  /** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
1378
1554
  interrupt(): void;
1379
1555
  stop(): void;
@@ -1387,12 +1563,67 @@ declare class VoiceEngine {
1387
1563
  * longer ones on count. */
1388
1564
  private genuine;
1389
1565
  private handlePartial;
1566
+ /** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
1567
+ * partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
1568
+ * onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
1569
+ * tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
1570
+ * dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
1571
+ * trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
1572
+ * utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
1573
+ private trackSpeculation;
1574
+ /** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
1575
+ * imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
1576
+ * every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
1577
+ * strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
1578
+ * 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
1579
+ * right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
1580
+ static readonly BACKCHANNELS: string[];
1581
+ /** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
1582
+ * that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
1583
+ * BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
1584
+ * Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
1585
+ * Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
1586
+ private trackBackchannel;
1587
+ /** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
1588
+ * markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
1589
+ * pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
1590
+ * never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
1591
+ * swallowed), and the echo window extends so echo-shaped finals stay gated. */
1592
+ private fireBackchannel;
1593
+ /** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
1594
+ private bcSupersede;
1595
+ /** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
1596
+ * phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
1597
+ * within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
1598
+ private stripBackchannelEcho;
1390
1599
  /** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
1391
1600
  * overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
1392
1601
  * utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
1393
1602
  * a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
1394
1603
  * check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
1395
1604
  private mergeUtterance;
1605
+ private static normWord;
1606
+ /** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
1607
+ * the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
1608
+ * may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
1609
+ * DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
1610
+ * "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
1611
+ * question already being answered).
1612
+ * DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
1613
+ * (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
1614
+ * previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
1615
+ * repliedSinceDispatch then picks the action:
1616
+ * • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
1617
+ * "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
1618
+ * worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
1619
+ * DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
1620
+ * Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
1621
+ * • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
1622
+ * truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
1623
+ * starves the conversation entirely — the fuller final IS the completion the Hold is waiting
1624
+ * for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
1625
+ * race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
1626
+ private refinalizes;
1396
1627
  private static readonly TRAIL_RE;
1397
1628
  /** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
1398
1629
  private looksIncomplete;
@@ -1430,6 +1661,15 @@ declare class SonioxSTT {
1430
1661
  /** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
1431
1662
  * loop). The host tears voice down instead of spinning forever. */
1432
1663
  onFatal: (message: string) => void;
1664
+ /** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
1665
+ * handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
1666
+ onDiag: (ev: {
1667
+ t: number;
1668
+ kind: string;
1669
+ [k: string]: unknown;
1670
+ }) => void;
1671
+ private diagOn;
1672
+ private diag;
1433
1673
  private lastChunkAt;
1434
1674
  private startedChunksAt;
1435
1675
  private noAudioTimer;
@@ -1462,6 +1702,21 @@ declare class CartesiaTTS {
1462
1702
  ctxId: string;
1463
1703
  onAudio: (chunk: Uint8Array) => void;
1464
1704
  onDone: () => void;
1705
+ /** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
1706
+ * across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
1707
+ onTimestamps: (words: string[], start: number[]) => void;
1708
+ /** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
1709
+ * when revealMode==='word'. */
1710
+ wantTimestamps: boolean;
1711
+ /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
1712
+ * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
1713
+ onDiag: (ev: {
1714
+ t: number;
1715
+ kind: string;
1716
+ [k: string]: unknown;
1717
+ }) => void;
1718
+ private diagOn;
1719
+ private diag;
1465
1720
  firstAudioAt: number;
1466
1721
  /** Circuit breaker: consecutive error count + down flag. */
1467
1722
  private consecutiveErrors;
@@ -1482,6 +1737,11 @@ declare class CartesiaTTS {
1482
1737
  private markRecovered;
1483
1738
  /** Ensure the WS is open before sending — reconnects if idle-closed. */
1484
1739
  private ensureConnected;
1740
+ /** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
1741
+ * a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
1742
+ * The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
1743
+ * newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
1744
+ warmup(): void;
1485
1745
  newContext(): string;
1486
1746
  private frame;
1487
1747
  speak(text: string, cont: boolean): void;