@livx.cc/agentx 0.99.5 → 0.99.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -974,8 +974,10 @@ declare class DuplexAgent {
974
974
  * left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
975
975
  * Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
976
976
  private turnDispatched;
977
+ private spokeBeforeDispatch;
977
978
  private turnBriefs;
978
979
  private spokeThisTurn;
980
+ private externalSpeech;
979
981
  private heldThisTurn;
980
982
  private nudging;
981
983
  private reflexBuf;
@@ -1001,6 +1003,18 @@ declare class DuplexAgent {
1001
1003
  question: string;
1002
1004
  resolve: (answer: string) => void;
1003
1005
  }>;
1006
+ /** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
1007
+ * PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
1008
+ * confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
1009
+ * aborts it: output dropped, history rolled back, the final dispatches normally. */
1010
+ private spec?;
1011
+ /** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
1012
+ speculativeAbortedCalls: number;
1013
+ /** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
1014
+ * intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
1015
+ * writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
1016
+ * instead (the endpointed final then dispatches normally and may use the tool for real). */
1017
+ private static readonly SPEC_SAFE_TOOLS;
1004
1018
  /** Lazily resolved memory tools (async loadMemory runs in initMemory). */
1005
1019
  private memoryReady;
1006
1020
  constructor(options?: Partial<DuplexAgentOptions>);
@@ -1009,6 +1023,10 @@ declare class DuplexAgent {
1009
1023
  /** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
1010
1024
  * turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
1011
1025
  private flushHeldReflexTail;
1026
+ /** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
1027
+ * Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
1028
+ * reflexForwarded would corrupt the forward offset. */
1029
+ private scrubStageDirections;
1012
1030
  /** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
1013
1031
  private resetTurn;
1014
1032
  /** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
@@ -1016,6 +1034,11 @@ declare class DuplexAgent {
1016
1034
  * re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
1017
1035
  * still allowed (parallel independent work). During a re-ack pass, block every tool. */
1018
1036
  private dispatchGuard;
1037
+ /** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
1038
+ * micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
1039
+ * inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
1040
+ * still dead air, so silentTurn ignores external speech unless work was dispatched. */
1041
+ noteExternalSpeech(): void;
1019
1042
  /** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
1020
1043
  * (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
1021
1044
  * gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
@@ -1026,8 +1049,27 @@ declare class DuplexAgent {
1026
1049
  * line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
1027
1050
  * Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
1028
1051
  private ackIfSilent;
1029
- /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns. */
1052
+ /** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
1053
+ * hosts/tests key on it. */
1054
+ private static readonly FALLBACK_ACKS;
1055
+ private static readonly FALLBACK_RETRY;
1056
+ private lastFallback;
1057
+ /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
1058
+ * If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
1059
+ * speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
1060
+ * host NOW (this is the latency win) and streaming continues live. Any other content aborts the
1061
+ * speculation first (rolled back silently) and runs a normal turn behind it. */
1030
1062
  send(content: MessageContent): Promise<RunResult>;
1063
+ /** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
1064
+ * output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
1065
+ * voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
1066
+ * into the transcript between the speculative messages and their rollback. No-op if a speculation
1067
+ * is already in flight. */
1068
+ speculate(text: string): void;
1069
+ /** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
1070
+ * Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
1071
+ * attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
1072
+ abortSpeculation(): void;
1031
1073
  /** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
1032
1074
  cancelTask(id: string): string;
1033
1075
  /** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
@@ -1040,6 +1082,11 @@ declare class DuplexAgent {
1040
1082
  /** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
1041
1083
  private enqueue;
1042
1084
  private notify;
1085
+ /** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
1086
+ * hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
1087
+ * order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
1088
+ * which bypasses this via host.notify directly) passes through untouched. */
1089
+ private emitHost;
1043
1090
  /** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
1044
1091
  * `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
1045
1092
  * marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
@@ -1221,6 +1268,22 @@ interface AudioSink {
1221
1268
  type AuthProvider = string | (() => string | Promise<string>);
1222
1269
  declare function resolveAuth(auth: AuthProvider): Promise<string>;
1223
1270
 
1271
+ /** Injectable time source — ALL engine timing (merge windows, barge grace, overlap resume, drain
1272
+ * settles) routes through it. Defaults to real timers; the deterministic bench swaps a virtual
1273
+ * clock so timing scenarios run instantly and exactly. */
1274
+ interface EngineClock {
1275
+ now(): number;
1276
+ setTimeout(fn: () => void, ms: number): unknown;
1277
+ clearTimeout(handle: unknown): void;
1278
+ }
1279
+ /** One structured diagnostic event (fire-and-forget, never affects behavior). `t` = engine clock ms.
1280
+ * Emitted at every DECISION point; kinds + fields are the per-session forensic record
1281
+ * (CLI: `<session>.voice.jsonl`; lab: `<session>.timeline.jsonl`). See mind/11-voice.md §Diagnostics. */
1282
+ interface DiagEvent {
1283
+ t: number;
1284
+ kind: string;
1285
+ [k: string]: unknown;
1286
+ }
1224
1287
  /** Structural contracts (satisfied by SonioxSTT/CartesiaTTS or test fakes). */
1225
1288
  interface SttLike {
1226
1289
  usingAec: boolean;
@@ -1237,6 +1300,9 @@ interface TtsLike {
1237
1300
  onAudio: (chunk: Uint8Array) => void;
1238
1301
  onDone: () => void;
1239
1302
  connect(): Promise<void> | void;
1303
+ /** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
1304
+ * so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
1305
+ warmup?(): void;
1240
1306
  newContext(): string;
1241
1307
  speak(text: string, cont: boolean): void;
1242
1308
  end(): void;
@@ -1258,6 +1324,12 @@ declare class VoiceEngineOptions {
1258
1324
  onBargeIn: (phase: 'speaking' | 'drain') => void;
1259
1325
  /** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
1260
1326
  ackPhrase: string;
1327
+ /** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
1328
+ * delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
1329
+ * per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
1330
+ adaptiveAckMs: number;
1331
+ /** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
1332
+ onAdaptiveAck: () => void;
1261
1333
  /** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
1262
1334
  * letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
1263
1335
  * ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
@@ -1303,11 +1375,46 @@ declare class VoiceEngineOptions {
1303
1375
  * speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
1304
1376
  * mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
1305
1377
  overlapEnergyHold: boolean;
1378
+ /** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
1379
+ * this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
1380
+ * start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
1381
+ * speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
1382
+ * confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
1383
+ speculativeMs: number;
1384
+ /** Minimum word count for a partial to qualify as a speculation trigger. */
1385
+ speculativeMinWords: number;
1386
+ /** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
1387
+ onSpeculate: (text: string) => void;
1388
+ /** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
1389
+ * partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
1390
+ * UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
1391
+ * quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
1392
+ * touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
1393
+ * context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
1394
+ * boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
1395
+ * 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
1396
+ * ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
1397
+ backchannelMs: number;
1398
+ /** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
1399
+ backchannelMinGapMs: number;
1400
+ /** Only multi-clause turns: the partial must carry at least this many words before a blip. */
1401
+ backchannelMinWords: number;
1402
+ /** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
1403
+ onBackchannel: (phrase: string) => void;
1404
+ /** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
1405
+ * quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
1406
+ onSpeculateAbort: () => void;
1306
1407
  /** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
1307
1408
  * tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
1308
1409
  emotions: boolean;
1309
1410
  /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
1310
1411
  showEmotions: boolean;
1412
+ /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
1413
+ clock: EngineClock;
1414
+ /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
1415
+ * barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
1416
+ * a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
1417
+ onDiag: (ev: DiagEvent) => void;
1311
1418
  }
1312
1419
  declare class VoiceEngine {
1313
1420
  options: VoiceEngineOptions;
@@ -1319,6 +1426,7 @@ declare class VoiceEngine {
1319
1426
  private ctxOpen;
1320
1427
  private interrupted;
1321
1428
  private spokeDeltas;
1429
+ private clock;
1322
1430
  private drainTimer;
1323
1431
  private echoWords;
1324
1432
  private prevReply;
@@ -1328,15 +1436,36 @@ declare class VoiceEngine {
1328
1436
  private hot;
1329
1437
  private suspectUntil;
1330
1438
  private ackAt;
1439
+ private lastAck;
1440
+ private ackTimer;
1331
1441
  private bargeGraceUntil;
1332
1442
  private pendingUtt;
1443
+ private mergePath;
1444
+ private lastGraceDiag;
1333
1445
  private pendingTimer;
1446
+ private lastDispatchFlat;
1447
+ private lastDispatchWords;
1448
+ private lastDispatchAt;
1449
+ private repliedSinceDispatch;
1450
+ private static readonly DUP_FINAL_MS;
1334
1451
  private lastInterrupted;
1335
1452
  private pausedAt;
1336
1453
  private lastResumeAt;
1337
1454
  private lastOverlapPartial;
1338
1455
  private resumeTimer;
1339
1456
  private turnStartAt;
1457
+ private specPartial;
1458
+ private specTimer;
1459
+ private specText;
1460
+ private specSpent;
1461
+ private bcPartial;
1462
+ private bcTimer;
1463
+ private bcCount;
1464
+ private bcActive;
1465
+ private bcActiveTimer;
1466
+ private lastBcAt;
1467
+ private lastBcPhrase;
1468
+ private recentBc;
1340
1469
  private uttQueue;
1341
1470
  private emo;
1342
1471
  constructor(options?: Partial<VoiceEngineOptions>);
@@ -1346,6 +1475,10 @@ declare class VoiceEngine {
1346
1475
  setBargeIn(on: boolean): void;
1347
1476
  /** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
1348
1477
  setShowEmotions(on: boolean): void;
1478
+ /** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
1479
+ * it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
1480
+ private diagOn;
1481
+ protected diag(kind: string, fields?: Record<string, unknown>): void;
1349
1482
  private idleWaiters;
1350
1483
  private setState;
1351
1484
  /** Resolve when the engine is no longer speaking (immediate if already idle). */
@@ -1374,6 +1507,20 @@ declare class VoiceEngine {
1374
1507
  /** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
1375
1508
  * next one from the settle completion (endSpeech), so utterances serialize without overlap. */
1376
1509
  private pumpQueue;
1510
+ /** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
1511
+ * utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
1512
+ * of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
1513
+ * many times"). All phrases are sub-second and semantically safe for their shape.
1514
+ * No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
1515
+ static readonly ACKS_NEUTRAL: string[];
1516
+ static readonly ACKS_QUESTION: string[];
1517
+ private recentAcks;
1518
+ private pickAck;
1519
+ private lastDispatchWasQuestion;
1520
+ private clearAckTimer;
1521
+ /** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
1522
+ * to be a Hold (intentionally silent; an ack would read as the start of an answer). */
1523
+ cancelPendingAck(): void;
1377
1524
  /** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
1378
1525
  interrupt(): void;
1379
1526
  stop(): void;
@@ -1387,12 +1534,67 @@ declare class VoiceEngine {
1387
1534
  * longer ones on count. */
1388
1535
  private genuine;
1389
1536
  private handlePartial;
1537
+ /** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
1538
+ * partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
1539
+ * onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
1540
+ * tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
1541
+ * dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
1542
+ * trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
1543
+ * utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
1544
+ private trackSpeculation;
1545
+ /** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
1546
+ * imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
1547
+ * every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
1548
+ * strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
1549
+ * 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
1550
+ * right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
1551
+ static readonly BACKCHANNELS: string[];
1552
+ /** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
1553
+ * that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
1554
+ * BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
1555
+ * Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
1556
+ * Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
1557
+ private trackBackchannel;
1558
+ /** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
1559
+ * markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
1560
+ * pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
1561
+ * never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
1562
+ * swallowed), and the echo window extends so echo-shaped finals stay gated. */
1563
+ private fireBackchannel;
1564
+ /** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
1565
+ private bcSupersede;
1566
+ /** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
1567
+ * phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
1568
+ * within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
1569
+ private stripBackchannelEcho;
1390
1570
  /** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
1391
1571
  * overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
1392
1572
  * utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
1393
1573
  * a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
1394
1574
  * check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
1395
1575
  private mergeUtterance;
1576
+ private static normWord;
1577
+ /** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
1578
+ * the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
1579
+ * may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
1580
+ * DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
1581
+ * "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
1582
+ * question already being answered).
1583
+ * DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
1584
+ * (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
1585
+ * previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
1586
+ * repliedSinceDispatch then picks the action:
1587
+ * • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
1588
+ * "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
1589
+ * worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
1590
+ * DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
1591
+ * Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
1592
+ * • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
1593
+ * truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
1594
+ * starves the conversation entirely — the fuller final IS the completion the Hold is waiting
1595
+ * for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
1596
+ * race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
1597
+ private refinalizes;
1396
1598
  private static readonly TRAIL_RE;
1397
1599
  /** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
1398
1600
  private looksIncomplete;
@@ -1430,6 +1632,15 @@ declare class SonioxSTT {
1430
1632
  /** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
1431
1633
  * loop). The host tears voice down instead of spinning forever. */
1432
1634
  onFatal: (message: string) => void;
1635
+ /** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
1636
+ * handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
1637
+ onDiag: (ev: {
1638
+ t: number;
1639
+ kind: string;
1640
+ [k: string]: unknown;
1641
+ }) => void;
1642
+ private diagOn;
1643
+ private diag;
1433
1644
  private lastChunkAt;
1434
1645
  private startedChunksAt;
1435
1646
  private noAudioTimer;
@@ -1462,6 +1673,15 @@ declare class CartesiaTTS {
1462
1673
  ctxId: string;
1463
1674
  onAudio: (chunk: Uint8Array) => void;
1464
1675
  onDone: () => void;
1676
+ /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
1677
+ * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
1678
+ onDiag: (ev: {
1679
+ t: number;
1680
+ kind: string;
1681
+ [k: string]: unknown;
1682
+ }) => void;
1683
+ private diagOn;
1684
+ private diag;
1465
1685
  firstAudioAt: number;
1466
1686
  /** Circuit breaker: consecutive error count + down flag. */
1467
1687
  private consecutiveErrors;
@@ -1482,6 +1702,11 @@ declare class CartesiaTTS {
1482
1702
  private markRecovered;
1483
1703
  /** Ensure the WS is open before sending — reconnects if idle-closed. */
1484
1704
  private ensureConnected;
1705
+ /** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
1706
+ * a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
1707
+ * The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
1708
+ * newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
1709
+ warmup(): void;
1485
1710
  newContext(): string;
1486
1711
  private frame;
1487
1712
  speak(text: string, cont: boolean): void;