@livx.cc/agentx 0.99.5 → 0.99.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +2368 -1566
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +226 -1
- package/dist/index.js +2160 -1513
- package/dist/index.js.map +1 -1
- package/package.json +3 -1
package/dist/index.d.ts
CHANGED
|
@@ -974,8 +974,10 @@ declare class DuplexAgent {
|
|
|
974
974
|
* left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
|
|
975
975
|
* Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
|
|
976
976
|
private turnDispatched;
|
|
977
|
+
private spokeBeforeDispatch;
|
|
977
978
|
private turnBriefs;
|
|
978
979
|
private spokeThisTurn;
|
|
980
|
+
private externalSpeech;
|
|
979
981
|
private heldThisTurn;
|
|
980
982
|
private nudging;
|
|
981
983
|
private reflexBuf;
|
|
@@ -1001,6 +1003,18 @@ declare class DuplexAgent {
|
|
|
1001
1003
|
question: string;
|
|
1002
1004
|
resolve: (answer: string) => void;
|
|
1003
1005
|
}>;
|
|
1006
|
+
/** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
|
|
1007
|
+
* PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
|
|
1008
|
+
* confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
|
|
1009
|
+
* aborts it: output dropped, history rolled back, the final dispatches normally. */
|
|
1010
|
+
private spec?;
|
|
1011
|
+
/** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
|
|
1012
|
+
speculativeAbortedCalls: number;
|
|
1013
|
+
/** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
|
|
1014
|
+
* intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
|
|
1015
|
+
* writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
|
|
1016
|
+
* instead (the endpointed final then dispatches normally and may use the tool for real). */
|
|
1017
|
+
private static readonly SPEC_SAFE_TOOLS;
|
|
1004
1018
|
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
1005
1019
|
private memoryReady;
|
|
1006
1020
|
constructor(options?: Partial<DuplexAgentOptions>);
|
|
@@ -1009,6 +1023,10 @@ declare class DuplexAgent {
|
|
|
1009
1023
|
/** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
|
|
1010
1024
|
* turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
|
|
1011
1025
|
private flushHeldReflexTail;
|
|
1026
|
+
/** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
|
|
1027
|
+
* Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
|
|
1028
|
+
* reflexForwarded would corrupt the forward offset. */
|
|
1029
|
+
private scrubStageDirections;
|
|
1012
1030
|
/** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
|
|
1013
1031
|
private resetTurn;
|
|
1014
1032
|
/** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
|
|
@@ -1016,6 +1034,11 @@ declare class DuplexAgent {
|
|
|
1016
1034
|
* re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
|
|
1017
1035
|
* still allowed (parallel independent work). During a re-ack pass, block every tool. */
|
|
1018
1036
|
private dispatchGuard;
|
|
1037
|
+
/** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
|
|
1038
|
+
* micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
|
|
1039
|
+
* inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
|
|
1040
|
+
* still dead air, so silentTurn ignores external speech unless work was dispatched. */
|
|
1041
|
+
noteExternalSpeech(): void;
|
|
1019
1042
|
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
1020
1043
|
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
1021
1044
|
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
@@ -1026,8 +1049,27 @@ declare class DuplexAgent {
|
|
|
1026
1049
|
* line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
|
|
1027
1050
|
* Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
|
|
1028
1051
|
private ackIfSilent;
|
|
1029
|
-
/**
|
|
1052
|
+
/** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
|
|
1053
|
+
* hosts/tests key on it. */
|
|
1054
|
+
private static readonly FALLBACK_ACKS;
|
|
1055
|
+
private static readonly FALLBACK_RETRY;
|
|
1056
|
+
private lastFallback;
|
|
1057
|
+
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
|
|
1058
|
+
* If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
|
|
1059
|
+
* speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
|
|
1060
|
+
* host NOW (this is the latency win) and streaming continues live. Any other content aborts the
|
|
1061
|
+
* speculation first (rolled back silently) and runs a normal turn behind it. */
|
|
1030
1062
|
send(content: MessageContent): Promise<RunResult>;
|
|
1063
|
+
/** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
|
|
1064
|
+
* output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
|
|
1065
|
+
* voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
|
|
1066
|
+
* into the transcript between the speculative messages and their rollback. No-op if a speculation
|
|
1067
|
+
* is already in flight. */
|
|
1068
|
+
speculate(text: string): void;
|
|
1069
|
+
/** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
|
|
1070
|
+
* Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
|
|
1071
|
+
* attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
|
|
1072
|
+
abortSpeculation(): void;
|
|
1031
1073
|
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
1032
1074
|
cancelTask(id: string): string;
|
|
1033
1075
|
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
@@ -1040,6 +1082,11 @@ declare class DuplexAgent {
|
|
|
1040
1082
|
/** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
|
|
1041
1083
|
private enqueue;
|
|
1042
1084
|
private notify;
|
|
1085
|
+
/** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
|
|
1086
|
+
* hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
|
|
1087
|
+
* order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
|
|
1088
|
+
* which bypasses this via host.notify directly) passes through untouched. */
|
|
1089
|
+
private emitHost;
|
|
1043
1090
|
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
1044
1091
|
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
1045
1092
|
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
@@ -1221,6 +1268,22 @@ interface AudioSink {
|
|
|
1221
1268
|
type AuthProvider = string | (() => string | Promise<string>);
|
|
1222
1269
|
declare function resolveAuth(auth: AuthProvider): Promise<string>;
|
|
1223
1270
|
|
|
1271
|
+
/** Injectable time source — ALL engine timing (merge windows, barge grace, overlap resume, drain
|
|
1272
|
+
* settles) routes through it. Defaults to real timers; the deterministic bench swaps a virtual
|
|
1273
|
+
* clock so timing scenarios run instantly and exactly. */
|
|
1274
|
+
interface EngineClock {
|
|
1275
|
+
now(): number;
|
|
1276
|
+
setTimeout(fn: () => void, ms: number): unknown;
|
|
1277
|
+
clearTimeout(handle: unknown): void;
|
|
1278
|
+
}
|
|
1279
|
+
/** One structured diagnostic event (fire-and-forget, never affects behavior). `t` = engine clock ms.
|
|
1280
|
+
* Emitted at every DECISION point; kinds + fields are the per-session forensic record
|
|
1281
|
+
* (CLI: `<session>.voice.jsonl`; lab: `<session>.timeline.jsonl`). See mind/11-voice.md §Diagnostics. */
|
|
1282
|
+
interface DiagEvent {
|
|
1283
|
+
t: number;
|
|
1284
|
+
kind: string;
|
|
1285
|
+
[k: string]: unknown;
|
|
1286
|
+
}
|
|
1224
1287
|
/** Structural contracts (satisfied by SonioxSTT/CartesiaTTS or test fakes). */
|
|
1225
1288
|
interface SttLike {
|
|
1226
1289
|
usingAec: boolean;
|
|
@@ -1237,6 +1300,9 @@ interface TtsLike {
|
|
|
1237
1300
|
onAudio: (chunk: Uint8Array) => void;
|
|
1238
1301
|
onDone: () => void;
|
|
1239
1302
|
connect(): Promise<void> | void;
|
|
1303
|
+
/** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
|
|
1304
|
+
* so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
|
|
1305
|
+
warmup?(): void;
|
|
1240
1306
|
newContext(): string;
|
|
1241
1307
|
speak(text: string, cont: boolean): void;
|
|
1242
1308
|
end(): void;
|
|
@@ -1258,6 +1324,12 @@ declare class VoiceEngineOptions {
|
|
|
1258
1324
|
onBargeIn: (phase: 'speaking' | 'drain') => void;
|
|
1259
1325
|
/** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
|
|
1260
1326
|
ackPhrase: string;
|
|
1327
|
+
/** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
|
|
1328
|
+
* delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
|
|
1329
|
+
* per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
|
|
1330
|
+
adaptiveAckMs: number;
|
|
1331
|
+
/** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
|
|
1332
|
+
onAdaptiveAck: () => void;
|
|
1261
1333
|
/** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
|
|
1262
1334
|
* letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
|
|
1263
1335
|
* ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
|
|
@@ -1303,11 +1375,46 @@ declare class VoiceEngineOptions {
|
|
|
1303
1375
|
* speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
|
|
1304
1376
|
* mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
|
|
1305
1377
|
overlapEnergyHold: boolean;
|
|
1378
|
+
/** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
|
|
1379
|
+
* this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
|
|
1380
|
+
* start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
|
|
1381
|
+
* speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
|
|
1382
|
+
* confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
|
|
1383
|
+
speculativeMs: number;
|
|
1384
|
+
/** Minimum word count for a partial to qualify as a speculation trigger. */
|
|
1385
|
+
speculativeMinWords: number;
|
|
1386
|
+
/** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
|
|
1387
|
+
onSpeculate: (text: string) => void;
|
|
1388
|
+
/** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
|
|
1389
|
+
* partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
|
|
1390
|
+
* UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
|
|
1391
|
+
* quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
|
|
1392
|
+
* touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
|
|
1393
|
+
* context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
|
|
1394
|
+
* boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
|
|
1395
|
+
* 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
|
|
1396
|
+
* ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
|
|
1397
|
+
backchannelMs: number;
|
|
1398
|
+
/** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
|
|
1399
|
+
backchannelMinGapMs: number;
|
|
1400
|
+
/** Only multi-clause turns: the partial must carry at least this many words before a blip. */
|
|
1401
|
+
backchannelMinWords: number;
|
|
1402
|
+
/** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
|
|
1403
|
+
onBackchannel: (phrase: string) => void;
|
|
1404
|
+
/** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
|
|
1405
|
+
* quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
|
|
1406
|
+
onSpeculateAbort: () => void;
|
|
1306
1407
|
/** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
|
|
1307
1408
|
* tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
|
|
1308
1409
|
emotions: boolean;
|
|
1309
1410
|
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
1310
1411
|
showEmotions: boolean;
|
|
1412
|
+
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
1413
|
+
clock: EngineClock;
|
|
1414
|
+
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
1415
|
+
* barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
|
|
1416
|
+
* a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
|
|
1417
|
+
onDiag: (ev: DiagEvent) => void;
|
|
1311
1418
|
}
|
|
1312
1419
|
declare class VoiceEngine {
|
|
1313
1420
|
options: VoiceEngineOptions;
|
|
@@ -1319,6 +1426,7 @@ declare class VoiceEngine {
|
|
|
1319
1426
|
private ctxOpen;
|
|
1320
1427
|
private interrupted;
|
|
1321
1428
|
private spokeDeltas;
|
|
1429
|
+
private clock;
|
|
1322
1430
|
private drainTimer;
|
|
1323
1431
|
private echoWords;
|
|
1324
1432
|
private prevReply;
|
|
@@ -1328,15 +1436,36 @@ declare class VoiceEngine {
|
|
|
1328
1436
|
private hot;
|
|
1329
1437
|
private suspectUntil;
|
|
1330
1438
|
private ackAt;
|
|
1439
|
+
private lastAck;
|
|
1440
|
+
private ackTimer;
|
|
1331
1441
|
private bargeGraceUntil;
|
|
1332
1442
|
private pendingUtt;
|
|
1443
|
+
private mergePath;
|
|
1444
|
+
private lastGraceDiag;
|
|
1333
1445
|
private pendingTimer;
|
|
1446
|
+
private lastDispatchFlat;
|
|
1447
|
+
private lastDispatchWords;
|
|
1448
|
+
private lastDispatchAt;
|
|
1449
|
+
private repliedSinceDispatch;
|
|
1450
|
+
private static readonly DUP_FINAL_MS;
|
|
1334
1451
|
private lastInterrupted;
|
|
1335
1452
|
private pausedAt;
|
|
1336
1453
|
private lastResumeAt;
|
|
1337
1454
|
private lastOverlapPartial;
|
|
1338
1455
|
private resumeTimer;
|
|
1339
1456
|
private turnStartAt;
|
|
1457
|
+
private specPartial;
|
|
1458
|
+
private specTimer;
|
|
1459
|
+
private specText;
|
|
1460
|
+
private specSpent;
|
|
1461
|
+
private bcPartial;
|
|
1462
|
+
private bcTimer;
|
|
1463
|
+
private bcCount;
|
|
1464
|
+
private bcActive;
|
|
1465
|
+
private bcActiveTimer;
|
|
1466
|
+
private lastBcAt;
|
|
1467
|
+
private lastBcPhrase;
|
|
1468
|
+
private recentBc;
|
|
1340
1469
|
private uttQueue;
|
|
1341
1470
|
private emo;
|
|
1342
1471
|
constructor(options?: Partial<VoiceEngineOptions>);
|
|
@@ -1346,6 +1475,10 @@ declare class VoiceEngine {
|
|
|
1346
1475
|
setBargeIn(on: boolean): void;
|
|
1347
1476
|
/** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
|
|
1348
1477
|
setShowEmotions(on: boolean): void;
|
|
1478
|
+
/** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
|
|
1479
|
+
* it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
|
|
1480
|
+
private diagOn;
|
|
1481
|
+
protected diag(kind: string, fields?: Record<string, unknown>): void;
|
|
1349
1482
|
private idleWaiters;
|
|
1350
1483
|
private setState;
|
|
1351
1484
|
/** Resolve when the engine is no longer speaking (immediate if already idle). */
|
|
@@ -1374,6 +1507,20 @@ declare class VoiceEngine {
|
|
|
1374
1507
|
/** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
|
|
1375
1508
|
* next one from the settle completion (endSpeech), so utterances serialize without overlap. */
|
|
1376
1509
|
private pumpQueue;
|
|
1510
|
+
/** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
|
|
1511
|
+
* utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
|
|
1512
|
+
* of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
|
|
1513
|
+
* many times"). All phrases are sub-second and semantically safe for their shape.
|
|
1514
|
+
* No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
|
|
1515
|
+
static readonly ACKS_NEUTRAL: string[];
|
|
1516
|
+
static readonly ACKS_QUESTION: string[];
|
|
1517
|
+
private recentAcks;
|
|
1518
|
+
private pickAck;
|
|
1519
|
+
private lastDispatchWasQuestion;
|
|
1520
|
+
private clearAckTimer;
|
|
1521
|
+
/** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
|
|
1522
|
+
* to be a Hold (intentionally silent; an ack would read as the start of an answer). */
|
|
1523
|
+
cancelPendingAck(): void;
|
|
1377
1524
|
/** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
|
|
1378
1525
|
interrupt(): void;
|
|
1379
1526
|
stop(): void;
|
|
@@ -1387,12 +1534,67 @@ declare class VoiceEngine {
|
|
|
1387
1534
|
* longer ones on count. */
|
|
1388
1535
|
private genuine;
|
|
1389
1536
|
private handlePartial;
|
|
1537
|
+
/** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
|
|
1538
|
+
* partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
|
|
1539
|
+
* onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
|
|
1540
|
+
* tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
|
|
1541
|
+
* dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
|
|
1542
|
+
* trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
|
|
1543
|
+
* utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
|
|
1544
|
+
private trackSpeculation;
|
|
1545
|
+
/** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
|
|
1546
|
+
* imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
|
|
1547
|
+
* every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
|
|
1548
|
+
* strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
|
|
1549
|
+
* 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
|
|
1550
|
+
* right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
|
|
1551
|
+
static readonly BACKCHANNELS: string[];
|
|
1552
|
+
/** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
|
|
1553
|
+
* that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
|
|
1554
|
+
* BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
|
|
1555
|
+
* Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
|
|
1556
|
+
* Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
|
|
1557
|
+
private trackBackchannel;
|
|
1558
|
+
/** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
|
|
1559
|
+
* markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
|
|
1560
|
+
* pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
|
|
1561
|
+
* never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
|
|
1562
|
+
* swallowed), and the echo window extends so echo-shaped finals stay gated. */
|
|
1563
|
+
private fireBackchannel;
|
|
1564
|
+
/** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
|
|
1565
|
+
private bcSupersede;
|
|
1566
|
+
/** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
|
|
1567
|
+
* phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
|
|
1568
|
+
* within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
|
|
1569
|
+
private stripBackchannelEcho;
|
|
1390
1570
|
/** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
|
|
1391
1571
|
* overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
|
|
1392
1572
|
* utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
|
|
1393
1573
|
* a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
|
|
1394
1574
|
* check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
|
|
1395
1575
|
private mergeUtterance;
|
|
1576
|
+
private static normWord;
|
|
1577
|
+
/** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
|
|
1578
|
+
* the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
|
|
1579
|
+
* may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
|
|
1580
|
+
* DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
|
|
1581
|
+
* "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
|
|
1582
|
+
* question already being answered).
|
|
1583
|
+
* DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
|
|
1584
|
+
* (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
|
|
1585
|
+
* previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
|
|
1586
|
+
* repliedSinceDispatch then picks the action:
|
|
1587
|
+
* • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
|
|
1588
|
+
* "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
|
|
1589
|
+
* worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
|
|
1590
|
+
* DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
|
|
1591
|
+
* Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
|
|
1592
|
+
* • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
|
|
1593
|
+
* truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
|
|
1594
|
+
* starves the conversation entirely — the fuller final IS the completion the Hold is waiting
|
|
1595
|
+
* for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
|
|
1596
|
+
* race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
|
|
1597
|
+
private refinalizes;
|
|
1396
1598
|
private static readonly TRAIL_RE;
|
|
1397
1599
|
/** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
|
|
1398
1600
|
private looksIncomplete;
|
|
@@ -1430,6 +1632,15 @@ declare class SonioxSTT {
|
|
|
1430
1632
|
/** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
|
|
1431
1633
|
* loop). The host tears voice down instead of spinning forever. */
|
|
1432
1634
|
onFatal: (message: string) => void;
|
|
1635
|
+
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
1636
|
+
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
1637
|
+
onDiag: (ev: {
|
|
1638
|
+
t: number;
|
|
1639
|
+
kind: string;
|
|
1640
|
+
[k: string]: unknown;
|
|
1641
|
+
}) => void;
|
|
1642
|
+
private diagOn;
|
|
1643
|
+
private diag;
|
|
1433
1644
|
private lastChunkAt;
|
|
1434
1645
|
private startedChunksAt;
|
|
1435
1646
|
private noAudioTimer;
|
|
@@ -1462,6 +1673,15 @@ declare class CartesiaTTS {
|
|
|
1462
1673
|
ctxId: string;
|
|
1463
1674
|
onAudio: (chunk: Uint8Array) => void;
|
|
1464
1675
|
onDone: () => void;
|
|
1676
|
+
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
1677
|
+
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
1678
|
+
onDiag: (ev: {
|
|
1679
|
+
t: number;
|
|
1680
|
+
kind: string;
|
|
1681
|
+
[k: string]: unknown;
|
|
1682
|
+
}) => void;
|
|
1683
|
+
private diagOn;
|
|
1684
|
+
private diag;
|
|
1465
1685
|
firstAudioAt: number;
|
|
1466
1686
|
/** Circuit breaker: consecutive error count + down flag. */
|
|
1467
1687
|
private consecutiveErrors;
|
|
@@ -1482,6 +1702,11 @@ declare class CartesiaTTS {
|
|
|
1482
1702
|
private markRecovered;
|
|
1483
1703
|
/** Ensure the WS is open before sending — reconnects if idle-closed. */
|
|
1484
1704
|
private ensureConnected;
|
|
1705
|
+
/** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
|
|
1706
|
+
* a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
|
|
1707
|
+
* The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
|
|
1708
|
+
* newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
|
|
1709
|
+
warmup(): void;
|
|
1485
1710
|
newContext(): string;
|
|
1486
1711
|
private frame;
|
|
1487
1712
|
speak(text: string, cont: boolean): void;
|