@livx.cc/agentx 0.99.6 → 0.99.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +2409 -1559
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +261 -1
- package/dist/index.js +2242 -1514
- package/dist/index.js.map +1 -1
- package/package.json +3 -1
package/dist/index.d.ts
CHANGED
|
@@ -974,8 +974,10 @@ declare class DuplexAgent {
|
|
|
974
974
|
* left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
|
|
975
975
|
* Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
|
|
976
976
|
private turnDispatched;
|
|
977
|
+
private spokeBeforeDispatch;
|
|
977
978
|
private turnBriefs;
|
|
978
979
|
private spokeThisTurn;
|
|
980
|
+
private externalSpeech;
|
|
979
981
|
private heldThisTurn;
|
|
980
982
|
private nudging;
|
|
981
983
|
private reflexBuf;
|
|
@@ -1001,6 +1003,18 @@ declare class DuplexAgent {
|
|
|
1001
1003
|
question: string;
|
|
1002
1004
|
resolve: (answer: string) => void;
|
|
1003
1005
|
}>;
|
|
1006
|
+
/** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
|
|
1007
|
+
* PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
|
|
1008
|
+
* confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
|
|
1009
|
+
* aborts it: output dropped, history rolled back, the final dispatches normally. */
|
|
1010
|
+
private spec?;
|
|
1011
|
+
/** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
|
|
1012
|
+
speculativeAbortedCalls: number;
|
|
1013
|
+
/** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
|
|
1014
|
+
* intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
|
|
1015
|
+
* writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
|
|
1016
|
+
* instead (the endpointed final then dispatches normally and may use the tool for real). */
|
|
1017
|
+
private static readonly SPEC_SAFE_TOOLS;
|
|
1004
1018
|
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
1005
1019
|
private memoryReady;
|
|
1006
1020
|
constructor(options?: Partial<DuplexAgentOptions>);
|
|
@@ -1009,6 +1023,10 @@ declare class DuplexAgent {
|
|
|
1009
1023
|
/** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
|
|
1010
1024
|
* turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
|
|
1011
1025
|
private flushHeldReflexTail;
|
|
1026
|
+
/** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
|
|
1027
|
+
* Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
|
|
1028
|
+
* reflexForwarded would corrupt the forward offset. */
|
|
1029
|
+
private scrubStageDirections;
|
|
1012
1030
|
/** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
|
|
1013
1031
|
private resetTurn;
|
|
1014
1032
|
/** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
|
|
@@ -1016,6 +1034,11 @@ declare class DuplexAgent {
|
|
|
1016
1034
|
* re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
|
|
1017
1035
|
* still allowed (parallel independent work). During a re-ack pass, block every tool. */
|
|
1018
1036
|
private dispatchGuard;
|
|
1037
|
+
/** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
|
|
1038
|
+
* micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
|
|
1039
|
+
* inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
|
|
1040
|
+
* still dead air, so silentTurn ignores external speech unless work was dispatched. */
|
|
1041
|
+
noteExternalSpeech(): void;
|
|
1019
1042
|
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
1020
1043
|
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
1021
1044
|
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
@@ -1026,8 +1049,27 @@ declare class DuplexAgent {
|
|
|
1026
1049
|
* line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
|
|
1027
1050
|
* Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
|
|
1028
1051
|
private ackIfSilent;
|
|
1029
|
-
/**
|
|
1052
|
+
/** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
|
|
1053
|
+
* hosts/tests key on it. */
|
|
1054
|
+
private static readonly FALLBACK_ACKS;
|
|
1055
|
+
private static readonly FALLBACK_RETRY;
|
|
1056
|
+
private lastFallback;
|
|
1057
|
+
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
|
|
1058
|
+
* If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
|
|
1059
|
+
* speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
|
|
1060
|
+
* host NOW (this is the latency win) and streaming continues live. Any other content aborts the
|
|
1061
|
+
* speculation first (rolled back silently) and runs a normal turn behind it. */
|
|
1030
1062
|
send(content: MessageContent): Promise<RunResult>;
|
|
1063
|
+
/** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
|
|
1064
|
+
* output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
|
|
1065
|
+
* voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
|
|
1066
|
+
* into the transcript between the speculative messages and their rollback. No-op if a speculation
|
|
1067
|
+
* is already in flight. */
|
|
1068
|
+
speculate(text: string): void;
|
|
1069
|
+
/** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
|
|
1070
|
+
* Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
|
|
1071
|
+
* attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
|
|
1072
|
+
abortSpeculation(): void;
|
|
1031
1073
|
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
1032
1074
|
cancelTask(id: string): string;
|
|
1033
1075
|
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
@@ -1040,6 +1082,11 @@ declare class DuplexAgent {
|
|
|
1040
1082
|
/** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
|
|
1041
1083
|
private enqueue;
|
|
1042
1084
|
private notify;
|
|
1085
|
+
/** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
|
|
1086
|
+
* hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
|
|
1087
|
+
* order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
|
|
1088
|
+
* which bypasses this via host.notify directly) passes through untouched. */
|
|
1089
|
+
private emitHost;
|
|
1043
1090
|
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
1044
1091
|
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
1045
1092
|
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
@@ -1221,6 +1268,22 @@ interface AudioSink {
|
|
|
1221
1268
|
type AuthProvider = string | (() => string | Promise<string>);
|
|
1222
1269
|
declare function resolveAuth(auth: AuthProvider): Promise<string>;
|
|
1223
1270
|
|
|
1271
|
+
/** Injectable time source — ALL engine timing (merge windows, barge grace, overlap resume, drain
|
|
1272
|
+
* settles) routes through it. Defaults to real timers; the deterministic bench swaps a virtual
|
|
1273
|
+
* clock so timing scenarios run instantly and exactly. */
|
|
1274
|
+
interface EngineClock {
|
|
1275
|
+
now(): number;
|
|
1276
|
+
setTimeout(fn: () => void, ms: number): unknown;
|
|
1277
|
+
clearTimeout(handle: unknown): void;
|
|
1278
|
+
}
|
|
1279
|
+
/** One structured diagnostic event (fire-and-forget, never affects behavior). `t` = engine clock ms.
|
|
1280
|
+
* Emitted at every DECISION point; kinds + fields are the per-session forensic record
|
|
1281
|
+
* (CLI: `<session>.voice.jsonl`; lab: `<session>.timeline.jsonl`). See mind/11-voice.md §Diagnostics. */
|
|
1282
|
+
interface DiagEvent {
|
|
1283
|
+
t: number;
|
|
1284
|
+
kind: string;
|
|
1285
|
+
[k: string]: unknown;
|
|
1286
|
+
}
|
|
1224
1287
|
/** Structural contracts (satisfied by SonioxSTT/CartesiaTTS or test fakes). */
|
|
1225
1288
|
interface SttLike {
|
|
1226
1289
|
usingAec: boolean;
|
|
@@ -1236,7 +1299,14 @@ interface SttLike {
|
|
|
1236
1299
|
interface TtsLike {
|
|
1237
1300
|
onAudio: (chunk: Uint8Array) => void;
|
|
1238
1301
|
onDone: () => void;
|
|
1302
|
+
/** Optional word-timestamp seam for karaoke reveal (revealMode==='word'). `start[i]` = seconds
|
|
1303
|
+
* from turn-audio start. Set `wantTimestamps` to request them from the provider. */
|
|
1304
|
+
onTimestamps?: (words: string[], start: number[]) => void;
|
|
1305
|
+
wantTimestamps?: boolean;
|
|
1239
1306
|
connect(): Promise<void> | void;
|
|
1307
|
+
/** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
|
|
1308
|
+
* so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
|
|
1309
|
+
warmup?(): void;
|
|
1240
1310
|
newContext(): string;
|
|
1241
1311
|
speak(text: string, cont: boolean): void;
|
|
1242
1312
|
end(): void;
|
|
@@ -1258,6 +1328,12 @@ declare class VoiceEngineOptions {
|
|
|
1258
1328
|
onBargeIn: (phase: 'speaking' | 'drain') => void;
|
|
1259
1329
|
/** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
|
|
1260
1330
|
ackPhrase: string;
|
|
1331
|
+
/** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
|
|
1332
|
+
* delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
|
|
1333
|
+
* per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
|
|
1334
|
+
adaptiveAckMs: number;
|
|
1335
|
+
/** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
|
|
1336
|
+
onAdaptiveAck: () => void;
|
|
1261
1337
|
/** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
|
|
1262
1338
|
* letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
|
|
1263
1339
|
* ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
|
|
@@ -1303,11 +1379,58 @@ declare class VoiceEngineOptions {
|
|
|
1303
1379
|
* speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
|
|
1304
1380
|
* mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
|
|
1305
1381
|
overlapEnergyHold: boolean;
|
|
1382
|
+
/** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
|
|
1383
|
+
* this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
|
|
1384
|
+
* start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
|
|
1385
|
+
* speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
|
|
1386
|
+
* confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
|
|
1387
|
+
speculativeMs: number;
|
|
1388
|
+
/** Minimum word count for a partial to qualify as a speculation trigger. */
|
|
1389
|
+
speculativeMinWords: number;
|
|
1390
|
+
/** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
|
|
1391
|
+
onSpeculate: (text: string) => void;
|
|
1392
|
+
/** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
|
|
1393
|
+
* partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
|
|
1394
|
+
* UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
|
|
1395
|
+
* quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
|
|
1396
|
+
* touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
|
|
1397
|
+
* context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
|
|
1398
|
+
* boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
|
|
1399
|
+
* 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
|
|
1400
|
+
* ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
|
|
1401
|
+
backchannelMs: number;
|
|
1402
|
+
/** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
|
|
1403
|
+
backchannelMinGapMs: number;
|
|
1404
|
+
/** Only multi-clause turns: the partial must carry at least this many words before a blip. */
|
|
1405
|
+
backchannelMinWords: number;
|
|
1406
|
+
/** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
|
|
1407
|
+
onBackchannel: (phrase: string) => void;
|
|
1408
|
+
/** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
|
|
1409
|
+
* quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
|
|
1410
|
+
onSpeculateAbort: () => void;
|
|
1306
1411
|
/** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
|
|
1307
1412
|
* tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
|
|
1308
1413
|
emotions: boolean;
|
|
1309
1414
|
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
1310
1415
|
showEmotions: boolean;
|
|
1416
|
+
/**
|
|
1417
|
+
* Progressive text reveal — the "karaoke" capability, opt-in.
|
|
1418
|
+
* 'off' — no reveal events (CLI default; the host renders text however it likes).
|
|
1419
|
+
* 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
|
|
1420
|
+
* in step with the model stream).
|
|
1421
|
+
* 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
|
|
1422
|
+
* timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
|
|
1423
|
+
*/
|
|
1424
|
+
revealMode: 'off' | 'delta' | 'word';
|
|
1425
|
+
/** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
|
|
1426
|
+
* Reset to '' at the start of each spoken turn. */
|
|
1427
|
+
onReveal: (revealed: string) => void;
|
|
1428
|
+
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
1429
|
+
clock: EngineClock;
|
|
1430
|
+
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
1431
|
+
* barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
|
|
1432
|
+
* a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
|
|
1433
|
+
onDiag: (ev: DiagEvent) => void;
|
|
1311
1434
|
}
|
|
1312
1435
|
declare class VoiceEngine {
|
|
1313
1436
|
options: VoiceEngineOptions;
|
|
@@ -1319,6 +1442,11 @@ declare class VoiceEngine {
|
|
|
1319
1442
|
private ctxOpen;
|
|
1320
1443
|
private interrupted;
|
|
1321
1444
|
private spokeDeltas;
|
|
1445
|
+
private revealText;
|
|
1446
|
+
private wordStarts;
|
|
1447
|
+
private revealedN;
|
|
1448
|
+
private revealPoll;
|
|
1449
|
+
private clock;
|
|
1322
1450
|
private drainTimer;
|
|
1323
1451
|
private echoWords;
|
|
1324
1452
|
private prevReply;
|
|
@@ -1328,15 +1456,36 @@ declare class VoiceEngine {
|
|
|
1328
1456
|
private hot;
|
|
1329
1457
|
private suspectUntil;
|
|
1330
1458
|
private ackAt;
|
|
1459
|
+
private lastAck;
|
|
1460
|
+
private ackTimer;
|
|
1331
1461
|
private bargeGraceUntil;
|
|
1332
1462
|
private pendingUtt;
|
|
1463
|
+
private mergePath;
|
|
1464
|
+
private lastGraceDiag;
|
|
1333
1465
|
private pendingTimer;
|
|
1466
|
+
private lastDispatchFlat;
|
|
1467
|
+
private lastDispatchWords;
|
|
1468
|
+
private lastDispatchAt;
|
|
1469
|
+
private repliedSinceDispatch;
|
|
1470
|
+
private static readonly DUP_FINAL_MS;
|
|
1334
1471
|
private lastInterrupted;
|
|
1335
1472
|
private pausedAt;
|
|
1336
1473
|
private lastResumeAt;
|
|
1337
1474
|
private lastOverlapPartial;
|
|
1338
1475
|
private resumeTimer;
|
|
1339
1476
|
private turnStartAt;
|
|
1477
|
+
private specPartial;
|
|
1478
|
+
private specTimer;
|
|
1479
|
+
private specText;
|
|
1480
|
+
private specSpent;
|
|
1481
|
+
private bcPartial;
|
|
1482
|
+
private bcTimer;
|
|
1483
|
+
private bcCount;
|
|
1484
|
+
private bcActive;
|
|
1485
|
+
private bcActiveTimer;
|
|
1486
|
+
private lastBcAt;
|
|
1487
|
+
private lastBcPhrase;
|
|
1488
|
+
private recentBc;
|
|
1340
1489
|
private uttQueue;
|
|
1341
1490
|
private emo;
|
|
1342
1491
|
constructor(options?: Partial<VoiceEngineOptions>);
|
|
@@ -1346,6 +1495,10 @@ declare class VoiceEngine {
|
|
|
1346
1495
|
setBargeIn(on: boolean): void;
|
|
1347
1496
|
/** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
|
|
1348
1497
|
setShowEmotions(on: boolean): void;
|
|
1498
|
+
/** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
|
|
1499
|
+
* it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
|
|
1500
|
+
private diagOn;
|
|
1501
|
+
protected diag(kind: string, fields?: Record<string, unknown>): void;
|
|
1349
1502
|
private idleWaiters;
|
|
1350
1503
|
private setState;
|
|
1351
1504
|
/** Resolve when the engine is no longer speaking (immediate if already idle). */
|
|
@@ -1357,6 +1510,15 @@ declare class VoiceEngine {
|
|
|
1357
1510
|
/** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
|
|
1358
1511
|
* host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
|
|
1359
1512
|
speakDelta(text: string): string;
|
|
1513
|
+
/** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
|
|
1514
|
+
* Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
|
|
1515
|
+
* paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
|
|
1516
|
+
* deterministic under the virtual clock. */
|
|
1517
|
+
private startWordReveal;
|
|
1518
|
+
/** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
|
|
1519
|
+
* undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
|
|
1520
|
+
* is what was actually spoken). */
|
|
1521
|
+
private stopWordReveal;
|
|
1360
1522
|
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
1361
1523
|
endSpeech(): void;
|
|
1362
1524
|
/** text of the reply cut by the last barge-in — consumed by the host to tell the model what
|
|
@@ -1374,6 +1536,20 @@ declare class VoiceEngine {
|
|
|
1374
1536
|
/** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
|
|
1375
1537
|
* next one from the settle completion (endSpeech), so utterances serialize without overlap. */
|
|
1376
1538
|
private pumpQueue;
|
|
1539
|
+
/** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
|
|
1540
|
+
* utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
|
|
1541
|
+
* of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
|
|
1542
|
+
* many times"). All phrases are sub-second and semantically safe for their shape.
|
|
1543
|
+
* No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
|
|
1544
|
+
static readonly ACKS_NEUTRAL: string[];
|
|
1545
|
+
static readonly ACKS_QUESTION: string[];
|
|
1546
|
+
private recentAcks;
|
|
1547
|
+
private pickAck;
|
|
1548
|
+
private lastDispatchWasQuestion;
|
|
1549
|
+
private clearAckTimer;
|
|
1550
|
+
/** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
|
|
1551
|
+
* to be a Hold (intentionally silent; an ack would read as the start of an answer). */
|
|
1552
|
+
cancelPendingAck(): void;
|
|
1377
1553
|
/** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
|
|
1378
1554
|
interrupt(): void;
|
|
1379
1555
|
stop(): void;
|
|
@@ -1387,12 +1563,67 @@ declare class VoiceEngine {
|
|
|
1387
1563
|
* longer ones on count. */
|
|
1388
1564
|
private genuine;
|
|
1389
1565
|
private handlePartial;
|
|
1566
|
+
/** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
|
|
1567
|
+
* partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
|
|
1568
|
+
* onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
|
|
1569
|
+
* tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
|
|
1570
|
+
* dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
|
|
1571
|
+
* trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
|
|
1572
|
+
* utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
|
|
1573
|
+
private trackSpeculation;
|
|
1574
|
+
/** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
|
|
1575
|
+
* imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
|
|
1576
|
+
* every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
|
|
1577
|
+
* strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
|
|
1578
|
+
* 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
|
|
1579
|
+
* right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
|
|
1580
|
+
static readonly BACKCHANNELS: string[];
|
|
1581
|
+
/** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
|
|
1582
|
+
* that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
|
|
1583
|
+
* BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
|
|
1584
|
+
* Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
|
|
1585
|
+
* Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
|
|
1586
|
+
private trackBackchannel;
|
|
1587
|
+
/** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
|
|
1588
|
+
* markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
|
|
1589
|
+
* pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
|
|
1590
|
+
* never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
|
|
1591
|
+
* swallowed), and the echo window extends so echo-shaped finals stay gated. */
|
|
1592
|
+
private fireBackchannel;
|
|
1593
|
+
/** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
|
|
1594
|
+
private bcSupersede;
|
|
1595
|
+
/** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
|
|
1596
|
+
* phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
|
|
1597
|
+
* within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
|
|
1598
|
+
private stripBackchannelEcho;
|
|
1390
1599
|
/** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
|
|
1391
1600
|
* overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
|
|
1392
1601
|
* utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
|
|
1393
1602
|
* a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
|
|
1394
1603
|
* check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
|
|
1395
1604
|
private mergeUtterance;
|
|
1605
|
+
private static normWord;
|
|
1606
|
+
/** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
|
|
1607
|
+
* the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
|
|
1608
|
+
* may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
|
|
1609
|
+
* DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
|
|
1610
|
+
* "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
|
|
1611
|
+
* question already being answered).
|
|
1612
|
+
* DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
|
|
1613
|
+
* (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
|
|
1614
|
+
* previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
|
|
1615
|
+
* repliedSinceDispatch then picks the action:
|
|
1616
|
+
* • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
|
|
1617
|
+
* "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
|
|
1618
|
+
* worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
|
|
1619
|
+
* DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
|
|
1620
|
+
* Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
|
|
1621
|
+
* • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
|
|
1622
|
+
* truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
|
|
1623
|
+
* starves the conversation entirely — the fuller final IS the completion the Hold is waiting
|
|
1624
|
+
* for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
|
|
1625
|
+
* race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
|
|
1626
|
+
private refinalizes;
|
|
1396
1627
|
private static readonly TRAIL_RE;
|
|
1397
1628
|
/** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
|
|
1398
1629
|
private looksIncomplete;
|
|
@@ -1430,6 +1661,15 @@ declare class SonioxSTT {
|
|
|
1430
1661
|
/** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
|
|
1431
1662
|
* loop). The host tears voice down instead of spinning forever. */
|
|
1432
1663
|
onFatal: (message: string) => void;
|
|
1664
|
+
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
1665
|
+
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
1666
|
+
onDiag: (ev: {
|
|
1667
|
+
t: number;
|
|
1668
|
+
kind: string;
|
|
1669
|
+
[k: string]: unknown;
|
|
1670
|
+
}) => void;
|
|
1671
|
+
private diagOn;
|
|
1672
|
+
private diag;
|
|
1433
1673
|
private lastChunkAt;
|
|
1434
1674
|
private startedChunksAt;
|
|
1435
1675
|
private noAudioTimer;
|
|
@@ -1462,6 +1702,21 @@ declare class CartesiaTTS {
|
|
|
1462
1702
|
ctxId: string;
|
|
1463
1703
|
onAudio: (chunk: Uint8Array) => void;
|
|
1464
1704
|
onDone: () => void;
|
|
1705
|
+
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
1706
|
+
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
1707
|
+
onTimestamps: (words: string[], start: number[]) => void;
|
|
1708
|
+
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
1709
|
+
* when revealMode==='word'. */
|
|
1710
|
+
wantTimestamps: boolean;
|
|
1711
|
+
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
1712
|
+
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
1713
|
+
onDiag: (ev: {
|
|
1714
|
+
t: number;
|
|
1715
|
+
kind: string;
|
|
1716
|
+
[k: string]: unknown;
|
|
1717
|
+
}) => void;
|
|
1718
|
+
private diagOn;
|
|
1719
|
+
private diag;
|
|
1465
1720
|
firstAudioAt: number;
|
|
1466
1721
|
/** Circuit breaker: consecutive error count + down flag. */
|
|
1467
1722
|
private consecutiveErrors;
|
|
@@ -1482,6 +1737,11 @@ declare class CartesiaTTS {
|
|
|
1482
1737
|
private markRecovered;
|
|
1483
1738
|
/** Ensure the WS is open before sending — reconnects if idle-closed. */
|
|
1484
1739
|
private ensureConnected;
|
|
1740
|
+
/** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
|
|
1741
|
+
* a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
|
|
1742
|
+
* The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
|
|
1743
|
+
* newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
|
|
1744
|
+
warmup(): void;
|
|
1485
1745
|
newContext(): string;
|
|
1486
1746
|
private frame;
|
|
1487
1747
|
speak(text: string, cont: boolean): void;
|