@livx.cc/agentx 0.99.14 → 0.99.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +1591 -1512
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +3 -549
- package/dist/index.js +1628 -1556
- package/dist/index.js.map +1 -1
- package/package.json +2 -1
package/dist/index.d.ts
CHANGED
|
@@ -6,6 +6,8 @@ export { CommandExecutor, FileMetadata, IFilesystem, IndexedDbFilesystem, MemFil
|
|
|
6
6
|
import { BodDB } from '@bod.ee/db';
|
|
7
7
|
import { A as AgentTool, C as ChatLike, a as ChatOptions, b as ChatResponse, h as ToolCall, H as HostBridge, U as UserQuestion, e as MessageContent } from './tools-DKf3hN4M.js';
|
|
8
8
|
export { c as ContentPart, d as HostEvent, M as Message, R as Role, S as SandboxJobRegistry, f as StreamChunk, T as TodoItem, g as Tool, i as ToolContext, j as bashTool, k as contentText, l as defaultTools, m as editTool, n as exitSessionTool, o as imagePart, p as makeContext, q as makeJobTools, r as readTool, t as toWireTools, s as todoWriteTool, u as toolRegistry, v as toolsByName } from './tools-DKf3hN4M.js';
|
|
9
|
+
import { SpokenSplitter } from '@bod.ee/voice';
|
|
10
|
+
export { AudioSink, AudioSource, AuthProvider, CartesiaTTS, CartesiaTTSOptions, STT_SAMPLE_RATE, SonioxSTT, SonioxSTTOptions, SttLike, TTS_SAMPLE_RATE, TtsLike, VoiceEngine, VoiceEngineOptions, VoiceState, resolveAuth } from '@bod.ee/voice';
|
|
9
11
|
export { M as McpCall, a as McpImage, b as McpRoute, c as McpRouteResolver, d as McpToolResult, e as McpToolSearchOptions, f as McpToolSpec, g as MountedMcpLike, h as buildMcpCatalog, m as makeLazyMcpToolSearch, i as makeMcpToolSearch, j as makeMcpToolSearchFromMounted, k as mcpToolToAgentTool, l as mcpToolsToAgentTools } from './mcp-DopVRDLx.js';
|
|
10
12
|
import * as libx_js_src_modules_log from 'libx.js/src/modules/log';
|
|
11
13
|
export { log } from 'libx.js/src/modules/log';
|
|
@@ -828,24 +830,6 @@ declare function reflectOnRun(o: ReflectOptions): Promise<string | null>;
|
|
|
828
830
|
*/
|
|
829
831
|
declare function loadInstructions(fs: IFilesystem, names?: string[]): Promise<string>;
|
|
830
832
|
|
|
831
|
-
declare class SpokenSplitter {
|
|
832
|
-
private buf;
|
|
833
|
-
private inSpoken;
|
|
834
|
-
/** True once any spoken char has ever been emitted (drives the no-spoken fallback). */
|
|
835
|
-
spokeAny: boolean;
|
|
836
|
-
/** Feed a delta; returns the spoken/detail spans completed by this chunk (either may be ''). */
|
|
837
|
-
feed(delta: string): {
|
|
838
|
-
spoken: string;
|
|
839
|
-
detail: string;
|
|
840
|
-
};
|
|
841
|
-
/** Drain any buffered partial. A trailing `<…` that never completed a tag is emitted as detail. */
|
|
842
|
-
flush(): {
|
|
843
|
-
spoken: string;
|
|
844
|
-
detail: string;
|
|
845
|
-
};
|
|
846
|
-
private drain;
|
|
847
|
-
}
|
|
848
|
-
|
|
849
833
|
/**
|
|
850
834
|
* DuplexAgent — voice-optimized three-tier conversational engine, composed on top of `Agent`.
|
|
851
835
|
*
|
|
@@ -1240,536 +1224,6 @@ declare function makeTaskBatchTool(opts: TaskToolOptions): AgentTool;
|
|
|
1240
1224
|
/** Component-scoped logger (libx.js). debug/verbose gated via DEBUG env/localStorage. */
|
|
1241
1225
|
declare const forComponent: (name: string) => libx_js_src_modules_log.ComponentLogger;
|
|
1242
1226
|
|
|
1243
|
-
/**
|
|
1244
|
-
* Portable voice I/O contracts — zero platform imports. The engine, STT and TTS clients in this
|
|
1245
|
-
* directory run anywhere with a `WebSocket` global (Bun, Node, browser); platform backends
|
|
1246
|
-
* (mic capture, audio playback) plug in through these seams.
|
|
1247
|
-
*/
|
|
1248
|
-
type VoiceState = 'idle' | 'listening' | 'thinking' | 'speaking';
|
|
1249
|
-
/** STT input sample format — sources MUST deliver this (downsample/convert in the adapter). */
|
|
1250
|
-
declare const STT_SAMPLE_RATE = 16000;
|
|
1251
|
-
/** Playback format the TTS requests and sinks must play. */
|
|
1252
|
-
declare const TTS_SAMPLE_RATE = 44100;
|
|
1253
|
-
/** A microphone (or any PCM) source. Emits s16le mono 16kHz chunks.
|
|
1254
|
-
* Node: mic-aec/ffmpeg child process. Web: getUserMedia({ echoCancellation: true }) + worklet. */
|
|
1255
|
-
interface AudioSource {
|
|
1256
|
-
/** `aec` = the source is echo-cancelled (assistant's own TTS never reaches the chunks) —
|
|
1257
|
-
* selects the engine's trivial barge-in path vs the heuristic echo tier. */
|
|
1258
|
-
readonly aec: boolean;
|
|
1259
|
-
start(onChunk: (pcm: Uint8Array) => void): void | Promise<void>;
|
|
1260
|
-
stop(): void;
|
|
1261
|
-
/** Unrecoverable source failure (e.g. mic permission denied → engine produces no audio AND, in the
|
|
1262
|
-
* duplex AEC engine, no playback). The host should tear voice down rather than spin reconnects. */
|
|
1263
|
-
onFatal?: (message: string) => void;
|
|
1264
|
-
}
|
|
1265
|
-
/** An audio playback sink for s16le mono 44.1kHz PCM.
|
|
1266
|
-
* Node: per-turn ffplay. Web: AudioContext/worklet ring buffer. */
|
|
1267
|
-
interface AudioSink {
|
|
1268
|
-
/** start a new spoken turn (may recycle or recreate the underlying player) */
|
|
1269
|
-
markTurn(): void;
|
|
1270
|
-
write(pcm: Uint8Array): void;
|
|
1271
|
-
/** estimated ms until queued audio finishes playing */
|
|
1272
|
-
drainMs(): number;
|
|
1273
|
-
/** ms of audio actually played this turn */
|
|
1274
|
-
playedMs(): number;
|
|
1275
|
-
/** stop playback NOW (barge-in primitive) */
|
|
1276
|
-
kill(): void;
|
|
1277
|
-
/** optional exact-sample pause/resume — enables the overlap trail-off tier (web: AudioContext
|
|
1278
|
-
* suspend/resume; CLI AEC helper: control frames). Sinks without it degrade to interrupt-only
|
|
1279
|
-
* turn-taking. Nothing is lost across a pause; playedMs/drainMs must exclude paused time. */
|
|
1280
|
-
pause?(): void;
|
|
1281
|
-
resume?(): void;
|
|
1282
|
-
}
|
|
1283
|
-
/** Static key (server/CLI) or an async getter (browser: fetch a short-lived token from YOUR
|
|
1284
|
-
* backend). Getters are invoked on EVERY (re)connect — temp tokens expire, so a reconnect
|
|
1285
|
-
* must re-mint, never reuse the boot-time token. */
|
|
1286
|
-
type AuthProvider = string | (() => string | Promise<string>);
|
|
1287
|
-
declare function resolveAuth(auth: AuthProvider): Promise<string>;
|
|
1288
|
-
|
|
1289
|
-
/** Injectable time source — ALL engine timing (merge windows, barge grace, overlap resume, drain
|
|
1290
|
-
* settles) routes through it. Defaults to real timers; the deterministic bench swaps a virtual
|
|
1291
|
-
* clock so timing scenarios run instantly and exactly. */
|
|
1292
|
-
interface EngineClock {
|
|
1293
|
-
now(): number;
|
|
1294
|
-
setTimeout(fn: () => void, ms: number): unknown;
|
|
1295
|
-
clearTimeout(handle: unknown): void;
|
|
1296
|
-
}
|
|
1297
|
-
/** One structured diagnostic event (fire-and-forget, never affects behavior). `t` = engine clock ms.
|
|
1298
|
-
* Emitted at every DECISION point; kinds + fields are the per-session forensic record
|
|
1299
|
-
* (CLI: `<session>.voice.jsonl`; lab: `<session>.timeline.jsonl`). See mind/11-voice.md §Diagnostics. */
|
|
1300
|
-
interface DiagEvent {
|
|
1301
|
-
t: number;
|
|
1302
|
-
kind: string;
|
|
1303
|
-
[k: string]: unknown;
|
|
1304
|
-
}
|
|
1305
|
-
/** Structural contracts (satisfied by SonioxSTT/CartesiaTTS or test fakes). */
|
|
1306
|
-
interface SttLike {
|
|
1307
|
-
usingAec: boolean;
|
|
1308
|
-
onPartial: (text: string) => void;
|
|
1309
|
-
onUtterance: (text: string, endpointAt: number) => void;
|
|
1310
|
-
onLevel: (rms: number) => void;
|
|
1311
|
-
/** Optional: unrecoverable capture failure (e.g. mic produced no audio) — host tears voice down. */
|
|
1312
|
-
onFatal?: (message: string) => void;
|
|
1313
|
-
start(): Promise<void> | void;
|
|
1314
|
-
reset(): void;
|
|
1315
|
-
stop(): void;
|
|
1316
|
-
}
|
|
1317
|
-
interface TtsLike {
|
|
1318
|
-
onAudio: (chunk: Uint8Array) => void;
|
|
1319
|
-
onDone: () => void;
|
|
1320
|
-
/** Optional word-timestamp seam for karaoke reveal (revealMode==='word'). `start[i]` = seconds
|
|
1321
|
-
* from turn-audio start. Set `wantTimestamps` to request them from the provider. */
|
|
1322
|
-
onTimestamps?: (words: string[], start: number[]) => void;
|
|
1323
|
-
wantTimestamps?: boolean;
|
|
1324
|
-
connect(): Promise<void> | void;
|
|
1325
|
-
/** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
|
|
1326
|
-
* so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
|
|
1327
|
-
warmup?(): void;
|
|
1328
|
-
newContext(): string;
|
|
1329
|
-
speak(text: string, cont: boolean): void;
|
|
1330
|
-
end(): void;
|
|
1331
|
-
cancel(): void;
|
|
1332
|
-
close(): void;
|
|
1333
|
-
}
|
|
1334
|
-
declare class VoiceEngineOptions {
|
|
1335
|
-
stt: SttLike;
|
|
1336
|
-
tts: TtsLike;
|
|
1337
|
-
player: AudioSink;
|
|
1338
|
-
/** a final utterance arrived (endpoint) — host dispatches it as a turn */
|
|
1339
|
-
onUtterance: (text: string) => void;
|
|
1340
|
-
/** live partial transcript while listening (host renders the 🎤 line) */
|
|
1341
|
-
onPartial: (text: string) => void;
|
|
1342
|
-
onState: (s: VoiceState) => void;
|
|
1343
|
-
/** user spoke/acted over playback — host aborts the in-flight turn (called AFTER audio is killed).
|
|
1344
|
-
* phase: 'speaking' = cut mid-speech (real interruption); 'drain' = in the final audio tail
|
|
1345
|
-
* (normal turn-taking — hosts shouldn't alarm). */
|
|
1346
|
-
onBargeIn: (phase: 'speaking' | 'drain') => void;
|
|
1347
|
-
/** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
|
|
1348
|
-
ackPhrase: string;
|
|
1349
|
-
/** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
|
|
1350
|
-
* delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
|
|
1351
|
-
* per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
|
|
1352
|
-
adaptiveAckMs: number;
|
|
1353
|
-
/** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
|
|
1354
|
-
onAdaptiveAck: () => void;
|
|
1355
|
-
/** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
|
|
1356
|
-
* letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
|
|
1357
|
-
* ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
|
|
1358
|
-
utteranceMergeMs: number;
|
|
1359
|
-
/** Extended merge window (ms) for utterances that look incomplete (trailing conjunction/filler).
|
|
1360
|
-
* Gives the user time to finish their thought without triggering a model call. */
|
|
1361
|
-
incompleteMergeMs: number;
|
|
1362
|
-
/** Grace window (ms) after an utterance dispatches, during which the user's own trailing audio cannot
|
|
1363
|
-
* barge the reply it requested. Soniox keeps finalizing partials past <end>; without this they read
|
|
1364
|
-
* as a barge and abort the fresh turn (live: mid-sentence self-interruption + steps=1→steps=0 double
|
|
1365
|
-
* abort). Short enough that a genuine immediate barge ("no wait—") still lands right after. */
|
|
1366
|
-
bargeGraceMs: number;
|
|
1367
|
-
/** Barge-in (talk over the assistant to interrupt). true = full-duplex (needs echo cancellation, or
|
|
1368
|
-
* the assistant's own TTS bleeds back and self-interrupts). false = HALF-DUPLEX: the engine is deaf
|
|
1369
|
-
* while audible (speaking + drain tail), so echo can never become a phantom turn — the right mode
|
|
1370
|
-
* when there's no AEC (e.g. the non-VPIO mic fallback) and no headphones. Cost: can't interrupt. */
|
|
1371
|
-
bargeIn: boolean;
|
|
1372
|
-
/** Filler phrase spoken when holding for an incomplete utterance ('' disables). */
|
|
1373
|
-
holdFiller: string;
|
|
1374
|
-
/** Called when the engine holds an incomplete utterance (host can render a visual cue). */
|
|
1375
|
-
onHold: () => void;
|
|
1376
|
-
/** heuristic (non-AEC) energy barge-in tuning */
|
|
1377
|
-
bargeRmsMult: number;
|
|
1378
|
-
bargeRmsFloor: number;
|
|
1379
|
-
/** Overlap turn-taking (AEC tier, needs player.pause/resume) — human phone-call model, driven by
|
|
1380
|
-
* the STT ITSELF (a trained speech classifier) instead of energy thresholds (energy could not
|
|
1381
|
-
* separate residue bursts from speech in every room — hiccup whack-a-mole): a GENUINE partial
|
|
1382
|
-
* (novel words dominate — echo of our own reply is inert) while speaking → PAUSE (exact-sample
|
|
1383
|
-
* hold); partial grows into dominant-novel ≥2 words → cede (interrupt; the LLM re-enters); partial
|
|
1384
|
-
* stalls/endpoints without ceding (backchannel by DURATION, not vocabulary) → resume + drop. false disables. */
|
|
1385
|
-
overlapPause: boolean;
|
|
1386
|
-
/** no new partial activity for this long while paused → resume, drop the interjection */
|
|
1387
|
-
overlapResumeMs: number;
|
|
1388
|
-
/** A genuine barge over a LONG reply is defeated by the dominant-novel gate: Meet echoes our own
|
|
1389
|
-
* speech back, so the partial is mostly our words + a few of hers → never "dominant novel" → it
|
|
1390
|
-
* resumes (replaying old audio — the audible "completes the buffer" blip) instead of ceding.
|
|
1391
|
-
* Mechanism-based discriminator: a re-PAUSE this soon after a resume = a persistent human, not an
|
|
1392
|
-
* echo blip (which pauses once and stalls). Cede on the re-pause regardless of the novel gate. */
|
|
1393
|
-
overlapRepauseCedeMs: number;
|
|
1394
|
-
/** Speculative ENERGY pre-pause while speaking (AEC tier): two residue gate-passes within 350ms →
|
|
1395
|
-
* pause ~300ms before the STT tokens land. But energy CANNOT separate residue bursts from speech
|
|
1396
|
-
* (the documented whack-a-mole) — so a residue spike during loud playback false-pauses with NO user
|
|
1397
|
-
* speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
|
|
1398
|
-
* mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
|
|
1399
|
-
overlapEnergyHold: boolean;
|
|
1400
|
-
/** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
|
|
1401
|
-
* this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
|
|
1402
|
-
* start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
|
|
1403
|
-
* speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
|
|
1404
|
-
* confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
|
|
1405
|
-
speculativeMs: number;
|
|
1406
|
-
/** Minimum word count for a partial to qualify as a speculation trigger. */
|
|
1407
|
-
speculativeMinWords: number;
|
|
1408
|
-
/** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
|
|
1409
|
-
onSpeculate: (text: string) => void;
|
|
1410
|
-
/** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
|
|
1411
|
-
* partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
|
|
1412
|
-
* UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
|
|
1413
|
-
* quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
|
|
1414
|
-
* touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
|
|
1415
|
-
* context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
|
|
1416
|
-
* boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
|
|
1417
|
-
* 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
|
|
1418
|
-
* ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
|
|
1419
|
-
backchannelMs: number;
|
|
1420
|
-
/** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
|
|
1421
|
-
backchannelMinGapMs: number;
|
|
1422
|
-
/** Only multi-clause turns: the partial must carry at least this many words before a blip. */
|
|
1423
|
-
backchannelMinWords: number;
|
|
1424
|
-
/** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
|
|
1425
|
-
onBackchannel: (phrase: string) => void;
|
|
1426
|
-
/** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
|
|
1427
|
-
* quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
|
|
1428
|
-
onSpeculateAbort: () => void;
|
|
1429
|
-
/** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
|
|
1430
|
-
* tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
|
|
1431
|
-
emotions: boolean;
|
|
1432
|
-
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
1433
|
-
showEmotions: boolean;
|
|
1434
|
-
/**
|
|
1435
|
-
* Progressive text reveal — the "karaoke" capability, opt-in.
|
|
1436
|
-
* 'off' — no reveal events (CLI default; the host renders text however it likes).
|
|
1437
|
-
* 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
|
|
1438
|
-
* in step with the model stream).
|
|
1439
|
-
* 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
|
|
1440
|
-
* timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
|
|
1441
|
-
*/
|
|
1442
|
-
revealMode: 'off' | 'delta' | 'word';
|
|
1443
|
-
/** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
|
|
1444
|
-
* Reset to '' at the start of each spoken turn. */
|
|
1445
|
-
onReveal: (revealed: string) => void;
|
|
1446
|
-
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
1447
|
-
clock: EngineClock;
|
|
1448
|
-
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
1449
|
-
* barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
|
|
1450
|
-
* a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
|
|
1451
|
-
onDiag: (ev: DiagEvent) => void;
|
|
1452
|
-
}
|
|
1453
|
-
declare class VoiceEngine {
|
|
1454
|
-
options: VoiceEngineOptions;
|
|
1455
|
-
state: VoiceState;
|
|
1456
|
-
protected stt: SttLike;
|
|
1457
|
-
protected tts: TtsLike;
|
|
1458
|
-
protected player: AudioSink;
|
|
1459
|
-
private speaking;
|
|
1460
|
-
private ctxOpen;
|
|
1461
|
-
private interrupted;
|
|
1462
|
-
private spokeDeltas;
|
|
1463
|
-
private revealText;
|
|
1464
|
-
private wordStarts;
|
|
1465
|
-
private revealedN;
|
|
1466
|
-
private revealPoll;
|
|
1467
|
-
private clock;
|
|
1468
|
-
private drainTimer;
|
|
1469
|
-
private echoWords;
|
|
1470
|
-
private prevReply;
|
|
1471
|
-
private reply;
|
|
1472
|
-
private echoUntil;
|
|
1473
|
-
private baseline;
|
|
1474
|
-
private hot;
|
|
1475
|
-
private suspectUntil;
|
|
1476
|
-
private ackAt;
|
|
1477
|
-
private lastAck;
|
|
1478
|
-
private ackTimer;
|
|
1479
|
-
private bargeGraceUntil;
|
|
1480
|
-
private pendingUtt;
|
|
1481
|
-
private mergePath;
|
|
1482
|
-
private lastGraceDiag;
|
|
1483
|
-
private pendingTimer;
|
|
1484
|
-
private lastDispatchFlat;
|
|
1485
|
-
private lastDispatchWords;
|
|
1486
|
-
private lastDispatchAt;
|
|
1487
|
-
private repliedSinceDispatch;
|
|
1488
|
-
private static readonly DUP_FINAL_MS;
|
|
1489
|
-
private lastInterrupted;
|
|
1490
|
-
private pausedAt;
|
|
1491
|
-
private lastResumeAt;
|
|
1492
|
-
private lastOverlapPartial;
|
|
1493
|
-
private resumeTimer;
|
|
1494
|
-
private turnStartAt;
|
|
1495
|
-
private specPartial;
|
|
1496
|
-
private specTimer;
|
|
1497
|
-
private specText;
|
|
1498
|
-
private specSpent;
|
|
1499
|
-
private bcPartial;
|
|
1500
|
-
private bcTimer;
|
|
1501
|
-
private bcCount;
|
|
1502
|
-
private bcActive;
|
|
1503
|
-
private bcActiveTimer;
|
|
1504
|
-
private lastBcAt;
|
|
1505
|
-
private lastBcPhrase;
|
|
1506
|
-
private recentBc;
|
|
1507
|
-
private uttQueue;
|
|
1508
|
-
private emo;
|
|
1509
|
-
constructor(options?: Partial<VoiceEngineOptions>);
|
|
1510
|
-
start(): Promise<void>;
|
|
1511
|
-
get usingAec(): boolean;
|
|
1512
|
-
/** Flip barge-in at runtime (e.g. the mic fell back to non-VPIO → go half-duplex so echo can't leak). */
|
|
1513
|
-
setBargeIn(on: boolean): void;
|
|
1514
|
-
/** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
|
|
1515
|
-
setShowEmotions(on: boolean): void;
|
|
1516
|
-
/** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
|
|
1517
|
-
* it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
|
|
1518
|
-
private diagOn;
|
|
1519
|
-
protected diag(kind: string, fields?: Record<string, unknown>): void;
|
|
1520
|
-
private idleWaiters;
|
|
1521
|
-
private setState;
|
|
1522
|
-
/** Resolve when the engine is no longer speaking (immediate if already idle). */
|
|
1523
|
-
awaitIdle(): Promise<void>;
|
|
1524
|
-
/** open a spoken turn (idempotent — safe from both onUtterance and first-delta paths).
|
|
1525
|
-
* `ack` speaks the configured micro-ack as the context opener (utterance path only —
|
|
1526
|
-
* masks LLM TTFT; re-voice turns begun by their first delta skip it). */
|
|
1527
|
-
beginSpeech(ack?: boolean): void;
|
|
1528
|
-
/** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
|
|
1529
|
-
* host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
|
|
1530
|
-
speakDelta(text: string): string;
|
|
1531
|
-
/** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
|
|
1532
|
-
* Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
|
|
1533
|
-
* paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
|
|
1534
|
-
* deterministic under the virtual clock. */
|
|
1535
|
-
private startWordReveal;
|
|
1536
|
-
/** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
|
|
1537
|
-
* undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
|
|
1538
|
-
* is what was actually spoken). */
|
|
1539
|
-
private stopWordReveal;
|
|
1540
|
-
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
1541
|
-
endSpeech(): void;
|
|
1542
|
-
/** text of the reply cut by the last barge-in — consumed by the host to tell the model what
|
|
1543
|
-
* the user did NOT hear. Cleared on read. */
|
|
1544
|
-
takeInterruptedReply(): {
|
|
1545
|
-
full: string;
|
|
1546
|
-
heard: string;
|
|
1547
|
-
} | null;
|
|
1548
|
-
/** Speak a short filler phrase without starting a model turn (stays in listening mode after). */
|
|
1549
|
-
speakFiller(text: string): void;
|
|
1550
|
-
/** Enqueue a COMPLETE worker utterance (already-split spoken text) onto the central speech queue.
|
|
1551
|
-
* If nothing is currently speaking it plays immediately; otherwise it queues and plays after the
|
|
1552
|
-
* current utterance fully ends (settle → pumpQueue) — never spliced into an open reflex utterance. */
|
|
1553
|
-
enqueueUtterance(text: string): void;
|
|
1554
|
-
/** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
|
|
1555
|
-
* next one from the settle completion (endSpeech), so utterances serialize without overlap. */
|
|
1556
|
-
private pumpQueue;
|
|
1557
|
-
/** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
|
|
1558
|
-
* utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
|
|
1559
|
-
* of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
|
|
1560
|
-
* many times"). All phrases are sub-second and semantically safe for their shape.
|
|
1561
|
-
* No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
|
|
1562
|
-
static readonly ACKS_NEUTRAL: string[];
|
|
1563
|
-
static readonly ACKS_QUESTION: string[];
|
|
1564
|
-
private recentAcks;
|
|
1565
|
-
private pickAck;
|
|
1566
|
-
private lastDispatchWasQuestion;
|
|
1567
|
-
private clearAckTimer;
|
|
1568
|
-
/** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
|
|
1569
|
-
* to be a Hold (intentionally silent; an ack would read as the start of an answer). */
|
|
1570
|
-
cancelPendingAck(): void;
|
|
1571
|
-
/** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
|
|
1572
|
-
interrupt(): void;
|
|
1573
|
-
stop(): void;
|
|
1574
|
-
private words;
|
|
1575
|
-
private novelWords;
|
|
1576
|
-
private echoActive;
|
|
1577
|
-
/** Genuine user speech vs our own bleed (AEC tier): novel words must DOMINATE, not merely exist.
|
|
1578
|
-
* Degraded AEC + an STT mis-hearing manufactures a single novel word out of pure echo (a name or
|
|
1579
|
-
* rare word in our own reply comes back transcribed slightly differently — 1 novel / N words).
|
|
1580
|
-
* A real interjection is mostly novel ("stop", "wait what") — short utterances pass on ratio,
|
|
1581
|
-
* longer ones on count. */
|
|
1582
|
-
private genuine;
|
|
1583
|
-
private handlePartial;
|
|
1584
|
-
/** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
|
|
1585
|
-
* partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
|
|
1586
|
-
* onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
|
|
1587
|
-
* tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
|
|
1588
|
-
* dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
|
|
1589
|
-
* trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
|
|
1590
|
-
* utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
|
|
1591
|
-
private trackSpeculation;
|
|
1592
|
-
/** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
|
|
1593
|
-
* imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
|
|
1594
|
-
* every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
|
|
1595
|
-
* strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
|
|
1596
|
-
* 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
|
|
1597
|
-
* right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
|
|
1598
|
-
static readonly BACKCHANNELS: string[];
|
|
1599
|
-
/** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
|
|
1600
|
-
* that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
|
|
1601
|
-
* BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
|
|
1602
|
-
* Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
|
|
1603
|
-
* Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
|
|
1604
|
-
private trackBackchannel;
|
|
1605
|
-
/** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
|
|
1606
|
-
* markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
|
|
1607
|
-
* pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
|
|
1608
|
-
* never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
|
|
1609
|
-
* swallowed), and the echo window extends so echo-shaped finals stay gated. */
|
|
1610
|
-
private fireBackchannel;
|
|
1611
|
-
/** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
|
|
1612
|
-
private bcSupersede;
|
|
1613
|
-
/** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
|
|
1614
|
-
* phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
|
|
1615
|
-
* within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
|
|
1616
|
-
private stripBackchannelEcho;
|
|
1617
|
-
/** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
|
|
1618
|
-
* overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
|
|
1619
|
-
* utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
|
|
1620
|
-
* a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
|
|
1621
|
-
* check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
|
|
1622
|
-
private mergeUtterance;
|
|
1623
|
-
private static normWord;
|
|
1624
|
-
/** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
|
|
1625
|
-
* the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
|
|
1626
|
-
* may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
|
|
1627
|
-
* DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
|
|
1628
|
-
* "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
|
|
1629
|
-
* question already being answered).
|
|
1630
|
-
* DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
|
|
1631
|
-
* (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
|
|
1632
|
-
* previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
|
|
1633
|
-
* repliedSinceDispatch then picks the action:
|
|
1634
|
-
* • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
|
|
1635
|
-
* "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
|
|
1636
|
-
* worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
|
|
1637
|
-
* DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
|
|
1638
|
-
* Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
|
|
1639
|
-
* • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
|
|
1640
|
-
* truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
|
|
1641
|
-
* starves the conversation entirely — the fuller final IS the completion the Hold is waiting
|
|
1642
|
-
* for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
|
|
1643
|
-
* race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
|
|
1644
|
-
private refinalizes;
|
|
1645
|
-
private static readonly TRAIL_RE;
|
|
1646
|
-
/** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
|
|
1647
|
-
private looksIncomplete;
|
|
1648
|
-
private handleUtterance;
|
|
1649
|
-
private flushUtterance;
|
|
1650
|
-
private get overlapCapable();
|
|
1651
|
-
private armResume;
|
|
1652
|
-
private resetOverlap;
|
|
1653
|
-
/** energy two-stage barge-in (heuristic tier only): spike over echo baseline → pause + confirm via STT */
|
|
1654
|
-
private gatePassTimes;
|
|
1655
|
-
private handleLevel;
|
|
1656
|
-
}
|
|
1657
|
-
|
|
1658
|
-
declare class SonioxSTTOptions {
|
|
1659
|
-
auth: AuthProvider;
|
|
1660
|
-
source: AudioSource;
|
|
1661
|
-
model: string;
|
|
1662
|
-
languageHints: string[];
|
|
1663
|
-
/** Client-side endpoint: finalized text + no new tokens for this long = utterance (don't wait for
|
|
1664
|
-
* Soniox's semantic <end>, which adds 0.5-1.5s — the difference between ping-pong and lag). */
|
|
1665
|
-
silenceEndpointMs: number;
|
|
1666
|
-
/** No-audio watchdog: if the mic source stops delivering chunks for this long, capture is dead →
|
|
1667
|
-
* fire onFatal + stop (else Soniox idle-timeouts and reconnect-loops forever). 0 = disable. */
|
|
1668
|
-
noAudioTimeoutMs: number;
|
|
1669
|
-
}
|
|
1670
|
-
declare class SonioxSTT {
|
|
1671
|
-
options: SonioxSTTOptions;
|
|
1672
|
-
private ws;
|
|
1673
|
-
private stopped;
|
|
1674
|
-
private sourceStarted;
|
|
1675
|
-
onPartial: (text: string) => void;
|
|
1676
|
-
onUtterance: (text: string, endpointAt: number) => void;
|
|
1677
|
-
/** mic energy (RMS) per chunk — drives the energy-based heuristic barge-in tier */
|
|
1678
|
-
onLevel: (rms: number) => void;
|
|
1679
|
-
/** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
|
|
1680
|
-
* loop). The host tears voice down instead of spinning forever. */
|
|
1681
|
-
onFatal: (message: string) => void;
|
|
1682
|
-
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
1683
|
-
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
1684
|
-
onDiag: (ev: {
|
|
1685
|
-
t: number;
|
|
1686
|
-
kind: string;
|
|
1687
|
-
[k: string]: unknown;
|
|
1688
|
-
}) => void;
|
|
1689
|
-
private diagOn;
|
|
1690
|
-
private diag;
|
|
1691
|
-
private lastChunkAt;
|
|
1692
|
-
private startedChunksAt;
|
|
1693
|
-
private noAudioTimer;
|
|
1694
|
-
private finalText;
|
|
1695
|
-
private partialText;
|
|
1696
|
-
private lastChangeAt;
|
|
1697
|
-
private lastCombined;
|
|
1698
|
-
private endpointTimer;
|
|
1699
|
-
private firstTokenAt;
|
|
1700
|
-
constructor(options?: Partial<SonioxSTTOptions>);
|
|
1701
|
-
get usingAec(): boolean;
|
|
1702
|
-
private connectWs;
|
|
1703
|
-
start(): Promise<void>;
|
|
1704
|
-
private handle;
|
|
1705
|
-
reset(): void;
|
|
1706
|
-
stop(): void;
|
|
1707
|
-
}
|
|
1708
|
-
|
|
1709
|
-
declare class CartesiaTTSOptions {
|
|
1710
|
-
auth: AuthProvider;
|
|
1711
|
-
voiceId: string;
|
|
1712
|
-
model: string;
|
|
1713
|
-
/** 'apiKey' (server/CLI) → `api_key=` URL param; 'token' (browser, BE-minted) → `access_token=`. */
|
|
1714
|
-
authMode: 'apiKey' | 'token';
|
|
1715
|
-
}
|
|
1716
|
-
declare class CartesiaTTS {
|
|
1717
|
-
options: CartesiaTTSOptions;
|
|
1718
|
-
private ws;
|
|
1719
|
-
private ctxSeq;
|
|
1720
|
-
ctxId: string;
|
|
1721
|
-
onAudio: (chunk: Uint8Array) => void;
|
|
1722
|
-
onDone: () => void;
|
|
1723
|
-
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
1724
|
-
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
1725
|
-
onTimestamps: (words: string[], start: number[]) => void;
|
|
1726
|
-
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
1727
|
-
* when revealMode==='word'. */
|
|
1728
|
-
wantTimestamps: boolean;
|
|
1729
|
-
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
1730
|
-
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
1731
|
-
onDiag: (ev: {
|
|
1732
|
-
t: number;
|
|
1733
|
-
kind: string;
|
|
1734
|
-
[k: string]: unknown;
|
|
1735
|
-
}) => void;
|
|
1736
|
-
private diagOn;
|
|
1737
|
-
private diag;
|
|
1738
|
-
firstAudioAt: number;
|
|
1739
|
-
/** Circuit breaker: consecutive error count + down flag. */
|
|
1740
|
-
private consecutiveErrors;
|
|
1741
|
-
private consecutiveOk;
|
|
1742
|
-
private down;
|
|
1743
|
-
private downAt;
|
|
1744
|
-
private probeTimer;
|
|
1745
|
-
private static readonly CB_THRESHOLD;
|
|
1746
|
-
private static readonly CB_RECOVER_OK;
|
|
1747
|
-
private static readonly CB_PROBE_MS;
|
|
1748
|
-
constructor(options?: Partial<CartesiaTTSOptions>);
|
|
1749
|
-
private closed;
|
|
1750
|
-
private connecting;
|
|
1751
|
-
connect(): Promise<void>;
|
|
1752
|
-
private doConnect;
|
|
1753
|
-
/** Close the breaker only after CB_RECOVER_OK consecutive good frames, so a single straggler chunk
|
|
1754
|
-
* after a 503 burst doesn't flap open→recover in <1s. A sub-2s down-window is a transient blip → debug. */
|
|
1755
|
-
private markRecovered;
|
|
1756
|
-
/** Ensure the WS is open before sending — reconnects if idle-closed. */
|
|
1757
|
-
private ensureConnected;
|
|
1758
|
-
/** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
|
|
1759
|
-
* a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
|
|
1760
|
-
* The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
|
|
1761
|
-
* newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
|
|
1762
|
-
warmup(): void;
|
|
1763
|
-
newContext(): string;
|
|
1764
|
-
private frame;
|
|
1765
|
-
speak(text: string, cont: boolean): void;
|
|
1766
|
-
end(): void;
|
|
1767
|
-
cancel(): void;
|
|
1768
|
-
private startProbe;
|
|
1769
|
-
private stopProbe;
|
|
1770
|
-
close(): void;
|
|
1771
|
-
}
|
|
1772
|
-
|
|
1773
1227
|
declare function validateWorktreeSlug(slug: string): void;
|
|
1774
1228
|
declare function worktreeBranchName(slug: string): string;
|
|
1775
1229
|
interface WorktreeInfo {
|
|
@@ -1801,4 +1255,4 @@ declare function exitWorktree(): {
|
|
|
1801
1255
|
reason?: string;
|
|
1802
1256
|
} | null;
|
|
1803
1257
|
|
|
1804
|
-
export { Agent, type AgentDef, AgentOptions, AgentTool, type AskOptions, type Attempt,
|
|
1258
|
+
export { Agent, type AgentDef, AgentOptions, AgentTool, type AskOptions, type Attempt, BodDbFilesystem, ChatLike, ChatOptions, ChatResponse, type CommandInfo, ConsoleHostBridge, DEFAULT_DENY, DuplexAgent, DuplexAgentOptions, type DuplexTaskStatus, FakeAIClient, Hooks, HostBridge, JailOptions, JailedFilesystem, type LessonOptions, LessonOptionsDefaults, type LoadMemoryOpts, MEMORY_PROMPT, MessageContent, type Mount, MountFilesystem, NodeDiskFilesystem, OverlayFilesystem, type ReflectOptions, RunResult, SCRATCH_DIR, type ScheduledJob, type ScheduledJobSnapshot, Scheduler, type SchedulerOptions, Scratch, type ScratchOptions, ScriptedHostBridge, type SiblingResolver, type SkillInfo, type TaskRecord, type TaskToolOptions, ToolCall, type ToolSpec, type Trigger, type TriggerCron, type TriggerInterval, type TriggerOneOff, UserQuestion, VOICE_MEMORY_PROMPT, VOICE_SYSTEM_PROMPT, type WebFetchOptions, type WebSearchOptions, type WorkerTier, type WorktreeInfo, type WorktreeSession, applyEditsTool, askUserQuestionTool, checkpointTool, checkpointTools, cleanupWorktree, compileSynthesizedTool, cronMatches, decodeDdgUrl, diskAgentOptions, enterWorktree, exitWorktree, expandCommand, expandEntry, expandTemplate, findGitRoot, forComponent, fullAgentOptions, getOrCreateWorktree, getWorktreeSession, globTool, grepTool, htmlToText, idfWeights, lessonCapture, loadAgents, loadCommands, loadInstructions, loadMemory, loadSkills, makeAskTool, makeScheduleTools, makeTaskBatchTool, makeTaskTool, makeWebFetchTool, makeWebSearchAnthropicTool, makeWebSearchTool, mkdirp, multiEditTool, nextCronAfter, parseCron, parseDdgHtml, raceAttempts, reflectOnRun, relevanceScore, repoIndex, repoMapTool, rollbackTool, sandboxAgentOptions, scanCommands, scanSkills, slugify, tokenize, toolCall, topByRelevance, validateToolCode, validateWorktreeSlug, webFetchTool, webSearchAnthropicTool, webSearchTool, worktreeBranchName, writeFact, writeTool };
|