@livx.cc/agentx 0.99.7 → 0.99.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +82 -1
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +35 -0
- package/dist/index.js +82 -1
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -1299,6 +1299,10 @@ interface SttLike {
|
|
|
1299
1299
|
interface TtsLike {
|
|
1300
1300
|
onAudio: (chunk: Uint8Array) => void;
|
|
1301
1301
|
onDone: () => void;
|
|
1302
|
+
/** Optional word-timestamp seam for karaoke reveal (revealMode==='word'). `start[i]` = seconds
|
|
1303
|
+
* from turn-audio start. Set `wantTimestamps` to request them from the provider. */
|
|
1304
|
+
onTimestamps?: (words: string[], start: number[]) => void;
|
|
1305
|
+
wantTimestamps?: boolean;
|
|
1302
1306
|
connect(): Promise<void> | void;
|
|
1303
1307
|
/** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
|
|
1304
1308
|
* so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
|
|
@@ -1409,6 +1413,18 @@ declare class VoiceEngineOptions {
|
|
|
1409
1413
|
emotions: boolean;
|
|
1410
1414
|
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
1411
1415
|
showEmotions: boolean;
|
|
1416
|
+
/**
|
|
1417
|
+
* Progressive text reveal — the "karaoke" capability, opt-in.
|
|
1418
|
+
* 'off' — no reveal events (CLI default; the host renders text however it likes).
|
|
1419
|
+
* 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
|
|
1420
|
+
* in step with the model stream).
|
|
1421
|
+
* 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
|
|
1422
|
+
* timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
|
|
1423
|
+
*/
|
|
1424
|
+
revealMode: 'off' | 'delta' | 'word';
|
|
1425
|
+
/** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
|
|
1426
|
+
* Reset to '' at the start of each spoken turn. */
|
|
1427
|
+
onReveal: (revealed: string) => void;
|
|
1412
1428
|
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
1413
1429
|
clock: EngineClock;
|
|
1414
1430
|
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
@@ -1426,6 +1442,10 @@ declare class VoiceEngine {
|
|
|
1426
1442
|
private ctxOpen;
|
|
1427
1443
|
private interrupted;
|
|
1428
1444
|
private spokeDeltas;
|
|
1445
|
+
private revealText;
|
|
1446
|
+
private wordStarts;
|
|
1447
|
+
private revealedN;
|
|
1448
|
+
private revealPoll;
|
|
1429
1449
|
private clock;
|
|
1430
1450
|
private drainTimer;
|
|
1431
1451
|
private echoWords;
|
|
@@ -1490,6 +1510,15 @@ declare class VoiceEngine {
|
|
|
1490
1510
|
/** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
|
|
1491
1511
|
* host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
|
|
1492
1512
|
speakDelta(text: string): string;
|
|
1513
|
+
/** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
|
|
1514
|
+
* Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
|
|
1515
|
+
* paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
|
|
1516
|
+
* deterministic under the virtual clock. */
|
|
1517
|
+
private startWordReveal;
|
|
1518
|
+
/** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
|
|
1519
|
+
* undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
|
|
1520
|
+
* is what was actually spoken). */
|
|
1521
|
+
private stopWordReveal;
|
|
1493
1522
|
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
1494
1523
|
endSpeech(): void;
|
|
1495
1524
|
/** text of the reply cut by the last barge-in — consumed by the host to tell the model what
|
|
@@ -1673,6 +1702,12 @@ declare class CartesiaTTS {
|
|
|
1673
1702
|
ctxId: string;
|
|
1674
1703
|
onAudio: (chunk: Uint8Array) => void;
|
|
1675
1704
|
onDone: () => void;
|
|
1705
|
+
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
1706
|
+
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
1707
|
+
onTimestamps: (words: string[], start: number[]) => void;
|
|
1708
|
+
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
1709
|
+
* when revealMode==='word'. */
|
|
1710
|
+
wantTimestamps: boolean;
|
|
1676
1711
|
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
1677
1712
|
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
1678
1713
|
onDiag: (ev: {
|
package/dist/index.js
CHANGED
|
@@ -5012,6 +5012,19 @@ var VoiceEngineOptions = class {
|
|
|
5012
5012
|
emotions = true;
|
|
5013
5013
|
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
5014
5014
|
showEmotions = false;
|
|
5015
|
+
/**
|
|
5016
|
+
* Progressive text reveal — the "karaoke" capability, opt-in.
|
|
5017
|
+
* 'off' — no reveal events (CLI default; the host renders text however it likes).
|
|
5018
|
+
* 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
|
|
5019
|
+
* in step with the model stream).
|
|
5020
|
+
* 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
|
|
5021
|
+
* timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
|
|
5022
|
+
*/
|
|
5023
|
+
revealMode = "off";
|
|
5024
|
+
/** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
|
|
5025
|
+
* Reset to '' at the start of each spoken turn. */
|
|
5026
|
+
onReveal = () => {
|
|
5027
|
+
};
|
|
5015
5028
|
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
5016
5029
|
clock = realClock;
|
|
5017
5030
|
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
@@ -5034,6 +5047,12 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5034
5047
|
// barge-in latch: drop in-flight deltas until the next legitimate turn
|
|
5035
5048
|
spokeDeltas = false;
|
|
5036
5049
|
// a TTS context is open for the current spoken turn
|
|
5050
|
+
revealText = "";
|
|
5051
|
+
// cumulative revealed (on-screen) text for the current turn — see revealMode
|
|
5052
|
+
// 'word' karaoke reveal: spoken-word start times (sec, absolute from turn-audio start) + poll state.
|
|
5053
|
+
wordStarts = [];
|
|
5054
|
+
revealedN = 0;
|
|
5055
|
+
revealPoll = null;
|
|
5037
5056
|
clock;
|
|
5038
5057
|
drainTimer = null;
|
|
5039
5058
|
// heuristic tier state (inert under AEC) — frozen as validated in the experiment
|
|
@@ -5121,6 +5140,13 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5121
5140
|
this.tts.onAudio = (c) => {
|
|
5122
5141
|
if (this.speaking || this.bcActive) this.player.write(c);
|
|
5123
5142
|
};
|
|
5143
|
+
if (this.options.revealMode === "word") {
|
|
5144
|
+
this.tts.wantTimestamps = true;
|
|
5145
|
+
this.tts.onTimestamps = (_words, start) => {
|
|
5146
|
+
if (this.interrupted) return;
|
|
5147
|
+
for (const s of start) this.wordStarts.push(s);
|
|
5148
|
+
};
|
|
5149
|
+
}
|
|
5124
5150
|
this.stt.onPartial = (text) => this.handlePartial(text);
|
|
5125
5151
|
this.stt.onUtterance = (text) => this.handleUtterance(text);
|
|
5126
5152
|
this.stt.onLevel = (rms) => this.handleLevel(rms);
|
|
@@ -5186,6 +5212,10 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5186
5212
|
this.ctxOpen = true;
|
|
5187
5213
|
this.spokeDeltas = false;
|
|
5188
5214
|
this.reply = "";
|
|
5215
|
+
this.revealText = "";
|
|
5216
|
+
this.wordStarts = [];
|
|
5217
|
+
this.revealedN = 0;
|
|
5218
|
+
if (this.options.revealMode === "word") this.startWordReveal();
|
|
5189
5219
|
this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
|
|
5190
5220
|
this.echoWords = new Set(this.words(this.prevReply));
|
|
5191
5221
|
this.tts.newContext();
|
|
@@ -5234,8 +5264,46 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5234
5264
|
if (!this.spokeDeltas && this.turnStartAt) log10.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
|
|
5235
5265
|
this.spokeDeltas = true;
|
|
5236
5266
|
this.setState("speaking");
|
|
5267
|
+
if (this.options.revealMode !== "off" && display) {
|
|
5268
|
+
this.revealText += display;
|
|
5269
|
+
if (this.options.revealMode === "delta") this.options.onReveal(this.revealText);
|
|
5270
|
+
}
|
|
5237
5271
|
return display;
|
|
5238
5272
|
}
|
|
5273
|
+
/** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
|
|
5274
|
+
* Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
|
|
5275
|
+
* paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
|
|
5276
|
+
* deterministic under the virtual clock. */
|
|
5277
|
+
startWordReveal() {
|
|
5278
|
+
if (this.revealPoll) return;
|
|
5279
|
+
const tick = () => {
|
|
5280
|
+
this.revealPoll = null;
|
|
5281
|
+
if (!this.speaking || this.interrupted || this.options.revealMode !== "word") return;
|
|
5282
|
+
const playedSec = this.player.playedMs() / 1e3;
|
|
5283
|
+
let n = this.revealedN;
|
|
5284
|
+
while (n < this.wordStarts.length && this.wordStarts[n] <= playedSec) n++;
|
|
5285
|
+
if (n !== this.revealedN) {
|
|
5286
|
+
this.revealedN = n;
|
|
5287
|
+
const words = this.revealText.trim().split(/\s+/).filter(Boolean);
|
|
5288
|
+
this.options.onReveal(words.slice(0, n).join(" "));
|
|
5289
|
+
}
|
|
5290
|
+
this.revealPoll = this.clock.setTimeout(tick, 60);
|
|
5291
|
+
};
|
|
5292
|
+
this.revealPoll = this.clock.setTimeout(tick, 60);
|
|
5293
|
+
}
|
|
5294
|
+
/** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
|
|
5295
|
+
* undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
|
|
5296
|
+
* is what was actually spoken). */
|
|
5297
|
+
stopWordReveal(finalFlush) {
|
|
5298
|
+
if (this.revealPoll) {
|
|
5299
|
+
this.clock.clearTimeout(this.revealPoll);
|
|
5300
|
+
this.revealPoll = null;
|
|
5301
|
+
}
|
|
5302
|
+
if (finalFlush && this.options.revealMode === "word" && !this.interrupted) {
|
|
5303
|
+
const full = this.revealText.trim();
|
|
5304
|
+
if (full) this.options.onReveal(full);
|
|
5305
|
+
}
|
|
5306
|
+
}
|
|
5239
5307
|
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
5240
5308
|
endSpeech() {
|
|
5241
5309
|
this.interrupted = false;
|
|
@@ -5261,6 +5329,7 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5261
5329
|
}
|
|
5262
5330
|
this.drainTimer = null;
|
|
5263
5331
|
this.speaking = false;
|
|
5332
|
+
this.stopWordReveal(true);
|
|
5264
5333
|
if (this.turnStartAt) log10.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
|
|
5265
5334
|
this.echoUntil = this.clock.now() + 2500;
|
|
5266
5335
|
if (!this.usingAec) this.stt.reset();
|
|
@@ -5348,6 +5417,7 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5348
5417
|
const droppedQueued = this.uttQueue.length;
|
|
5349
5418
|
this.uttQueue = [];
|
|
5350
5419
|
if (!this.speaking && !this.drainTimer) return;
|
|
5420
|
+
this.stopWordReveal(false);
|
|
5351
5421
|
this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
|
|
5352
5422
|
if (this.drainTimer) {
|
|
5353
5423
|
this.clock.clearTimeout(this.drainTimer);
|
|
@@ -7272,6 +7342,13 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7272
7342
|
};
|
|
7273
7343
|
onDone = () => {
|
|
7274
7344
|
};
|
|
7345
|
+
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
7346
|
+
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
7347
|
+
onTimestamps = () => {
|
|
7348
|
+
};
|
|
7349
|
+
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
7350
|
+
* when revealMode==='word'. */
|
|
7351
|
+
wantTimestamps = false;
|
|
7275
7352
|
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
7276
7353
|
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
7277
7354
|
onDiag = () => {
|
|
@@ -7339,6 +7416,9 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7339
7416
|
this.consecutiveErrors = 0;
|
|
7340
7417
|
this.markRecovered();
|
|
7341
7418
|
this.onDone();
|
|
7419
|
+
} else if (m.type === "timestamps" && m.word_timestamps) {
|
|
7420
|
+
const wt = m.word_timestamps;
|
|
7421
|
+
if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
|
|
7342
7422
|
} else if (m.type === "error") {
|
|
7343
7423
|
if (/already been cancelled|does not exist/.test(m.message || "")) return;
|
|
7344
7424
|
this.consecutiveErrors++;
|
|
@@ -7395,7 +7475,8 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7395
7475
|
voice: { mode: "id", id: this.options.voiceId },
|
|
7396
7476
|
output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
|
|
7397
7477
|
context_id: this.ctxId,
|
|
7398
|
-
continue: cont
|
|
7478
|
+
continue: cont,
|
|
7479
|
+
...this.wantTimestamps ? { add_timestamps: true } : {}
|
|
7399
7480
|
});
|
|
7400
7481
|
}
|
|
7401
7482
|
speak(text, cont) {
|