@livx.cc/agentx 0.99.7 → 0.99.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1299,6 +1299,10 @@ interface SttLike {
1299
1299
  interface TtsLike {
1300
1300
  onAudio: (chunk: Uint8Array) => void;
1301
1301
  onDone: () => void;
1302
+ /** Optional word-timestamp seam for karaoke reveal (revealMode==='word'). `start[i]` = seconds
1303
+ * from turn-audio start. Set `wantTimestamps` to request them from the provider. */
1304
+ onTimestamps?: (words: string[], start: number[]) => void;
1305
+ wantTimestamps?: boolean;
1302
1306
  connect(): Promise<void> | void;
1303
1307
  /** Optional: prime the synthesis pipeline right after connect (throwaway context, audio discarded)
1304
1308
  * so the FIRST real turn doesn't pay the provider's cold-synthesis spin-up. */
@@ -1409,6 +1413,18 @@ declare class VoiceEngineOptions {
1409
1413
  emotions: boolean;
1410
1414
  /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
1411
1415
  showEmotions: boolean;
1416
+ /**
1417
+ * Progressive text reveal — the "karaoke" capability, opt-in.
1418
+ * 'off' — no reveal events (CLI default; the host renders text however it likes).
1419
+ * 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
1420
+ * in step with the model stream).
1421
+ * 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
1422
+ * timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
1423
+ */
1424
+ revealMode: 'off' | 'delta' | 'word';
1425
+ /** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
1426
+ * Reset to '' at the start of each spoken turn. */
1427
+ onReveal: (revealed: string) => void;
1412
1428
  /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
1413
1429
  clock: EngineClock;
1414
1430
  /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
@@ -1426,6 +1442,10 @@ declare class VoiceEngine {
1426
1442
  private ctxOpen;
1427
1443
  private interrupted;
1428
1444
  private spokeDeltas;
1445
+ private revealText;
1446
+ private wordStarts;
1447
+ private revealedN;
1448
+ private revealPoll;
1429
1449
  private clock;
1430
1450
  private drainTimer;
1431
1451
  private echoWords;
@@ -1490,6 +1510,15 @@ declare class VoiceEngine {
1490
1510
  /** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
1491
1511
  * host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
1492
1512
  speakDelta(text: string): string;
1513
+ /** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
1514
+ * Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
1515
+ * paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
1516
+ * deterministic under the virtual clock. */
1517
+ private startWordReveal;
1518
+ /** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
1519
+ * undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
1520
+ * is what was actually spoken). */
1521
+ private stopWordReveal;
1493
1522
  /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
1494
1523
  endSpeech(): void;
1495
1524
  /** text of the reply cut by the last barge-in — consumed by the host to tell the model what
@@ -1673,6 +1702,12 @@ declare class CartesiaTTS {
1673
1702
  ctxId: string;
1674
1703
  onAudio: (chunk: Uint8Array) => void;
1675
1704
  onDone: () => void;
1705
+ /** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
1706
+ * across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
1707
+ onTimestamps: (words: string[], start: number[]) => void;
1708
+ /** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
1709
+ * when revealMode==='word'. */
1710
+ wantTimestamps: boolean;
1676
1711
  /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
1677
1712
  * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
1678
1713
  onDiag: (ev: {
package/dist/index.js CHANGED
@@ -5012,6 +5012,19 @@ var VoiceEngineOptions = class {
5012
5012
  emotions = true;
5013
5013
  /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
5014
5014
  showEmotions = false;
5015
+ /**
5016
+ * Progressive text reveal — the "karaoke" capability, opt-in.
5017
+ * 'off' — no reveal events (CLI default; the host renders text however it likes).
5018
+ * 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
5019
+ * in step with the model stream).
5020
+ * 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
5021
+ * timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
5022
+ */
5023
+ revealMode = "off";
5024
+ /** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
5025
+ * Reset to '' at the start of each spoken turn. */
5026
+ onReveal = () => {
5027
+ };
5015
5028
  /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
5016
5029
  clock = realClock;
5017
5030
  /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
@@ -5034,6 +5047,12 @@ var VoiceEngine = class _VoiceEngine {
5034
5047
  // barge-in latch: drop in-flight deltas until the next legitimate turn
5035
5048
  spokeDeltas = false;
5036
5049
  // a TTS context is open for the current spoken turn
5050
+ revealText = "";
5051
+ // cumulative revealed (on-screen) text for the current turn — see revealMode
5052
+ // 'word' karaoke reveal: spoken-word start times (sec, absolute from turn-audio start) + poll state.
5053
+ wordStarts = [];
5054
+ revealedN = 0;
5055
+ revealPoll = null;
5037
5056
  clock;
5038
5057
  drainTimer = null;
5039
5058
  // heuristic tier state (inert under AEC) — frozen as validated in the experiment
@@ -5121,6 +5140,13 @@ var VoiceEngine = class _VoiceEngine {
5121
5140
  this.tts.onAudio = (c) => {
5122
5141
  if (this.speaking || this.bcActive) this.player.write(c);
5123
5142
  };
5143
+ if (this.options.revealMode === "word") {
5144
+ this.tts.wantTimestamps = true;
5145
+ this.tts.onTimestamps = (_words, start) => {
5146
+ if (this.interrupted) return;
5147
+ for (const s of start) this.wordStarts.push(s);
5148
+ };
5149
+ }
5124
5150
  this.stt.onPartial = (text) => this.handlePartial(text);
5125
5151
  this.stt.onUtterance = (text) => this.handleUtterance(text);
5126
5152
  this.stt.onLevel = (rms) => this.handleLevel(rms);
@@ -5186,6 +5212,10 @@ var VoiceEngine = class _VoiceEngine {
5186
5212
  this.ctxOpen = true;
5187
5213
  this.spokeDeltas = false;
5188
5214
  this.reply = "";
5215
+ this.revealText = "";
5216
+ this.wordStarts = [];
5217
+ this.revealedN = 0;
5218
+ if (this.options.revealMode === "word") this.startWordReveal();
5189
5219
  this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
5190
5220
  this.echoWords = new Set(this.words(this.prevReply));
5191
5221
  this.tts.newContext();
@@ -5234,8 +5264,46 @@ var VoiceEngine = class _VoiceEngine {
5234
5264
  if (!this.spokeDeltas && this.turnStartAt) log10.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
5235
5265
  this.spokeDeltas = true;
5236
5266
  this.setState("speaking");
5267
+ if (this.options.revealMode !== "off" && display) {
5268
+ this.revealText += display;
5269
+ if (this.options.revealMode === "delta") this.options.onReveal(this.revealText);
5270
+ }
5237
5271
  return display;
5238
5272
  }
5273
+ /** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
5274
+ * Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
5275
+ * paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
5276
+ * deterministic under the virtual clock. */
5277
+ startWordReveal() {
5278
+ if (this.revealPoll) return;
5279
+ const tick = () => {
5280
+ this.revealPoll = null;
5281
+ if (!this.speaking || this.interrupted || this.options.revealMode !== "word") return;
5282
+ const playedSec = this.player.playedMs() / 1e3;
5283
+ let n = this.revealedN;
5284
+ while (n < this.wordStarts.length && this.wordStarts[n] <= playedSec) n++;
5285
+ if (n !== this.revealedN) {
5286
+ this.revealedN = n;
5287
+ const words = this.revealText.trim().split(/\s+/).filter(Boolean);
5288
+ this.options.onReveal(words.slice(0, n).join(" "));
5289
+ }
5290
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5291
+ };
5292
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5293
+ }
5294
+ /** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
5295
+ * undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
5296
+ * is what was actually spoken). */
5297
+ stopWordReveal(finalFlush) {
5298
+ if (this.revealPoll) {
5299
+ this.clock.clearTimeout(this.revealPoll);
5300
+ this.revealPoll = null;
5301
+ }
5302
+ if (finalFlush && this.options.revealMode === "word" && !this.interrupted) {
5303
+ const full = this.revealText.trim();
5304
+ if (full) this.options.onReveal(full);
5305
+ }
5306
+ }
5239
5307
  /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
5240
5308
  endSpeech() {
5241
5309
  this.interrupted = false;
@@ -5261,6 +5329,7 @@ var VoiceEngine = class _VoiceEngine {
5261
5329
  }
5262
5330
  this.drainTimer = null;
5263
5331
  this.speaking = false;
5332
+ this.stopWordReveal(true);
5264
5333
  if (this.turnStartAt) log10.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
5265
5334
  this.echoUntil = this.clock.now() + 2500;
5266
5335
  if (!this.usingAec) this.stt.reset();
@@ -5348,6 +5417,7 @@ var VoiceEngine = class _VoiceEngine {
5348
5417
  const droppedQueued = this.uttQueue.length;
5349
5418
  this.uttQueue = [];
5350
5419
  if (!this.speaking && !this.drainTimer) return;
5420
+ this.stopWordReveal(false);
5351
5421
  this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
5352
5422
  if (this.drainTimer) {
5353
5423
  this.clock.clearTimeout(this.drainTimer);
@@ -7272,6 +7342,13 @@ var CartesiaTTS = class _CartesiaTTS {
7272
7342
  };
7273
7343
  onDone = () => {
7274
7344
  };
7345
+ /** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
7346
+ * across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
7347
+ onTimestamps = () => {
7348
+ };
7349
+ /** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
7350
+ * when revealMode==='word'. */
7351
+ wantTimestamps = false;
7275
7352
  /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
7276
7353
  * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
7277
7354
  onDiag = () => {
@@ -7339,6 +7416,9 @@ var CartesiaTTS = class _CartesiaTTS {
7339
7416
  this.consecutiveErrors = 0;
7340
7417
  this.markRecovered();
7341
7418
  this.onDone();
7419
+ } else if (m.type === "timestamps" && m.word_timestamps) {
7420
+ const wt = m.word_timestamps;
7421
+ if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
7342
7422
  } else if (m.type === "error") {
7343
7423
  if (/already been cancelled|does not exist/.test(m.message || "")) return;
7344
7424
  this.consecutiveErrors++;
@@ -7395,7 +7475,8 @@ var CartesiaTTS = class _CartesiaTTS {
7395
7475
  voice: { mode: "id", id: this.options.voiceId },
7396
7476
  output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
7397
7477
  context_id: this.ctxId,
7398
- continue: cont
7478
+ continue: cont,
7479
+ ...this.wantTimestamps ? { add_timestamps: true } : {}
7399
7480
  });
7400
7481
  }
7401
7482
  speak(text, cont) {