@livx.cc/agentx 0.99.7 → 0.99.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -4966,6 +4966,19 @@ var VoiceEngineOptions = class {
4966
4966
  emotions = true;
4967
4967
  /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
4968
4968
  showEmotions = false;
4969
+ /**
4970
+ * Progressive text reveal — the "karaoke" capability, opt-in.
4971
+ * 'off' — no reveal events (CLI default; the host renders text however it likes).
4972
+ * 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
4973
+ * in step with the model stream).
4974
+ * 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
4975
+ * timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
4976
+ */
4977
+ revealMode = "off";
4978
+ /** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
4979
+ * Reset to '' at the start of each spoken turn. */
4980
+ onReveal = () => {
4981
+ };
4969
4982
  /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
4970
4983
  clock = realClock;
4971
4984
  /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
@@ -4988,6 +5001,12 @@ var VoiceEngine = class _VoiceEngine {
4988
5001
  // barge-in latch: drop in-flight deltas until the next legitimate turn
4989
5002
  spokeDeltas = false;
4990
5003
  // a TTS context is open for the current spoken turn
5004
+ revealText = "";
5005
+ // cumulative revealed (on-screen) text for the current turn — see revealMode
5006
+ // 'word' karaoke reveal: spoken-word start times (sec, absolute from turn-audio start) + poll state.
5007
+ wordStarts = [];
5008
+ revealedN = 0;
5009
+ revealPoll = null;
4991
5010
  clock;
4992
5011
  drainTimer = null;
4993
5012
  // heuristic tier state (inert under AEC) — frozen as validated in the experiment
@@ -5075,6 +5094,13 @@ var VoiceEngine = class _VoiceEngine {
5075
5094
  this.tts.onAudio = (c) => {
5076
5095
  if (this.speaking || this.bcActive) this.player.write(c);
5077
5096
  };
5097
+ if (this.options.revealMode === "word") {
5098
+ this.tts.wantTimestamps = true;
5099
+ this.tts.onTimestamps = (_words, start) => {
5100
+ if (this.interrupted) return;
5101
+ for (const s of start) this.wordStarts.push(s);
5102
+ };
5103
+ }
5078
5104
  this.stt.onPartial = (text) => this.handlePartial(text);
5079
5105
  this.stt.onUtterance = (text) => this.handleUtterance(text);
5080
5106
  this.stt.onLevel = (rms) => this.handleLevel(rms);
@@ -5140,6 +5166,10 @@ var VoiceEngine = class _VoiceEngine {
5140
5166
  this.ctxOpen = true;
5141
5167
  this.spokeDeltas = false;
5142
5168
  this.reply = "";
5169
+ this.revealText = "";
5170
+ this.wordStarts = [];
5171
+ this.revealedN = 0;
5172
+ if (this.options.revealMode === "word") this.startWordReveal();
5143
5173
  this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
5144
5174
  this.echoWords = new Set(this.words(this.prevReply));
5145
5175
  this.tts.newContext();
@@ -5188,8 +5218,46 @@ var VoiceEngine = class _VoiceEngine {
5188
5218
  if (!this.spokeDeltas && this.turnStartAt) log9.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
5189
5219
  this.spokeDeltas = true;
5190
5220
  this.setState("speaking");
5221
+ if (this.options.revealMode !== "off" && display) {
5222
+ this.revealText += display;
5223
+ if (this.options.revealMode === "delta") this.options.onReveal(this.revealText);
5224
+ }
5191
5225
  return display;
5192
5226
  }
5227
+ /** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
5228
+ * Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
5229
+ * paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
5230
+ * deterministic under the virtual clock. */
5231
+ startWordReveal() {
5232
+ if (this.revealPoll) return;
5233
+ const tick = () => {
5234
+ this.revealPoll = null;
5235
+ if (!this.speaking || this.interrupted || this.options.revealMode !== "word") return;
5236
+ const playedSec = this.player.playedMs() / 1e3;
5237
+ let n = this.revealedN;
5238
+ while (n < this.wordStarts.length && this.wordStarts[n] <= playedSec) n++;
5239
+ if (n !== this.revealedN) {
5240
+ this.revealedN = n;
5241
+ const words = this.revealText.trim().split(/\s+/).filter(Boolean);
5242
+ this.options.onReveal(words.slice(0, n).join(" "));
5243
+ }
5244
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5245
+ };
5246
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5247
+ }
5248
+ /** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
5249
+ * undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
5250
+ * is what was actually spoken). */
5251
+ stopWordReveal(finalFlush) {
5252
+ if (this.revealPoll) {
5253
+ this.clock.clearTimeout(this.revealPoll);
5254
+ this.revealPoll = null;
5255
+ }
5256
+ if (finalFlush && this.options.revealMode === "word" && !this.interrupted) {
5257
+ const full = this.revealText.trim();
5258
+ if (full) this.options.onReveal(full);
5259
+ }
5260
+ }
5193
5261
  /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
5194
5262
  endSpeech() {
5195
5263
  this.interrupted = false;
@@ -5215,6 +5283,7 @@ var VoiceEngine = class _VoiceEngine {
5215
5283
  }
5216
5284
  this.drainTimer = null;
5217
5285
  this.speaking = false;
5286
+ this.stopWordReveal(true);
5218
5287
  if (this.turnStartAt) log9.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
5219
5288
  this.echoUntil = this.clock.now() + 2500;
5220
5289
  if (!this.usingAec) this.stt.reset();
@@ -5302,6 +5371,7 @@ var VoiceEngine = class _VoiceEngine {
5302
5371
  const droppedQueued = this.uttQueue.length;
5303
5372
  this.uttQueue = [];
5304
5373
  if (!this.speaking && !this.drainTimer) return;
5374
+ this.stopWordReveal(false);
5305
5375
  this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
5306
5376
  if (this.drainTimer) {
5307
5377
  this.clock.clearTimeout(this.drainTimer);
@@ -7168,6 +7238,13 @@ var CartesiaTTS = class _CartesiaTTS {
7168
7238
  };
7169
7239
  onDone = () => {
7170
7240
  };
7241
+ /** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
7242
+ * across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
7243
+ onTimestamps = () => {
7244
+ };
7245
+ /** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
7246
+ * when revealMode==='word'. */
7247
+ wantTimestamps = false;
7171
7248
  /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
7172
7249
  * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
7173
7250
  onDiag = () => {
@@ -7235,6 +7312,9 @@ var CartesiaTTS = class _CartesiaTTS {
7235
7312
  this.consecutiveErrors = 0;
7236
7313
  this.markRecovered();
7237
7314
  this.onDone();
7315
+ } else if (m.type === "timestamps" && m.word_timestamps) {
7316
+ const wt = m.word_timestamps;
7317
+ if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
7238
7318
  } else if (m.type === "error") {
7239
7319
  if (/already been cancelled|does not exist/.test(m.message || "")) return;
7240
7320
  this.consecutiveErrors++;
@@ -7291,7 +7371,8 @@ var CartesiaTTS = class _CartesiaTTS {
7291
7371
  voice: { mode: "id", id: this.options.voiceId },
7292
7372
  output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
7293
7373
  context_id: this.ctxId,
7294
- continue: cont
7374
+ continue: cont,
7375
+ ...this.wantTimestamps ? { add_timestamps: true } : {}
7295
7376
  });
7296
7377
  }
7297
7378
  speak(text, cont) {