@livx.cc/agentx 0.99.7 → 0.99.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +82 -1
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +35 -0
- package/dist/index.js +82 -1
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -4966,6 +4966,19 @@ var VoiceEngineOptions = class {
|
|
|
4966
4966
|
emotions = true;
|
|
4967
4967
|
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
4968
4968
|
showEmotions = false;
|
|
4969
|
+
/**
|
|
4970
|
+
* Progressive text reveal — the "karaoke" capability, opt-in.
|
|
4971
|
+
* 'off' — no reveal events (CLI default; the host renders text however it likes).
|
|
4972
|
+
* 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
|
|
4973
|
+
* in step with the model stream).
|
|
4974
|
+
* 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
|
|
4975
|
+
* timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
|
|
4976
|
+
*/
|
|
4977
|
+
revealMode = "off";
|
|
4978
|
+
/** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
|
|
4979
|
+
* Reset to '' at the start of each spoken turn. */
|
|
4980
|
+
onReveal = () => {
|
|
4981
|
+
};
|
|
4969
4982
|
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
4970
4983
|
clock = realClock;
|
|
4971
4984
|
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
@@ -4988,6 +5001,12 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
4988
5001
|
// barge-in latch: drop in-flight deltas until the next legitimate turn
|
|
4989
5002
|
spokeDeltas = false;
|
|
4990
5003
|
// a TTS context is open for the current spoken turn
|
|
5004
|
+
revealText = "";
|
|
5005
|
+
// cumulative revealed (on-screen) text for the current turn — see revealMode
|
|
5006
|
+
// 'word' karaoke reveal: spoken-word start times (sec, absolute from turn-audio start) + poll state.
|
|
5007
|
+
wordStarts = [];
|
|
5008
|
+
revealedN = 0;
|
|
5009
|
+
revealPoll = null;
|
|
4991
5010
|
clock;
|
|
4992
5011
|
drainTimer = null;
|
|
4993
5012
|
// heuristic tier state (inert under AEC) — frozen as validated in the experiment
|
|
@@ -5075,6 +5094,13 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5075
5094
|
this.tts.onAudio = (c) => {
|
|
5076
5095
|
if (this.speaking || this.bcActive) this.player.write(c);
|
|
5077
5096
|
};
|
|
5097
|
+
if (this.options.revealMode === "word") {
|
|
5098
|
+
this.tts.wantTimestamps = true;
|
|
5099
|
+
this.tts.onTimestamps = (_words, start) => {
|
|
5100
|
+
if (this.interrupted) return;
|
|
5101
|
+
for (const s of start) this.wordStarts.push(s);
|
|
5102
|
+
};
|
|
5103
|
+
}
|
|
5078
5104
|
this.stt.onPartial = (text) => this.handlePartial(text);
|
|
5079
5105
|
this.stt.onUtterance = (text) => this.handleUtterance(text);
|
|
5080
5106
|
this.stt.onLevel = (rms) => this.handleLevel(rms);
|
|
@@ -5140,6 +5166,10 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5140
5166
|
this.ctxOpen = true;
|
|
5141
5167
|
this.spokeDeltas = false;
|
|
5142
5168
|
this.reply = "";
|
|
5169
|
+
this.revealText = "";
|
|
5170
|
+
this.wordStarts = [];
|
|
5171
|
+
this.revealedN = 0;
|
|
5172
|
+
if (this.options.revealMode === "word") this.startWordReveal();
|
|
5143
5173
|
this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
|
|
5144
5174
|
this.echoWords = new Set(this.words(this.prevReply));
|
|
5145
5175
|
this.tts.newContext();
|
|
@@ -5188,8 +5218,46 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5188
5218
|
if (!this.spokeDeltas && this.turnStartAt) log9.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
|
|
5189
5219
|
this.spokeDeltas = true;
|
|
5190
5220
|
this.setState("speaking");
|
|
5221
|
+
if (this.options.revealMode !== "off" && display) {
|
|
5222
|
+
this.revealText += display;
|
|
5223
|
+
if (this.options.revealMode === "delta") this.options.onReveal(this.revealText);
|
|
5224
|
+
}
|
|
5191
5225
|
return display;
|
|
5192
5226
|
}
|
|
5227
|
+
/** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
|
|
5228
|
+
* Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
|
|
5229
|
+
* paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
|
|
5230
|
+
* deterministic under the virtual clock. */
|
|
5231
|
+
startWordReveal() {
|
|
5232
|
+
if (this.revealPoll) return;
|
|
5233
|
+
const tick = () => {
|
|
5234
|
+
this.revealPoll = null;
|
|
5235
|
+
if (!this.speaking || this.interrupted || this.options.revealMode !== "word") return;
|
|
5236
|
+
const playedSec = this.player.playedMs() / 1e3;
|
|
5237
|
+
let n = this.revealedN;
|
|
5238
|
+
while (n < this.wordStarts.length && this.wordStarts[n] <= playedSec) n++;
|
|
5239
|
+
if (n !== this.revealedN) {
|
|
5240
|
+
this.revealedN = n;
|
|
5241
|
+
const words = this.revealText.trim().split(/\s+/).filter(Boolean);
|
|
5242
|
+
this.options.onReveal(words.slice(0, n).join(" "));
|
|
5243
|
+
}
|
|
5244
|
+
this.revealPoll = this.clock.setTimeout(tick, 60);
|
|
5245
|
+
};
|
|
5246
|
+
this.revealPoll = this.clock.setTimeout(tick, 60);
|
|
5247
|
+
}
|
|
5248
|
+
/** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
|
|
5249
|
+
* undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
|
|
5250
|
+
* is what was actually spoken). */
|
|
5251
|
+
stopWordReveal(finalFlush) {
|
|
5252
|
+
if (this.revealPoll) {
|
|
5253
|
+
this.clock.clearTimeout(this.revealPoll);
|
|
5254
|
+
this.revealPoll = null;
|
|
5255
|
+
}
|
|
5256
|
+
if (finalFlush && this.options.revealMode === "word" && !this.interrupted) {
|
|
5257
|
+
const full = this.revealText.trim();
|
|
5258
|
+
if (full) this.options.onReveal(full);
|
|
5259
|
+
}
|
|
5260
|
+
}
|
|
5193
5261
|
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
5194
5262
|
endSpeech() {
|
|
5195
5263
|
this.interrupted = false;
|
|
@@ -5215,6 +5283,7 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5215
5283
|
}
|
|
5216
5284
|
this.drainTimer = null;
|
|
5217
5285
|
this.speaking = false;
|
|
5286
|
+
this.stopWordReveal(true);
|
|
5218
5287
|
if (this.turnStartAt) log9.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
|
|
5219
5288
|
this.echoUntil = this.clock.now() + 2500;
|
|
5220
5289
|
if (!this.usingAec) this.stt.reset();
|
|
@@ -5302,6 +5371,7 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5302
5371
|
const droppedQueued = this.uttQueue.length;
|
|
5303
5372
|
this.uttQueue = [];
|
|
5304
5373
|
if (!this.speaking && !this.drainTimer) return;
|
|
5374
|
+
this.stopWordReveal(false);
|
|
5305
5375
|
this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
|
|
5306
5376
|
if (this.drainTimer) {
|
|
5307
5377
|
this.clock.clearTimeout(this.drainTimer);
|
|
@@ -7168,6 +7238,13 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7168
7238
|
};
|
|
7169
7239
|
onDone = () => {
|
|
7170
7240
|
};
|
|
7241
|
+
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
7242
|
+
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
7243
|
+
onTimestamps = () => {
|
|
7244
|
+
};
|
|
7245
|
+
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
7246
|
+
* when revealMode==='word'. */
|
|
7247
|
+
wantTimestamps = false;
|
|
7171
7248
|
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
7172
7249
|
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
7173
7250
|
onDiag = () => {
|
|
@@ -7235,6 +7312,9 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7235
7312
|
this.consecutiveErrors = 0;
|
|
7236
7313
|
this.markRecovered();
|
|
7237
7314
|
this.onDone();
|
|
7315
|
+
} else if (m.type === "timestamps" && m.word_timestamps) {
|
|
7316
|
+
const wt = m.word_timestamps;
|
|
7317
|
+
if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
|
|
7238
7318
|
} else if (m.type === "error") {
|
|
7239
7319
|
if (/already been cancelled|does not exist/.test(m.message || "")) return;
|
|
7240
7320
|
this.consecutiveErrors++;
|
|
@@ -7291,7 +7371,8 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
7291
7371
|
voice: { mode: "id", id: this.options.voiceId },
|
|
7292
7372
|
output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
|
|
7293
7373
|
context_id: this.ctxId,
|
|
7294
|
-
continue: cont
|
|
7374
|
+
continue: cont,
|
|
7375
|
+
...this.wantTimestamps ? { add_timestamps: true } : {}
|
|
7295
7376
|
});
|
|
7296
7377
|
}
|
|
7297
7378
|
speak(text, cont) {
|