@livx.cc/agentx 0.99.6 → 0.99.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1149,11 +1149,11 @@ var init_NodeDiskFilesystem = __esm({
1149
1149
  if (rel === "" || rel.startsWith("..")) return;
1150
1150
  const parts = rel.split(np.sep);
1151
1151
  let cur = this.baseDir;
1152
- const now4 = Date.now();
1152
+ const now3 = Date.now();
1153
1153
  for (let i = 0; i < parts.length; i++) {
1154
1154
  cur = np.join(cur, parts[i]);
1155
1155
  const isLeaf = i === parts.length - 1;
1156
- if (!isLeaf && (this.verified.get(cur) ?? 0) > now4) continue;
1156
+ if (!isLeaf && (this.verified.get(cur) ?? 0) > now3) continue;
1157
1157
  let st;
1158
1158
  try {
1159
1159
  st = await fsp.lstat(cur);
@@ -1163,7 +1163,7 @@ var init_NodeDiskFilesystem = __esm({
1163
1163
  if (st.isSymbolicLink()) throw new Error("File not found: symlink not permitted");
1164
1164
  if (!isLeaf) {
1165
1165
  if (this.verified.size > 1e4) this.verified.clear();
1166
- this.verified.set(cur, now4 + _NodeDiskFilesystem.VERIFY_TTL_MS);
1166
+ this.verified.set(cur, now3 + _NodeDiskFilesystem.VERIFY_TTL_MS);
1167
1167
  }
1168
1168
  }
1169
1169
  }
@@ -4345,12 +4345,12 @@ var Scheduler = class {
4345
4345
  if (this.firing) return;
4346
4346
  this.firing = true;
4347
4347
  try {
4348
- const now4 = this.now();
4348
+ const now3 = this.now();
4349
4349
  for (const job of this.jobs.values()) {
4350
4350
  if (job.status !== "active") continue;
4351
4351
  const due = this.nextFire(job);
4352
- if (due == null || due > now4) continue;
4353
- job.lastRun = now4;
4352
+ if (due == null || due > now3) continue;
4353
+ job.lastRun = now3;
4354
4354
  job.runs++;
4355
4355
  if ("at" in job.trigger) job.status = "done";
4356
4356
  try {
@@ -4883,1597 +4883,2262 @@ var EmotionStream = class {
4883
4883
  }
4884
4884
  };
4885
4885
 
4886
- // src/voice/spokenSplitter.ts
4887
- var OPEN = "<spoken>";
4888
- var CLOSE = "</spoken>";
4889
- var CLOSERS = `"')]}\xBB\u201D\u2019`;
4890
- var hasSpeech = (s) => /[\p{L}\p{N}]/u.test(s);
4891
- var SentenceCoalescer = class _SentenceCoalescer {
4892
- buf = "";
4893
- static isEnd(c) {
4894
- return c === "\n" || c === "." || c === "!" || c === "?" || c === "\u2026";
4886
+ // src/voice/engine.ts
4887
+ init_logging();
4888
+ var log10 = forComponent("VoiceEngine");
4889
+ var realClock = {
4890
+ now: () => performance.now(),
4891
+ setTimeout: (fn, ms) => setTimeout(fn, ms),
4892
+ clearTimeout: (h) => clearTimeout(h)
4893
+ };
4894
+ var forSpeech = (t) => t.replace(/[*`#]+/g, "").replace(/(?<![\p{L}\p{N}])_([^_\n]+)_(?![\p{L}\p{N}])/gu, "$1").replace(/^[ \t]*[-•]\s+/gm, "").replace(/\s*[\u2013\u2014]\s*/g, ", ").replace(/[\u2010\u2011]/g, "-").replace(/\s*\|\s*/g, ", ").replace(/(\d)\s+%/g, "$1%").replace(/\.{3,}/g, ".");
4895
+ var normWord = (w) => w.toLowerCase().replace(/[^a-z0-9]/g, "");
4896
+ var SPEC_TRAILING_OK = /* @__PURE__ */ new Set(["please", "thanks", "thank", "you", "now", "okay", "ok", "kindly", "alright", "then", "though", "right", "yeah"]);
4897
+ function speculationConfirms(spec, final) {
4898
+ const sw = spec.trim().split(/\s+/).map(normWord).filter(Boolean);
4899
+ const fw = final.trim().split(/\s+/).map(normWord).filter(Boolean);
4900
+ if (!sw.length || fw.length < sw.length || fw.length > sw.length + 2) return false;
4901
+ for (let i = 0; i < sw.length - 1; i++) if (sw[i] !== fw[i]) return false;
4902
+ if (!fw[sw.length - 1].startsWith(sw[sw.length - 1])) return false;
4903
+ for (let i = sw.length; i < fw.length; i++) if (!SPEC_TRAILING_OK.has(fw[i])) return false;
4904
+ return true;
4905
+ }
4906
+ var VoiceEngineOptions = class {
4907
+ stt;
4908
+ tts;
4909
+ player;
4910
+ /** a final utterance arrived (endpoint) — host dispatches it as a turn */
4911
+ onUtterance = () => {
4912
+ };
4913
+ /** live partial transcript while listening (host renders the 🎤 line) */
4914
+ onPartial = () => {
4915
+ };
4916
+ onState = () => {
4917
+ };
4918
+ /** user spoke/acted over playback — host aborts the in-flight turn (called AFTER audio is killed).
4919
+ * phase: 'speaking' = cut mid-speech (real interruption); 'drain' = in the final audio tail
4920
+ * (normal turn-taking — hosts shouldn't alarm). */
4921
+ onBargeIn = () => {
4922
+ };
4923
+ /** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
4924
+ ackPhrase = "";
4925
+ /** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
4926
+ * delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
4927
+ * per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
4928
+ adaptiveAckMs = 0;
4929
+ /** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
4930
+ onAdaptiveAck = () => {
4931
+ };
4932
+ /** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
4933
+ * letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
4934
+ * ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
4935
+ utteranceMergeMs = 350;
4936
+ /** Extended merge window (ms) for utterances that look incomplete (trailing conjunction/filler).
4937
+ * Gives the user time to finish their thought without triggering a model call. */
4938
+ incompleteMergeMs = 1500;
4939
+ /** Grace window (ms) after an utterance dispatches, during which the user's own trailing audio cannot
4940
+ * barge the reply it requested. Soniox keeps finalizing partials past <end>; without this they read
4941
+ * as a barge and abort the fresh turn (live: mid-sentence self-interruption + steps=1→steps=0 double
4942
+ * abort). Short enough that a genuine immediate barge ("no wait—") still lands right after. */
4943
+ bargeGraceMs = 600;
4944
+ /** Barge-in (talk over the assistant to interrupt). true = full-duplex (needs echo cancellation, or
4945
+ * the assistant's own TTS bleeds back and self-interrupts). false = HALF-DUPLEX: the engine is deaf
4946
+ * while audible (speaking + drain tail), so echo can never become a phantom turn — the right mode
4947
+ * when there's no AEC (e.g. the non-VPIO mic fallback) and no headphones. Cost: can't interrupt. */
4948
+ bargeIn = true;
4949
+ /** Filler phrase spoken when holding for an incomplete utterance ('' disables). */
4950
+ holdFiller = "";
4951
+ /** Called when the engine holds an incomplete utterance (host can render a visual cue). */
4952
+ onHold = () => {
4953
+ };
4954
+ /** heuristic (non-AEC) energy barge-in tuning */
4955
+ bargeRmsMult = 2;
4956
+ bargeRmsFloor = 500;
4957
+ /** Overlap turn-taking (AEC tier, needs player.pause/resume) — human phone-call model, driven by
4958
+ * the STT ITSELF (a trained speech classifier) instead of energy thresholds (energy could not
4959
+ * separate residue bursts from speech in every room — hiccup whack-a-mole): a GENUINE partial
4960
+ * (novel words dominate — echo of our own reply is inert) while speaking → PAUSE (exact-sample
4961
+ * hold); partial grows into dominant-novel ≥2 words → cede (interrupt; the LLM re-enters); partial
4962
+ * stalls/endpoints without ceding (backchannel by DURATION, not vocabulary) → resume + drop. false disables. */
4963
+ overlapPause = true;
4964
+ /** no new partial activity for this long while paused → resume, drop the interjection */
4965
+ overlapResumeMs = 700;
4966
+ /** A genuine barge over a LONG reply is defeated by the dominant-novel gate: Meet echoes our own
4967
+ * speech back, so the partial is mostly our words + a few of hers → never "dominant novel" → it
4968
+ * resumes (replaying old audio — the audible "completes the buffer" blip) instead of ceding.
4969
+ * Mechanism-based discriminator: a re-PAUSE this soon after a resume = a persistent human, not an
4970
+ * echo blip (which pauses once and stalls). Cede on the re-pause regardless of the novel gate. */
4971
+ overlapRepauseCedeMs = 1500;
4972
+ /** Speculative ENERGY pre-pause while speaking (AEC tier): two residue gate-passes within 350ms →
4973
+ * pause ~300ms before the STT tokens land. But energy CANNOT separate residue bursts from speech
4974
+ * (the documented whack-a-mole) — so a residue spike during loud playback false-pauses with NO user
4975
+ * speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
4976
+ * mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
4977
+ overlapEnergyHold = false;
4978
+ /** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
4979
+ * this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
4980
+ * start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
4981
+ * speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
4982
+ * confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
4983
+ speculativeMs = 0;
4984
+ /** Minimum word count for a partial to qualify as a speculation trigger. */
4985
+ speculativeMinWords = 4;
4986
+ /** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
4987
+ onSpeculate = () => {
4988
+ };
4989
+ /** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
4990
+ * partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
4991
+ * UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
4992
+ * quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
4993
+ * touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
4994
+ * context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
4995
+ * boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
4996
+ * 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
4997
+ * ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
4998
+ backchannelMs = 0;
4999
+ /** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
5000
+ backchannelMinGapMs = 8e3;
5001
+ /** Only multi-clause turns: the partial must carry at least this many words before a blip. */
5002
+ backchannelMinWords = 8;
5003
+ /** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
5004
+ onBackchannel = () => {
5005
+ };
5006
+ /** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
5007
+ * quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
5008
+ onSpeculateAbort = () => {
5009
+ };
5010
+ /** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
5011
+ * tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
5012
+ emotions = true;
5013
+ /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
5014
+ showEmotions = false;
5015
+ /**
5016
+ * Progressive text reveal — the "karaoke" capability, opt-in.
5017
+ * 'off' — no reveal events (CLI default; the host renders text however it likes).
5018
+ * 'delta' — emit `onReveal(fullText)` as each spoken delta lands (near-free; text grows
5019
+ * in step with the model stream).
5020
+ * 'word' — reveal word-by-word synced to the TTS playback CLOCK (true karaoke). Requires a
5021
+ * timestamp-emitting TTS (CartesiaTTS.wantTimestamps) driven over `AudioSink.playedMs`.
5022
+ */
5023
+ revealMode = "off";
5024
+ /** Called with the cumulative revealed text for the current assistant turn (see `revealMode`).
5025
+ * Reset to '' at the start of each spoken turn. */
5026
+ onReveal = () => {
5027
+ };
5028
+ /** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
5029
+ clock = realClock;
5030
+ /** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
5031
+ * barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
5032
+ * a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
5033
+ onDiag = () => {
5034
+ };
5035
+ };
5036
+ var VoiceEngine = class _VoiceEngine {
5037
+ options;
5038
+ state = "idle";
5039
+ stt;
5040
+ tts;
5041
+ player;
5042
+ speaking = false;
5043
+ // audible (deltas flowing OR audio draining)
5044
+ ctxOpen = false;
5045
+ // the current TTS context still accepts deltas (false once end-frame sent)
5046
+ interrupted = false;
5047
+ // barge-in latch: drop in-flight deltas until the next legitimate turn
5048
+ spokeDeltas = false;
5049
+ // a TTS context is open for the current spoken turn
5050
+ revealText = "";
5051
+ // cumulative revealed (on-screen) text for the current turn — see revealMode
5052
+ // 'word' karaoke reveal: spoken-word start times (sec, absolute from turn-audio start) + poll state.
5053
+ wordStarts = [];
5054
+ revealedN = 0;
5055
+ revealPoll = null;
5056
+ clock;
5057
+ drainTimer = null;
5058
+ // heuristic tier state (inert under AEC) — frozen as validated in the experiment
5059
+ echoWords = /* @__PURE__ */ new Set();
5060
+ prevReply = "";
5061
+ reply = "";
5062
+ echoUntil = 0;
5063
+ baseline = 0;
5064
+ hot = 0;
5065
+ suspectUntil = 0;
5066
+ ackAt = 0;
5067
+ // when the micro-ack was spoken — its echo can leak before the AEC filter converges
5068
+ lastAck = "";
5069
+ // the exact ack text last spoken (fixed OR adaptive) — the echo-leak guard matches it
5070
+ ackTimer = null;
5071
+ // one-shot adaptive-ack timer (armed at dispatch, cancelled on first delta)
5072
+ bargeGraceUntil = 0;
5073
+ // no barge-in until this time — the user's OWN trailing audio (after the
5074
+ // utterance that JUST dispatched this turn) must not immediately re-interrupt the reply it requested.
5075
+ pendingUtt = "";
5076
+ // endpointed text held for the merge window
5077
+ mergePath = "direct";
5078
+ // how the pending utterance was assembled (diag)
5079
+ lastGraceDiag = 0;
5080
+ // grace_suppress emitted once per grace window
5081
+ pendingTimer = null;
5082
+ // Duplicate-final guard: STT sometimes re-finalizes the SAME audio a beat later (past the merge
5083
+ // window) — the identical utterance dispatched twice with no reply in between (live: "When you
5084
+ // start" twice). Conservative: only an IDENTICAL (normalized) text, within dupFinalMs, with zero
5085
+ // reply deltas since the first dispatch, is dropped. A user genuinely repeating themselves after
5086
+ // the agent replied (or after 3s) still dispatches.
5087
+ lastDispatchFlat = "";
5088
+ lastDispatchWords = [];
5089
+ lastDispatchAt = 0;
5090
+ repliedSinceDispatch = false;
5091
+ static DUP_FINAL_MS = 3e3;
5092
+ lastInterrupted = null;
5093
+ // overlap (pause) tier state — AEC + pause-capable sinks only
5094
+ pausedAt = 0;
5095
+ lastResumeAt = 0;
5096
+ // when the overlap last resumed from a false alarm — a quick re-pause cedes
5097
+ lastOverlapPartial = "";
5098
+ // change-detection: only NEW partial text counts as activity
5099
+ resumeTimer = null;
5100
+ turnStartAt = 0;
5101
+ // timestamp when the current turn began (for TTFT logging)
5102
+ // speculative reflex trigger state (options.speculativeMs) — see trackSpeculation
5103
+ specPartial = "";
5104
+ // last partial observed (stability = unchanged for speculativeMs)
5105
+ specTimer = null;
5106
+ specText = "";
5107
+ // text handed to onSpeculate ('' = no speculation in flight)
5108
+ specSpent = false;
5109
+ // at most one speculation per turn-in-progress
5110
+ // backchannel blip state (options.backchannelMs) — see trackBackchannel
5111
+ bcPartial = "";
5112
+ // last partial observed (stability = unchanged for backchannelMs)
5113
+ bcTimer = null;
5114
+ bcCount = 0;
5115
+ // blips this user turn-in-progress (max 2; reset at dispatch)
5116
+ bcActive = false;
5117
+ // NARROW audio-gate bypass: blip audio may reach the sink while listening
5118
+ bcActiveTimer = null;
5119
+ // safety: clear the bypass even if the blip's 'done' is lost
5120
+ lastBcAt = 0;
5121
+ lastBcPhrase = "";
5122
+ // exact blip text last spoken — flushUtterance strips its mic echo
5123
+ recentBc = [];
5124
+ // Central speech queue (above the TTS context): complete worker utterances serialize into ONE
5125
+ // playback stream, one-at-a-time, never splicing into the live reflex's open utterance.
5126
+ uttQueue = [];
5127
+ // Per-turn emotion-tag parser (reset on beginSpeech) — converts `[emotion]` → Cartesia inline tags
5128
+ // for TTS, tracks tag-free prose for echo discrimination, and surfaces display text for the screen.
5129
+ emo = null;
5130
+ constructor(options) {
5131
+ this.options = { ...new VoiceEngineOptions(), ...options };
5132
+ const o = this.options;
5133
+ if (!o.stt || !o.tts || !o.player) throw new Error("VoiceEngine needs stt, tts and player (see cli/voice.ts VoiceIO for platform defaults)");
5134
+ this.stt = o.stt;
5135
+ this.tts = o.tts;
5136
+ this.player = o.player;
5137
+ this.clock = o.clock;
4895
5138
  }
4896
- feed(delta) {
4897
- if (delta) this.buf += delta;
4898
- let cut = -1;
4899
- for (let i = 0; i < this.buf.length; i++) if (_SentenceCoalescer.isEnd(this.buf[i])) cut = i;
4900
- if (cut < 0) return "";
4901
- if (this.buf[cut] !== "\n") while (cut + 1 < this.buf.length && CLOSERS.includes(this.buf[cut + 1])) cut++;
4902
- const ready = this.buf.slice(0, cut + 1).trim();
4903
- this.buf = this.buf.slice(cut + 1);
4904
- return hasSpeech(ready) ? ready : "";
5139
+ async start() {
5140
+ this.tts.onAudio = (c) => {
5141
+ if (this.speaking || this.bcActive) this.player.write(c);
5142
+ };
5143
+ if (this.options.revealMode === "word") {
5144
+ this.tts.wantTimestamps = true;
5145
+ this.tts.onTimestamps = (_words, start) => {
5146
+ if (this.interrupted) return;
5147
+ for (const s of start) this.wordStarts.push(s);
5148
+ };
5149
+ }
5150
+ this.stt.onPartial = (text) => this.handlePartial(text);
5151
+ this.stt.onUtterance = (text) => this.handleUtterance(text);
5152
+ this.stt.onLevel = (rms) => this.handleLevel(rms);
5153
+ await Promise.all([this.tts.connect(), this.stt.start()]);
5154
+ this.tts.warmup?.();
5155
+ this.setState("listening");
5156
+ log10.debug(`voice I/O up (${this.stt.usingAec ? "AEC" : "heuristic echo"} capture)`);
4905
5157
  }
4906
- flush() {
4907
- const s = this.buf.trim();
4908
- this.buf = "";
4909
- return hasSpeech(s) ? s : "";
5158
+ get usingAec() {
5159
+ return this.stt.usingAec;
4910
5160
  }
4911
- };
4912
- var SpokenSplitter = class {
4913
- buf = "";
4914
- inSpoken = false;
4915
- /** True once any spoken char has ever been emitted (drives the no-spoken fallback). */
4916
- spokeAny = false;
4917
- /** Feed a delta; returns the spoken/detail spans completed by this chunk (either may be ''). */
4918
- feed(delta) {
4919
- this.buf += delta;
4920
- return this.drain(false);
5161
+ /** Flip barge-in at runtime (e.g. the mic fell back to non-VPIO → go half-duplex so echo can't leak). */
5162
+ setBargeIn(on) {
5163
+ this.options.bargeIn = on;
4921
5164
  }
4922
- /** Drain any buffered partial. A trailing `<…` that never completed a tag is emitted as detail. */
4923
- flush() {
4924
- return this.drain(true);
5165
+ /** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
5166
+ setShowEmotions(on) {
5167
+ this.options.showEmotions = on;
4925
5168
  }
4926
- drain(final) {
4927
- let spoken = "";
4928
- let detail = "";
4929
- while (this.buf.length) {
4930
- const tag = this.inSpoken ? CLOSE : OPEN;
4931
- const idx = this.buf.indexOf(tag);
4932
- if (idx >= 0) {
4933
- const text2 = this.buf.slice(0, idx);
4934
- if (this.inSpoken) spoken += text2;
4935
- else detail += text2;
4936
- this.buf = this.buf.slice(idx + tag.length);
4937
- this.inSpoken = !this.inSpoken;
4938
- continue;
4939
- }
4940
- const lt = this.buf.lastIndexOf("<");
4941
- const holdStart = lt >= 0 && tag.startsWith(this.buf.slice(lt)) ? lt : this.buf.length;
4942
- const text = this.buf.slice(0, holdStart);
4943
- if (this.inSpoken) spoken += text;
4944
- else detail += text;
4945
- this.buf = this.buf.slice(holdStart);
4946
- break;
5169
+ /** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
5170
+ * it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
5171
+ diagOn = true;
5172
+ diag(kind, fields) {
5173
+ if (!this.diagOn) return;
5174
+ try {
5175
+ this.options.onDiag({ t: this.clock.now(), kind, ...fields });
5176
+ } catch (e) {
5177
+ this.diagOn = false;
5178
+ log10.debug(`onDiag threw \u2014 diagnostics disabled: ${e instanceof Error ? e.message : e}`);
4947
5179
  }
4948
- if (final && this.buf) {
4949
- if (this.inSpoken) spoken += this.buf;
4950
- else detail += this.buf;
4951
- this.buf = "";
5180
+ }
5181
+ idleWaiters = [];
5182
+ setState(s) {
5183
+ if (this.state === s) return;
5184
+ this.diag("state", { from: this.state, to: s });
5185
+ this.state = s;
5186
+ this.options.onState(s);
5187
+ if (s !== "speaking" && s !== "thinking") {
5188
+ for (const r of this.idleWaiters.splice(0)) r();
4952
5189
  }
4953
- if (spoken.trim()) this.spokeAny = true;
4954
- return { spoken, detail };
4955
5190
  }
4956
- };
4957
-
4958
- // src/duplex.ts
4959
- var log10 = forComponent("DuplexAgent");
4960
- function describeCall(call) {
4961
- const v = call.args && Object.values(call.args).find((x) => typeof x === "string" && x.trim());
4962
- const hint = v ? ` (${String(v).replace(/\s+/g, " ").trim().slice(0, 48)})` : "";
4963
- return `${call.name}${hint}`;
4964
- }
4965
- var DuplexAgentOptions = class {
4966
- /** Any ai.libx.js AIClient — shared by all tiers (routed by model). */
4967
- ai;
4968
- /** The WORKER's filesystem (act + think). If omitted the worker keeps Agent's jailed-disk-at-cwd default. */
4969
- fs;
4970
- // The reflex IS the voice. 120b (not 20b) for channel discipline + instruction-following: the 20b
4971
- // mislabels gpt-oss harmony channels under load, leaking raw analysis into the spoken `final` channel
4972
- // (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
4973
- reflexModel = "groq/openai/gpt-oss-120b";
4974
- actModel = "anthropic/claude-sonnet-4-6";
4975
- /** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
4976
- thinkModel = "anthropic/claude-opus-4-8";
4977
- /** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
4978
- * provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
4979
- * cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
4980
- * leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
4981
- providerOptionsFor;
4982
- /** Escape hatches merged over the derived per-agent options. */
4983
- reflexOptions;
4984
- actOptions;
4985
- thinkOptions;
4986
- /** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
4987
- * the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
4988
- * pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
4989
- * this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
4990
- verifyActTasks = false;
4991
- /** Receives the voice text_delta stream + task lifecycle events. */
4992
- host;
4993
- /** How many recent transcript messages are rendered into a worker's brief. */
4994
- excerptTurns = 6;
4995
- /** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
4996
- * backchannels, impulsive first reactions before content (mimics real duplex conversation). */
4997
- voiceStyle = "neutral";
4998
- /** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
4999
- * TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
5000
- emotionTags = false;
5001
- /** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
5002
- * (post-spawn would race the worker's first edits). */
5003
- onTaskStart;
5004
- /** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
5005
- * air. Off by default — each update costs a voice turn (LLM call + speech). */
5006
- progressUpdates = false;
5007
- /** Min ms between progress re-voices per task. */
5008
- progressIntervalMs = 25e3;
5009
- /** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
5010
- * the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
5011
- * voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
5012
- askRelay = false;
5013
- /** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
5014
- askTimeoutMs = 12e4;
5015
- /** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
5016
- * bounding memory over a long-lived session. Running tasks are never evicted. */
5017
- maxTaskRecords = 50;
5018
- /** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
5019
- * (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
5020
- quickLook;
5021
- /** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
5022
- * tools directly (no delegation needed) and implicit capture guidance. */
5023
- memoryDir;
5024
- /** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
5025
- memoryUserDir;
5026
- };
5027
- var RESERVED_EVENT_MARKER = /\[task\b[^\]\n]*\b(?:completed|failed|progress|asks)\b/i;
5028
- var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
5029
- var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 give at most a half-sentence aside ("still on it \u2014 running tests now") and end your turn. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, just say "still on it" and end the turn \u2014 never poll it again and again in a loop. Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
5030
- var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
5031
- var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
5032
- var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
5033
- var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
5034
- var DuplexAgent = class _DuplexAgent {
5035
- options;
5036
- voice;
5037
- tasks = /* @__PURE__ */ new Map();
5038
- queue = Promise.resolve();
5039
- seq = 0;
5040
- pendingEvents = [];
5041
- /** Out-of-band follow-up attribution for the events coalescing into the next flush turn: TRUE iff ≥1 of
5042
- * the tasks being integrated was NON-CLEAN (early-stop/failure). Carried out-of-band on the enqueue call
5043
- * by the caller that KNOWS the outcome — a plain boolean the MODEL CANNOT PERTURB. It is NOT scanned from
5044
- * worker-authored event text (v1: an "Outcome:" substring over-stamped siblings) and NOT keyed on a brief
5045
- * string the reflex re-authors (v2: a paraphrased escalation brief missed the Set → followUp:false →
5046
- * RE-ENABLED unbounded auto-escalation, the dangerous runaway direction). See [[wrong-discriminator]] /
5047
- * [[drive-real-reflex]] / [[fakeaiclient-blind-to-wire-format]]. */
5048
- pendingNonClean = false;
5049
- flushQueued = false;
5050
- /** Per-voice-turn guards (reset by resetTurn at each turn's start). The reflex is a weak model:
5051
- * left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
5052
- * Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
5053
- turnDispatched = false;
5054
- // an Act/Think fired this turn
5055
- turnBriefs = /* @__PURE__ */ new Set();
5056
- // briefs dispatched this turn (detect identical re-dispatch)
5057
- spokeThisTurn = false;
5058
- // any non-empty text_delta streamed this turn
5059
- heldThisTurn = false;
5060
- // Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
5061
- nudging = false;
5062
- // re-ack pass in flight: block ALL tools, prevent recursion
5063
- reflexBuf = "";
5064
- // accumulated reflex text this turn (fabricated-event detection)
5065
- reflexForwarded = 0;
5066
- // chars of reflexBuf already forwarded to the host/TTS
5067
- fabricationCut = false;
5068
- // reflex emitted a reserved [task …] marker → suppress its tail
5069
- /** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
5070
- * carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
5071
- * turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
5072
- * escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
5073
- * ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
5074
- * means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
5075
- * ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
5076
- * flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
5077
- * fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
5078
- * ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
5079
- turnFollowUp = false;
5080
- /** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
5081
- * the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
5082
- autoEscalations = 0;
5083
- static MAX_AUTO_ESCALATIONS = 8;
5084
- /** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
5085
- pendingAsks = /* @__PURE__ */ new Map();
5086
- /** Lazily resolved memory tools (async loadMemory runs in initMemory). */
5087
- memoryReady;
5088
- constructor(options) {
5089
- this.options = { ...new DuplexAgentOptions(), ...options };
5090
- const o = this.options;
5091
- if (o.memoryDir && o.fs) {
5092
- this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
5093
- }
5094
- const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
5095
- const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
5096
- const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
5097
- const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
5098
- const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
5099
- const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
5100
- const mcpNames = [
5101
- ...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
5102
- ...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
5103
- ];
5104
- const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
5105
- const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
5106
- Today's date: ${(/* @__PURE__ */ new Date()).toDateString()}.`;
5107
- const tools = [
5108
- ...o.reflexOptions?.tools ?? [],
5109
- this.actTool(),
5110
- ...o.thinkModel !== false ? [this.thinkTool()] : [],
5111
- this.taskStatusTool(),
5112
- this.cancelTaskTool(),
5113
- this.quickLookTool(),
5114
- this.answerTaskTool(),
5115
- this.holdTool()
5116
- ];
5117
- const host = o.host;
5118
- const voiceHost = host && {
5119
- ask: host.ask ? (q) => host.ask(q) : void 0,
5120
- confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
5121
- notify: (ev) => {
5122
- if (ev?.kind === "text_delta" && typeof ev.message === "string") {
5123
- if (this.heldThisTurn) return;
5124
- if (this.fabricationCut) return;
5125
- const msg = ev.message;
5126
- this.reflexBuf += msg;
5127
- const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
5128
- if (m) {
5129
- this.fabricationCut = true;
5130
- log10.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
5131
- const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
5132
- if (!safe) return;
5133
- if (safe.trim()) this.spokeThisTurn = true;
5134
- host.notify?.({ ...ev, message: safe });
5135
- return;
5136
- }
5137
- const held = this.reflexBuf.length - this.reflexForwarded;
5138
- const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
5139
- const upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
5140
- const out = this.reflexBuf.slice(this.reflexForwarded, upto);
5141
- this.reflexForwarded = upto;
5142
- if (!out) return;
5143
- if (out.trim()) this.spokeThisTurn = true;
5144
- host.notify?.({ ...ev, message: out });
5145
- return;
5146
- }
5147
- host.notify?.(ev);
5148
- }
5149
- };
5150
- this.voice = new Agent({
5151
- ai: o.ai,
5152
- fs: new MemFilesystem2(),
5153
- model: o.reflexModel,
5154
- stream: true,
5155
- host: voiceHost,
5156
- // The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
5157
- // never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
5158
- // it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
5159
- askUserQuestion: false,
5160
- systemPrompt: prompt,
5161
- instructionFiles: false,
5162
- maxSteps: 8,
5163
- timeoutMs: 3e4,
5164
- ...o.reflexOptions,
5165
- tools,
5166
- // Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
5167
- hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
5168
- });
5169
- }
5170
- /** Resolve memory tools + inject index into voice system prompt (once). */
5171
- async initMemory() {
5172
- if (!this.memoryReady) return;
5173
- const mem = await this.memoryReady;
5174
- this.memoryReady = void 0;
5175
- this.voice.options.tools.push(...mem.tools);
5176
- if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
5177
- }
5178
- /** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
5179
- * turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
5180
- flushHeldReflexTail() {
5181
- if (this.fabricationCut) return;
5182
- const tail = this.reflexBuf.slice(this.reflexForwarded);
5183
- this.reflexForwarded = this.reflexBuf.length;
5184
- if (!tail) return;
5185
- if (tail.trim()) this.spokeThisTurn = true;
5186
- this.options.host?.notify?.({ kind: "text_delta", message: tail });
5187
- }
5188
- /** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
5189
- resetTurn() {
5190
- this.turnDispatched = false;
5191
- this.turnBriefs.clear();
5192
- this.spokeThisTurn = false;
5193
- this.heldThisTurn = false;
5194
- this.reflexBuf = "";
5195
- this.reflexForwarded = 0;
5196
- this.fabricationCut = false;
5197
- this.turnFollowUp = false;
5198
- this.voice.options.toolChoice = void 0;
5199
- }
5200
- /** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
5201
- * now wait for the push" (CC's Task model). Block the temptations — TaskStatus polling and identical
5202
- * re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
5203
- * still allowed (parallel independent work). During a re-ack pass, block every tool. */
5204
- dispatchGuard() {
5205
- return {
5206
- preToolUse: (call) => {
5207
- if (this.nudging) return { block: true, reason: "Just say one short spoken acknowledgement \u2014 no tools this turn." };
5208
- if (!this.turnDispatched) return;
5209
- if (call.name === "TaskStatus")
5210
- return { block: true, reason: "You just dispatched a task this turn \u2014 do NOT poll. Give one short spoken acknowledgement and end your turn; the result arrives later as a [task \u2026] event." };
5211
- if ((call.name === "Act" || call.name === "Think") && this.turnBriefs.has(String(call.args?.brief ?? "")))
5212
- return { block: true, reason: "You already dispatched this exact task \u2014 acknowledge briefly and end your turn." };
5213
- }
5214
- };
5215
- }
5216
- /** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
5217
- * (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
5218
- * gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
5219
- * voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
5220
- * Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
5221
- get silentTurn() {
5222
- return !!this.options.host && !this.spokeThisTurn && !this.heldThisTurn;
5223
- }
5224
- /** A turn that voiced nothing is dead air. Re-prompt the reflex ONCE so the LLM itself voices a short
5225
- * line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
5226
- * Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
5227
- async ackIfSilent(fallback) {
5228
- const dispatched = this.turnDispatched;
5229
- this.nudging = true;
5230
- try {
5231
- await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
5232
- } catch (e) {
5233
- log10.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
5234
- } finally {
5235
- this.nudging = false;
5236
- }
5237
- if (!this.spokeThisTurn)
5238
- this.options.host?.notify?.({ kind: "text_delta", message: fallback ?? (dispatched ? "Okay, on it." : "Sorry, could you say that again?") });
5239
- }
5240
- /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns. */
5241
- send(content) {
5242
- return this.enqueue(async () => {
5243
- await this.initMemory();
5244
- this.resetTurn();
5245
- const res = await this.voice.send(content);
5246
- this.flushHeldReflexTail();
5247
- if (this.silentTurn) await this.ackIfSilent();
5248
- return res;
5249
- });
5250
- }
5251
- /** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
5252
- cancelTask(id) {
5253
- const rec = this.tasks.get(id);
5254
- if (!rec) return `No task '${id}'.`;
5255
- if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
5256
- rec.status = "cancelled";
5257
- rec.controller.abort();
5258
- return `Task ${rec.id} (${rec.label}) cancelled.`;
5259
- }
5260
- /** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
5261
- * delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
5262
- * tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
5263
- * Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
5264
- parkInFlightDeliveries() {
5265
- const parked = [];
5266
- for (const rec of this.tasks.values())
5267
- if (rec.status === "running" && !rec.deliveryParked) {
5268
- rec.deliveryParked = true;
5269
- parked.push(rec.id);
5270
- }
5271
- return parked;
5272
- }
5273
- /** Resolve when all queued voice turns AND all in-flight worker tasks have settled (tests, graceful shutdown). */
5274
- async idle() {
5275
- while (true) {
5276
- const q = this.queue;
5277
- await q.catch(() => {
5278
- });
5279
- await Promise.all([...this.tasks.values()].map((t) => t.promise));
5280
- if (this.queue === q && ![...this.tasks.values()].some((t) => t.status === "running")) return;
5281
- }
5282
- }
5283
- /** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
5284
- enqueue(fn) {
5285
- const run = this.queue.then(fn, fn);
5286
- this.queue = run.then(() => {
5287
- }, () => {
5288
- });
5289
- return run;
5290
- }
5291
- notify(kind, message, data) {
5292
- this.options.host?.notify?.({ kind, message, data });
5293
- }
5294
- /** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
5295
- * `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
5296
- * marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
5297
- * text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
5298
- queueRevoice(event, nonClean = false) {
5299
- this.pendingEvents.push(event);
5300
- if (nonClean) this.pendingNonClean = true;
5301
- if (this.flushQueued) return;
5302
- this.flushQueued = true;
5303
- void this.enqueue(async () => {
5304
- this.flushQueued = false;
5305
- const events = this.pendingEvents.splice(0);
5306
- const nonCleanTurn = this.pendingNonClean;
5307
- this.pendingNonClean = false;
5308
- if (!events.length) return;
5309
- const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
5310
- this.resetTurn();
5311
- this.turnFollowUp = nonCleanTurn;
5312
- await this.voice.send(events.join("\n"));
5313
- this.flushHeldReflexTail();
5314
- if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
5315
- this.notify("revoice_done", "");
5316
- });
5317
- }
5318
- /** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
5319
- * Act briefs get a self-verify footer — the worker's report is trusted without review, so it
5320
- * must check its own work before reporting (nearly free under prompt caching; measured honest:
5321
- * it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
5322
- buildBrief(brief, tier = "act", deliver = true) {
5323
- const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
5324
- const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
5325
- const deliverContract = deliver ? "\n\n## DELIVER (spoken delivery)\nYou are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown." + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
5326
- return (recent ? `${brief}
5327
-
5328
- ## Recent conversation (for context)
5329
- ${recent}` : brief) + verify + deliverContract;
5330
- }
5331
- /** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
5332
- spawnWorker(id, label, briefText, tier, brief, followUp) {
5333
- const o = this.options;
5334
- const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
5335
- const tierModel = tier === "think" ? o.thinkModel : o.actModel;
5336
- const controller = new AbortController();
5337
- const base = tierOpts?.hooks ?? o.actOptions?.hooks;
5338
- const report = o.progressUpdates ? this.progressReporter(id) : void 0;
5339
- const tail = [];
5340
- const pushTail = (line) => {
5341
- tail.push(line.slice(0, 200));
5342
- if (tail.length > 120) tail.splice(0, tail.length - 120);
5343
- };
5344
- const hooks = {
5345
- ...base,
5346
- preToolUse: async (call, meta) => {
5347
- const d = await base?.preToolUse?.(call, meta);
5348
- pushTail(`\u2699 ${describeCall(call)}`);
5349
- report?.pre(call);
5350
- return d;
5351
- },
5352
- postToolUse: async (call, result, meta) => {
5353
- await base?.postToolUse?.(call, result, meta);
5354
- const last = result?.trim().split("\n").filter(Boolean).pop();
5355
- if (last) pushTail(` \u21B3 ${last}`);
5356
- report?.post(call);
5357
- },
5358
- onToolOutput: (call, chunk, meta) => {
5359
- base?.onToolOutput?.(call, chunk, meta);
5360
- report?.output(chunk);
5361
- }
5362
- };
5363
- const relayAsk = async (q) => {
5364
- const opts = q.options?.length ? ` Options: ${q.options.map((x) => x.label).join(", ")}.` : "";
5365
- const a = await this.parkQuestion(id, `${q.question}${opts}`);
5366
- return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
5367
- };
5368
- const splitter = new SpokenSplitter();
5369
- const speak = (seg) => {
5370
- if (seg && !this.tasks.get(id)?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
5371
- };
5372
- const coalescer = new SentenceCoalescer();
5373
- const feedSpoken = (s) => {
5374
- const ready = coalescer.feed(s);
5375
- if (ready) speak(ready);
5376
- };
5377
- const flushSpoken = () => speak(coalescer.flush());
5378
- const askBridge = o.askRelay ? { ask: relayAsk } : o.host?.ask ? { ask: (q) => o.host.ask(q) } : {};
5379
- const workerHost = {
5380
- ...askBridge,
5381
- notify: (ev) => {
5382
- if (ev?.kind === "text_delta" && typeof ev.message === "string") {
5383
- const { spoken, detail } = splitter.feed(ev.message);
5384
- feedSpoken(spoken);
5385
- if (detail.trim()) pushTail(detail.trim());
5386
- return;
5387
- }
5388
- }
5389
- };
5390
- const agentOpts = {
5391
- ai: o.ai,
5392
- fs: o.fs,
5393
- model: tierModel,
5394
- ...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
5395
- ...tierOpts,
5396
- // Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
5397
- // main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
5398
- providerOptions: o.providerOptionsFor?.(tierModel),
5399
- stream: true,
5400
- // worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
5401
- host: workerHost,
5402
- // carries BOTH ask AND the <spoken>-splitting notify
5403
- ...hooks ? { hooks } : {},
5404
- signal: controller.signal
5405
- // shared with the checker so a cancel tears down both
5406
- };
5407
- const promise = new Agent(agentOpts).run(briefText).then((res) => {
5408
- const { spoken, detail } = splitter.flush();
5409
- feedSpoken(spoken);
5410
- if (detail.trim()) pushTail(detail.trim());
5411
- flushSpoken();
5412
- return res;
5413
- }).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err) => this.onWorkerFailed(id, err));
5414
- this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
5415
- if (this.tasks.size > this.options.maxTaskRecords)
5416
- for (const [tid, rec] of this.tasks) {
5417
- if (this.tasks.size <= this.options.maxTaskRecords) break;
5418
- if (rec.status !== "running") this.tasks.delete(tid);
5419
- }
5191
+ /** Resolve when the engine is no longer speaking (immediate if already idle). */
5192
+ awaitIdle() {
5193
+ if (this.state !== "speaking" && this.state !== "thinking") return Promise.resolve();
5194
+ return new Promise((r) => this.idleWaiters.push(r));
5420
5195
  }
5421
- /** Fresh-context check of a successful Act task: a NEW agent (same model/fs/tools, but NO shared
5422
- * conversation context) re-reads the file state against the brief and fixes any gap. The fix lands
5423
- * on the shared fs automatically (workers write fs directly, no overlay), so grading sees the
5424
- * corrected state. Bounded to ONE pass. Off unless `verifyActTasks`; never runs for think/failed/
5425
- * cancelled tasks. Usage is merged so /cost reflects the real (worker + checker) spend. */
5426
- async maybeVerify(id, brief, res, tier, agentOpts, askBridge) {
5427
- if (!this.options.verifyActTasks || tier !== "act" || res.finishReason !== "stop") return res;
5428
- if (this.tasks.get(id)?.status === "cancelled") return res;
5429
- const { stream: _stream, host: _host, ...restOpts } = agentOpts;
5430
- const checkerOpts = {
5431
- ...restOpts,
5432
- ...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
5433
- };
5434
- const checkBrief = `${this.buildBrief(brief, tier, false)}
5435
-
5436
- ## VERIFY MODE
5437
- Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
5438
- this.notify("task_verify", `task ${id}: verifying`, { id });
5439
- const cres = await new Agent(checkerOpts).run(checkBrief);
5440
- if (cres.finishReason !== "stop") {
5441
- log10.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
5442
- this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
5196
+ // --- speaking side (host-driven) ---
5197
+ /** open a spoken turn (idempotent — safe from both onUtterance and first-delta paths).
5198
+ * `ack` speaks the configured micro-ack as the context opener (utterance path only —
5199
+ * masks LLM TTFT; re-voice turns begun by their first delta skip it). */
5200
+ beginSpeech(ack = false) {
5201
+ if (this.speaking && this.ctxOpen) return;
5202
+ if (this.drainTimer) {
5203
+ this.clock.clearTimeout(this.drainTimer);
5204
+ this.drainTimer = null;
5443
5205
  }
5444
- const sum = (a = 0, b = 0) => a + b;
5445
- return {
5446
- ...res,
5447
- steps: res.steps + cres.steps,
5448
- // Merge the checker's messages so downstream tool-call/step accounting includes BOTH agents
5449
- // (else a verified task's toolCalls would undercount vs its steps/usage).
5450
- messages: [...res.messages, ...cres.messages],
5451
- usageEstimated: res.usageEstimated || cres.usageEstimated,
5452
- usage: res.usage && cres.usage ? {
5453
- promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
5454
- completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
5455
- totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
5456
- cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
5457
- cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
5458
- } : res.usage ?? cres.usage
5459
- };
5460
- }
5461
- /** Throttled per-task progress: worker tool calls → at most one progress re-voice per interval.
5462
- * Two sources, one throttle: completed steps (post) and a heartbeat for a SINGLE long tool call
5463
- * (pre records the in-flight call; a self-cleaning timer narrates "still inside Bash — 70s").
5464
- * Completion supersedes: nothing is emitted once the task has settled. */
5465
- progressReporter(id) {
5466
- let lastAt = Date.now();
5467
- let steps = 0;
5468
- let inflight = null;
5469
- const due = () => {
5470
- if (this.pendingAsks.size) return void 0;
5471
- const rec = this.tasks.get(id);
5472
- return rec && rec.status === "running" && Date.now() - lastAt >= this.options.progressIntervalMs ? rec : void 0;
5473
- };
5474
- const emit = (rec, line, call) => {
5475
- lastAt = Date.now();
5476
- this.notify("task_progress", `task ${id} (${rec.label}): ${line}`, { id, steps, call: call.name });
5477
- this.queueRevoice(`[task ${id} progress] ${line}`);
5478
- };
5479
- const timer = setInterval(() => {
5480
- const rec = this.tasks.get(id);
5481
- if (!rec || rec.status !== "running") return clearInterval(timer);
5482
- if (!inflight || !due()) return;
5483
- const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
5484
- emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
5485
- }, Math.max(this.options.progressIntervalMs, 250));
5486
- timer.unref?.();
5487
- return {
5488
- pre: (call) => {
5489
- inflight = { call, at: Date.now(), tail: "" };
5490
- },
5491
- output: (chunk) => {
5492
- if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
5493
- },
5494
- // digest only — NEVER re-voices directly
5495
- post: (call) => {
5496
- steps++;
5497
- inflight = null;
5498
- const rec = due();
5499
- if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
5500
- }
5501
- };
5502
- }
5503
- /** Park a question under `askId` (a task id, or any unique key for permission asks): re-voices
5504
- * '[task <id> asks] …' and resolves with the user's answer via AnswerTask — or '' on timeout/
5505
- * task settle (callers map '' to deny / best-judgment). Workers never block forever. */
5506
- parkQuestion(askId, question) {
5507
- return new Promise((resolve) => {
5508
- let settled = false;
5509
- const finish = (answer) => {
5510
- if (settled) return;
5511
- settled = true;
5512
- clearTimeout(timer);
5513
- this.pendingAsks.delete(askId);
5514
- resolve(answer);
5515
- };
5516
- const timer = setTimeout(() => {
5517
- this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
5518
- finish("");
5519
- }, this.options.askTimeoutMs);
5520
- this.pendingAsks.set(askId, { question, resolve: finish });
5521
- this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
5522
- this.queueRevoice(`[task ${askId} asks] ${question}
5523
- (Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
5524
- });
5525
- }
5526
- /** Resolve any question a settling/cancelled task left parked (its answer can no longer matter). */
5527
- dropAsk(id) {
5528
- this.pendingAsks.get(id)?.resolve("");
5529
- }
5530
- /** Build the INTEGRATION TURN prompt for a NON-CLEAN settled worker (early stop / failure). A clean
5531
- * success never reaches here — it streams its own `<spoken>` delivery during the run. For a partial
5532
- * or failed result the outcome re-enters the reflex as a decision (like a tool_result flowing back
5533
- * into a normal agent loop): the reflex evaluates the outcome against the original intent and chooses
5534
- * what to do next.
5535
- *
5536
- * Decision branches (the reflex acts on them with EXISTING tools — no new surface):
5537
- * • accept → SPEAK the (partial) result plainly — don't dress a failure up as success.
5538
- * • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
5539
- * exists AND this task wasn't already a follow-up (one hop max). Wires the dead
5540
- * "Reserve Think for a problem Act already FAILED at" promise.
5541
- * • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
5542
- * • ask → ask the user ONE concrete question if genuinely blocked.
5543
- *
5544
- * Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
5545
- * failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
5546
- integrationPrompt(rec, outcome, body, finishReason) {
5547
- const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
5548
- const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
5549
- const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
5550
- const hasThink = this.options.thinkModel !== false;
5551
- const options = [];
5552
- if (!rec.followUp && canEscalate && hasThink)
5553
- options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
5554
- if (!rec.followUp && canEscalate)
5555
- options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
5556
- options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
5557
- options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
5558
- const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
5559
- const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
5560
- return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
5561
- }
5562
- onWorkerSettled(id, res) {
5563
- this.dropAsk(id);
5564
- const rec = this.tasks.get(id);
5565
- if (res.finishReason === "aborted" || rec.status === "cancelled") {
5566
- rec.status = "cancelled";
5567
- this.notify("task_cancelled", `task ${id} (${rec.label}) cancelled`);
5568
- return;
5206
+ this.interrupted = false;
5207
+ this.bcSupersede();
5208
+ this.resetOverlap(true);
5209
+ this.diag("tts_turn_begin", { ack, gapless: this.speaking });
5210
+ if (!this.speaking) this.player.markTurn();
5211
+ this.speaking = true;
5212
+ this.ctxOpen = true;
5213
+ this.spokeDeltas = false;
5214
+ this.reply = "";
5215
+ this.revealText = "";
5216
+ this.wordStarts = [];
5217
+ this.revealedN = 0;
5218
+ if (this.options.revealMode === "word") this.startWordReveal();
5219
+ this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
5220
+ this.echoWords = new Set(this.words(this.prevReply));
5221
+ this.tts.newContext();
5222
+ this.clearAckTimer("new_turn");
5223
+ if (ack && this.options.ackPhrase) {
5224
+ this.diag("ack_fired", { phrase: this.options.ackPhrase, adaptive: false });
5225
+ this.tts.speak(this.options.ackPhrase + " ", true);
5226
+ this.spokeDeltas = true;
5227
+ this.ackAt = this.clock.now();
5228
+ this.lastAck = this.options.ackPhrase;
5229
+ for (const w of this.words(this.options.ackPhrase)) this.echoWords.add(w);
5230
+ } else if (ack && this.options.adaptiveAckMs > 0) {
5231
+ this.ackTimer = this.clock.setTimeout(() => {
5232
+ this.ackTimer = null;
5233
+ if (!this.speaking || !this.ctxOpen || this.spokeDeltas || this.interrupted) return;
5234
+ const phrase = this.pickAck();
5235
+ this.diag("ack_fired", { phrase, adaptive: true, waitedMs: this.options.adaptiveAckMs });
5236
+ this.tts.speak(phrase + " ", true);
5237
+ this.spokeDeltas = true;
5238
+ this.ackAt = this.clock.now();
5239
+ this.lastAck = phrase;
5240
+ for (const w of this.words(phrase)) this.echoWords.add(w);
5241
+ this.setState("speaking");
5242
+ this.options.onAdaptiveAck();
5243
+ }, this.options.adaptiveAckMs);
5569
5244
  }
5570
- if (res.finishReason === "error") {
5571
- const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
5572
- return this.failTask(rec, msg);
5245
+ if (!this.turnStartAt) this.turnStartAt = this.clock.now();
5246
+ this.setState("thinking");
5247
+ }
5248
+ /** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
5249
+ * host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
5250
+ speakDelta(text) {
5251
+ if (this.interrupted) return "";
5252
+ this.clearAckTimer("first_delta");
5253
+ if (!this.speaking || !this.ctxOpen) this.beginSpeech();
5254
+ let { speech, display, prose } = this.emo ? this.emo.feed(text) : { speech: text, display: text, prose: text };
5255
+ if (this.reply && /[.!?…]$/.test(this.reply) && /^[A-Z]/.test(prose)) {
5256
+ speech = " " + speech;
5257
+ display = " " + display;
5258
+ prose = " " + prose;
5573
5259
  }
5574
- rec.status = "done";
5575
- rec.result = res.text;
5576
- const incomplete = res.finishReason !== "stop";
5577
- log10.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
5578
- this.notify("task_done", `task ${id} (${rec.label}) completed`, {
5579
- id,
5580
- text: res.text,
5581
- usage: res.usage,
5582
- usageEstimated: res.usageEstimated,
5583
- finishReason: res.finishReason,
5584
- steps: res.steps,
5585
- toolCalls: res.messages.filter((m) => m.role === "tool").length
5586
- });
5587
- if (incomplete) {
5588
- return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
5260
+ this.reply += prose;
5261
+ if (prose.trim()) this.repliedSinceDispatch = true;
5262
+ for (const w of this.words(this.reply)) this.echoWords.add(w);
5263
+ this.tts.speak(forSpeech(speech), true);
5264
+ if (!this.spokeDeltas && this.turnStartAt) log10.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
5265
+ this.spokeDeltas = true;
5266
+ this.setState("speaking");
5267
+ if (this.options.revealMode !== "off" && display) {
5268
+ this.revealText += display;
5269
+ if (this.options.revealMode === "delta") this.options.onReveal(this.revealText);
5589
5270
  }
5590
- const tail = rec.splitter?.flush();
5591
- if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
5592
- if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
5593
- if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
5594
- this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
5271
+ return display;
5595
5272
  }
5596
- onWorkerFailed(id, err) {
5597
- this.failTask(this.tasks.get(id), err instanceof Error ? err.message : String(err));
5273
+ /** Karaoke reveal poll: reveal the display-word prefix up to the count of spoken words whose
5274
+ * Cartesia start time has been reached on the playback clock (`AudioSink.playedMs`, which excludes
5275
+ * paused time). Poll-driven (not AudioContext timers) so it's portable across CLI/browser sinks and
5276
+ * deterministic under the virtual clock. */
5277
+ startWordReveal() {
5278
+ if (this.revealPoll) return;
5279
+ const tick = () => {
5280
+ this.revealPoll = null;
5281
+ if (!this.speaking || this.interrupted || this.options.revealMode !== "word") return;
5282
+ const playedSec = this.player.playedMs() / 1e3;
5283
+ let n = this.revealedN;
5284
+ while (n < this.wordStarts.length && this.wordStarts[n] <= playedSec) n++;
5285
+ if (n !== this.revealedN) {
5286
+ this.revealedN = n;
5287
+ const words = this.revealText.trim().split(/\s+/).filter(Boolean);
5288
+ this.options.onReveal(words.slice(0, n).join(" "));
5289
+ }
5290
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5291
+ };
5292
+ this.revealPoll = this.clock.setTimeout(tick, 60);
5293
+ }
5294
+ /** Stop the poll. `finalFlush` reveals the whole line once audio has finished, so a timing
5295
+ * undershoot never leaves the tail unrevealed (skipped after a barge-in — the truncated prefix
5296
+ * is what was actually spoken). */
5297
+ stopWordReveal(finalFlush) {
5298
+ if (this.revealPoll) {
5299
+ this.clock.clearTimeout(this.revealPoll);
5300
+ this.revealPoll = null;
5301
+ }
5302
+ if (finalFlush && this.options.revealMode === "word" && !this.interrupted) {
5303
+ const full = this.revealText.trim();
5304
+ if (full) this.options.onReveal(full);
5305
+ }
5598
5306
  }
5599
- failTask(rec, msg) {
5600
- this.dropAsk(rec.id);
5601
- rec.status = "error";
5602
- rec.result = msg;
5603
- log10.warn(`task ${rec.id} failed: ${msg}`);
5604
- this.notify("task_error", `task ${rec.id} (${rec.label}) failed: ${msg}`);
5605
- this.queueRevoice(this.integrationPrompt(rec, "error", msg, "error"), true);
5307
+ /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
5308
+ endSpeech() {
5309
+ this.interrupted = false;
5310
+ this.clearAckTimer("turn_end");
5311
+ if (!this.speaking) return;
5312
+ this.diag("tts_turn_end", { spoke: this.spokeDeltas, replyChars: this.reply.length });
5313
+ this.ctxOpen = false;
5314
+ if (this.emo) {
5315
+ const t = this.emo.flush();
5316
+ this.emo = null;
5317
+ if (t.prose) this.reply += t.prose;
5318
+ if (t.speech) this.tts.speak(forSpeech(t.speech), true);
5319
+ }
5320
+ if (this.reply) this.prevReply = this.reply;
5321
+ const settle = () => {
5322
+ if (this.ctxOpen) {
5323
+ this.drainTimer = null;
5324
+ return;
5325
+ }
5326
+ if (this.pausedAt) {
5327
+ this.drainTimer = this.clock.setTimeout(settle, 250);
5328
+ return;
5329
+ }
5330
+ this.drainTimer = null;
5331
+ this.speaking = false;
5332
+ this.stopWordReveal(true);
5333
+ if (this.turnStartAt) log10.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
5334
+ this.echoUntil = this.clock.now() + 2500;
5335
+ if (!this.usingAec) this.stt.reset();
5336
+ this.setState("listening");
5337
+ if (this.uttQueue.length) this.pumpQueue();
5338
+ };
5339
+ const drainThenSettle = () => {
5340
+ if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
5341
+ this.drainTimer = this.clock.setTimeout(settle, this.player.drainMs() + 300);
5342
+ };
5343
+ if (this.spokeDeltas) {
5344
+ this.tts.onDone = drainThenSettle;
5345
+ this.tts.end();
5346
+ if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
5347
+ this.drainTimer = this.clock.setTimeout(drainThenSettle, 15e3);
5348
+ } else drainThenSettle();
5606
5349
  }
5607
- // --- voice tools (closures over this instance) ---
5608
- /** Live-switch the think tier: `false` disables (removes the Think tool from the voice agent),
5609
- * a model id enables (adds the tool if missing). The system-prompt THINK_SLOT text is frozen at
5610
- * construction — the tool's own description carries the routing guidance, so a live enable works;
5611
- * dispatch()'s think→act fallback covers any straggler calls after a live disable. */
5612
- setThinkModel(model) {
5613
- this.options.thinkModel = model;
5614
- const tools = this.voice.options.tools;
5615
- const i = tools.findIndex((t) => t.name === "Think");
5616
- if (model === false && i >= 0) tools.splice(i, 1);
5617
- else if (model !== false && i < 0) tools.push(this.thinkTool());
5350
+ /** text of the reply cut by the last barge-in — consumed by the host to tell the model what
5351
+ * the user did NOT hear. Cleared on read. */
5352
+ takeInterruptedReply() {
5353
+ const r = this.lastInterrupted;
5354
+ this.lastInterrupted = null;
5355
+ return r;
5618
5356
  }
5619
- /** User/programmatic spawn: the CLI's /act and /think commands. Returns the task id.
5620
- * `followUp` marks an automatic escalation/re-delegation (set by the integration turn) so the new
5621
- * task's own integration turn won't escalate again — capping auto-follow-ups to one hop. */
5622
- async dispatch(brief, tier = "act", label, followUp = false) {
5623
- if (tier === "think" && this.options.thinkModel === false) tier = "act";
5624
- if (followUp) this.autoEscalations++;
5625
- const id = `t${++this.seq}`;
5626
- const lbl = label ?? tier;
5627
- await this.options.onTaskStart?.(id, lbl);
5628
- this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
5629
- this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
5630
- return id;
5357
+ /** Speak a short filler phrase without starting a model turn (stays in listening mode after). */
5358
+ speakFiller(text) {
5359
+ if (!text || this.speaking) return;
5360
+ const replied = this.repliedSinceDispatch;
5361
+ this.beginSpeech();
5362
+ this.speakDelta(text);
5363
+ this.endSpeech();
5364
+ this.repliedSinceDispatch = replied;
5631
5365
  }
5632
- actTool() {
5633
- return {
5634
- name: "Act",
5635
- description: 'Escalate real work (reading/editing files, searching, running tasks, building) to a standard background worker. Returns immediately with a task id; the result arrives later as a "[task <id> completed]" event. Provide a clear, self-contained `brief` (the worker does not hear the live conversation).',
5636
- parameters: {
5637
- type: "object",
5638
- required: ["brief"],
5639
- properties: {
5640
- brief: { type: "string", description: "full, self-contained instructions for the worker" },
5641
- label: { type: "string", description: "a short (2-4 word) label for the task" }
5642
- }
5643
- },
5644
- run: async ({ brief, label }) => {
5645
- this.turnDispatched = true;
5646
- this.turnBriefs.add(String(brief ?? ""));
5647
- this.voice.options.toolChoice = "none";
5648
- const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
5649
- return `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
5650
- }
5651
- };
5366
+ /** Enqueue a COMPLETE worker utterance (already-split spoken text) onto the central speech queue.
5367
+ * If nothing is currently speaking it plays immediately; otherwise it queues and plays after the
5368
+ * current utterance fully ends (settle → pumpQueue) — never spliced into an open reflex utterance. */
5369
+ enqueueUtterance(text) {
5370
+ if (!text || !/[\p{L}\p{N}]/u.test(text)) return;
5371
+ this.diag("enqueue_utterance", { chars: text.length, queued: this.uttQueue.length, speaking: this.speaking });
5372
+ this.uttQueue.push(text);
5373
+ if (!this.speaking) this.pumpQueue();
5652
5374
  }
5653
- thinkTool() {
5654
- return {
5655
- name: "Think",
5656
- description: "Escalate to a premium deep-reasoning agent for complex analysis, architecture decisions, hard debugging, or planning. Same async pattern as Act \u2014 returns a task id. Use when the problem needs careful thought before (or instead of) action. Do not use Think for simple tasks \u2014 Act is cheaper and faster.",
5657
- parameters: {
5658
- type: "object",
5659
- required: ["brief"],
5660
- properties: {
5661
- brief: { type: "string", description: "the question or problem to reason about deeply" },
5662
- label: { type: "string", description: "a short (2-4 word) label for the task" }
5663
- }
5664
- },
5665
- run: async ({ brief, label }) => {
5666
- this.turnDispatched = true;
5667
- this.turnBriefs.add(String(brief ?? ""));
5668
- this.voice.options.toolChoice = "none";
5669
- const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
5670
- return `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
5671
- }
5672
- };
5375
+ /** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
5376
+ * next one from the settle completion (endSpeech), so utterances serialize without overlap. */
5377
+ pumpQueue() {
5378
+ if (this.speaking) return;
5379
+ const text = this.uttQueue.shift();
5380
+ if (text == null) return;
5381
+ this.beginSpeech();
5382
+ this.speakDelta(text);
5383
+ this.endSpeech();
5673
5384
  }
5674
- taskStatusTool() {
5675
- return {
5676
- name: "TaskStatus",
5677
- description: "Status of background tasks. Pass `id` for one task, or omit it to list all.",
5678
- parameters: { type: "object", properties: { id: { type: "string" } } },
5679
- run: async ({ id }) => {
5680
- const list = id ? [this.tasks.get(String(id))].filter(Boolean) : [...this.tasks.values()];
5681
- if (!list.length) return id ? `No task '${id}'.` : "No background tasks.";
5682
- return list.map((t) => `${t.id} (${t.label}): ${t.status}`).join("\n");
5683
- }
5684
- };
5385
+ /** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
5386
+ * utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
5387
+ * of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
5388
+ * many times"). All phrases are sub-second and semantically safe for their shape.
5389
+ * No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
5390
+ static ACKS_NEUTRAL = ["Mm-hm.", "One sec.", "Right.", "Sec.", "On it.", "Sure, moment.", "Uh, one moment.", "Let me see."];
5391
+ static ACKS_QUESTION = ["Hmm.", "Let me think.", "Hm, let me see.", "Good question.", "Mm, one sec.", "Let's see."];
5392
+ recentAcks = [];
5393
+ pickAck() {
5394
+ const pool = this.lastDispatchWasQuestion ? _VoiceEngine.ACKS_QUESTION : _VoiceEngine.ACKS_NEUTRAL;
5395
+ const fresh = pool.filter((p) => !this.recentAcks.includes(p));
5396
+ const phrase = (fresh.length ? fresh : pool)[Math.floor(Math.random() * (fresh.length ? fresh.length : pool.length))];
5397
+ this.recentAcks.push(phrase);
5398
+ if (this.recentAcks.length > 4) this.recentAcks.shift();
5399
+ return phrase;
5400
+ }
5401
+ lastDispatchWasQuestion = false;
5402
+ clearAckTimer(reason) {
5403
+ if (!this.ackTimer) return;
5404
+ this.clock.clearTimeout(this.ackTimer);
5405
+ this.ackTimer = null;
5406
+ if (reason) this.diag("ack_cancelled", { reason });
5407
+ }
5408
+ /** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
5409
+ * to be a Hold (intentionally silent; an ack would read as the start of an answer). */
5410
+ cancelPendingAck() {
5411
+ this.clearAckTimer("hold");
5685
5412
  }
5686
- /** Sub-100ms read-only lookups the voice may do itself — everything else stays Act-only.
5687
- * fs-only (no shell; the engine is VFS-abstracted): time, git branch (.git/HEAD read), ls, file
5688
- * head. Output is hard-capped so a lookup can never bloat the skinny voice context. */
5689
- quickLookTool() {
5690
- const CAP = 2e3;
5691
- const kinds = [.../* @__PURE__ */ new Set(["time", "branch", "ls", "file", "capabilities", ...Object.keys(this.options.quickLook ?? {})])];
5692
- return {
5693
- name: "QuickLook",
5694
- description: `Instant read-only lookup \u2014 one of: ${kinds.join(", ")}. For trivial facts only; anything needing search, commands, or reasoning goes through Act.`,
5695
- parameters: {
5696
- type: "object",
5697
- required: ["what"],
5698
- properties: {
5699
- what: { type: "string", enum: kinds, description: "what to look up" },
5700
- path: { type: "string", description: "for ls/file: the path to look at" }
5701
- }
5702
- },
5703
- run: async ({ what, path }) => {
5704
- const fs = this.options.fs;
5705
- try {
5706
- const over = this.options.quickLook?.[String(what)];
5707
- if (over) return await over(path ? String(path) : void 0);
5708
- switch (String(what)) {
5709
- case "capabilities": {
5710
- const actTools = this.options.actOptions?.tools ?? [];
5711
- const names = actTools.map((t) => t.name);
5712
- const mcpServers = Object.keys(this.options.actOptions?.providerOptions?.mcpServers ?? {});
5713
- const mcpNote = mcpServers.length ? ` Plus MCP servers your worker can use: ${mcpServers.join(", ")} (e.g. browser-bridge \u2192 drive a real browser: open tabs, navigate, click, screenshot).` : "";
5714
- if (!names.length)
5715
- return "Your worker uses Act's default local toolset (reading/editing files, running shell commands). No extra tools (e.g. web/internet) are configured; if a request is not a basic file or shell operation, assume you can't do it and say so." + mcpNote;
5716
- const hasFetch = names.some((n) => /WebFetch/i.test(n));
5717
- const hasBrowser = names.some((n) => /browser.*(navigate|click|page|type)/i.test(n));
5718
- const hasSearch = names.some((n) => /(^|_)WebSearch$|search/i.test(n) && !/WebFetch|browser/i.test(n));
5719
- const notes = [];
5720
- if (hasFetch) notes.push("WebFetch retrieves ONE specific URL you are given \u2014 it is not a search engine.");
5721
- if (hasBrowser) notes.push("The browser tools drive a real browser: you CAN open a site and, if needed, navigate to a search engine and search there \u2014 but it is manual and takes a moment, not an instant lookup.");
5722
- else if (!hasSearch && hasFetch) notes.push('You have no general web-search tool, so for an instant "search the web" you can only fetch a URL they provide.');
5723
- const webNote = notes.length ? " NOTE: " + notes.join(" ") : "";
5724
- return `Tools your background worker (Act) can actually use: ${names.join(", ")}. Read each name literally and match the request to a SPECIFIC tool; if none fits, you do NOT have that ability \u2014 say so honestly.` + webNote + mcpNote;
5725
- }
5726
- case "time":
5727
- return (/* @__PURE__ */ new Date()).toString();
5728
- case "branch": {
5729
- if (!fs) return "unavailable (no filesystem)";
5730
- const head = (await fs.readFile(".git/HEAD")).trim();
5731
- return head.startsWith("ref: refs/heads/") ? `branch: ${head.slice("ref: refs/heads/".length)}` : `detached HEAD at ${head.slice(0, 12)}`;
5732
- }
5733
- case "ls": {
5734
- if (!fs) return "unavailable (no filesystem)";
5735
- const names = await fs.readDir(String(path ?? "."));
5736
- return names.slice(0, 50).join("\n") + (names.length > 50 ? `
5737
- \u2026 (+${names.length - 50} more)` : "");
5738
- }
5739
- case "file": {
5740
- if (!fs) return "unavailable (no filesystem)";
5741
- if (!path) return "file lookup needs a path";
5742
- const text = await fs.readFile(String(path));
5743
- return text.length > CAP ? text.slice(0, CAP) + `
5744
- \u2026 (truncated \u2014 ${text.length} chars total; Act for the full file)` : text;
5745
- }
5746
- default:
5747
- return `unknown lookup '${what}'`;
5413
+ /** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
5414
+ interrupt() {
5415
+ this.clearAckTimer("interrupt");
5416
+ if (this.uttQueue.length) log10.info(`barge-in dropped ${this.uttQueue.length} queued worker utterance(s)`);
5417
+ const droppedQueued = this.uttQueue.length;
5418
+ this.uttQueue = [];
5419
+ if (!this.speaking && !this.drainTimer) return;
5420
+ this.stopWordReveal(false);
5421
+ this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
5422
+ if (this.drainTimer) {
5423
+ this.clock.clearTimeout(this.drainTimer);
5424
+ this.drainTimer = null;
5425
+ }
5426
+ this.resetOverlap(false);
5427
+ this.lastResumeAt = 0;
5428
+ const heardChars = Math.round(Math.max(0, this.player.playedMs()) / 1e3 * 15);
5429
+ if (this.reply) this.lastInterrupted = { full: this.reply, heard: this.reply.slice(0, heardChars) };
5430
+ this.speaking = false;
5431
+ this.ctxOpen = false;
5432
+ this.interrupted = true;
5433
+ this.suspectUntil = 0;
5434
+ this.echoUntil = this.clock.now() + Math.max(2500, this.player.drainMs() + 3e3);
5435
+ this.tts.cancel();
5436
+ this.player.kill();
5437
+ if (!this.usingAec) this.stt.reset();
5438
+ if (this.reply) this.prevReply = this.reply;
5439
+ this.setState("listening");
5440
+ }
5441
+ stop() {
5442
+ this.uttQueue = [];
5443
+ this.clearAckTimer();
5444
+ if (this.resumeTimer) this.clock.clearTimeout(this.resumeTimer);
5445
+ if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
5446
+ if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
5447
+ if (this.specTimer) this.clock.clearTimeout(this.specTimer);
5448
+ this.bcSupersede();
5449
+ this.stt.stop();
5450
+ this.player.kill();
5451
+ this.tts.close();
5452
+ this.setState("idle");
5453
+ }
5454
+ // --- listening side (STT-driven) ---
5455
+ words(s) {
5456
+ return s.toLowerCase().replace(/[^a-z0-9\s]/g, "").split(/\s+/).filter((w) => w.length >= 2);
5457
+ }
5458
+ novelWords(text) {
5459
+ return this.words(text).filter((w) => !this.echoWords.has(w));
5460
+ }
5461
+ echoActive() {
5462
+ return this.speaking || this.clock.now() < this.echoUntil;
5463
+ }
5464
+ /** Genuine user speech vs our own bleed (AEC tier): novel words must DOMINATE, not merely exist.
5465
+ * Degraded AEC + an STT mis-hearing manufactures a single novel word out of pure echo (a name or
5466
+ * rare word in our own reply comes back transcribed slightly differently — 1 novel / N words).
5467
+ * A real interjection is mostly novel ("stop", "wait what") — short utterances pass on ratio,
5468
+ * longer ones on count. */
5469
+ genuine(text) {
5470
+ const total = this.words(text).length;
5471
+ const novel = this.novelWords(text).length;
5472
+ return novel > 0 && novel / Math.max(1, total) > 0.5;
5473
+ }
5474
+ handlePartial(text) {
5475
+ if (this.speaking) {
5476
+ if (!this.options.bargeIn) return;
5477
+ if (this.clock.now() < this.bargeGraceUntil) {
5478
+ if (this.lastGraceDiag !== this.bargeGraceUntil) {
5479
+ this.lastGraceDiag = this.bargeGraceUntil;
5480
+ this.diag("grace_suppress", { text: text.slice(0, 60) });
5481
+ }
5482
+ if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
5483
+ return;
5484
+ }
5485
+ if (this.overlapCapable) {
5486
+ const txt = text.trim();
5487
+ if (!txt || txt === this.lastOverlapPartial) return;
5488
+ this.lastOverlapPartial = txt;
5489
+ if (!this.genuine(txt)) {
5490
+ if (this.pausedAt) this.armResume();
5491
+ return;
5492
+ }
5493
+ if (!this.pausedAt) {
5494
+ this.pausedAt = this.clock.now();
5495
+ const sinceResumeMs = this.lastResumeAt ? Math.round(this.clock.now() - this.lastResumeAt) : void 0;
5496
+ this.diag("overlap_pause", { text: txt.slice(0, 60), sinceResumeMs });
5497
+ this.player.pause();
5498
+ if (this.lastResumeAt && this.clock.now() - this.lastResumeAt < this.options.overlapRepauseCedeMs) {
5499
+ const phase = this.ctxOpen ? "speaking" : "drain";
5500
+ this.diag("barge_in", { phase, trigger: "overlap_repause", sinceResumeMs });
5501
+ this.interrupt();
5502
+ this.options.onBargeIn(phase);
5503
+ return;
5748
5504
  }
5749
- } catch (e) {
5750
- return `lookup failed: ${e?.message ?? e}`;
5751
5505
  }
5506
+ if (this.words(txt).length >= 2) {
5507
+ const phase = this.ctxOpen ? "speaking" : "drain";
5508
+ this.diag("barge_in", { phase, trigger: "overlap_cede", novel: this.novelWords(txt).length, total: this.words(txt).length });
5509
+ this.interrupt();
5510
+ this.options.onBargeIn(phase);
5511
+ return;
5512
+ }
5513
+ this.armResume();
5514
+ return;
5752
5515
  }
5753
- };
5754
- }
5755
- answerTaskTool() {
5756
- return {
5757
- name: "AnswerTask",
5758
- description: `Relay the user's answer to a pending question from a background task (the "[task <id> asks]" events). Pass the id from the event and the user's answer.`,
5759
- parameters: {
5760
- type: "object",
5761
- required: ["id", "answer"],
5762
- properties: { id: { type: "string" }, answer: { type: "string", description: "the user's answer, verbatim or faithfully summarized" } }
5763
- },
5764
- run: async ({ id, answer }) => {
5765
- const ask = this.pendingAsks.get(String(id));
5766
- if (!ask) return `No pending question for '${id}' \u2014 it may have been answered already or timed out.`;
5767
- ask.resolve(String(answer ?? ""));
5768
- return `Answer relayed \u2014 task ${id} resumes.`;
5516
+ const barge = this.usingAec ? this.genuine(text) : this.novelWords(text).length >= (this.suspectUntil ? 1 : 2);
5517
+ if (barge) {
5518
+ const phase = this.ctxOpen ? "speaking" : "drain";
5519
+ this.diag("barge_in", { phase, trigger: this.usingAec ? "genuine" : "heuristic_novel", novel: this.novelWords(text).length, total: this.words(text).length });
5520
+ this.interrupt();
5521
+ this.options.onBargeIn(phase);
5769
5522
  }
5770
- };
5523
+ return;
5524
+ }
5525
+ if (this.pendingUtt && text.trim()) {
5526
+ if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
5527
+ this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), Math.max(800, this.options.utteranceMergeMs));
5528
+ }
5529
+ this.trackSpeculation(text);
5530
+ this.trackBackchannel(text);
5531
+ if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
5771
5532
  }
5772
- holdTool() {
5773
- return {
5774
- name: "Hold",
5775
- description: 'The user seems mid-thought \u2014 hold the turn (stay listening) instead of answering. Optionally pass a short filler ("mhm", "go on") to speak while waiting. Use when the message sounds incomplete, trailing off, or like they paused to think.',
5776
- parameters: {
5777
- type: "object",
5778
- properties: {
5779
- filler: { type: "string", description: 'optional short filler to speak ("mhm", "go on", "mm-hm")' }
5780
- }
5781
- },
5782
- run: async ({ filler }) => {
5783
- this.heldThisTurn = true;
5784
- if (filler) this.notify("hold_filler", String(filler));
5785
- return "Holding \u2014 listening for the rest of the user's thought. Do not respond further this turn.";
5533
+ /** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
5534
+ * partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
5535
+ * onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
5536
+ * tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
5537
+ * dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
5538
+ * trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
5539
+ * utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
5540
+ trackSpeculation(text) {
5541
+ if (!this.options.speculativeMs) return;
5542
+ const t = text.trim();
5543
+ if (this.specText) {
5544
+ if (this.words(t).length > this.words(this.specText).length + 2) {
5545
+ this.specText = "";
5546
+ this.diag("speculate_abort", { reason: "partial_grew", partial: t.slice(0, 60) });
5547
+ this.options.onSpeculateAbort();
5786
5548
  }
5549
+ return;
5550
+ }
5551
+ if (this.specSpent || this.pendingUtt || !t || t === this.specPartial) return;
5552
+ this.specPartial = t;
5553
+ if (this.specTimer) this.clock.clearTimeout(this.specTimer);
5554
+ this.specTimer = this.clock.setTimeout(() => {
5555
+ this.specTimer = null;
5556
+ if (this.speaking || this.specSpent || this.pendingUtt) return;
5557
+ if (this.words(this.specPartial).length < this.options.speculativeMinWords) return;
5558
+ const flat = this.specPartial.toLowerCase().replace(/[^a-z0-9]/g, "");
5559
+ if (flat && this.lastDispatchFlat && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS && (this.lastDispatchFlat.startsWith(flat) || flat.startsWith(this.lastDispatchFlat))) return;
5560
+ this.specSpent = true;
5561
+ this.specText = this.specPartial;
5562
+ log10.debug(`speculate: "${this.specText.slice(0, 60)}"`);
5563
+ this.diag("speculate", { text: this.specText });
5564
+ this.options.onSpeculate(this.specText);
5565
+ }, this.options.speculativeMs);
5566
+ }
5567
+ /** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
5568
+ * imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
5569
+ * every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
5570
+ * strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
5571
+ * 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
5572
+ * right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
5573
+ static BACKCHANNELS = ["Mm-hm.", "Uh-huh.", "Right.", "Okay."];
5574
+ /** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
5575
+ * that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
5576
+ * BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
5577
+ * Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
5578
+ * Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
5579
+ trackBackchannel(text) {
5580
+ if (!this.options.backchannelMs) return;
5581
+ const t = text.trim();
5582
+ if (!t || t === this.bcPartial) return;
5583
+ this.bcPartial = t;
5584
+ if (this.bcTimer) this.clock.clearTimeout(this.bcTimer);
5585
+ this.bcTimer = this.clock.setTimeout(() => {
5586
+ this.bcTimer = null;
5587
+ const p = this.bcPartial;
5588
+ const veto = (reason) => this.diag("backchannel_suppress", { reason });
5589
+ if (this.speaking || this.pendingUtt || this.bcActive) return veto("busy");
5590
+ if (!this.usingAec || !this.options.bargeIn) return veto("half_duplex");
5591
+ if (this.bcCount >= 2 || this.lastBcAt && this.clock.now() - this.lastBcAt < this.options.backchannelMinGapMs) return veto("rate_limited");
5592
+ if (this.words(p).length < this.options.backchannelMinWords) return veto("short");
5593
+ if (!/[,.;:!?…]$/.test(p) && !this.looksIncomplete(p)) return veto("no_boundary");
5594
+ const letters = p.replace(/[^\p{L}]/gu, "");
5595
+ const latin = p.replace(/[^A-Za-z]/g, "");
5596
+ if (!letters || latin.length / letters.length < 0.7) return veto("non_latin");
5597
+ const flat = p.toLowerCase().replace(/[^a-z0-9]/g, "");
5598
+ if (flat && this.lastDispatchFlat && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS && (this.lastDispatchFlat.startsWith(flat) || flat.startsWith(this.lastDispatchFlat))) return;
5599
+ this.fireBackchannel();
5600
+ }, this.options.backchannelMs);
5601
+ }
5602
+ /** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
5603
+ * markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
5604
+ * pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
5605
+ * never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
5606
+ * swallowed), and the echo window extends so echo-shaped finals stay gated. */
5607
+ fireBackchannel() {
5608
+ const fresh = _VoiceEngine.BACKCHANNELS.filter((x) => !this.recentBc.includes(x));
5609
+ const pool = fresh.length ? fresh : _VoiceEngine.BACKCHANNELS;
5610
+ const phrase = pool[Math.floor(Math.random() * pool.length)];
5611
+ this.recentBc.push(phrase);
5612
+ if (this.recentBc.length > 2) this.recentBc.shift();
5613
+ const now3 = this.clock.now();
5614
+ for (const w of this.words(phrase)) this.echoWords.add(w);
5615
+ this.ackAt = now3;
5616
+ this.lastAck = phrase;
5617
+ this.lastBcAt = now3;
5618
+ this.lastBcPhrase = phrase;
5619
+ this.echoUntil = Math.max(this.echoUntil, now3 + 2500);
5620
+ this.bcCount++;
5621
+ this.bcActive = true;
5622
+ if (this.bcActiveTimer) this.clock.clearTimeout(this.bcActiveTimer);
5623
+ this.bcActiveTimer = this.clock.setTimeout(() => {
5624
+ this.bcActive = false;
5625
+ }, 3e3);
5626
+ this.tts.onDone = () => {
5627
+ this.bcActive = false;
5787
5628
  };
5629
+ this.tts.newContext();
5630
+ this.tts.speak(phrase, false);
5631
+ log10.debug(`backchannel: "${phrase}"`);
5632
+ this.diag("backchannel", { phrase });
5633
+ this.options.onBackchannel(phrase);
5634
+ }
5635
+ /** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
5636
+ bcSupersede() {
5637
+ this.bcActive = false;
5638
+ if (this.bcActiveTimer) {
5639
+ this.clock.clearTimeout(this.bcActiveTimer);
5640
+ this.bcActiveTimer = null;
5641
+ }
5642
+ if (this.bcTimer) {
5643
+ this.clock.clearTimeout(this.bcTimer);
5644
+ this.bcTimer = null;
5645
+ }
5788
5646
  }
5789
- cancelTaskTool() {
5790
- return {
5791
- name: "CancelTask",
5792
- description: "Cancel a running background task by id.",
5793
- parameters: { type: "object", required: ["id"], properties: { id: { type: "string" } } },
5794
- run: async ({ id }) => this.cancelTask(String(id))
5795
- };
5647
+ /** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
5648
+ * phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
5649
+ * within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
5650
+ stripBackchannelEcho(text) {
5651
+ if (!this.lastBcPhrase || this.clock.now() - this.lastBcAt > 2e3) return text;
5652
+ const target = normWord(this.lastBcPhrase);
5653
+ if (!target) return text;
5654
+ const toks = text.split(/\s+/);
5655
+ const idx = toks.findIndex((tok, i) => {
5656
+ if (normWord(tok) !== target) return false;
5657
+ if (/\?/.test(tok)) return false;
5658
+ const edgeBefore = i === 0 || /[,.;:!?…]$/.test(toks[i - 1]);
5659
+ const edgeAfter = i === toks.length - 1 || /[,.;:!?…]$/.test(tok);
5660
+ return edgeBefore && edgeAfter;
5661
+ });
5662
+ if (idx < 0) return text;
5663
+ toks.splice(idx, 1);
5664
+ log10.verbose(`stripped backchannel echo "${this.lastBcPhrase}" from dispatching final`);
5665
+ return toks.join(" ");
5796
5666
  }
5797
- };
5798
-
5799
- // src/mcp.ts
5800
- function toResult(result) {
5801
- if (result == null) return { text: "" };
5802
- if (typeof result === "string") return { text: result };
5803
- const content = result.content;
5804
- if (Array.isArray(content)) {
5805
- const texts = [];
5806
- const images = [];
5807
- for (const c of content) {
5808
- if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
5809
- images.push({ mimeType: c.mimeType, data: c.data });
5810
- } else if (typeof c?.text === "string") {
5811
- texts.push(c.text);
5812
- } else {
5813
- texts.push(JSON.stringify(c));
5667
+ /** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
5668
+ * overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
5669
+ * utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
5670
+ * a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
5671
+ * check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
5672
+ mergeUtterance(prev, next) {
5673
+ if (!prev) return next;
5674
+ if (!next) return prev;
5675
+ const pw = prev.split(/\s+/), nw = next.split(/\s+/);
5676
+ const norm2 = _VoiceEngine.normWord;
5677
+ if (pw.length >= 2 && pw.length <= nw.length) {
5678
+ let pfx = true;
5679
+ for (let i = 0; i < pw.length - 1; i++) if (norm2(pw[i]) !== norm2(nw[i])) {
5680
+ pfx = false;
5681
+ break;
5814
5682
  }
5683
+ if (pfx && norm2(nw[pw.length - 1]).startsWith(norm2(pw[pw.length - 1]))) return next;
5815
5684
  }
5816
- const text = texts.join("\n");
5817
- if (text || images.length) return { text, ...images.length ? { images } : {} };
5685
+ const max = Math.min(pw.length, nw.length);
5686
+ for (let k = max; k > 0; k--) {
5687
+ let match = true;
5688
+ for (let i = 0; i < k; i++) if (norm2(pw[pw.length - k + i]) !== norm2(nw[i])) {
5689
+ match = false;
5690
+ break;
5691
+ }
5692
+ if (match) return [...pw, ...nw.slice(k)].join(" ");
5693
+ }
5694
+ return `${prev} ${next}`;
5818
5695
  }
5819
- return { text: JSON.stringify(result) };
5820
- }
5821
- function mcpToolToAgentTool(spec, callTool, prefix = "mcp__") {
5822
- return {
5823
- name: `${prefix}${spec.name}`,
5824
- description: spec.description ?? `MCP tool ${spec.name}`,
5825
- parameters: spec.inputSchema ?? { type: "object", properties: {} },
5826
- async run(args, _ctx) {
5827
- const r = toResult(await callTool(spec.name, args ?? {}));
5828
- return r.images?.length ? r : r.text;
5696
+ static normWord(w) {
5697
+ return normWord(w);
5698
+ }
5699
+ /** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
5700
+ * the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
5701
+ * may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
5702
+ * DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
5703
+ * "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
5704
+ * question already being answered).
5705
+ * DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
5706
+ * (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
5707
+ * previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
5708
+ * repliedSinceDispatch then picks the action:
5709
+ * • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
5710
+ * "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
5711
+ * worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
5712
+ * DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
5713
+ * Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
5714
+ * • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
5715
+ * truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
5716
+ * starves the conversation entirely — the fuller final IS the completion the Hold is waiting
5717
+ * for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
5718
+ * race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
5719
+ refinalizes(next) {
5720
+ const pw = this.lastDispatchWords, nw = next.trim().split(/\s+/);
5721
+ const norm2 = _VoiceEngine.normWord;
5722
+ if (pw.length < 3 || nw.length < pw.length) return false;
5723
+ if (this.clock.now() - this.lastDispatchAt >= _VoiceEngine.DUP_FINAL_MS) return false;
5724
+ for (let i = 0; i < pw.length - 1; i++) if (norm2(pw[i]) !== norm2(nw[i])) return false;
5725
+ const pLast = norm2(pw[pw.length - 1]), nSame = norm2(nw[pw.length - 1]);
5726
+ if (!nSame.startsWith(pLast)) return false;
5727
+ return nw.length > pw.length || nSame.length > pLast.length;
5728
+ }
5729
+ static TRAIL_RE = /(?:^|\s)(?:and|but|or|so|to|the|a|an|of|in|for|with|that|if|uh|um|like|about|from|into|on|is|are|was|were|,)$/i;
5730
+ /** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
5731
+ looksIncomplete(text) {
5732
+ return _VoiceEngine.TRAIL_RE.test(text.trim());
5733
+ }
5734
+ handleUtterance(text) {
5735
+ if (this.speaking && (this.ctxOpen || this.pausedAt) && this.overlapCapable) {
5736
+ this.diag("utterance_drop", { reason: "overlap_noise", text: text.slice(0, 60) });
5737
+ this.stt.reset();
5738
+ return;
5829
5739
  }
5830
- };
5831
- }
5832
- function mcpToolsToAgentTools(specs, callTool, prefix = "mcp__", filter) {
5833
- return (filter ? specs.filter(filter) : specs).map((s) => mcpToolToAgentTool(s, callTool, prefix));
5834
- }
5835
- function describeSpec(s) {
5836
- const schema = s.inputSchema ? `
5837
- args: ${JSON.stringify(s.inputSchema)}` : "";
5838
- return `${s.name} \u2014 ${s.description ?? "(no description)"}${schema}`;
5839
- }
5840
- function makeMcpToolSearch(specs, callTool, options = {}) {
5841
- const maxResults = options.maxResults ?? 10;
5842
- const byName = new Map(specs.map((s) => [s.name, s]));
5843
- const catalogLine = `${specs.length} MCP tool(s) available \u2014 search by keyword, then call by exact name.`;
5844
- const searchTool = {
5845
- name: "ToolSearch",
5846
- description: `Search the available MCP tools by keyword (${catalogLine}). Returns matching tool names + their argument schemas; call one with \`McpCall\`.`,
5847
- parameters: { type: "object", required: ["query"], properties: { query: { type: "string", description: "keywords to match against tool name + description" } } },
5848
- async run({ query }) {
5849
- const q = String(query ?? "").trim();
5850
- if (!q) return catalogLine;
5851
- const { kept } = topByRelevance(specs, q, (s) => `${s.name} ${s.description ?? ""}`, maxResults);
5852
- if (!kept.length) return `(no MCP tool matches "${q}" \u2014 try broader keywords)`;
5853
- return kept.map(describeSpec).join("\n");
5740
+ if (this.echoActive() && (!this.options.bargeIn || (this.usingAec ? !this.genuine(text) : this.novelWords(text).length < 2))) {
5741
+ this.diag("echo_swallow", { guard: "echo_gate", text: text.slice(0, 60) });
5742
+ this.stt.reset();
5743
+ return;
5854
5744
  }
5855
- };
5856
- const callMcpTool = {
5857
- name: "McpCall",
5858
- description: "Call an MCP tool discovered via `ToolSearch`, by its exact name. Pass its arguments as `args`.",
5859
- parameters: {
5860
- type: "object",
5861
- required: ["name"],
5862
- properties: {
5863
- name: { type: "string", description: "exact tool name from ToolSearch" },
5864
- args: { type: "object", description: "arguments object for the tool (per its schema)" }
5745
+ const squash = (t) => t.toLowerCase().replace(/[^a-z]/g, "").replace(/(.)\1+/g, "$1");
5746
+ if (this.ackAt && this.lastAck && this.clock.now() - this.ackAt < 6e3 && squash(text) === squash(this.lastAck)) {
5747
+ this.diag("echo_swallow", { guard: "ack_squash", text: text.slice(0, 60) });
5748
+ this.ackAt = 0;
5749
+ return;
5750
+ }
5751
+ if (this.pendingUtt) this.mergePath = "merged";
5752
+ this.pendingUtt = this.mergeUtterance(this.pendingUtt, text);
5753
+ if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
5754
+ if (this.options.incompleteMergeMs && this.looksIncomplete(this.pendingUtt)) {
5755
+ log10.verbose(`hold: incomplete utterance "${this.pendingUtt.slice(-40)}"`);
5756
+ this.diag("hold", { reason: "incomplete", tail: this.pendingUtt.slice(-40) });
5757
+ this.options.onHold();
5758
+ if (this.options.holdFiller && !this.speaking) this.speakFiller(this.options.holdFiller);
5759
+ this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), this.options.incompleteMergeMs);
5760
+ return;
5761
+ }
5762
+ if (!this.options.utteranceMergeMs || this.words(this.pendingUtt).length >= 4) return this.flushUtterance();
5763
+ this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), this.options.utteranceMergeMs);
5764
+ }
5765
+ flushUtterance() {
5766
+ if (this.pendingTimer) {
5767
+ this.clock.clearTimeout(this.pendingTimer);
5768
+ this.pendingTimer = null;
5769
+ }
5770
+ if (this.specTimer) {
5771
+ this.clock.clearTimeout(this.specTimer);
5772
+ this.specTimer = null;
5773
+ }
5774
+ this.specSpent = false;
5775
+ this.specPartial = "";
5776
+ this.specText = "";
5777
+ if (this.bcTimer) {
5778
+ this.clock.clearTimeout(this.bcTimer);
5779
+ this.bcTimer = null;
5780
+ }
5781
+ this.bcCount = 0;
5782
+ this.bcPartial = "";
5783
+ const text = this.stripBackchannelEcho(this.pendingUtt);
5784
+ this.pendingUtt = "";
5785
+ const path = this.mergePath;
5786
+ this.mergePath = "direct";
5787
+ if (text) {
5788
+ const flat = text.toLowerCase().replace(/[^a-z0-9]/g, "");
5789
+ if (flat && flat === this.lastDispatchFlat && !this.repliedSinceDispatch && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS) {
5790
+ log10.verbose(`dropped duplicate final "${text.slice(0, 40)}" (identical to the just-dispatched utterance, no reply between)`);
5791
+ this.diag("utterance_drop", { reason: "dup_final", text: text.slice(0, 60) });
5792
+ return;
5865
5793
  }
5866
- },
5867
- async run({ name, args }) {
5868
- const n = String(name ?? "");
5869
- if (!byName.has(n)) return `Error: unknown MCP tool '${n}'. Use ToolSearch to find valid names.`;
5870
- const r = toResult(await callTool(n, args ?? {}));
5871
- return r.images?.length ? r : r.text;
5794
+ const refinal = this.refinalizes(text);
5795
+ if (refinal && this.repliedSinceDispatch) {
5796
+ log10.verbose(`dropped re-finalization "${text.slice(0, 60)}" (superset of the just-dispatched utterance, reply already flowing)`);
5797
+ this.diag("utterance_drop", { reason: "refinalized", text: text.slice(0, 60) });
5798
+ this.lastDispatchFlat = flat;
5799
+ this.lastDispatchWords = text.trim().split(/\s+/);
5800
+ return;
5801
+ }
5802
+ this.lastDispatchFlat = flat;
5803
+ this.lastDispatchWords = text.trim().split(/\s+/);
5804
+ this.lastDispatchAt = this.clock.now();
5805
+ this.repliedSinceDispatch = false;
5806
+ this.lastDispatchWasQuestion = /\?\s*$/.test(text);
5807
+ this.turnStartAt = this.clock.now();
5808
+ this.bargeGraceUntil = this.clock.now() + this.options.bargeGraceMs;
5809
+ this.diag("dispatch", { text, path: refinal ? "refinalized" : path, question: this.lastDispatchWasQuestion });
5810
+ this.options.onUtterance(text);
5811
+ }
5812
+ }
5813
+ get overlapCapable() {
5814
+ return this.usingAec && this.options.overlapPause && !!this.player.pause && !!this.player.resume;
5815
+ }
5816
+ armResume() {
5817
+ if (this.resumeTimer) this.clock.clearTimeout(this.resumeTimer);
5818
+ this.resumeTimer = this.clock.setTimeout(() => {
5819
+ this.resumeTimer = null;
5820
+ if (!this.pausedAt) return;
5821
+ this.diag("overlap_resume", { heldMs: Math.round(this.clock.now() - this.pausedAt) });
5822
+ this.stt.reset();
5823
+ this.resetOverlap(true);
5824
+ }, this.options.overlapResumeMs);
5825
+ }
5826
+ resetOverlap(resume) {
5827
+ if (this.resumeTimer) {
5828
+ this.clock.clearTimeout(this.resumeTimer);
5829
+ this.resumeTimer = null;
5830
+ }
5831
+ if (this.pausedAt && resume) {
5832
+ this.player.resume?.();
5833
+ this.lastResumeAt = this.clock.now();
5834
+ }
5835
+ this.pausedAt = 0;
5836
+ this.lastOverlapPartial = "";
5837
+ this.gatePassTimes = [];
5838
+ }
5839
+ /** energy two-stage barge-in (heuristic tier only): spike over echo baseline → pause + confirm via STT */
5840
+ gatePassTimes = [];
5841
+ // recent gate-PASSING chunks (helper zeroes residue — nonzero = vetted)
5842
+ handleLevel(rms) {
5843
+ if (this.usingAec) {
5844
+ if (!this.options.overlapEnergyHold || !this.speaking || !this.overlapCapable || this.pausedAt || rms < 50) return;
5845
+ const t = this.clock.now();
5846
+ this.gatePassTimes = this.gatePassTimes.filter((x) => t - x < 350);
5847
+ this.gatePassTimes.push(t);
5848
+ if (this.gatePassTimes.length < 2) return;
5849
+ this.gatePassTimes = [];
5850
+ this.pausedAt = t;
5851
+ this.player.pause();
5852
+ this.armResume();
5853
+ return;
5854
+ }
5855
+ if (!this.speaking) {
5856
+ this.baseline = 0;
5857
+ this.hot = 0;
5858
+ return;
5859
+ }
5860
+ if (!this.baseline) {
5861
+ this.baseline = rms;
5862
+ return;
5872
5863
  }
5873
- };
5874
- return [searchTool, callMcpTool];
5875
- }
5876
- function buildMcpCatalog(servers) {
5877
- const specs = [];
5878
- const routes = /* @__PURE__ */ new Map();
5879
- for (const m of servers) {
5880
- for (const s of m.specs) {
5881
- const base = `mcp__${m.name}__${s.name}`.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
5882
- let display = base;
5883
- for (let i = 2; routes.has(display); i++) display = `${base.slice(0, 128 - String(i).length - 1)}_${i}`;
5884
- specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
5885
- routes.set(display, { server: m.name, rawName: s.name });
5864
+ this.baseline = this.baseline * 0.9 + rms * 0.1;
5865
+ if (rms > Math.max(this.baseline * this.options.bargeRmsMult, this.options.bargeRmsFloor)) this.hot++;
5866
+ else this.hot = 0;
5867
+ if (this.hot >= 2 && !this.suspectUntil) {
5868
+ this.suspectUntil = this.clock.now() + 1300;
5869
+ this.clock.setTimeout(() => {
5870
+ this.suspectUntil = 0;
5871
+ }, 1350);
5886
5872
  }
5887
5873
  }
5888
- return { specs, routes };
5889
- }
5890
- function searchOverCatalog(servers, specs, routes, resolve, options) {
5891
- const tools = specs.length ? makeMcpToolSearch(specs, (name, args) => {
5892
- const r = routes.get(name);
5893
- if (!r) throw new Error(`unknown MCP tool '${name}' \u2014 use ToolSearch to find valid names`);
5894
- return resolve(r.server, r.rawName, args ?? {});
5895
- }, options) : [];
5896
- return { tools, serverNames: servers, toolCount: specs.length };
5897
- }
5898
- function makeMcpToolSearchFromMounted(mounted, options) {
5899
- const { specs, routes } = buildMcpCatalog(mounted);
5900
- const byName = new Map(mounted.map((m) => [m.name, m]));
5901
- return searchOverCatalog(mounted.map((m) => m.name), specs, routes, (server, rawName, args) => byName.get(server).client.callTool(rawName, args), options);
5902
- }
5903
- function makeLazyMcpToolSearch(servers, resolve, options) {
5904
- const { specs, routes } = buildMcpCatalog(servers);
5905
- return searchOverCatalog(servers.map((s) => s.name), specs, routes, resolve, options);
5906
- }
5874
+ };
5907
5875
 
5908
- // src/hooks.ts
5909
- var RecordingHooks = class {
5910
- /** tool name -> reason; a matching preToolUse call is blocked with that reason. */
5911
- constructor(blocks = {}) {
5912
- this.blocks = blocks;
5913
- }
5914
- blocks;
5915
- pre = [];
5916
- post = [];
5917
- outputs = [];
5918
- stops = [];
5919
- preToolUse(call, meta) {
5920
- this.pre.push({ call, meta });
5921
- const reason = this.blocks[call.name];
5922
- if (reason != null) return { block: true, reason };
5923
- }
5924
- postToolUse(call, result, meta) {
5925
- this.post.push({ call, result, meta });
5876
+ // src/voice/spokenSplitter.ts
5877
+ var OPEN = "<spoken>";
5878
+ var CLOSE = "</spoken>";
5879
+ var CLOSERS = `"')]}\xBB\u201D\u2019`;
5880
+ var hasSpeech = (s) => /[\p{L}\p{N}]/u.test(s);
5881
+ var SentenceCoalescer = class _SentenceCoalescer {
5882
+ buf = "";
5883
+ static isEnd(c) {
5884
+ return c === "\n" || c === "." || c === "!" || c === "?" || c === "\u2026";
5926
5885
  }
5927
- onToolOutput(call, chunk, meta) {
5928
- this.outputs.push({ call, chunk, meta });
5886
+ feed(delta) {
5887
+ if (delta) this.buf += delta;
5888
+ let cut = -1;
5889
+ for (let i = 0; i < this.buf.length; i++) if (_SentenceCoalescer.isEnd(this.buf[i])) cut = i;
5890
+ if (cut < 0) return "";
5891
+ if (this.buf[cut] !== "\n") while (cut + 1 < this.buf.length && CLOSERS.includes(this.buf[cut + 1])) cut++;
5892
+ const ready = this.buf.slice(0, cut + 1).trim();
5893
+ this.buf = this.buf.slice(cut + 1);
5894
+ return hasSpeech(ready) ? ready : "";
5929
5895
  }
5930
- onStop(finalText) {
5931
- this.stops.push(finalText);
5896
+ flush() {
5897
+ const s = this.buf.trim();
5898
+ this.buf = "";
5899
+ return hasSpeech(s) ? s : "";
5932
5900
  }
5933
5901
  };
5934
- var RecordingLifecycle = class {
5935
- /** @param startContext injected at session start; @param rewrite maps a submitted prompt to a new one. */
5936
- constructor(startContext, rewrite) {
5937
- this.startContext = startContext;
5938
- this.rewrite = rewrite;
5939
- }
5940
- startContext;
5941
- rewrite;
5942
- starts = 0;
5943
- prompts = [];
5944
- compactions = [];
5945
- subagentStops = [];
5946
- onSessionStart() {
5947
- this.starts++;
5948
- return this.startContext;
5949
- }
5950
- onUserPromptSubmit(text) {
5951
- this.prompts.push(text);
5952
- return this.rewrite?.(text);
5902
+ var SpokenSplitter = class {
5903
+ buf = "";
5904
+ inSpoken = false;
5905
+ /** True once any spoken char has ever been emitted (drives the no-spoken fallback). */
5906
+ spokeAny = false;
5907
+ /** Feed a delta; returns the spoken/detail spans completed by this chunk (either may be ''). */
5908
+ feed(delta) {
5909
+ this.buf += delta;
5910
+ return this.drain(false);
5953
5911
  }
5954
- onPreCompact(messages) {
5955
- this.compactions.push(messages.length);
5912
+ /** Drain any buffered partial. A trailing `<…` that never completed a tag is emitted as detail. */
5913
+ flush() {
5914
+ return this.drain(true);
5956
5915
  }
5957
- onSubagentStop(summary, info) {
5958
- this.subagentStops.push({ summary, label: info?.label });
5916
+ drain(final) {
5917
+ let spoken = "";
5918
+ let detail = "";
5919
+ while (this.buf.length) {
5920
+ const tag = this.inSpoken ? CLOSE : OPEN;
5921
+ const idx = this.buf.indexOf(tag);
5922
+ if (idx >= 0) {
5923
+ const text2 = this.buf.slice(0, idx);
5924
+ if (this.inSpoken) spoken += text2;
5925
+ else detail += text2;
5926
+ this.buf = this.buf.slice(idx + tag.length);
5927
+ this.inSpoken = !this.inSpoken;
5928
+ continue;
5929
+ }
5930
+ const lt = this.buf.lastIndexOf("<");
5931
+ const holdStart = lt >= 0 && tag.startsWith(this.buf.slice(lt)) ? lt : this.buf.length;
5932
+ const text = this.buf.slice(0, holdStart);
5933
+ if (this.inSpoken) spoken += text;
5934
+ else detail += text;
5935
+ this.buf = this.buf.slice(holdStart);
5936
+ break;
5937
+ }
5938
+ if (final && this.buf) {
5939
+ if (this.inSpoken) spoken += this.buf;
5940
+ else detail += this.buf;
5941
+ this.buf = "";
5942
+ }
5943
+ if (spoken.trim()) this.spokeAny = true;
5944
+ return { spoken, detail };
5959
5945
  }
5960
5946
  };
5961
5947
 
5962
- // src/index.ts
5963
- init_logging();
5964
-
5965
- // src/voice/engine.ts
5966
- init_logging();
5967
- var log11 = forComponent("VoiceEngine");
5968
- var now = () => performance.now();
5969
- var forSpeech = (t) => t.replace(/[*_`#]+/g, "").replace(/^[ \t]*[-•]\s+/gm, "").replace(/\s*[\u2013\u2014]\s*/g, ", ").replace(/[\u2010\u2011]/g, "-").replace(/\s*\|\s*/g, ", ").replace(/(\d)\s+%/g, "$1%").replace(/\.{3,}/g, ".");
5970
- var VoiceEngineOptions = class {
5971
- stt;
5972
- tts;
5973
- player;
5974
- /** a final utterance arrived (endpoint) — host dispatches it as a turn */
5975
- onUtterance = () => {
5976
- };
5977
- /** live partial transcript while listening (host renders the 🎤 line) */
5978
- onPartial = () => {
5979
- };
5980
- onState = () => {
5981
- };
5982
- /** user spoke/acted over playback — host aborts the in-flight turn (called AFTER audio is killed).
5983
- * phase: 'speaking' = cut mid-speech (real interruption); 'drain' = in the final audio tail
5984
- * (normal turn-taking — hosts shouldn't alarm). */
5985
- onBargeIn = () => {
5986
- };
5987
- /** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
5988
- ackPhrase = "";
5989
- /** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
5990
- * letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
5991
- * ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
5992
- utteranceMergeMs = 350;
5993
- /** Extended merge window (ms) for utterances that look incomplete (trailing conjunction/filler).
5994
- * Gives the user time to finish their thought without triggering a model call. */
5995
- incompleteMergeMs = 1500;
5996
- /** Grace window (ms) after an utterance dispatches, during which the user's own trailing audio cannot
5997
- * barge the reply it requested. Soniox keeps finalizing partials past <end>; without this they read
5998
- * as a barge and abort the fresh turn (live: mid-sentence self-interruption + steps=1→steps=0 double
5999
- * abort). Short enough that a genuine immediate barge ("no wait—") still lands right after. */
6000
- bargeGraceMs = 600;
6001
- /** Barge-in (talk over the assistant to interrupt). true = full-duplex (needs echo cancellation, or
6002
- * the assistant's own TTS bleeds back and self-interrupts). false = HALF-DUPLEX: the engine is deaf
6003
- * while audible (speaking + drain tail), so echo can never become a phantom turn — the right mode
6004
- * when there's no AEC (e.g. the non-VPIO mic fallback) and no headphones. Cost: can't interrupt. */
6005
- bargeIn = true;
6006
- /** Filler phrase spoken when holding for an incomplete utterance ('' disables). */
6007
- holdFiller = "";
6008
- /** Called when the engine holds an incomplete utterance (host can render a visual cue). */
6009
- onHold = () => {
6010
- };
6011
- /** heuristic (non-AEC) energy barge-in tuning */
6012
- bargeRmsMult = 2;
6013
- bargeRmsFloor = 500;
6014
- /** Overlap turn-taking (AEC tier, needs player.pause/resume) — human phone-call model, driven by
6015
- * the STT ITSELF (a trained speech classifier) instead of energy thresholds (energy could not
6016
- * separate residue bursts from speech in every room — hiccup whack-a-mole): a GENUINE partial
6017
- * (novel words dominate — echo of our own reply is inert) while speaking → PAUSE (exact-sample
6018
- * hold); partial grows into dominant-novel ≥2 words → cede (interrupt; the LLM re-enters); partial
6019
- * stalls/endpoints without ceding (backchannel by DURATION, not vocabulary) → resume + drop. false disables. */
6020
- overlapPause = true;
6021
- /** no new partial activity for this long while paused → resume, drop the interjection */
6022
- overlapResumeMs = 700;
6023
- /** A genuine barge over a LONG reply is defeated by the dominant-novel gate: Meet echoes our own
6024
- * speech back, so the partial is mostly our words + a few of hers → never "dominant novel" → it
6025
- * resumes (replaying old audio — the audible "completes the buffer" blip) instead of ceding.
6026
- * Mechanism-based discriminator: a re-PAUSE this soon after a resume = a persistent human, not an
6027
- * echo blip (which pauses once and stalls). Cede on the re-pause regardless of the novel gate. */
6028
- overlapRepauseCedeMs = 1500;
6029
- /** Speculative ENERGY pre-pause while speaking (AEC tier): two residue gate-passes within 350ms →
6030
- * pause ~300ms before the STT tokens land. But energy CANNOT separate residue bursts from speech
6031
- * (the documented whack-a-mole) — so a residue spike during loud playback false-pauses with NO user
6032
- * speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
6033
- * mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
6034
- overlapEnergyHold = false;
6035
- /** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
6036
- * tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
6037
- emotions = true;
6038
- /** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
6039
- showEmotions = false;
5948
+ // src/duplex.ts
5949
+ var log11 = forComponent("DuplexAgent");
5950
+ function describeCall(call) {
5951
+ const v = call.args && Object.values(call.args).find((x) => typeof x === "string" && x.trim());
5952
+ const hint = v ? ` (${String(v).replace(/\s+/g, " ").trim().slice(0, 48)})` : "";
5953
+ return `${call.name}${hint}`;
5954
+ }
5955
+ var DuplexAgentOptions = class {
5956
+ /** Any ai.libx.js AIClient — shared by all tiers (routed by model). */
5957
+ ai;
5958
+ /** The WORKER's filesystem (act + think). If omitted the worker keeps Agent's jailed-disk-at-cwd default. */
5959
+ fs;
5960
+ // The reflex IS the voice. 120b (not 20b) for channel discipline + instruction-following: the 20b
5961
+ // mislabels gpt-oss harmony channels under load, leaking raw analysis into the spoken `final` channel
5962
+ // (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
5963
+ reflexModel = "groq/openai/gpt-oss-120b";
5964
+ actModel = "anthropic/claude-sonnet-4-6";
5965
+ /** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
5966
+ thinkModel = "anthropic/claude-opus-4-8";
5967
+ /** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
5968
+ * provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
5969
+ * cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
5970
+ * leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
5971
+ providerOptionsFor;
5972
+ /** Escape hatches merged over the derived per-agent options. */
5973
+ reflexOptions;
5974
+ actOptions;
5975
+ thinkOptions;
5976
+ /** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
5977
+ * the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
5978
+ * pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
5979
+ * this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
5980
+ verifyActTasks = false;
5981
+ /** Receives the voice text_delta stream + task lifecycle events. */
5982
+ host;
5983
+ /** How many recent transcript messages are rendered into a worker's brief. */
5984
+ excerptTurns = 6;
5985
+ /** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
5986
+ * backchannels, impulsive first reactions before content (mimics real duplex conversation). */
5987
+ voiceStyle = "neutral";
5988
+ /** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
5989
+ * TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
5990
+ emotionTags = false;
5991
+ /** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
5992
+ * (post-spawn would race the worker's first edits). */
5993
+ onTaskStart;
5994
+ /** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
5995
+ * air. Off by default — each update costs a voice turn (LLM call + speech). */
5996
+ progressUpdates = false;
5997
+ /** Min ms between progress re-voices per task. */
5998
+ progressIntervalMs = 25e3;
5999
+ /** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
6000
+ * the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
6001
+ * voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
6002
+ askRelay = false;
6003
+ /** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
6004
+ askTimeoutMs = 12e4;
6005
+ /** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
6006
+ * bounding memory over a long-lived session. Running tasks are never evicted. */
6007
+ maxTaskRecords = 50;
6008
+ /** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
6009
+ * (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
6010
+ quickLook;
6011
+ /** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
6012
+ * tools directly (no delegation needed) and implicit capture guidance. */
6013
+ memoryDir;
6014
+ /** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
6015
+ memoryUserDir;
6040
6016
  };
6041
- var VoiceEngine = class _VoiceEngine {
6017
+ var RESERVED_EVENT_MARKER = /\[task\b[^\]\n]*\b(?:completed|failed|progress|asks)\b/i;
6018
+ var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
6019
+ var STAGE_DIRECTION_RE = /^\(\s*(?:(?:waiting|checking|searching|thinking|processing|loading|working|fetching|looking)\b[^)]*|[^)]*(?:\.\.\.|…)\s*)\)$/i;
6020
+ var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji. Never emit stage directions or parenthetical asides about your own process \u2014 nothing like "(waiting for the result...)" or "(checking)"; while work runs, either say it as plain speech or end your turn.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. DELEGATION RULE \u2014 decide for yourself, the user never has to push: if you cannot answer confidently from the conversation plus trivial well-known knowledge, do NOT refuse and do NOT guess \u2014 dispatch `Act` immediately with a clear brief and say you are checking. Anything needing CURRENT data (weather, news, prices, dates, sky/astronomy, "right now"), real computation, or verification is an automatic dispatch \u2014 never a refusal. The user should never need to say "search the web" or "think harder" to make you act; needing fresh or verified information IS the trigger. Refuse only what your worker genuinely cannot do (check `capabilities`), and say why. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 give at most a half-sentence aside ("still on it \u2014 running tests now") and end your turn. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, just say "still on it" and end the turn \u2014 never poll it again and again in a loop. Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
6021
+ var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
6022
+ var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
6023
+ var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
6024
+ var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
6025
+ var DuplexAgent = class _DuplexAgent {
6042
6026
  options;
6043
- state = "idle";
6044
- stt;
6045
- tts;
6046
- player;
6047
- speaking = false;
6048
- // audible (deltas flowing OR audio draining)
6049
- ctxOpen = false;
6050
- // the current TTS context still accepts deltas (false once end-frame sent)
6051
- interrupted = false;
6052
- // barge-in latch: drop in-flight deltas until the next legitimate turn
6053
- spokeDeltas = false;
6054
- // a TTS context is open for the current spoken turn
6055
- drainTimer = null;
6056
- // heuristic tier state (inert under AEC) — frozen as validated in the experiment
6057
- echoWords = /* @__PURE__ */ new Set();
6058
- prevReply = "";
6059
- reply = "";
6060
- echoUntil = 0;
6061
- baseline = 0;
6062
- hot = 0;
6063
- suspectUntil = 0;
6064
- ackAt = 0;
6065
- // when the micro-ack was spoken — its echo can leak before the AEC filter converges
6066
- bargeGraceUntil = 0;
6067
- // no barge-in until this time — the user's OWN trailing audio (after the
6068
- // utterance that JUST dispatched this turn) must not immediately re-interrupt the reply it requested.
6069
- pendingUtt = "";
6070
- // endpointed text held for the merge window
6071
- pendingTimer = null;
6072
- lastInterrupted = null;
6073
- // overlap (pause) tier state — AEC + pause-capable sinks only
6074
- pausedAt = 0;
6075
- lastResumeAt = 0;
6076
- // when the overlap last resumed from a false alarm — a quick re-pause cedes
6077
- lastOverlapPartial = "";
6078
- // change-detection: only NEW partial text counts as activity
6079
- resumeTimer = null;
6080
- turnStartAt = 0;
6081
- // timestamp when the current turn began (for TTFT logging)
6082
- // Central speech queue (above the TTS context): complete worker utterances serialize into ONE
6083
- // playback stream, one-at-a-time, never splicing into the live reflex's open utterance.
6084
- uttQueue = [];
6085
- // Per-turn emotion-tag parser (reset on beginSpeech) — converts `[emotion]` → Cartesia inline tags
6086
- // for TTS, tracks tag-free prose for echo discrimination, and surfaces display text for the screen.
6087
- emo = null;
6027
+ voice;
6028
+ tasks = /* @__PURE__ */ new Map();
6029
+ queue = Promise.resolve();
6030
+ seq = 0;
6031
+ pendingEvents = [];
6032
+ /** Out-of-band follow-up attribution for the events coalescing into the next flush turn: TRUE iff ≥1 of
6033
+ * the tasks being integrated was NON-CLEAN (early-stop/failure). Carried out-of-band on the enqueue call
6034
+ * by the caller that KNOWS the outcome — a plain boolean the MODEL CANNOT PERTURB. It is NOT scanned from
6035
+ * worker-authored event text (v1: an "Outcome:" substring over-stamped siblings) and NOT keyed on a brief
6036
+ * string the reflex re-authors (v2: a paraphrased escalation brief missed the Set → followUp:false →
6037
+ * RE-ENABLED unbounded auto-escalation, the dangerous runaway direction). See [[wrong-discriminator]] /
6038
+ * [[drive-real-reflex]] / [[fakeaiclient-blind-to-wire-format]]. */
6039
+ pendingNonClean = false;
6040
+ flushQueued = false;
6041
+ /** Per-voice-turn guards (reset by resetTurn at each turn's start). The reflex is a weak model:
6042
+ * left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
6043
+ * Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
6044
+ turnDispatched = false;
6045
+ // an Act/Think fired this turn
6046
+ spokeBeforeDispatch = false;
6047
+ // the reflex ALREADY acked before dispatching — the forced post-dispatch text step must stay silent (live: "Got it…" twice)
6048
+ turnBriefs = /* @__PURE__ */ new Set();
6049
+ // briefs dispatched this turn (detect identical re-dispatch)
6050
+ spokeThisTurn = false;
6051
+ // any non-empty text_delta streamed this turn
6052
+ externalSpeech = false;
6053
+ // host spoke on our behalf (adaptive micro-ack) — not reflex output
6054
+ heldThisTurn = false;
6055
+ // Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
6056
+ nudging = false;
6057
+ // re-ack pass in flight: block ALL tools, prevent recursion
6058
+ reflexBuf = "";
6059
+ // accumulated reflex text this turn (fabricated-event detection)
6060
+ reflexForwarded = 0;
6061
+ // chars of reflexBuf already forwarded to the host/TTS
6062
+ fabricationCut = false;
6063
+ // reflex emitted a reserved [task …] marker → suppress its tail
6064
+ /** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
6065
+ * carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
6066
+ * turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
6067
+ * escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
6068
+ * ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
6069
+ * means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
6070
+ * ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
6071
+ * flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
6072
+ * fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
6073
+ * ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
6074
+ turnFollowUp = false;
6075
+ /** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
6076
+ * the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
6077
+ autoEscalations = 0;
6078
+ static MAX_AUTO_ESCALATIONS = 8;
6079
+ /** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
6080
+ pendingAsks = /* @__PURE__ */ new Map();
6081
+ /** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
6082
+ * PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
6083
+ * confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
6084
+ * aborts it: output dropped, history rolled back, the final dispatches normally. */
6085
+ spec;
6086
+ /** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
6087
+ speculativeAbortedCalls = 0;
6088
+ /** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
6089
+ * intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
6090
+ * writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
6091
+ * instead (the endpointed final then dispatches normally and may use the tool for real). */
6092
+ static SPEC_SAFE_TOOLS = /* @__PURE__ */ new Set(["QuickLook", "TaskStatus", "Hold"]);
6093
+ /** Lazily resolved memory tools (async loadMemory runs in initMemory). */
6094
+ memoryReady;
6088
6095
  constructor(options) {
6089
- this.options = { ...new VoiceEngineOptions(), ...options };
6096
+ this.options = { ...new DuplexAgentOptions(), ...options };
6090
6097
  const o = this.options;
6091
- if (!o.stt || !o.tts || !o.player) throw new Error("VoiceEngine needs stt, tts and player (see cli/voice.ts VoiceIO for platform defaults)");
6092
- this.stt = o.stt;
6093
- this.tts = o.tts;
6094
- this.player = o.player;
6098
+ if (o.memoryDir && o.fs) {
6099
+ this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
6100
+ }
6101
+ const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
6102
+ const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
6103
+ const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
6104
+ const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
6105
+ const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
6106
+ const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
6107
+ const mcpNames = [
6108
+ ...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
6109
+ ...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
6110
+ ];
6111
+ const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
6112
+ const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
6113
+ Today's date: ${(/* @__PURE__ */ new Date()).toDateString()}.`;
6114
+ const tools = [
6115
+ ...o.reflexOptions?.tools ?? [],
6116
+ this.actTool(),
6117
+ ...o.thinkModel !== false ? [this.thinkTool()] : [],
6118
+ this.taskStatusTool(),
6119
+ this.cancelTaskTool(),
6120
+ this.quickLookTool(),
6121
+ this.answerTaskTool(),
6122
+ this.holdTool()
6123
+ ];
6124
+ const host = o.host;
6125
+ const voiceHost = host && {
6126
+ ask: host.ask ? (q) => host.ask(q) : void 0,
6127
+ confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
6128
+ notify: (ev) => {
6129
+ if (ev?.kind === "text_delta" && typeof ev.message === "string") {
6130
+ if (this.heldThisTurn) return;
6131
+ if (this.fabricationCut) return;
6132
+ if (this.turnDispatched && this.spokeBeforeDispatch) return;
6133
+ const msg = ev.message;
6134
+ this.reflexBuf += msg;
6135
+ this.scrubStageDirections();
6136
+ const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
6137
+ if (m) {
6138
+ this.fabricationCut = true;
6139
+ log11.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
6140
+ const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
6141
+ if (!safe) return;
6142
+ if (safe.trim()) this.spokeThisTurn = true;
6143
+ this.emitHost({ ...ev, message: safe });
6144
+ return;
6145
+ }
6146
+ const held = this.reflexBuf.length - this.reflexForwarded;
6147
+ const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
6148
+ let upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
6149
+ const paren = this.reflexBuf.lastIndexOf("(");
6150
+ if (paren >= this.reflexForwarded && !this.reflexBuf.includes(")", paren) && this.reflexBuf.length - paren <= 80)
6151
+ upto = Math.min(upto, paren);
6152
+ const out = this.reflexBuf.slice(this.reflexForwarded, upto);
6153
+ this.reflexForwarded = upto;
6154
+ if (!out) return;
6155
+ if (out.trim()) this.spokeThisTurn = true;
6156
+ this.emitHost({ ...ev, message: out });
6157
+ return;
6158
+ }
6159
+ host.notify?.(ev);
6160
+ }
6161
+ };
6162
+ this.voice = new Agent({
6163
+ ai: o.ai,
6164
+ fs: new MemFilesystem2(),
6165
+ model: o.reflexModel,
6166
+ stream: true,
6167
+ host: voiceHost,
6168
+ // The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
6169
+ // never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
6170
+ // it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
6171
+ askUserQuestion: false,
6172
+ systemPrompt: prompt,
6173
+ instructionFiles: false,
6174
+ maxSteps: 8,
6175
+ timeoutMs: 3e4,
6176
+ ...o.reflexOptions,
6177
+ tools,
6178
+ // Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
6179
+ hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
6180
+ });
6095
6181
  }
6096
- async start() {
6097
- this.tts.onAudio = (c) => {
6098
- if (this.speaking) this.player.write(c);
6182
+ /** Resolve memory tools + inject index into voice system prompt (once). */
6183
+ async initMemory() {
6184
+ if (!this.memoryReady) return;
6185
+ const mem = await this.memoryReady;
6186
+ this.memoryReady = void 0;
6187
+ this.voice.options.tools.push(...mem.tools);
6188
+ if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
6189
+ }
6190
+ /** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
6191
+ * turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
6192
+ flushHeldReflexTail() {
6193
+ if (this.fabricationCut) return;
6194
+ this.scrubStageDirections();
6195
+ const tail = this.reflexBuf.slice(this.reflexForwarded);
6196
+ this.reflexForwarded = this.reflexBuf.length;
6197
+ if (!tail) return;
6198
+ if (tail.trim()) this.spokeThisTurn = true;
6199
+ this.emitHost({ kind: "text_delta", message: tail });
6200
+ }
6201
+ /** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
6202
+ * Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
6203
+ * reflexForwarded would corrupt the forward offset. */
6204
+ scrubStageDirections() {
6205
+ if (this.reflexForwarded >= this.reflexBuf.length) return;
6206
+ const region = this.reflexBuf.slice(this.reflexForwarded);
6207
+ const scrubbed = region.replace(/\([^()]*\)/g, (s) => STAGE_DIRECTION_RE.test(s) ? "" : s);
6208
+ if (scrubbed !== region) this.reflexBuf = this.reflexBuf.slice(0, this.reflexForwarded) + scrubbed;
6209
+ }
6210
+ /** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
6211
+ resetTurn() {
6212
+ this.turnDispatched = false;
6213
+ this.spokeBeforeDispatch = false;
6214
+ this.turnBriefs.clear();
6215
+ this.spokeThisTurn = false;
6216
+ this.externalSpeech = false;
6217
+ this.heldThisTurn = false;
6218
+ this.reflexBuf = "";
6219
+ this.reflexForwarded = 0;
6220
+ this.fabricationCut = false;
6221
+ this.turnFollowUp = false;
6222
+ this.voice.options.toolChoice = void 0;
6223
+ }
6224
+ /** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
6225
+ * now wait for the push" (CC's Task model). Block the temptations — TaskStatus polling and identical
6226
+ * re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
6227
+ * still allowed (parallel independent work). During a re-ack pass, block every tool. */
6228
+ dispatchGuard() {
6229
+ return {
6230
+ preToolUse: (call) => {
6231
+ if (this.spec?.state === "pending" && !_DuplexAgent.SPEC_SAFE_TOOLS.has(call.name)) {
6232
+ log11.verbose(`speculation aborted: reflex called ${call.name} on unconfirmed input`);
6233
+ this.abortSpeculation();
6234
+ return { block: true, reason: "Speculative turn aborted." };
6235
+ }
6236
+ if (this.nudging) return { block: true, reason: "Just say one short spoken acknowledgement \u2014 no tools this turn." };
6237
+ if (!this.turnDispatched) return;
6238
+ if (call.name === "TaskStatus")
6239
+ return { block: true, reason: "You just dispatched a task this turn \u2014 do NOT poll. Give one short spoken acknowledgement and end your turn; the result arrives later as a [task \u2026] event." };
6240
+ if ((call.name === "Act" || call.name === "Think") && this.turnBriefs.has(String(call.args?.brief ?? "")))
6241
+ return { block: true, reason: "You already dispatched this exact task \u2014 acknowledge briefly and end your turn." };
6242
+ }
6099
6243
  };
6100
- this.stt.onPartial = (text) => this.handlePartial(text);
6101
- this.stt.onUtterance = (text) => this.handleUtterance(text);
6102
- this.stt.onLevel = (rms) => this.handleLevel(rms);
6103
- await Promise.all([this.tts.connect(), this.stt.start()]);
6104
- this.setState("listening");
6105
- log11.debug(`voice I/O up (${this.stt.usingAec ? "AEC" : "heuristic echo"} capture)`);
6106
6244
  }
6107
- get usingAec() {
6108
- return this.stt.usingAec;
6245
+ /** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
6246
+ * micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
6247
+ * inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
6248
+ * still dead air, so silentTurn ignores external speech unless work was dispatched. */
6249
+ noteExternalSpeech() {
6250
+ this.externalSpeech = true;
6109
6251
  }
6110
- /** Flip barge-in at runtime (e.g. the mic fell back to non-VPIO → go half-duplex so echo can't leak). */
6111
- setBargeIn(on) {
6112
- this.options.bargeIn = on;
6252
+ /** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
6253
+ * (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
6254
+ * gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
6255
+ * voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
6256
+ * Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
6257
+ get silentTurn() {
6258
+ const ackedByHost = this.externalSpeech && this.turnDispatched;
6259
+ return !!this.options.host && !this.spokeThisTurn && !ackedByHost && !this.heldThisTurn;
6113
6260
  }
6114
- /** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
6115
- setShowEmotions(on) {
6116
- this.options.showEmotions = on;
6261
+ /** A turn that voiced nothing is dead air. Re-prompt the reflex ONCE so the LLM itself voices a short
6262
+ * line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
6263
+ * Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
6264
+ async ackIfSilent(fallback) {
6265
+ const dispatched = this.turnDispatched;
6266
+ this.nudging = true;
6267
+ try {
6268
+ await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
6269
+ } catch (e) {
6270
+ log11.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
6271
+ } finally {
6272
+ this.nudging = false;
6273
+ }
6274
+ if (!this.spokeThisTurn) {
6275
+ const pool = fallback ? [fallback] : dispatched ? _DuplexAgent.FALLBACK_ACKS : _DuplexAgent.FALLBACK_RETRY;
6276
+ const fresh = pool.filter((p) => p !== this.lastFallback);
6277
+ const line = (fresh.length ? fresh : pool)[Math.floor(Math.random() * (fresh.length || pool.length))];
6278
+ this.lastFallback = line;
6279
+ this.emitHost({ kind: "text_delta", message: line });
6280
+ }
6117
6281
  }
6118
- idleWaiters = [];
6119
- setState(s) {
6120
- if (this.state === s) return;
6121
- this.state = s;
6122
- this.options.onState(s);
6123
- if (s !== "speaking" && s !== "thinking") {
6124
- for (const r of this.idleWaiters.splice(0)) r();
6282
+ /** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
6283
+ * hosts/tests key on it. */
6284
+ static FALLBACK_ACKS = ["Okay, on it.", "On it.", "Alright, working on it."];
6285
+ static FALLBACK_RETRY = ["Sorry, could you say that again?", "Hm, I missed that \u2014 could you say that again?"];
6286
+ lastFallback = "";
6287
+ /** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
6288
+ * If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
6289
+ * speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
6290
+ * host NOW (this is the latency win) and streaming continues live. Any other content aborts the
6291
+ * speculation first (rolled back silently) and runs a normal turn behind it. */
6292
+ send(content) {
6293
+ const spec = this.spec;
6294
+ if (spec?.state === "pending") {
6295
+ if (typeof content === "string" && speculationConfirms(spec.text, content)) {
6296
+ spec.state = "confirmed";
6297
+ spec.finalText = content;
6298
+ for (const ev of spec.buf.splice(0)) this.options.host?.notify?.(ev);
6299
+ spec.decide("confirm");
6300
+ this.notify("diag", "speculation_confirmed", { text: spec.text.slice(0, 80) });
6301
+ return spec.done;
6302
+ }
6303
+ this.abortSpeculation();
6125
6304
  }
6305
+ return this.enqueue(async () => {
6306
+ await this.initMemory();
6307
+ this.resetTurn();
6308
+ const res = await this.voice.send(content);
6309
+ this.flushHeldReflexTail();
6310
+ if (this.silentTurn) await this.ackIfSilent();
6311
+ return res;
6312
+ });
6126
6313
  }
6127
- /** Resolve when the engine is no longer speaking (immediate if already idle). */
6128
- awaitIdle() {
6129
- if (this.state !== "speaking" && this.state !== "thinking") return Promise.resolve();
6130
- return new Promise((r) => this.idleWaiters.push(r));
6314
+ /** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
6315
+ * output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
6316
+ * voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
6317
+ * into the transcript between the speculative messages and their rollback. No-op if a speculation
6318
+ * is already in flight. */
6319
+ speculate(text) {
6320
+ if (!text.trim() || this.spec) return;
6321
+ let decide;
6322
+ const decision = new Promise((r) => {
6323
+ decide = r;
6324
+ });
6325
+ const spec = {
6326
+ text,
6327
+ state: "pending",
6328
+ buf: [],
6329
+ ctl: new AbortController(),
6330
+ decide,
6331
+ decision,
6332
+ done: void 0
6333
+ };
6334
+ this.spec = spec;
6335
+ spec.done = this.enqueue(async () => {
6336
+ const empty = { text: "", steps: 0, finishReason: "aborted", messages: this.voice.transcript };
6337
+ if (spec.state === "aborted") {
6338
+ this.speculativeAbortedCalls++;
6339
+ if (this.spec === spec) this.spec = void 0;
6340
+ return empty;
6341
+ }
6342
+ await this.initMemory();
6343
+ this.resetTurn();
6344
+ const base = this.voice.transcript.length;
6345
+ const prevSignal = this.voice.options.signal;
6346
+ this.voice.options.signal = spec.ctl.signal;
6347
+ let res;
6348
+ try {
6349
+ res = await this.voice.send(spec.text);
6350
+ } catch (e) {
6351
+ log11.warn(`speculative turn failed: ${e instanceof Error ? e.message : e}`);
6352
+ } finally {
6353
+ this.voice.options.signal = prevSignal;
6354
+ }
6355
+ const timer = setTimeout(() => {
6356
+ spec.state = spec.state === "pending" ? "aborted" : spec.state;
6357
+ spec.decide("abort");
6358
+ }, 1e4);
6359
+ timer.unref?.();
6360
+ const d = await spec.decision;
6361
+ clearTimeout(timer);
6362
+ if (d === "abort") {
6363
+ if (this.voice.transcript.length > base) this.voice.transcript.length = base;
6364
+ this.speculativeAbortedCalls++;
6365
+ log11.verbose(`speculation aborted (${this.speculativeAbortedCalls} total): "${spec.text.slice(0, 50)}"`);
6366
+ if (this.spec === spec) this.spec = void 0;
6367
+ return res ?? empty;
6368
+ }
6369
+ for (let i = base; i < this.voice.transcript.length; i++) {
6370
+ const m = this.voice.transcript[i];
6371
+ if (m.role === "user" && contentText(m.content) === spec.text) {
6372
+ m.content = spec.finalText;
6373
+ break;
6374
+ }
6375
+ }
6376
+ this.spec = void 0;
6377
+ this.flushHeldReflexTail();
6378
+ if (this.silentTurn) await this.ackIfSilent();
6379
+ return res ?? empty;
6380
+ });
6131
6381
  }
6132
- // --- speaking side (host-driven) ---
6133
- /** open a spoken turn (idempotent — safe from both onUtterance and first-delta paths).
6134
- * `ack` speaks the configured micro-ack as the context opener (utterance path only —
6135
- * masks LLM TTFT; re-voice turns begun by their first delta skip it). */
6136
- beginSpeech(ack = false) {
6137
- if (this.speaking && this.ctxOpen) return;
6138
- if (this.drainTimer) {
6139
- clearTimeout(this.drainTimer);
6140
- this.drainTimer = null;
6141
- }
6142
- this.interrupted = false;
6143
- this.resetOverlap(true);
6144
- if (!this.speaking) this.player.markTurn();
6145
- this.speaking = true;
6146
- this.ctxOpen = true;
6147
- this.spokeDeltas = false;
6148
- this.reply = "";
6149
- this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
6150
- this.echoWords = new Set(this.words(this.prevReply));
6151
- this.tts.newContext();
6152
- if (ack && this.options.ackPhrase) {
6153
- this.tts.speak(this.options.ackPhrase + " ", true);
6154
- this.spokeDeltas = true;
6155
- this.ackAt = now();
6156
- }
6157
- if (!this.turnStartAt) this.turnStartAt = now();
6158
- this.setState("thinking");
6382
+ /** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
6383
+ * Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
6384
+ * attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
6385
+ abortSpeculation() {
6386
+ const spec = this.spec;
6387
+ if (spec?.state !== "pending") return;
6388
+ spec.state = "aborted";
6389
+ spec.buf.length = 0;
6390
+ spec.ctl.abort();
6391
+ spec.decide("abort");
6392
+ this.notify("diag", "speculation_aborted", { text: spec.text.slice(0, 80) });
6159
6393
  }
6160
- /** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
6161
- * host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
6162
- speakDelta(text) {
6163
- if (this.interrupted) return "";
6164
- if (!this.speaking || !this.ctxOpen) this.beginSpeech();
6165
- const { speech, display, prose } = this.emo ? this.emo.feed(text) : { speech: text, display: text, prose: text };
6166
- this.reply += prose;
6167
- for (const w of this.words(this.reply)) this.echoWords.add(w);
6168
- this.tts.speak(forSpeech(speech), true);
6169
- if (!this.spokeDeltas && this.turnStartAt) log11.debug(`ttft: ${Math.round(now() - this.turnStartAt)}ms`);
6170
- this.spokeDeltas = true;
6171
- this.setState("speaking");
6172
- return display;
6394
+ /** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
6395
+ cancelTask(id) {
6396
+ const rec = this.tasks.get(id);
6397
+ if (!rec) return `No task '${id}'.`;
6398
+ if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
6399
+ rec.status = "cancelled";
6400
+ rec.controller.abort();
6401
+ return `Task ${rec.id} (${rec.label}) cancelled.`;
6173
6402
  }
6174
- /** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
6175
- endSpeech() {
6176
- this.interrupted = false;
6177
- if (!this.speaking) return;
6178
- this.ctxOpen = false;
6179
- if (this.emo) {
6180
- const t = this.emo.flush();
6181
- this.emo = null;
6182
- if (t.prose) this.reply += t.prose;
6183
- if (t.speech) this.tts.speak(forSpeech(t.speech), true);
6403
+ /** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
6404
+ * delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
6405
+ * tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
6406
+ * Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
6407
+ parkInFlightDeliveries() {
6408
+ const parked = [];
6409
+ for (const rec of this.tasks.values())
6410
+ if (rec.status === "running" && !rec.deliveryParked) {
6411
+ rec.deliveryParked = true;
6412
+ parked.push(rec.id);
6413
+ }
6414
+ return parked;
6415
+ }
6416
+ /** Resolve when all queued voice turns AND all in-flight worker tasks have settled (tests, graceful shutdown). */
6417
+ async idle() {
6418
+ while (true) {
6419
+ const q = this.queue;
6420
+ await q.catch(() => {
6421
+ });
6422
+ await Promise.all([...this.tasks.values()].map((t) => t.promise));
6423
+ if (this.queue === q && ![...this.tasks.values()].some((t) => t.status === "running")) return;
6184
6424
  }
6185
- if (this.reply) this.prevReply = this.reply;
6186
- const settle = () => {
6187
- if (this.ctxOpen) {
6188
- this.drainTimer = null;
6425
+ }
6426
+ /** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
6427
+ enqueue(fn) {
6428
+ const run = this.queue.then(fn, fn);
6429
+ this.queue = run.then(() => {
6430
+ }, () => {
6431
+ });
6432
+ return run;
6433
+ }
6434
+ notify(kind, message, data) {
6435
+ this.emitHost({ kind, message, data });
6436
+ }
6437
+ /** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
6438
+ * hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
6439
+ * order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
6440
+ * which bypasses this via host.notify directly) passes through untouched. */
6441
+ emitHost(ev) {
6442
+ const spec = this.spec;
6443
+ if (spec && (ev.kind === "text_delta" || ev.kind === "hold_filler")) {
6444
+ if (spec.state === "pending") {
6445
+ spec.buf.push(ev);
6189
6446
  return;
6190
6447
  }
6191
- if (this.pausedAt) {
6192
- this.drainTimer = setTimeout(settle, 250);
6193
- return;
6448
+ if (spec.state === "aborted") return;
6449
+ }
6450
+ this.options.host?.notify?.(ev);
6451
+ }
6452
+ /** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
6453
+ * `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
6454
+ * marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
6455
+ * text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
6456
+ queueRevoice(event, nonClean = false) {
6457
+ this.pendingEvents.push(event);
6458
+ if (nonClean) this.pendingNonClean = true;
6459
+ if (this.flushQueued) return;
6460
+ this.flushQueued = true;
6461
+ void this.enqueue(async () => {
6462
+ this.flushQueued = false;
6463
+ const events = this.pendingEvents.splice(0);
6464
+ const nonCleanTurn = this.pendingNonClean;
6465
+ this.pendingNonClean = false;
6466
+ if (!events.length) return;
6467
+ const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
6468
+ this.resetTurn();
6469
+ this.turnFollowUp = nonCleanTurn;
6470
+ await this.voice.send(events.join("\n"));
6471
+ this.flushHeldReflexTail();
6472
+ if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
6473
+ this.notify("revoice_done", "");
6474
+ });
6475
+ }
6476
+ /** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
6477
+ * Act briefs get a self-verify footer — the worker's report is trusted without review, so it
6478
+ * must check its own work before reporting (nearly free under prompt caching; measured honest:
6479
+ * it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
6480
+ buildBrief(brief, tier = "act", deliver = true) {
6481
+ const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
6482
+ const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
6483
+ const deliverContract = deliver ? `
6484
+
6485
+ ## DELIVER (spoken delivery)
6486
+ You are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown. NEVER enumerate in speech \u2014 no numbered or bulleted lists ("One. \u2026 Two. \u2026" is robotic). Deliver multiple items as flowing conversation with brief connective phrasing ("here's one\u2026", "and another\u2026", "oh, and\u2026"), pausing between items with sentence breaks, not numbers.` + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
6487
+ return (recent ? `${brief}
6488
+
6489
+ ## Recent conversation (for context)
6490
+ ${recent}` : brief) + verify + deliverContract;
6491
+ }
6492
+ /** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
6493
+ spawnWorker(id, label, briefText, tier, brief, followUp) {
6494
+ const o = this.options;
6495
+ const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
6496
+ const tierModel = tier === "think" ? o.thinkModel : o.actModel;
6497
+ const controller = new AbortController();
6498
+ const base = tierOpts?.hooks ?? o.actOptions?.hooks;
6499
+ const report = o.progressUpdates ? this.progressReporter(id) : void 0;
6500
+ const tail = [];
6501
+ const pushTail = (line) => {
6502
+ tail.push(line.slice(0, 200));
6503
+ if (tail.length > 120) tail.splice(0, tail.length - 120);
6504
+ };
6505
+ const hooks = {
6506
+ ...base,
6507
+ preToolUse: async (call, meta) => {
6508
+ const d = await base?.preToolUse?.(call, meta);
6509
+ pushTail(`\u2699 ${describeCall(call)}`);
6510
+ report?.pre(call);
6511
+ return d;
6512
+ },
6513
+ postToolUse: async (call, result, meta) => {
6514
+ await base?.postToolUse?.(call, result, meta);
6515
+ const last = result?.trim().split("\n").filter(Boolean).pop();
6516
+ if (last) pushTail(` \u21B3 ${last}`);
6517
+ report?.post(call);
6518
+ },
6519
+ onToolOutput: (call, chunk, meta) => {
6520
+ base?.onToolOutput?.(call, chunk, meta);
6521
+ report?.output(chunk);
6522
+ }
6523
+ };
6524
+ const relayAsk = async (q) => {
6525
+ const opts = q.options?.length ? ` Options: ${q.options.map((x) => x.label).join(", ")}.` : "";
6526
+ const a = await this.parkQuestion(id, `${q.question}${opts}`);
6527
+ return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
6528
+ };
6529
+ const splitter = new SpokenSplitter();
6530
+ const speak = (seg) => {
6531
+ if (seg && !this.tasks.get(id)?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
6532
+ };
6533
+ const coalescer = new SentenceCoalescer();
6534
+ const feedSpoken = (s) => {
6535
+ const ready = coalescer.feed(s);
6536
+ if (ready) speak(ready);
6537
+ };
6538
+ const flushSpoken = () => speak(coalescer.flush());
6539
+ const askBridge = o.askRelay ? { ask: relayAsk } : o.host?.ask ? { ask: (q) => o.host.ask(q) } : {};
6540
+ const workerHost = {
6541
+ ...askBridge,
6542
+ notify: (ev) => {
6543
+ if (ev?.kind === "text_delta" && typeof ev.message === "string") {
6544
+ const { spoken, detail } = splitter.feed(ev.message);
6545
+ feedSpoken(spoken);
6546
+ if (detail.trim()) pushTail(detail.trim());
6547
+ return;
6548
+ }
6549
+ }
6550
+ };
6551
+ const agentOpts = {
6552
+ ai: o.ai,
6553
+ fs: o.fs,
6554
+ model: tierModel,
6555
+ ...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
6556
+ ...tierOpts,
6557
+ // Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
6558
+ // main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
6559
+ providerOptions: o.providerOptionsFor?.(tierModel),
6560
+ stream: true,
6561
+ // worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
6562
+ host: workerHost,
6563
+ // carries BOTH ask AND the <spoken>-splitting notify
6564
+ ...hooks ? { hooks } : {},
6565
+ signal: controller.signal
6566
+ // shared with the checker so a cancel tears down both
6567
+ };
6568
+ const promise = new Agent(agentOpts).run(briefText).then((res) => {
6569
+ const { spoken, detail } = splitter.flush();
6570
+ feedSpoken(spoken);
6571
+ if (detail.trim()) pushTail(detail.trim());
6572
+ flushSpoken();
6573
+ return res;
6574
+ }).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err) => this.onWorkerFailed(id, err));
6575
+ this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
6576
+ if (this.tasks.size > this.options.maxTaskRecords)
6577
+ for (const [tid, rec] of this.tasks) {
6578
+ if (this.tasks.size <= this.options.maxTaskRecords) break;
6579
+ if (rec.status !== "running") this.tasks.delete(tid);
6194
6580
  }
6195
- this.drainTimer = null;
6196
- this.speaking = false;
6197
- if (this.turnStartAt) log11.debug(`turn: ${Math.round(now() - this.turnStartAt)}ms (incl. playback)`);
6198
- this.echoUntil = now() + 2500;
6199
- if (!this.usingAec) this.stt.reset();
6200
- this.setState("listening");
6201
- if (this.uttQueue.length) this.pumpQueue();
6581
+ }
6582
+ /** Fresh-context check of a successful Act task: a NEW agent (same model/fs/tools, but NO shared
6583
+ * conversation context) re-reads the file state against the brief and fixes any gap. The fix lands
6584
+ * on the shared fs automatically (workers write fs directly, no overlay), so grading sees the
6585
+ * corrected state. Bounded to ONE pass. Off unless `verifyActTasks`; never runs for think/failed/
6586
+ * cancelled tasks. Usage is merged so /cost reflects the real (worker + checker) spend. */
6587
+ async maybeVerify(id, brief, res, tier, agentOpts, askBridge) {
6588
+ if (!this.options.verifyActTasks || tier !== "act" || res.finishReason !== "stop") return res;
6589
+ if (this.tasks.get(id)?.status === "cancelled") return res;
6590
+ const { stream: _stream, host: _host, ...restOpts } = agentOpts;
6591
+ const checkerOpts = {
6592
+ ...restOpts,
6593
+ ...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
6202
6594
  };
6203
- const drainThenSettle = () => {
6204
- if (this.drainTimer) clearTimeout(this.drainTimer);
6205
- this.drainTimer = setTimeout(settle, this.player.drainMs() + 300);
6595
+ const checkBrief = `${this.buildBrief(brief, tier, false)}
6596
+
6597
+ ## VERIFY MODE
6598
+ Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
6599
+ this.notify("task_verify", `task ${id}: verifying`, { id });
6600
+ const cres = await new Agent(checkerOpts).run(checkBrief);
6601
+ if (cres.finishReason !== "stop") {
6602
+ log11.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
6603
+ this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
6604
+ }
6605
+ const sum = (a = 0, b = 0) => a + b;
6606
+ return {
6607
+ ...res,
6608
+ steps: res.steps + cres.steps,
6609
+ // Merge the checker's messages so downstream tool-call/step accounting includes BOTH agents
6610
+ // (else a verified task's toolCalls would undercount vs its steps/usage).
6611
+ messages: [...res.messages, ...cres.messages],
6612
+ usageEstimated: res.usageEstimated || cres.usageEstimated,
6613
+ usage: res.usage && cres.usage ? {
6614
+ promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
6615
+ completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
6616
+ totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
6617
+ cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
6618
+ cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
6619
+ } : res.usage ?? cres.usage
6206
6620
  };
6207
- if (this.spokeDeltas) {
6208
- this.tts.onDone = drainThenSettle;
6209
- this.tts.end();
6210
- if (this.drainTimer) clearTimeout(this.drainTimer);
6211
- this.drainTimer = setTimeout(drainThenSettle, 15e3);
6212
- } else drainThenSettle();
6213
6621
  }
6214
- /** text of the reply cut by the last barge-in — consumed by the host to tell the model what
6215
- * the user did NOT hear. Cleared on read. */
6216
- takeInterruptedReply() {
6217
- const r = this.lastInterrupted;
6218
- this.lastInterrupted = null;
6219
- return r;
6622
+ /** Throttled per-task progress: worker tool calls → at most one progress re-voice per interval.
6623
+ * Two sources, one throttle: completed steps (post) and a heartbeat for a SINGLE long tool call
6624
+ * (pre records the in-flight call; a self-cleaning timer narrates "still inside Bash — 70s").
6625
+ * Completion supersedes: nothing is emitted once the task has settled. */
6626
+ progressReporter(id) {
6627
+ let lastAt = Date.now();
6628
+ let steps = 0;
6629
+ let inflight = null;
6630
+ const due = () => {
6631
+ if (this.pendingAsks.size) return void 0;
6632
+ const rec = this.tasks.get(id);
6633
+ return rec && rec.status === "running" && Date.now() - lastAt >= this.options.progressIntervalMs ? rec : void 0;
6634
+ };
6635
+ const emit = (rec, line, call) => {
6636
+ lastAt = Date.now();
6637
+ this.notify("task_progress", `task ${id} (${rec.label}): ${line}`, { id, steps, call: call.name });
6638
+ this.queueRevoice(`[task ${id} progress] ${line}`);
6639
+ };
6640
+ const timer = setInterval(() => {
6641
+ const rec = this.tasks.get(id);
6642
+ if (!rec || rec.status !== "running") return clearInterval(timer);
6643
+ if (!inflight || !due()) return;
6644
+ const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
6645
+ emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
6646
+ }, Math.max(this.options.progressIntervalMs, 250));
6647
+ timer.unref?.();
6648
+ return {
6649
+ pre: (call) => {
6650
+ inflight = { call, at: Date.now(), tail: "" };
6651
+ },
6652
+ output: (chunk) => {
6653
+ if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
6654
+ },
6655
+ // digest only — NEVER re-voices directly
6656
+ post: (call) => {
6657
+ steps++;
6658
+ inflight = null;
6659
+ const rec = due();
6660
+ if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
6661
+ }
6662
+ };
6220
6663
  }
6221
- /** Speak a short filler phrase without starting a model turn (stays in listening mode after). */
6222
- speakFiller(text) {
6223
- if (!text || this.speaking) return;
6224
- this.beginSpeech();
6225
- this.speakDelta(text);
6226
- this.endSpeech();
6664
+ /** Park a question under `askId` (a task id, or any unique key for permission asks): re-voices
6665
+ * '[task <id> asks] …' and resolves with the user's answer via AnswerTask — or '' on timeout/
6666
+ * task settle (callers map '' to deny / best-judgment). Workers never block forever. */
6667
+ parkQuestion(askId, question) {
6668
+ return new Promise((resolve) => {
6669
+ let settled = false;
6670
+ const finish = (answer) => {
6671
+ if (settled) return;
6672
+ settled = true;
6673
+ clearTimeout(timer);
6674
+ this.pendingAsks.delete(askId);
6675
+ resolve(answer);
6676
+ };
6677
+ const timer = setTimeout(() => {
6678
+ this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
6679
+ finish("");
6680
+ }, this.options.askTimeoutMs);
6681
+ this.pendingAsks.set(askId, { question, resolve: finish });
6682
+ this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
6683
+ this.queueRevoice(`[task ${askId} asks] ${question}
6684
+ (Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
6685
+ });
6227
6686
  }
6228
- /** Enqueue a COMPLETE worker utterance (already-split spoken text) onto the central speech queue.
6229
- * If nothing is currently speaking it plays immediately; otherwise it queues and plays after the
6230
- * current utterance fully ends (settle → pumpQueue) — never spliced into an open reflex utterance. */
6231
- enqueueUtterance(text) {
6232
- if (!text || !/[\p{L}\p{N}]/u.test(text)) return;
6233
- this.uttQueue.push(text);
6234
- if (!this.speaking) this.pumpQueue();
6687
+ /** Resolve any question a settling/cancelled task left parked (its answer can no longer matter). */
6688
+ dropAsk(id) {
6689
+ this.pendingAsks.get(id)?.resolve("");
6235
6690
  }
6236
- /** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
6237
- * next one from the settle completion (endSpeech), so utterances serialize without overlap. */
6238
- pumpQueue() {
6239
- if (this.speaking) return;
6240
- const text = this.uttQueue.shift();
6241
- if (text == null) return;
6242
- this.beginSpeech();
6243
- this.speakDelta(text);
6244
- this.endSpeech();
6691
+ /** Build the INTEGRATION TURN prompt for a NON-CLEAN settled worker (early stop / failure). A clean
6692
+ * success never reaches here — it streams its own `<spoken>` delivery during the run. For a partial
6693
+ * or failed result the outcome re-enters the reflex as a decision (like a tool_result flowing back
6694
+ * into a normal agent loop): the reflex evaluates the outcome against the original intent and chooses
6695
+ * what to do next.
6696
+ *
6697
+ * Decision branches (the reflex acts on them with EXISTING tools — no new surface):
6698
+ * • accept → SPEAK the (partial) result plainly — don't dress a failure up as success.
6699
+ * • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
6700
+ * exists AND this task wasn't already a follow-up (one hop max). Wires the dead
6701
+ * "Reserve Think for a problem Act already FAILED at" promise.
6702
+ * • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
6703
+ * • ask → ask the user ONE concrete question if genuinely blocked.
6704
+ *
6705
+ * Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
6706
+ * failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
6707
+ integrationPrompt(rec, outcome, body, finishReason) {
6708
+ const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
6709
+ const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
6710
+ const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
6711
+ const hasThink = this.options.thinkModel !== false;
6712
+ const options = [];
6713
+ if (!rec.followUp && canEscalate && hasThink)
6714
+ options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
6715
+ if (!rec.followUp && canEscalate)
6716
+ options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
6717
+ options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
6718
+ options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
6719
+ const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
6720
+ const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
6721
+ return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
6245
6722
  }
6246
- /** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
6247
- interrupt() {
6248
- this.uttQueue = [];
6249
- if (!this.speaking && !this.drainTimer) return;
6250
- if (this.drainTimer) {
6251
- clearTimeout(this.drainTimer);
6252
- this.drainTimer = null;
6723
+ onWorkerSettled(id, res) {
6724
+ this.dropAsk(id);
6725
+ const rec = this.tasks.get(id);
6726
+ if (res.finishReason === "aborted" || rec.status === "cancelled") {
6727
+ rec.status = "cancelled";
6728
+ this.notify("task_cancelled", `task ${id} (${rec.label}) cancelled`);
6729
+ return;
6253
6730
  }
6254
- this.resetOverlap(false);
6255
- this.lastResumeAt = 0;
6256
- const heardChars = Math.round(Math.max(0, this.player.playedMs()) / 1e3 * 15);
6257
- if (this.reply) this.lastInterrupted = { full: this.reply, heard: this.reply.slice(0, heardChars) };
6258
- this.speaking = false;
6259
- this.ctxOpen = false;
6260
- this.interrupted = true;
6261
- this.suspectUntil = 0;
6262
- this.echoUntil = now() + Math.max(2500, this.player.drainMs() + 3e3);
6263
- this.tts.cancel();
6264
- this.player.kill();
6265
- if (!this.usingAec) this.stt.reset();
6266
- if (this.reply) this.prevReply = this.reply;
6267
- this.setState("listening");
6731
+ if (res.finishReason === "error") {
6732
+ const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
6733
+ return this.failTask(rec, msg);
6734
+ }
6735
+ rec.status = "done";
6736
+ rec.result = res.text;
6737
+ const incomplete = res.finishReason !== "stop";
6738
+ log11.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
6739
+ this.notify("task_done", `task ${id} (${rec.label}) completed`, {
6740
+ id,
6741
+ text: res.text,
6742
+ usage: res.usage,
6743
+ usageEstimated: res.usageEstimated,
6744
+ finishReason: res.finishReason,
6745
+ steps: res.steps,
6746
+ toolCalls: res.messages.filter((m) => m.role === "tool").length
6747
+ });
6748
+ if (incomplete) {
6749
+ return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
6750
+ }
6751
+ const tail = rec.splitter?.flush();
6752
+ if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
6753
+ if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
6754
+ if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
6755
+ this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
6268
6756
  }
6269
- stop() {
6270
- this.uttQueue = [];
6271
- if (this.resumeTimer) clearTimeout(this.resumeTimer);
6272
- if (this.pendingTimer) clearTimeout(this.pendingTimer);
6273
- if (this.drainTimer) clearTimeout(this.drainTimer);
6274
- this.stt.stop();
6275
- this.player.kill();
6276
- this.tts.close();
6277
- this.setState("idle");
6757
+ onWorkerFailed(id, err) {
6758
+ this.failTask(this.tasks.get(id), err instanceof Error ? err.message : String(err));
6278
6759
  }
6279
- // --- listening side (STT-driven) ---
6280
- words(s) {
6281
- return s.toLowerCase().replace(/[^a-z0-9\s]/g, "").split(/\s+/).filter((w) => w.length >= 2);
6760
+ failTask(rec, msg) {
6761
+ this.dropAsk(rec.id);
6762
+ rec.status = "error";
6763
+ rec.result = msg;
6764
+ log11.warn(`task ${rec.id} failed: ${msg}`);
6765
+ this.notify("task_error", `task ${rec.id} (${rec.label}) failed: ${msg}`);
6766
+ this.queueRevoice(this.integrationPrompt(rec, "error", msg, "error"), true);
6282
6767
  }
6283
- novelWords(text) {
6284
- return this.words(text).filter((w) => !this.echoWords.has(w));
6768
+ // --- voice tools (closures over this instance) ---
6769
+ /** Live-switch the think tier: `false` disables (removes the Think tool from the voice agent),
6770
+ * a model id enables (adds the tool if missing). The system-prompt THINK_SLOT text is frozen at
6771
+ * construction — the tool's own description carries the routing guidance, so a live enable works;
6772
+ * dispatch()'s think→act fallback covers any straggler calls after a live disable. */
6773
+ setThinkModel(model) {
6774
+ this.options.thinkModel = model;
6775
+ const tools = this.voice.options.tools;
6776
+ const i = tools.findIndex((t) => t.name === "Think");
6777
+ if (model === false && i >= 0) tools.splice(i, 1);
6778
+ else if (model !== false && i < 0) tools.push(this.thinkTool());
6285
6779
  }
6286
- echoActive() {
6287
- return this.speaking || now() < this.echoUntil;
6780
+ /** User/programmatic spawn: the CLI's /act and /think commands. Returns the task id.
6781
+ * `followUp` marks an automatic escalation/re-delegation (set by the integration turn) so the new
6782
+ * task's own integration turn won't escalate again — capping auto-follow-ups to one hop. */
6783
+ async dispatch(brief, tier = "act", label, followUp = false) {
6784
+ if (tier === "think" && this.options.thinkModel === false) tier = "act";
6785
+ if (followUp) this.autoEscalations++;
6786
+ const id = `t${++this.seq}`;
6787
+ const lbl = label ?? tier;
6788
+ await this.options.onTaskStart?.(id, lbl);
6789
+ this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
6790
+ this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
6791
+ return id;
6288
6792
  }
6289
- /** Genuine user speech vs our own bleed (AEC tier): novel words must DOMINATE, not merely exist.
6290
- * Degraded AEC + an STT mis-hearing manufactures a single novel word out of pure echo (a name or
6291
- * rare word in our own reply comes back transcribed slightly differently — 1 novel / N words).
6292
- * A real interjection is mostly novel ("stop", "wait what") — short utterances pass on ratio,
6293
- * longer ones on count. */
6294
- genuine(text) {
6295
- const total = this.words(text).length;
6296
- const novel = this.novelWords(text).length;
6297
- return novel > 0 && novel / Math.max(1, total) > 0.5;
6793
+ actTool() {
6794
+ return {
6795
+ name: "Act",
6796
+ description: 'Escalate real work (reading/editing files, searching, running tasks, building) to a standard background worker. Returns immediately with a task id; the result arrives later as a "[task <id> completed]" event. Provide a clear, self-contained `brief` (the worker does not hear the live conversation).',
6797
+ parameters: {
6798
+ type: "object",
6799
+ required: ["brief"],
6800
+ properties: {
6801
+ brief: { type: "string", description: "full, self-contained instructions for the worker" },
6802
+ label: { type: "string", description: "a short (2-4 word) label for the task" }
6803
+ }
6804
+ },
6805
+ run: async ({ brief, label }) => {
6806
+ this.spokeBeforeDispatch = this.spokeThisTurn;
6807
+ this.turnDispatched = true;
6808
+ this.turnBriefs.add(String(brief ?? ""));
6809
+ this.voice.options.toolChoice = "none";
6810
+ const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
6811
+ return this.spokeBeforeDispatch ? `Acting on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
6812
+ }
6813
+ };
6298
6814
  }
6299
- handlePartial(text) {
6300
- if (this.speaking) {
6301
- if (!this.options.bargeIn) return;
6302
- if (now() < this.bargeGraceUntil) {
6303
- if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
6304
- return;
6815
+ thinkTool() {
6816
+ return {
6817
+ name: "Think",
6818
+ description: "Escalate to a premium deep-reasoning agent for complex analysis, architecture decisions, hard debugging, or planning. Same async pattern as Act \u2014 returns a task id. Use when the problem needs careful thought before (or instead of) action. Do not use Think for simple tasks \u2014 Act is cheaper and faster.",
6819
+ parameters: {
6820
+ type: "object",
6821
+ required: ["brief"],
6822
+ properties: {
6823
+ brief: { type: "string", description: "the question or problem to reason about deeply" },
6824
+ label: { type: "string", description: "a short (2-4 word) label for the task" }
6825
+ }
6826
+ },
6827
+ run: async ({ brief, label }) => {
6828
+ this.spokeBeforeDispatch = this.spokeThisTurn;
6829
+ this.turnDispatched = true;
6830
+ this.turnBriefs.add(String(brief ?? ""));
6831
+ this.voice.options.toolChoice = "none";
6832
+ const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
6833
+ return this.spokeBeforeDispatch ? `Thinking on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
6305
6834
  }
6306
- if (this.overlapCapable) {
6307
- const txt = text.trim();
6308
- if (!txt || txt === this.lastOverlapPartial) return;
6309
- this.lastOverlapPartial = txt;
6310
- if (!this.genuine(txt)) {
6311
- if (this.pausedAt) this.armResume();
6312
- return;
6835
+ };
6836
+ }
6837
+ taskStatusTool() {
6838
+ return {
6839
+ name: "TaskStatus",
6840
+ description: "Status of background tasks. Pass `id` for one task, or omit it to list all.",
6841
+ parameters: { type: "object", properties: { id: { type: "string" } } },
6842
+ run: async ({ id }) => {
6843
+ const list = id ? [this.tasks.get(String(id))].filter(Boolean) : [...this.tasks.values()];
6844
+ if (!list.length) return id ? `No task '${id}'.` : "No background tasks.";
6845
+ return list.map((t) => `${t.id} (${t.label}): ${t.status}`).join("\n");
6846
+ }
6847
+ };
6848
+ }
6849
+ /** Sub-100ms read-only lookups the voice may do itself — everything else stays Act-only.
6850
+ * fs-only (no shell; the engine is VFS-abstracted): time, git branch (.git/HEAD read), ls, file
6851
+ * head. Output is hard-capped so a lookup can never bloat the skinny voice context. */
6852
+ quickLookTool() {
6853
+ const CAP = 2e3;
6854
+ const kinds = [.../* @__PURE__ */ new Set(["time", "branch", "ls", "file", "capabilities", ...Object.keys(this.options.quickLook ?? {})])];
6855
+ return {
6856
+ name: "QuickLook",
6857
+ description: `Instant read-only lookup \u2014 one of: ${kinds.join(", ")}. For trivial facts only; anything needing search, commands, or reasoning goes through Act.`,
6858
+ parameters: {
6859
+ type: "object",
6860
+ required: ["what"],
6861
+ properties: {
6862
+ what: { type: "string", enum: kinds, description: "what to look up" },
6863
+ path: { type: "string", description: "for ls/file: the path to look at" }
6313
6864
  }
6314
- if (!this.pausedAt) {
6315
- this.pausedAt = now();
6316
- this.player.pause();
6317
- if (this.lastResumeAt && now() - this.lastResumeAt < this.options.overlapRepauseCedeMs) {
6318
- this.interrupt();
6319
- this.options.onBargeIn(this.ctxOpen ? "speaking" : "drain");
6320
- return;
6865
+ },
6866
+ run: async ({ what, path }) => {
6867
+ const fs = this.options.fs;
6868
+ try {
6869
+ const over = this.options.quickLook?.[String(what)];
6870
+ if (over) return await over(path ? String(path) : void 0);
6871
+ switch (String(what)) {
6872
+ case "capabilities": {
6873
+ const actTools = this.options.actOptions?.tools ?? [];
6874
+ const names = actTools.map((t) => t.name);
6875
+ const mcpServers = Object.keys(this.options.actOptions?.providerOptions?.mcpServers ?? {});
6876
+ const mcpNote = mcpServers.length ? ` Plus MCP servers your worker can use: ${mcpServers.join(", ")} (e.g. browser-bridge \u2192 drive a real browser: open tabs, navigate, click, screenshot).` : "";
6877
+ if (!names.length)
6878
+ return "Your worker uses Act's default local toolset (reading/editing files, running shell commands). No extra tools (e.g. web/internet) are configured; if a request is not a basic file or shell operation, assume you can't do it and say so." + mcpNote;
6879
+ const hasFetch = names.some((n) => /WebFetch/i.test(n));
6880
+ const hasBrowser = names.some((n) => /browser.*(navigate|click|page|type)/i.test(n));
6881
+ const hasSearch = names.some((n) => /(^|_)WebSearch$|search/i.test(n) && !/WebFetch|browser/i.test(n));
6882
+ const notes = [];
6883
+ if (hasFetch) notes.push("WebFetch retrieves ONE specific URL you are given \u2014 it is not a search engine.");
6884
+ if (hasBrowser) notes.push("The browser tools drive a real browser: you CAN open a site and, if needed, navigate to a search engine and search there \u2014 but it is manual and takes a moment, not an instant lookup.");
6885
+ else if (!hasSearch && hasFetch) notes.push('You have no general web-search tool, so for an instant "search the web" you can only fetch a URL they provide.');
6886
+ const webNote = notes.length ? " NOTE: " + notes.join(" ") : "";
6887
+ return `Tools your background worker (Act) can actually use: ${names.join(", ")}. Read each name literally and match the request to a SPECIFIC tool; if none fits, you do NOT have that ability \u2014 say so honestly.` + webNote + mcpNote;
6888
+ }
6889
+ case "time":
6890
+ return (/* @__PURE__ */ new Date()).toString();
6891
+ case "branch": {
6892
+ if (!fs) return "unavailable (no filesystem)";
6893
+ const head = (await fs.readFile(".git/HEAD")).trim();
6894
+ return head.startsWith("ref: refs/heads/") ? `branch: ${head.slice("ref: refs/heads/".length)}` : `detached HEAD at ${head.slice(0, 12)}`;
6895
+ }
6896
+ case "ls": {
6897
+ if (!fs) return "unavailable (no filesystem)";
6898
+ const p = String(path ?? ".");
6899
+ try {
6900
+ const names = await fs.readDir(p);
6901
+ return names.slice(0, 50).join("\n") + (names.length > 50 ? `
6902
+ \u2026 (+${names.length - 50} more)` : "");
6903
+ } catch {
6904
+ const names = await fs.readDir(".").catch(() => []);
6905
+ return `'${p}' not found here \u2014 you are likely already inside it. Current directory listing:
6906
+ ` + names.slice(0, 50).join("\n") + (names.length > 50 ? `
6907
+ \u2026 (+${names.length - 50} more)` : "");
6908
+ }
6909
+ }
6910
+ case "file": {
6911
+ if (!fs) return "unavailable (no filesystem)";
6912
+ if (!path) return "file lookup needs a path";
6913
+ try {
6914
+ const text = await fs.readFile(String(path));
6915
+ return text.length > CAP ? text.slice(0, CAP) + `
6916
+ \u2026 (truncated \u2014 ${text.length} chars total; Act for the full file)` : text;
6917
+ } catch {
6918
+ const names = await fs.readDir(".").catch(() => []);
6919
+ return `'${path}' not found. Current directory contains:
6920
+ ` + names.slice(0, 50).join("\n");
6921
+ }
6922
+ }
6923
+ default:
6924
+ return `unknown lookup '${what}'`;
6321
6925
  }
6926
+ } catch (e) {
6927
+ return `lookup failed: ${e?.message ?? e}`;
6322
6928
  }
6323
- if (this.words(txt).length >= 2) {
6324
- const phase = this.ctxOpen ? "speaking" : "drain";
6325
- this.interrupt();
6326
- this.options.onBargeIn(phase);
6327
- return;
6328
- }
6329
- this.armResume();
6330
- return;
6331
6929
  }
6332
- const barge = this.usingAec ? this.genuine(text) : this.novelWords(text).length >= (this.suspectUntil ? 1 : 2);
6333
- if (barge) {
6334
- const phase = this.ctxOpen ? "speaking" : "drain";
6335
- this.interrupt();
6336
- this.options.onBargeIn(phase);
6930
+ };
6931
+ }
6932
+ answerTaskTool() {
6933
+ return {
6934
+ name: "AnswerTask",
6935
+ description: `Relay the user's answer to a pending question from a background task (the "[task <id> asks]" events). Pass the id from the event and the user's answer.`,
6936
+ parameters: {
6937
+ type: "object",
6938
+ required: ["id", "answer"],
6939
+ properties: { id: { type: "string" }, answer: { type: "string", description: "the user's answer, verbatim or faithfully summarized" } }
6940
+ },
6941
+ run: async ({ id, answer }) => {
6942
+ const ask = this.pendingAsks.get(String(id));
6943
+ if (!ask) return `No pending question for '${id}' \u2014 it may have been answered already or timed out.`;
6944
+ ask.resolve(String(answer ?? ""));
6945
+ return `Answer relayed \u2014 task ${id} resumes.`;
6337
6946
  }
6338
- return;
6339
- }
6340
- if (this.pendingUtt && text.trim()) {
6341
- if (this.pendingTimer) clearTimeout(this.pendingTimer);
6342
- this.pendingTimer = setTimeout(() => this.flushUtterance(), Math.max(800, this.options.utteranceMergeMs));
6343
- }
6344
- if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
6947
+ };
6345
6948
  }
6346
- /** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
6347
- * overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
6348
- * utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
6349
- * a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
6350
- * check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
6351
- mergeUtterance(prev, next) {
6352
- if (!prev) return next;
6353
- if (!next) return prev;
6354
- const pw = prev.split(/\s+/), nw = next.split(/\s+/);
6355
- const norm2 = (w) => w.toLowerCase().replace(/[^a-z0-9]/g, "");
6356
- const max = Math.min(pw.length, nw.length);
6357
- for (let k = max; k > 0; k--) {
6358
- let match = true;
6359
- for (let i = 0; i < k; i++) if (norm2(pw[pw.length - k + i]) !== norm2(nw[i])) {
6360
- match = false;
6361
- break;
6949
+ holdTool() {
6950
+ return {
6951
+ name: "Hold",
6952
+ description: 'The user seems mid-thought \u2014 hold the turn (stay listening) instead of answering. Optionally pass a short filler ("mhm", "go on") to speak while waiting. Use when the message sounds incomplete, trailing off, or like they paused to think.',
6953
+ parameters: {
6954
+ type: "object",
6955
+ properties: {
6956
+ filler: { type: "string", description: 'optional short filler to speak ("mhm", "go on", "mm-hm")' }
6957
+ }
6958
+ },
6959
+ run: async ({ filler }) => {
6960
+ this.heldThisTurn = true;
6961
+ this.notify("hold_filler", filler ? String(filler) : "");
6962
+ return "Holding \u2014 listening for the rest of the user's thought. Do not respond further this turn.";
6963
+ }
6964
+ };
6965
+ }
6966
+ cancelTaskTool() {
6967
+ return {
6968
+ name: "CancelTask",
6969
+ description: "Cancel a running background task by id.",
6970
+ parameters: { type: "object", required: ["id"], properties: { id: { type: "string" } } },
6971
+ run: async ({ id }) => this.cancelTask(String(id))
6972
+ };
6973
+ }
6974
+ };
6975
+
6976
+ // src/mcp.ts
6977
+ function toResult(result) {
6978
+ if (result == null) return { text: "" };
6979
+ if (typeof result === "string") return { text: result };
6980
+ const content = result.content;
6981
+ if (Array.isArray(content)) {
6982
+ const texts = [];
6983
+ const images = [];
6984
+ for (const c of content) {
6985
+ if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
6986
+ images.push({ mimeType: c.mimeType, data: c.data });
6987
+ } else if (typeof c?.text === "string") {
6988
+ texts.push(c.text);
6989
+ } else {
6990
+ texts.push(JSON.stringify(c));
6362
6991
  }
6363
- if (match) return [...pw, ...nw.slice(k)].join(" ");
6364
6992
  }
6365
- return `${prev} ${next}`;
6366
- }
6367
- static TRAIL_RE = /(?:^|\s)(?:and|but|or|so|to|the|a|an|of|in|for|with|that|if|uh|um|like|about|from|into|on|is|are|was|were|,)$/i;
6368
- /** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
6369
- looksIncomplete(text) {
6370
- return _VoiceEngine.TRAIL_RE.test(text.trim());
6993
+ const text = texts.join("\n");
6994
+ if (text || images.length) return { text, ...images.length ? { images } : {} };
6371
6995
  }
6372
- handleUtterance(text) {
6373
- if (this.speaking && (this.ctxOpen || this.pausedAt) && this.overlapCapable) {
6374
- this.stt.reset();
6375
- return;
6376
- }
6377
- if (this.echoActive() && (!this.options.bargeIn || (this.usingAec ? !this.genuine(text) : this.novelWords(text).length < 2))) {
6378
- this.stt.reset();
6379
- return;
6996
+ return { text: JSON.stringify(result) };
6997
+ }
6998
+ function mcpToolToAgentTool(spec, callTool, prefix = "mcp__") {
6999
+ return {
7000
+ name: `${prefix}${spec.name}`,
7001
+ description: spec.description ?? `MCP tool ${spec.name}`,
7002
+ parameters: spec.inputSchema ?? { type: "object", properties: {} },
7003
+ async run(args, _ctx) {
7004
+ const r = toResult(await callTool(spec.name, args ?? {}));
7005
+ return r.images?.length ? r : r.text;
6380
7006
  }
6381
- const squash = (t) => t.toLowerCase().replace(/[^a-z]/g, "").replace(/(.)\1+/g, "$1");
6382
- if (this.ackAt && now() - this.ackAt < 6e3 && squash(text) === squash(this.options.ackPhrase)) {
6383
- this.ackAt = 0;
6384
- return;
7007
+ };
7008
+ }
7009
+ function mcpToolsToAgentTools(specs, callTool, prefix = "mcp__", filter) {
7010
+ return (filter ? specs.filter(filter) : specs).map((s) => mcpToolToAgentTool(s, callTool, prefix));
7011
+ }
7012
+ function describeSpec(s) {
7013
+ const schema = s.inputSchema ? `
7014
+ args: ${JSON.stringify(s.inputSchema)}` : "";
7015
+ return `${s.name} \u2014 ${s.description ?? "(no description)"}${schema}`;
7016
+ }
7017
+ function makeMcpToolSearch(specs, callTool, options = {}) {
7018
+ const maxResults = options.maxResults ?? 10;
7019
+ const byName = new Map(specs.map((s) => [s.name, s]));
7020
+ const catalogLine = `${specs.length} MCP tool(s) available \u2014 search by keyword, then call by exact name.`;
7021
+ const searchTool = {
7022
+ name: "ToolSearch",
7023
+ description: `Search the available MCP tools by keyword (${catalogLine}). Returns matching tool names + their argument schemas; call one with \`McpCall\`.`,
7024
+ parameters: { type: "object", required: ["query"], properties: { query: { type: "string", description: "keywords to match against tool name + description" } } },
7025
+ async run({ query }) {
7026
+ const q = String(query ?? "").trim();
7027
+ if (!q) return catalogLine;
7028
+ const { kept } = topByRelevance(specs, q, (s) => `${s.name} ${s.description ?? ""}`, maxResults);
7029
+ if (!kept.length) return `(no MCP tool matches "${q}" \u2014 try broader keywords)`;
7030
+ return kept.map(describeSpec).join("\n");
6385
7031
  }
6386
- this.pendingUtt = this.mergeUtterance(this.pendingUtt, text);
6387
- if (this.pendingTimer) clearTimeout(this.pendingTimer);
6388
- if (this.options.incompleteMergeMs && this.looksIncomplete(this.pendingUtt)) {
6389
- log11.verbose(`hold: incomplete utterance "${this.pendingUtt.slice(-40)}"`);
6390
- this.options.onHold();
6391
- if (this.options.holdFiller && !this.speaking) {
6392
- this.beginSpeech();
6393
- this.speakDelta(this.options.holdFiller);
6394
- this.endSpeech();
7032
+ };
7033
+ const callMcpTool = {
7034
+ name: "McpCall",
7035
+ description: "Call an MCP tool discovered via `ToolSearch`, by its exact name. Pass its arguments as `args`.",
7036
+ parameters: {
7037
+ type: "object",
7038
+ required: ["name"],
7039
+ properties: {
7040
+ name: { type: "string", description: "exact tool name from ToolSearch" },
7041
+ args: { type: "object", description: "arguments object for the tool (per its schema)" }
6395
7042
  }
6396
- this.pendingTimer = setTimeout(() => this.flushUtterance(), this.options.incompleteMergeMs);
6397
- return;
6398
- }
6399
- if (!this.options.utteranceMergeMs || this.words(this.pendingUtt).length >= 4) return this.flushUtterance();
6400
- this.pendingTimer = setTimeout(() => this.flushUtterance(), this.options.utteranceMergeMs);
6401
- }
6402
- flushUtterance() {
6403
- if (this.pendingTimer) {
6404
- clearTimeout(this.pendingTimer);
6405
- this.pendingTimer = null;
7043
+ },
7044
+ async run({ name, args }) {
7045
+ const n = String(name ?? "");
7046
+ if (!byName.has(n)) return `Error: unknown MCP tool '${n}'. Use ToolSearch to find valid names.`;
7047
+ const r = toResult(await callTool(n, args ?? {}));
7048
+ return r.images?.length ? r : r.text;
6406
7049
  }
6407
- const text = this.pendingUtt;
6408
- this.pendingUtt = "";
6409
- if (text) {
6410
- this.turnStartAt = now();
6411
- this.bargeGraceUntil = now() + this.options.bargeGraceMs;
6412
- this.options.onUtterance(text);
7050
+ };
7051
+ return [searchTool, callMcpTool];
7052
+ }
7053
+ function buildMcpCatalog(servers) {
7054
+ const specs = [];
7055
+ const routes = /* @__PURE__ */ new Map();
7056
+ for (const m of servers) {
7057
+ for (const s of m.specs) {
7058
+ const base = `mcp__${m.name}__${s.name}`.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
7059
+ let display = base;
7060
+ for (let i = 2; routes.has(display); i++) display = `${base.slice(0, 128 - String(i).length - 1)}_${i}`;
7061
+ specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
7062
+ routes.set(display, { server: m.name, rawName: s.name });
6413
7063
  }
6414
7064
  }
6415
- get overlapCapable() {
6416
- return this.usingAec && this.options.overlapPause && !!this.player.pause && !!this.player.resume;
7065
+ return { specs, routes };
7066
+ }
7067
+ function searchOverCatalog(servers, specs, routes, resolve, options) {
7068
+ const tools = specs.length ? makeMcpToolSearch(specs, (name, args) => {
7069
+ const r = routes.get(name);
7070
+ if (!r) throw new Error(`unknown MCP tool '${name}' \u2014 use ToolSearch to find valid names`);
7071
+ return resolve(r.server, r.rawName, args ?? {});
7072
+ }, options) : [];
7073
+ return { tools, serverNames: servers, toolCount: specs.length };
7074
+ }
7075
+ function makeMcpToolSearchFromMounted(mounted, options) {
7076
+ const { specs, routes } = buildMcpCatalog(mounted);
7077
+ const byName = new Map(mounted.map((m) => [m.name, m]));
7078
+ return searchOverCatalog(mounted.map((m) => m.name), specs, routes, (server, rawName, args) => byName.get(server).client.callTool(rawName, args), options);
7079
+ }
7080
+ function makeLazyMcpToolSearch(servers, resolve, options) {
7081
+ const { specs, routes } = buildMcpCatalog(servers);
7082
+ return searchOverCatalog(servers.map((s) => s.name), specs, routes, resolve, options);
7083
+ }
7084
+
7085
+ // src/hooks.ts
7086
+ var RecordingHooks = class {
7087
+ /** tool name -> reason; a matching preToolUse call is blocked with that reason. */
7088
+ constructor(blocks = {}) {
7089
+ this.blocks = blocks;
6417
7090
  }
6418
- armResume() {
6419
- if (this.resumeTimer) clearTimeout(this.resumeTimer);
6420
- this.resumeTimer = setTimeout(() => {
6421
- this.resumeTimer = null;
6422
- if (!this.pausedAt) return;
6423
- this.stt.reset();
6424
- this.resetOverlap(true);
6425
- }, this.options.overlapResumeMs);
7091
+ blocks;
7092
+ pre = [];
7093
+ post = [];
7094
+ outputs = [];
7095
+ stops = [];
7096
+ preToolUse(call, meta) {
7097
+ this.pre.push({ call, meta });
7098
+ const reason = this.blocks[call.name];
7099
+ if (reason != null) return { block: true, reason };
6426
7100
  }
6427
- resetOverlap(resume) {
6428
- if (this.resumeTimer) {
6429
- clearTimeout(this.resumeTimer);
6430
- this.resumeTimer = null;
6431
- }
6432
- if (this.pausedAt && resume) {
6433
- this.player.resume?.();
6434
- this.lastResumeAt = now();
6435
- }
6436
- this.pausedAt = 0;
6437
- this.lastOverlapPartial = "";
6438
- this.gatePassTimes = [];
7101
+ postToolUse(call, result, meta) {
7102
+ this.post.push({ call, result, meta });
6439
7103
  }
6440
- /** energy two-stage barge-in (heuristic tier only): spike over echo baseline → pause + confirm via STT */
6441
- gatePassTimes = [];
6442
- // recent gate-PASSING chunks (helper zeroes residue — nonzero = vetted)
6443
- handleLevel(rms) {
6444
- if (this.usingAec) {
6445
- if (!this.options.overlapEnergyHold || !this.speaking || !this.overlapCapable || this.pausedAt || rms < 50) return;
6446
- const t = now();
6447
- this.gatePassTimes = this.gatePassTimes.filter((x) => t - x < 350);
6448
- this.gatePassTimes.push(t);
6449
- if (this.gatePassTimes.length < 2) return;
6450
- this.gatePassTimes = [];
6451
- this.pausedAt = t;
6452
- this.player.pause();
6453
- this.armResume();
6454
- return;
6455
- }
6456
- if (!this.speaking) {
6457
- this.baseline = 0;
6458
- this.hot = 0;
6459
- return;
6460
- }
6461
- if (!this.baseline) {
6462
- this.baseline = rms;
6463
- return;
6464
- }
6465
- this.baseline = this.baseline * 0.9 + rms * 0.1;
6466
- if (rms > Math.max(this.baseline * this.options.bargeRmsMult, this.options.bargeRmsFloor)) this.hot++;
6467
- else this.hot = 0;
6468
- if (this.hot >= 2 && !this.suspectUntil) {
6469
- this.suspectUntil = now() + 1300;
6470
- setTimeout(() => {
6471
- this.suspectUntil = 0;
6472
- }, 1350);
6473
- }
7104
+ onToolOutput(call, chunk, meta) {
7105
+ this.outputs.push({ call, chunk, meta });
7106
+ }
7107
+ onStop(finalText) {
7108
+ this.stops.push(finalText);
7109
+ }
7110
+ };
7111
+ var RecordingLifecycle = class {
7112
+ /** @param startContext injected at session start; @param rewrite maps a submitted prompt to a new one. */
7113
+ constructor(startContext, rewrite) {
7114
+ this.startContext = startContext;
7115
+ this.rewrite = rewrite;
7116
+ }
7117
+ startContext;
7118
+ rewrite;
7119
+ starts = 0;
7120
+ prompts = [];
7121
+ compactions = [];
7122
+ subagentStops = [];
7123
+ onSessionStart() {
7124
+ this.starts++;
7125
+ return this.startContext;
7126
+ }
7127
+ onUserPromptSubmit(text) {
7128
+ this.prompts.push(text);
7129
+ return this.rewrite?.(text);
7130
+ }
7131
+ onPreCompact(messages) {
7132
+ this.compactions.push(messages.length);
7133
+ }
7134
+ onSubagentStop(summary, info) {
7135
+ this.subagentStops.push({ summary, label: info?.label });
6474
7136
  }
6475
7137
  };
6476
7138
 
7139
+ // src/index.ts
7140
+ init_logging();
7141
+
6477
7142
  // src/voice/soniox.ts
6478
7143
  init_logging();
6479
7144
 
@@ -6486,7 +7151,7 @@ async function resolveAuth(auth) {
6486
7151
 
6487
7152
  // src/voice/soniox.ts
6488
7153
  var log12 = forComponent("SonioxSTT");
6489
- var now2 = () => performance.now();
7154
+ var now = () => performance.now();
6490
7155
  var SonioxSTTOptions = class {
6491
7156
  auth = "";
6492
7157
  source;
@@ -6515,6 +7180,20 @@ var SonioxSTT = class {
6515
7180
  * loop). The host tears voice down instead of spinning forever. */
6516
7181
  onFatal = () => {
6517
7182
  };
7183
+ /** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
7184
+ * handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
7185
+ onDiag = () => {
7186
+ };
7187
+ diagOn = true;
7188
+ diag(kind, fields) {
7189
+ if (!this.diagOn) return;
7190
+ try {
7191
+ this.onDiag({ t: now(), kind, ...fields });
7192
+ } catch (e) {
7193
+ this.diagOn = false;
7194
+ log12.debug(`onDiag threw \u2014 STT diagnostics disabled: ${e instanceof Error ? e.message : e}`);
7195
+ }
7196
+ }
6518
7197
  lastChunkAt = 0;
6519
7198
  // timestamp of the most recent mic chunk (0 = none yet)
6520
7199
  startedChunksAt = 0;
@@ -6555,8 +7234,12 @@ var SonioxSTT = class {
6555
7234
  this.ws.onclose = (ev) => {
6556
7235
  if (this.stopped) return;
6557
7236
  log12.warn(`soniox ws closed (${ev.code} ${ev.reason || ""}) \u2014 reconnecting`);
7237
+ this.diag("stt_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: true });
6558
7238
  this.reset();
6559
- this.connectWs().catch((e) => log12.error(`soniox reconnect failed: ${e.message}`));
7239
+ this.connectWs().catch((e) => {
7240
+ log12.error(`soniox reconnect failed: ${e.message}`);
7241
+ this.diag("stt_reconnect_failed", { message: e.message });
7242
+ });
6560
7243
  };
6561
7244
  }
6562
7245
  async start() {
@@ -6565,20 +7248,21 @@ var SonioxSTT = class {
6565
7248
  this.sourceStarted = true;
6566
7249
  this.endpointTimer = setInterval(() => {
6567
7250
  const combined = (this.finalText + this.partialText).trim();
6568
- if (!combined || now2() - this.lastChangeAt < this.options.silenceEndpointMs) return;
6569
- if (this.firstTokenAt) log12.debug(`stt: ${Math.round(now2() - this.firstTokenAt)}ms first-token\u2192silence-endpoint, "${combined.slice(0, 60)}"`);
7251
+ if (!combined || now() - this.lastChangeAt < this.options.silenceEndpointMs) return;
7252
+ if (this.firstTokenAt) log12.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192silence-endpoint, "${combined.slice(0, 60)}"`);
6570
7253
  this.reset();
6571
- this.onUtterance(combined, now2());
7254
+ this.onUtterance(combined, now());
6572
7255
  }, 120);
6573
7256
  this.endpointTimer.unref?.();
6574
- this.startedChunksAt = now2();
7257
+ this.startedChunksAt = now();
6575
7258
  const noAudioMs = this.options.noAudioTimeoutMs;
6576
7259
  if (noAudioMs > 0) {
6577
7260
  this.noAudioTimer = setInterval(() => {
6578
7261
  if (this.stopped) return;
6579
7262
  const ref = this.lastChunkAt || this.startedChunksAt;
6580
- if (now2() - ref > noAudioMs) {
7263
+ if (now() - ref > noAudioMs) {
6581
7264
  log12.error(`stt: no mic audio for >${Math.round(noAudioMs / 1e3)}s \u2014 capture device stopped delivering`);
7265
+ this.diag("stt_watchdog_fatal", { noAudioMs });
6582
7266
  this.onFatal("microphone stopped delivering audio (try a different input device, e.g. AirPods, or check System Settings \u2192 Sound \u2192 Input)");
6583
7267
  this.stop();
6584
7268
  }
@@ -6586,7 +7270,7 @@ var SonioxSTT = class {
6586
7270
  this.noAudioTimer.unref?.();
6587
7271
  }
6588
7272
  await this.options.source.start((chunk) => {
6589
- this.lastChunkAt = now2();
7273
+ this.lastChunkAt = now();
6590
7274
  let sum = 0;
6591
7275
  const view = new DataView(chunk.buffer, chunk.byteOffset, chunk.byteLength);
6592
7276
  for (let i = 0; i + 1 < chunk.byteLength; i += 2) {
@@ -6598,7 +7282,10 @@ var SonioxSTT = class {
6598
7282
  });
6599
7283
  }
6600
7284
  handle(m) {
6601
- if (m.error_message) return log12.error(`soniox: ${m.error_message}`);
7285
+ if (m.error_message) {
7286
+ this.diag("stt_error", { message: String(m.error_message), code: m.error_code });
7287
+ return log12.error(`soniox: ${m.error_message}`);
7288
+ }
6602
7289
  let endpoint = false;
6603
7290
  for (const t of m.tokens ?? []) {
6604
7291
  if (t.text === "<end>") endpoint = true;
@@ -6608,15 +7295,15 @@ var SonioxSTT = class {
6608
7295
  const combined = this.finalText + this.partialText;
6609
7296
  if (combined !== this.lastCombined) {
6610
7297
  this.lastCombined = combined;
6611
- this.lastChangeAt = now2();
6612
- if (!this.firstTokenAt && combined.trim()) this.firstTokenAt = now2();
7298
+ this.lastChangeAt = now();
7299
+ if (!this.firstTokenAt && combined.trim()) this.firstTokenAt = now();
6613
7300
  }
6614
7301
  this.onPartial(combined);
6615
7302
  if (endpoint && this.finalText.trim()) {
6616
7303
  const utterance = this.finalText.trim();
6617
- if (this.firstTokenAt) log12.debug(`stt: ${Math.round(now2() - this.firstTokenAt)}ms first-token\u2192endpoint, "${utterance.slice(0, 60)}"`);
7304
+ if (this.firstTokenAt) log12.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192endpoint, "${utterance.slice(0, 60)}"`);
6618
7305
  this.reset();
6619
- this.onUtterance(utterance, now2());
7306
+ this.onUtterance(utterance, now());
6620
7307
  }
6621
7308
  }
6622
7309
  reset() {
@@ -6638,7 +7325,7 @@ var SonioxSTT = class {
6638
7325
  // src/voice/cartesia.ts
6639
7326
  init_logging();
6640
7327
  var log13 = forComponent("CartesiaTTS");
6641
- var now3 = () => performance.now();
7328
+ var now2 = () => performance.now();
6642
7329
  var CartesiaTTSOptions = class {
6643
7330
  auth = "";
6644
7331
  voiceId = "";
@@ -6655,6 +7342,27 @@ var CartesiaTTS = class _CartesiaTTS {
6655
7342
  };
6656
7343
  onDone = () => {
6657
7344
  };
7345
+ /** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
7346
+ * across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
7347
+ onTimestamps = () => {
7348
+ };
7349
+ /** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
7350
+ * when revealMode==='word'. */
7351
+ wantTimestamps = false;
7352
+ /** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
7353
+ * a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
7354
+ onDiag = () => {
7355
+ };
7356
+ diagOn = true;
7357
+ diag(kind, fields) {
7358
+ if (!this.diagOn) return;
7359
+ try {
7360
+ this.onDiag({ t: now2(), kind, ...fields });
7361
+ } catch (e) {
7362
+ this.diagOn = false;
7363
+ log13.debug(`onDiag threw \u2014 TTS diagnostics disabled: ${e instanceof Error ? e.message : e}`);
7364
+ }
7365
+ }
6658
7366
  firstAudioAt = 0;
6659
7367
  /** Circuit breaker: consecutive error count + down flag. */
6660
7368
  consecutiveErrors = 0;
@@ -6688,8 +7396,12 @@ var CartesiaTTS = class _CartesiaTTS {
6688
7396
  });
6689
7397
  this.ws.onclose = (ev) => {
6690
7398
  log13.warn(`cartesia ws closed (${ev.code} ${ev.reason || ""})`);
7399
+ this.diag("tts_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: !this.closed });
6691
7400
  if (!this.closed) {
6692
- this.connecting = this.doConnect().catch((e) => log13.error(`cartesia reconnect failed: ${e.message}`));
7401
+ this.connecting = this.doConnect().catch((e) => {
7402
+ log13.error(`cartesia reconnect failed: ${e.message}`);
7403
+ this.diag("tts_reconnect_failed", { message: e.message });
7404
+ });
6693
7405
  }
6694
7406
  };
6695
7407
  this.ws.onmessage = (ev) => {
@@ -6698,24 +7410,29 @@ var CartesiaTTS = class _CartesiaTTS {
6698
7410
  if (m.type === "chunk" && m.data) {
6699
7411
  this.consecutiveErrors = 0;
6700
7412
  this.markRecovered();
6701
- if (!this.firstAudioAt) this.firstAudioAt = now3();
7413
+ if (!this.firstAudioAt) this.firstAudioAt = now2();
6702
7414
  this.onAudio(base64ToBytes(m.data));
6703
7415
  } else if (m.type === "done") {
6704
7416
  this.consecutiveErrors = 0;
6705
7417
  this.markRecovered();
6706
7418
  this.onDone();
7419
+ } else if (m.type === "timestamps" && m.word_timestamps) {
7420
+ const wt = m.word_timestamps;
7421
+ if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
6707
7422
  } else if (m.type === "error") {
6708
7423
  if (/already been cancelled|does not exist/.test(m.message || "")) return;
6709
7424
  this.consecutiveErrors++;
7425
+ this.diag("tts_error", { message: String(m.message || ""), code: m.status_code, contextId: m.context_id, consecutive: this.consecutiveErrors });
6710
7426
  if (!this.down && this.consecutiveErrors >= _CartesiaTTS.CB_THRESHOLD) {
6711
7427
  this.down = true;
6712
- this.downAt = now3();
7428
+ this.downAt = now2();
6713
7429
  this.consecutiveOk = 0;
6714
7430
  log13.warn(`TTS circuit breaker open \u2014 ${this.consecutiveErrors} consecutive errors, switching to text-only`);
7431
+ this.diag("tts_breaker_open", { errors: this.consecutiveErrors });
6715
7432
  this.onDone();
6716
7433
  this.startProbe();
6717
7434
  } else if (!this.down) {
6718
- log13.warn(`cartesia: ${JSON.stringify(m)}`);
7435
+ (/No valid transcripts/i.test(m.message || "") ? log13.debug : log13.warn)(`cartesia: ${JSON.stringify(m)}`);
6719
7436
  }
6720
7437
  }
6721
7438
  };
@@ -6728,7 +7445,8 @@ var CartesiaTTS = class _CartesiaTTS {
6728
7445
  this.down = false;
6729
7446
  this.consecutiveOk = 0;
6730
7447
  this.stopProbe();
6731
- const downMs = this.downAt ? now3() - this.downAt : 0;
7448
+ const downMs = this.downAt ? now2() - this.downAt : 0;
7449
+ this.diag("tts_breaker_close", { downMs: Math.round(downMs) });
6732
7450
  (downMs < 2e3 ? log13.debug : log13.info)(`TTS recovered${downMs ? ` (down ${downMs}ms)` : ""}`);
6733
7451
  }
6734
7452
  /** Ensure the WS is open before sending — reconnects if idle-closed. */
@@ -6736,6 +7454,15 @@ var CartesiaTTS = class _CartesiaTTS {
6736
7454
  if (this.connecting) await this.connecting;
6737
7455
  if (this.ws?.readyState !== WebSocket.OPEN) await this.connect();
6738
7456
  }
7457
+ /** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
7458
+ * a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
7459
+ * The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
7460
+ * newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
7461
+ warmup() {
7462
+ if (this.down) return;
7463
+ this.newContext();
7464
+ if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
7465
+ }
6739
7466
  newContext() {
6740
7467
  this.ctxId = `ctx-${++this.ctxSeq}`;
6741
7468
  this.firstAudioAt = 0;
@@ -6748,7 +7475,8 @@ var CartesiaTTS = class _CartesiaTTS {
6748
7475
  voice: { mode: "id", id: this.options.voiceId },
6749
7476
  output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
6750
7477
  context_id: this.ctxId,
6751
- continue: cont
7478
+ continue: cont,
7479
+ ...this.wantTimestamps ? { add_timestamps: true } : {}
6752
7480
  });
6753
7481
  }
6754
7482
  speak(text, cont) {
@@ -6776,7 +7504,7 @@ var CartesiaTTS = class _CartesiaTTS {
6776
7504
  }
6777
7505
  this.consecutiveErrors = 0;
6778
7506
  this.newContext();
6779
- if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame(".", false));
7507
+ if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
6780
7508
  }, _CartesiaTTS.CB_PROBE_MS);
6781
7509
  this.probeTimer.unref?.();
6782
7510
  }