@livx.cc/agentx 0.99.6 → 0.99.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +2324 -1555
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +226 -1
- package/dist/index.js +2160 -1513
- package/dist/index.js.map +1 -1
- package/package.json +3 -1
package/dist/cli.js
CHANGED
|
@@ -1316,11 +1316,11 @@ var init_NodeDiskFilesystem = __esm({
|
|
|
1316
1316
|
if (rel === "" || rel.startsWith("..")) return;
|
|
1317
1317
|
const parts = rel.split(np.sep);
|
|
1318
1318
|
let cur = this.baseDir;
|
|
1319
|
-
const
|
|
1319
|
+
const now4 = Date.now();
|
|
1320
1320
|
for (let i = 0; i < parts.length; i++) {
|
|
1321
1321
|
cur = np.join(cur, parts[i]);
|
|
1322
1322
|
const isLeaf = i === parts.length - 1;
|
|
1323
|
-
if (!isLeaf && (this.verified.get(cur) ?? 0) >
|
|
1323
|
+
if (!isLeaf && (this.verified.get(cur) ?? 0) > now4) continue;
|
|
1324
1324
|
let st;
|
|
1325
1325
|
try {
|
|
1326
1326
|
st = await fsp.lstat(cur);
|
|
@@ -1330,7 +1330,7 @@ var init_NodeDiskFilesystem = __esm({
|
|
|
1330
1330
|
if (st.isSymbolicLink()) throw new Error("File not found: symlink not permitted");
|
|
1331
1331
|
if (!isLeaf) {
|
|
1332
1332
|
if (this.verified.size > 1e4) this.verified.clear();
|
|
1333
|
-
this.verified.set(cur,
|
|
1333
|
+
this.verified.set(cur, now4 + _NodeDiskFilesystem.VERIFY_TTL_MS);
|
|
1334
1334
|
}
|
|
1335
1335
|
}
|
|
1336
1336
|
}
|
|
@@ -1958,7 +1958,7 @@ var init_tools_shell = __esm({
|
|
|
1958
1958
|
|
|
1959
1959
|
// cli/cli.ts
|
|
1960
1960
|
import { createInterface } from "readline/promises";
|
|
1961
|
-
import { existsSync as existsSync9, readFileSync as readFileSync8, appendFileSync, mkdirSync as
|
|
1961
|
+
import { existsSync as existsSync9, readFileSync as readFileSync8, appendFileSync, mkdirSync as mkdirSync12, writeFileSync as writeFileSync9, readdirSync as readdirSync4, statSync as statSync4, unlinkSync as unlinkSync5 } from "fs";
|
|
1962
1962
|
import { homedir as homedir10, tmpdir as tmpdir3 } from "os";
|
|
1963
1963
|
import { spawnSync as spawnSync6 } from "child_process";
|
|
1964
1964
|
|
|
@@ -2027,7 +2027,7 @@ function copyTextToClipboard(text, platform2 = process.platform) {
|
|
|
2027
2027
|
}
|
|
2028
2028
|
|
|
2029
2029
|
// cli/cli.ts
|
|
2030
|
-
import { join as join14, resolve as resolve3, basename as basename2, extname, dirname as
|
|
2030
|
+
import { join as join14, resolve as resolve3, basename as basename2, extname, dirname as dirname5 } from "path";
|
|
2031
2031
|
import { AIClient, listModels, listProviders, getProviderFromModel, getModelInfo, resolveModel, isModelSupported, disposeCursorSessions, disposeClaudeCodeSessions } from "ai.libx.js";
|
|
2032
2032
|
|
|
2033
2033
|
// src/llm.ts
|
|
@@ -4292,12 +4292,12 @@ var Scheduler = class {
|
|
|
4292
4292
|
if (this.firing) return;
|
|
4293
4293
|
this.firing = true;
|
|
4294
4294
|
try {
|
|
4295
|
-
const
|
|
4295
|
+
const now4 = this.now();
|
|
4296
4296
|
for (const job of this.jobs.values()) {
|
|
4297
4297
|
if (job.status !== "active") continue;
|
|
4298
4298
|
const due = this.nextFire(job);
|
|
4299
|
-
if (due == null || due >
|
|
4300
|
-
job.lastRun =
|
|
4299
|
+
if (due == null || due > now4) continue;
|
|
4300
|
+
job.lastRun = now4;
|
|
4301
4301
|
job.runs++;
|
|
4302
4302
|
if ("at" in job.trigger) job.status = "done";
|
|
4303
4303
|
try {
|
|
@@ -4837,1538 +4837,2133 @@ var EmotionStream = class {
|
|
|
4837
4837
|
}
|
|
4838
4838
|
};
|
|
4839
4839
|
|
|
4840
|
-
// src/voice/
|
|
4841
|
-
|
|
4842
|
-
var
|
|
4843
|
-
var
|
|
4844
|
-
|
|
4845
|
-
|
|
4846
|
-
|
|
4847
|
-
|
|
4848
|
-
|
|
4840
|
+
// src/voice/engine.ts
|
|
4841
|
+
init_logging();
|
|
4842
|
+
var log9 = forComponent("VoiceEngine");
|
|
4843
|
+
var realClock = {
|
|
4844
|
+
now: () => performance.now(),
|
|
4845
|
+
setTimeout: (fn, ms) => setTimeout(fn, ms),
|
|
4846
|
+
clearTimeout: (h) => clearTimeout(h)
|
|
4847
|
+
};
|
|
4848
|
+
var forSpeech = (t) => t.replace(/[*`#]+/g, "").replace(/(?<![\p{L}\p{N}])_([^_\n]+)_(?![\p{L}\p{N}])/gu, "$1").replace(/^[ \t]*[-•]\s+/gm, "").replace(/\s*[\u2013\u2014]\s*/g, ", ").replace(/[\u2010\u2011]/g, "-").replace(/\s*\|\s*/g, ", ").replace(/(\d)\s+%/g, "$1%").replace(/\.{3,}/g, ".");
|
|
4849
|
+
var normWord = (w) => w.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
4850
|
+
var SPEC_TRAILING_OK = /* @__PURE__ */ new Set(["please", "thanks", "thank", "you", "now", "okay", "ok", "kindly", "alright", "then", "though", "right", "yeah"]);
|
|
4851
|
+
function speculationConfirms(spec, final) {
|
|
4852
|
+
const sw = spec.trim().split(/\s+/).map(normWord).filter(Boolean);
|
|
4853
|
+
const fw = final.trim().split(/\s+/).map(normWord).filter(Boolean);
|
|
4854
|
+
if (!sw.length || fw.length < sw.length || fw.length > sw.length + 2) return false;
|
|
4855
|
+
for (let i = 0; i < sw.length - 1; i++) if (sw[i] !== fw[i]) return false;
|
|
4856
|
+
if (!fw[sw.length - 1].startsWith(sw[sw.length - 1])) return false;
|
|
4857
|
+
for (let i = sw.length; i < fw.length; i++) if (!SPEC_TRAILING_OK.has(fw[i])) return false;
|
|
4858
|
+
return true;
|
|
4859
|
+
}
|
|
4860
|
+
var VoiceEngineOptions = class {
|
|
4861
|
+
stt;
|
|
4862
|
+
tts;
|
|
4863
|
+
player;
|
|
4864
|
+
/** a final utterance arrived (endpoint) — host dispatches it as a turn */
|
|
4865
|
+
onUtterance = () => {
|
|
4866
|
+
};
|
|
4867
|
+
/** live partial transcript while listening (host renders the 🎤 line) */
|
|
4868
|
+
onPartial = () => {
|
|
4869
|
+
};
|
|
4870
|
+
onState = () => {
|
|
4871
|
+
};
|
|
4872
|
+
/** user spoke/acted over playback — host aborts the in-flight turn (called AFTER audio is killed).
|
|
4873
|
+
* phase: 'speaking' = cut mid-speech (real interruption); 'drain' = in the final audio tail
|
|
4874
|
+
* (normal turn-taking — hosts shouldn't alarm). */
|
|
4875
|
+
onBargeIn = () => {
|
|
4876
|
+
};
|
|
4877
|
+
/** spoken micro-ack on utterance endpoint (masks LLM TTFT); '' disables */
|
|
4878
|
+
ackPhrase = "";
|
|
4879
|
+
/** ADAPTIVE micro-ack: on an utterance-dispatched turn, speak a short varied ack ONLY if no reflex
|
|
4880
|
+
* delta has arrived after this many ms (masks a slow TTFT without acking every turn — a fixed
|
|
4881
|
+
* per-turn ack was rejected as robotic). First delta / interrupt / hold cancels it. 0 = off. */
|
|
4882
|
+
adaptiveAckMs = 0;
|
|
4883
|
+
/** the adaptive ack actually fired (host can mark the turn as spoken — e.g. suppress dead-air repair) */
|
|
4884
|
+
onAdaptiveAck = () => {
|
|
4885
|
+
};
|
|
4886
|
+
/** Endpoint merge window (ms): hold an endpointed utterance briefly — if speech resumes (spelled
|
|
4887
|
+
* letters, mid-thought pauses), the next utterance MERGES instead of dispatching a truncated one
|
|
4888
|
+
* ("E-L-Y." / "A."). Costs this much latency per turn; 0 disables. */
|
|
4889
|
+
utteranceMergeMs = 350;
|
|
4890
|
+
/** Extended merge window (ms) for utterances that look incomplete (trailing conjunction/filler).
|
|
4891
|
+
* Gives the user time to finish their thought without triggering a model call. */
|
|
4892
|
+
incompleteMergeMs = 1500;
|
|
4893
|
+
/** Grace window (ms) after an utterance dispatches, during which the user's own trailing audio cannot
|
|
4894
|
+
* barge the reply it requested. Soniox keeps finalizing partials past <end>; without this they read
|
|
4895
|
+
* as a barge and abort the fresh turn (live: mid-sentence self-interruption + steps=1→steps=0 double
|
|
4896
|
+
* abort). Short enough that a genuine immediate barge ("no wait—") still lands right after. */
|
|
4897
|
+
bargeGraceMs = 600;
|
|
4898
|
+
/** Barge-in (talk over the assistant to interrupt). true = full-duplex (needs echo cancellation, or
|
|
4899
|
+
* the assistant's own TTS bleeds back and self-interrupts). false = HALF-DUPLEX: the engine is deaf
|
|
4900
|
+
* while audible (speaking + drain tail), so echo can never become a phantom turn — the right mode
|
|
4901
|
+
* when there's no AEC (e.g. the non-VPIO mic fallback) and no headphones. Cost: can't interrupt. */
|
|
4902
|
+
bargeIn = true;
|
|
4903
|
+
/** Filler phrase spoken when holding for an incomplete utterance ('' disables). */
|
|
4904
|
+
holdFiller = "";
|
|
4905
|
+
/** Called when the engine holds an incomplete utterance (host can render a visual cue). */
|
|
4906
|
+
onHold = () => {
|
|
4907
|
+
};
|
|
4908
|
+
/** heuristic (non-AEC) energy barge-in tuning */
|
|
4909
|
+
bargeRmsMult = 2;
|
|
4910
|
+
bargeRmsFloor = 500;
|
|
4911
|
+
/** Overlap turn-taking (AEC tier, needs player.pause/resume) — human phone-call model, driven by
|
|
4912
|
+
* the STT ITSELF (a trained speech classifier) instead of energy thresholds (energy could not
|
|
4913
|
+
* separate residue bursts from speech in every room — hiccup whack-a-mole): a GENUINE partial
|
|
4914
|
+
* (novel words dominate — echo of our own reply is inert) while speaking → PAUSE (exact-sample
|
|
4915
|
+
* hold); partial grows into dominant-novel ≥2 words → cede (interrupt; the LLM re-enters); partial
|
|
4916
|
+
* stalls/endpoints without ceding (backchannel by DURATION, not vocabulary) → resume + drop. false disables. */
|
|
4917
|
+
overlapPause = true;
|
|
4918
|
+
/** no new partial activity for this long while paused → resume, drop the interjection */
|
|
4919
|
+
overlapResumeMs = 700;
|
|
4920
|
+
/** A genuine barge over a LONG reply is defeated by the dominant-novel gate: Meet echoes our own
|
|
4921
|
+
* speech back, so the partial is mostly our words + a few of hers → never "dominant novel" → it
|
|
4922
|
+
* resumes (replaying old audio — the audible "completes the buffer" blip) instead of ceding.
|
|
4923
|
+
* Mechanism-based discriminator: a re-PAUSE this soon after a resume = a persistent human, not an
|
|
4924
|
+
* echo blip (which pauses once and stalls). Cede on the re-pause regardless of the novel gate. */
|
|
4925
|
+
overlapRepauseCedeMs = 1500;
|
|
4926
|
+
/** Speculative ENERGY pre-pause while speaking (AEC tier): two residue gate-passes within 350ms →
|
|
4927
|
+
* pause ~300ms before the STT tokens land. But energy CANNOT separate residue bursts from speech
|
|
4928
|
+
* (the documented whack-a-mole) — so a residue spike during loud playback false-pauses with NO user
|
|
4929
|
+
* speech at all, an audible hiccup. Default OFF: the genuine-gated STT partial is the
|
|
4930
|
+
* mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
|
|
4931
|
+
overlapEnergyHold = false;
|
|
4932
|
+
/** SPECULATIVE REFLEX START (the root TTFT fix): a partial transcript that has stopped changing for
|
|
4933
|
+
* this many ms AND carries ≥ speculativeMinWords is "stable" — `onSpeculate` fires so the host can
|
|
4934
|
+
* start the reflex EARLY, ~endpoint+merge (500-850ms) before the final would dispatch. The
|
|
4935
|
+
* speculative call's output is HELD by the host (nothing reaches TTS) until the endpointed final
|
|
4936
|
+
* confirms it (see speculationConfirms). At most one speculation per turn-in-progress. 0 = off. */
|
|
4937
|
+
speculativeMs = 0;
|
|
4938
|
+
/** Minimum word count for a partial to qualify as a speculation trigger. */
|
|
4939
|
+
speculativeMinWords = 4;
|
|
4940
|
+
/** A stable partial (speculativeMs) — the host starts a HELD speculative reflex call. */
|
|
4941
|
+
onSpeculate = () => {
|
|
4942
|
+
};
|
|
4943
|
+
/** AGENT-SIDE BACKCHANNELING (rule-based v1): while LISTENING to a long multi-clause user turn, a
|
|
4944
|
+
* partial that reaches a clause boundary (trailing [,.;!?] or conjunction/filler) and then stays
|
|
4945
|
+
* UNCHANGED for this many ms (a micro-pause — before the silence endpoint fires) triggers a short
|
|
4946
|
+
* quiet TTS blip ("Mm-hm.") on a throwaway context. ZERO floor-claim: no state change, no timers
|
|
4947
|
+
* touched, no turn context — audio passes a narrow gate bypass and a real turn supersedes it via
|
|
4948
|
+
* context rotation. Latin-predominant partials only (Hebrew/mixed text never misfires — the
|
|
4949
|
+
* boundary/conjunction heuristics are English-tuned, so non-Latin turns simply get no blips).
|
|
4950
|
+
* 0 = off (default). ~200-300 recommended: live, Soniox's SEMANTIC endpoint (<end>) lands within
|
|
4951
|
+
* ~300-400ms of a clause pause — a longer stability window loses the race and never fires. */
|
|
4952
|
+
backchannelMs = 0;
|
|
4953
|
+
/** Min gap between blips (rate limit); additionally max 2 blips per user turn-in-progress. */
|
|
4954
|
+
backchannelMinGapMs = 8e3;
|
|
4955
|
+
/** Only multi-clause turns: the partial must carry at least this many words before a blip. */
|
|
4956
|
+
backchannelMinWords = 8;
|
|
4957
|
+
/** A backchannel blip was spoken (host renders a timeline event; the blip is NOT a reply). */
|
|
4958
|
+
onBackchannel = () => {
|
|
4959
|
+
};
|
|
4960
|
+
/** The partial outgrew the speculated text (user kept talking) — the host aborts the speculation
|
|
4961
|
+
* quietly (the endpointed final will also refuse to confirm; this just stops the billing earlier). */
|
|
4962
|
+
onSpeculateAbort = () => {
|
|
4963
|
+
};
|
|
4964
|
+
/** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
|
|
4965
|
+
* tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
|
|
4966
|
+
emotions = true;
|
|
4967
|
+
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
4968
|
+
showEmotions = false;
|
|
4969
|
+
/** Time source for every engine timer/timestamp. Default: real time (identical behavior). */
|
|
4970
|
+
clock = realClock;
|
|
4971
|
+
/** Structured diagnostic event at every decision point (state transitions, dispatch/drop paths,
|
|
4972
|
+
* barge-in, overlap pause/resume, acks, speculation, backchannels, echo swallows). Fire-and-forget:
|
|
4973
|
+
* a throwing handler is caught once and diagnostics disable — engine behavior is never affected. */
|
|
4974
|
+
onDiag = () => {
|
|
4975
|
+
};
|
|
4976
|
+
};
|
|
4977
|
+
var VoiceEngine = class _VoiceEngine {
|
|
4978
|
+
options;
|
|
4979
|
+
state = "idle";
|
|
4980
|
+
stt;
|
|
4981
|
+
tts;
|
|
4982
|
+
player;
|
|
4983
|
+
speaking = false;
|
|
4984
|
+
// audible (deltas flowing OR audio draining)
|
|
4985
|
+
ctxOpen = false;
|
|
4986
|
+
// the current TTS context still accepts deltas (false once end-frame sent)
|
|
4987
|
+
interrupted = false;
|
|
4988
|
+
// barge-in latch: drop in-flight deltas until the next legitimate turn
|
|
4989
|
+
spokeDeltas = false;
|
|
4990
|
+
// a TTS context is open for the current spoken turn
|
|
4991
|
+
clock;
|
|
4992
|
+
drainTimer = null;
|
|
4993
|
+
// heuristic tier state (inert under AEC) — frozen as validated in the experiment
|
|
4994
|
+
echoWords = /* @__PURE__ */ new Set();
|
|
4995
|
+
prevReply = "";
|
|
4996
|
+
reply = "";
|
|
4997
|
+
echoUntil = 0;
|
|
4998
|
+
baseline = 0;
|
|
4999
|
+
hot = 0;
|
|
5000
|
+
suspectUntil = 0;
|
|
5001
|
+
ackAt = 0;
|
|
5002
|
+
// when the micro-ack was spoken — its echo can leak before the AEC filter converges
|
|
5003
|
+
lastAck = "";
|
|
5004
|
+
// the exact ack text last spoken (fixed OR adaptive) — the echo-leak guard matches it
|
|
5005
|
+
ackTimer = null;
|
|
5006
|
+
// one-shot adaptive-ack timer (armed at dispatch, cancelled on first delta)
|
|
5007
|
+
bargeGraceUntil = 0;
|
|
5008
|
+
// no barge-in until this time — the user's OWN trailing audio (after the
|
|
5009
|
+
// utterance that JUST dispatched this turn) must not immediately re-interrupt the reply it requested.
|
|
5010
|
+
pendingUtt = "";
|
|
5011
|
+
// endpointed text held for the merge window
|
|
5012
|
+
mergePath = "direct";
|
|
5013
|
+
// how the pending utterance was assembled (diag)
|
|
5014
|
+
lastGraceDiag = 0;
|
|
5015
|
+
// grace_suppress emitted once per grace window
|
|
5016
|
+
pendingTimer = null;
|
|
5017
|
+
// Duplicate-final guard: STT sometimes re-finalizes the SAME audio a beat later (past the merge
|
|
5018
|
+
// window) — the identical utterance dispatched twice with no reply in between (live: "When you
|
|
5019
|
+
// start" twice). Conservative: only an IDENTICAL (normalized) text, within dupFinalMs, with zero
|
|
5020
|
+
// reply deltas since the first dispatch, is dropped. A user genuinely repeating themselves after
|
|
5021
|
+
// the agent replied (or after 3s) still dispatches.
|
|
5022
|
+
lastDispatchFlat = "";
|
|
5023
|
+
lastDispatchWords = [];
|
|
5024
|
+
lastDispatchAt = 0;
|
|
5025
|
+
repliedSinceDispatch = false;
|
|
5026
|
+
static DUP_FINAL_MS = 3e3;
|
|
5027
|
+
lastInterrupted = null;
|
|
5028
|
+
// overlap (pause) tier state — AEC + pause-capable sinks only
|
|
5029
|
+
pausedAt = 0;
|
|
5030
|
+
lastResumeAt = 0;
|
|
5031
|
+
// when the overlap last resumed from a false alarm — a quick re-pause cedes
|
|
5032
|
+
lastOverlapPartial = "";
|
|
5033
|
+
// change-detection: only NEW partial text counts as activity
|
|
5034
|
+
resumeTimer = null;
|
|
5035
|
+
turnStartAt = 0;
|
|
5036
|
+
// timestamp when the current turn began (for TTFT logging)
|
|
5037
|
+
// speculative reflex trigger state (options.speculativeMs) — see trackSpeculation
|
|
5038
|
+
specPartial = "";
|
|
5039
|
+
// last partial observed (stability = unchanged for speculativeMs)
|
|
5040
|
+
specTimer = null;
|
|
5041
|
+
specText = "";
|
|
5042
|
+
// text handed to onSpeculate ('' = no speculation in flight)
|
|
5043
|
+
specSpent = false;
|
|
5044
|
+
// at most one speculation per turn-in-progress
|
|
5045
|
+
// backchannel blip state (options.backchannelMs) — see trackBackchannel
|
|
5046
|
+
bcPartial = "";
|
|
5047
|
+
// last partial observed (stability = unchanged for backchannelMs)
|
|
5048
|
+
bcTimer = null;
|
|
5049
|
+
bcCount = 0;
|
|
5050
|
+
// blips this user turn-in-progress (max 2; reset at dispatch)
|
|
5051
|
+
bcActive = false;
|
|
5052
|
+
// NARROW audio-gate bypass: blip audio may reach the sink while listening
|
|
5053
|
+
bcActiveTimer = null;
|
|
5054
|
+
// safety: clear the bypass even if the blip's 'done' is lost
|
|
5055
|
+
lastBcAt = 0;
|
|
5056
|
+
lastBcPhrase = "";
|
|
5057
|
+
// exact blip text last spoken — flushUtterance strips its mic echo
|
|
5058
|
+
recentBc = [];
|
|
5059
|
+
// Central speech queue (above the TTS context): complete worker utterances serialize into ONE
|
|
5060
|
+
// playback stream, one-at-a-time, never splicing into the live reflex's open utterance.
|
|
5061
|
+
uttQueue = [];
|
|
5062
|
+
// Per-turn emotion-tag parser (reset on beginSpeech) — converts `[emotion]` → Cartesia inline tags
|
|
5063
|
+
// for TTS, tracks tag-free prose for echo discrimination, and surfaces display text for the screen.
|
|
5064
|
+
emo = null;
|
|
5065
|
+
constructor(options) {
|
|
5066
|
+
this.options = { ...new VoiceEngineOptions(), ...options };
|
|
5067
|
+
const o = this.options;
|
|
5068
|
+
if (!o.stt || !o.tts || !o.player) throw new Error("VoiceEngine needs stt, tts and player (see cli/voice.ts VoiceIO for platform defaults)");
|
|
5069
|
+
this.stt = o.stt;
|
|
5070
|
+
this.tts = o.tts;
|
|
5071
|
+
this.player = o.player;
|
|
5072
|
+
this.clock = o.clock;
|
|
4849
5073
|
}
|
|
4850
|
-
|
|
4851
|
-
|
|
4852
|
-
|
|
4853
|
-
|
|
4854
|
-
|
|
4855
|
-
|
|
4856
|
-
|
|
4857
|
-
this.
|
|
4858
|
-
|
|
5074
|
+
async start() {
|
|
5075
|
+
this.tts.onAudio = (c) => {
|
|
5076
|
+
if (this.speaking || this.bcActive) this.player.write(c);
|
|
5077
|
+
};
|
|
5078
|
+
this.stt.onPartial = (text) => this.handlePartial(text);
|
|
5079
|
+
this.stt.onUtterance = (text) => this.handleUtterance(text);
|
|
5080
|
+
this.stt.onLevel = (rms) => this.handleLevel(rms);
|
|
5081
|
+
await Promise.all([this.tts.connect(), this.stt.start()]);
|
|
5082
|
+
this.tts.warmup?.();
|
|
5083
|
+
this.setState("listening");
|
|
5084
|
+
log9.debug(`voice I/O up (${this.stt.usingAec ? "AEC" : "heuristic echo"} capture)`);
|
|
4859
5085
|
}
|
|
4860
|
-
|
|
4861
|
-
|
|
4862
|
-
this.buf = "";
|
|
4863
|
-
return hasSpeech(s) ? s : "";
|
|
5086
|
+
get usingAec() {
|
|
5087
|
+
return this.stt.usingAec;
|
|
4864
5088
|
}
|
|
4865
|
-
|
|
4866
|
-
|
|
4867
|
-
|
|
4868
|
-
inSpoken = false;
|
|
4869
|
-
/** True once any spoken char has ever been emitted (drives the no-spoken fallback). */
|
|
4870
|
-
spokeAny = false;
|
|
4871
|
-
/** Feed a delta; returns the spoken/detail spans completed by this chunk (either may be ''). */
|
|
4872
|
-
feed(delta) {
|
|
4873
|
-
this.buf += delta;
|
|
4874
|
-
return this.drain(false);
|
|
5089
|
+
/** Flip barge-in at runtime (e.g. the mic fell back to non-VPIO → go half-duplex so echo can't leak). */
|
|
5090
|
+
setBargeIn(on) {
|
|
5091
|
+
this.options.bargeIn = on;
|
|
4875
5092
|
}
|
|
4876
|
-
/**
|
|
4877
|
-
|
|
4878
|
-
|
|
5093
|
+
/** Show/hide the `[emotion]` debug tags in the echo (next turn's stream picks it up). */
|
|
5094
|
+
setShowEmotions(on) {
|
|
5095
|
+
this.options.showEmotions = on;
|
|
4879
5096
|
}
|
|
4880
|
-
|
|
4881
|
-
|
|
4882
|
-
|
|
4883
|
-
|
|
4884
|
-
|
|
4885
|
-
|
|
4886
|
-
|
|
4887
|
-
|
|
4888
|
-
|
|
4889
|
-
|
|
4890
|
-
this.buf = this.buf.slice(idx + tag.length);
|
|
4891
|
-
this.inSpoken = !this.inSpoken;
|
|
4892
|
-
continue;
|
|
4893
|
-
}
|
|
4894
|
-
const lt = this.buf.lastIndexOf("<");
|
|
4895
|
-
const holdStart = lt >= 0 && tag.startsWith(this.buf.slice(lt)) ? lt : this.buf.length;
|
|
4896
|
-
const text = this.buf.slice(0, holdStart);
|
|
4897
|
-
if (this.inSpoken) spoken += text;
|
|
4898
|
-
else detail += text;
|
|
4899
|
-
this.buf = this.buf.slice(holdStart);
|
|
4900
|
-
break;
|
|
5097
|
+
/** Diagnostics tap (options.onDiag). Fire-and-forget: a throwing handler disables the tap once —
|
|
5098
|
+
* it can NEVER perturb engine behavior. Protected so VoiceIO can route provider events through it. */
|
|
5099
|
+
diagOn = true;
|
|
5100
|
+
diag(kind, fields) {
|
|
5101
|
+
if (!this.diagOn) return;
|
|
5102
|
+
try {
|
|
5103
|
+
this.options.onDiag({ t: this.clock.now(), kind, ...fields });
|
|
5104
|
+
} catch (e) {
|
|
5105
|
+
this.diagOn = false;
|
|
5106
|
+
log9.debug(`onDiag threw \u2014 diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
4901
5107
|
}
|
|
4902
|
-
|
|
4903
|
-
|
|
4904
|
-
|
|
4905
|
-
|
|
5108
|
+
}
|
|
5109
|
+
idleWaiters = [];
|
|
5110
|
+
setState(s) {
|
|
5111
|
+
if (this.state === s) return;
|
|
5112
|
+
this.diag("state", { from: this.state, to: s });
|
|
5113
|
+
this.state = s;
|
|
5114
|
+
this.options.onState(s);
|
|
5115
|
+
if (s !== "speaking" && s !== "thinking") {
|
|
5116
|
+
for (const r of this.idleWaiters.splice(0)) r();
|
|
4906
5117
|
}
|
|
4907
|
-
if (spoken.trim()) this.spokeAny = true;
|
|
4908
|
-
return { spoken, detail };
|
|
4909
5118
|
}
|
|
4910
|
-
|
|
4911
|
-
|
|
4912
|
-
|
|
4913
|
-
|
|
4914
|
-
function describeCall(call) {
|
|
4915
|
-
const v = call.args && Object.values(call.args).find((x) => typeof x === "string" && x.trim());
|
|
4916
|
-
const hint = v ? ` (${String(v).replace(/\s+/g, " ").trim().slice(0, 48)})` : "";
|
|
4917
|
-
return `${call.name}${hint}`;
|
|
4918
|
-
}
|
|
4919
|
-
var DuplexAgentOptions = class {
|
|
4920
|
-
/** Any ai.libx.js AIClient — shared by all tiers (routed by model). */
|
|
4921
|
-
ai;
|
|
4922
|
-
/** The WORKER's filesystem (act + think). If omitted the worker keeps Agent's jailed-disk-at-cwd default. */
|
|
4923
|
-
fs;
|
|
4924
|
-
// The reflex IS the voice. 120b (not 20b) for channel discipline + instruction-following: the 20b
|
|
4925
|
-
// mislabels gpt-oss harmony channels under load, leaking raw analysis into the spoken `final` channel
|
|
4926
|
-
// (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
|
|
4927
|
-
reflexModel = "groq/openai/gpt-oss-120b";
|
|
4928
|
-
actModel = "anthropic/claude-sonnet-4-6";
|
|
4929
|
-
/** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
|
|
4930
|
-
thinkModel = "anthropic/claude-opus-4-8";
|
|
4931
|
-
/** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
|
|
4932
|
-
* provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
|
|
4933
|
-
* cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
|
|
4934
|
-
* leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
|
|
4935
|
-
providerOptionsFor;
|
|
4936
|
-
/** Escape hatches merged over the derived per-agent options. */
|
|
4937
|
-
reflexOptions;
|
|
4938
|
-
actOptions;
|
|
4939
|
-
thinkOptions;
|
|
4940
|
-
/** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
|
|
4941
|
-
* the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
|
|
4942
|
-
* pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
|
|
4943
|
-
* this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
|
|
4944
|
-
verifyActTasks = false;
|
|
4945
|
-
/** Receives the voice text_delta stream + task lifecycle events. */
|
|
4946
|
-
host;
|
|
4947
|
-
/** How many recent transcript messages are rendered into a worker's brief. */
|
|
4948
|
-
excerptTurns = 6;
|
|
4949
|
-
/** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
|
|
4950
|
-
* backchannels, impulsive first reactions before content (mimics real duplex conversation). */
|
|
4951
|
-
voiceStyle = "neutral";
|
|
4952
|
-
/** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
|
|
4953
|
-
* TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
|
|
4954
|
-
emotionTags = false;
|
|
4955
|
-
/** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
|
|
4956
|
-
* (post-spawn would race the worker's first edits). */
|
|
4957
|
-
onTaskStart;
|
|
4958
|
-
/** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
|
|
4959
|
-
* air. Off by default — each update costs a voice turn (LLM call + speech). */
|
|
4960
|
-
progressUpdates = false;
|
|
4961
|
-
/** Min ms between progress re-voices per task. */
|
|
4962
|
-
progressIntervalMs = 25e3;
|
|
4963
|
-
/** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
|
|
4964
|
-
* the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
|
|
4965
|
-
* voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
|
|
4966
|
-
askRelay = false;
|
|
4967
|
-
/** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
|
|
4968
|
-
askTimeoutMs = 12e4;
|
|
4969
|
-
/** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
|
|
4970
|
-
* bounding memory over a long-lived session. Running tasks are never evicted. */
|
|
4971
|
-
maxTaskRecords = 50;
|
|
4972
|
-
/** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
|
|
4973
|
-
* (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
|
|
4974
|
-
quickLook;
|
|
4975
|
-
/** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
|
|
4976
|
-
* tools directly (no delegation needed) and implicit capture guidance. */
|
|
4977
|
-
memoryDir;
|
|
4978
|
-
/** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
|
|
4979
|
-
memoryUserDir;
|
|
4980
|
-
};
|
|
4981
|
-
var RESERVED_EVENT_MARKER = /\[task\b[^\]\n]*\b(?:completed|failed|progress|asks)\b/i;
|
|
4982
|
-
var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
|
|
4983
|
-
var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 give at most a half-sentence aside ("still on it \u2014 running tests now") and end your turn. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, just say "still on it" and end the turn \u2014 never poll it again and again in a loop. Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
|
|
4984
|
-
var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
|
|
4985
|
-
var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
|
|
4986
|
-
var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
|
|
4987
|
-
var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
|
|
4988
|
-
var DuplexAgent = class _DuplexAgent {
|
|
4989
|
-
options;
|
|
4990
|
-
voice;
|
|
4991
|
-
tasks = /* @__PURE__ */ new Map();
|
|
4992
|
-
queue = Promise.resolve();
|
|
4993
|
-
seq = 0;
|
|
4994
|
-
pendingEvents = [];
|
|
4995
|
-
/** Out-of-band follow-up attribution for the events coalescing into the next flush turn: TRUE iff ≥1 of
|
|
4996
|
-
* the tasks being integrated was NON-CLEAN (early-stop/failure). Carried out-of-band on the enqueue call
|
|
4997
|
-
* by the caller that KNOWS the outcome — a plain boolean the MODEL CANNOT PERTURB. It is NOT scanned from
|
|
4998
|
-
* worker-authored event text (v1: an "Outcome:" substring over-stamped siblings) and NOT keyed on a brief
|
|
4999
|
-
* string the reflex re-authors (v2: a paraphrased escalation brief missed the Set → followUp:false →
|
|
5000
|
-
* RE-ENABLED unbounded auto-escalation, the dangerous runaway direction). See [[wrong-discriminator]] /
|
|
5001
|
-
* [[drive-real-reflex]] / [[fakeaiclient-blind-to-wire-format]]. */
|
|
5002
|
-
pendingNonClean = false;
|
|
5003
|
-
flushQueued = false;
|
|
5004
|
-
/** Per-voice-turn guards (reset by resetTurn at each turn's start). The reflex is a weak model:
|
|
5005
|
-
* left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
|
|
5006
|
-
* Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
|
|
5007
|
-
turnDispatched = false;
|
|
5008
|
-
// an Act/Think fired this turn
|
|
5009
|
-
turnBriefs = /* @__PURE__ */ new Set();
|
|
5010
|
-
// briefs dispatched this turn (detect identical re-dispatch)
|
|
5011
|
-
spokeThisTurn = false;
|
|
5012
|
-
// any non-empty text_delta streamed this turn
|
|
5013
|
-
heldThisTurn = false;
|
|
5014
|
-
// Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
|
|
5015
|
-
nudging = false;
|
|
5016
|
-
// re-ack pass in flight: block ALL tools, prevent recursion
|
|
5017
|
-
reflexBuf = "";
|
|
5018
|
-
// accumulated reflex text this turn (fabricated-event detection)
|
|
5019
|
-
reflexForwarded = 0;
|
|
5020
|
-
// chars of reflexBuf already forwarded to the host/TTS
|
|
5021
|
-
fabricationCut = false;
|
|
5022
|
-
// reflex emitted a reserved [task …] marker → suppress its tail
|
|
5023
|
-
/** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
|
|
5024
|
-
* carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
|
|
5025
|
-
* turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
|
|
5026
|
-
* escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
|
|
5027
|
-
* ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
|
|
5028
|
-
* means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
|
|
5029
|
-
* ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
|
|
5030
|
-
* flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
|
|
5031
|
-
* fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
|
|
5032
|
-
* ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
|
|
5033
|
-
turnFollowUp = false;
|
|
5034
|
-
/** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
|
|
5035
|
-
* the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
|
|
5036
|
-
autoEscalations = 0;
|
|
5037
|
-
static MAX_AUTO_ESCALATIONS = 8;
|
|
5038
|
-
/** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
|
|
5039
|
-
pendingAsks = /* @__PURE__ */ new Map();
|
|
5040
|
-
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
5041
|
-
memoryReady;
|
|
5042
|
-
constructor(options) {
|
|
5043
|
-
this.options = { ...new DuplexAgentOptions(), ...options };
|
|
5044
|
-
const o = this.options;
|
|
5045
|
-
if (o.memoryDir && o.fs) {
|
|
5046
|
-
this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
|
|
5047
|
-
}
|
|
5048
|
-
const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
|
|
5049
|
-
const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
|
|
5050
|
-
const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
|
|
5051
|
-
const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
|
|
5052
|
-
const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
|
|
5053
|
-
const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
|
|
5054
|
-
const mcpNames = [
|
|
5055
|
-
...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
|
|
5056
|
-
...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
|
|
5057
|
-
];
|
|
5058
|
-
const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
|
|
5059
|
-
const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
|
|
5060
|
-
Today's date: ${(/* @__PURE__ */ new Date()).toDateString()}.`;
|
|
5061
|
-
const tools = [
|
|
5062
|
-
...o.reflexOptions?.tools ?? [],
|
|
5063
|
-
this.actTool(),
|
|
5064
|
-
...o.thinkModel !== false ? [this.thinkTool()] : [],
|
|
5065
|
-
this.taskStatusTool(),
|
|
5066
|
-
this.cancelTaskTool(),
|
|
5067
|
-
this.quickLookTool(),
|
|
5068
|
-
this.answerTaskTool(),
|
|
5069
|
-
this.holdTool()
|
|
5070
|
-
];
|
|
5071
|
-
const host = o.host;
|
|
5072
|
-
const voiceHost = host && {
|
|
5073
|
-
ask: host.ask ? (q2) => host.ask(q2) : void 0,
|
|
5074
|
-
confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
|
|
5075
|
-
notify: (ev) => {
|
|
5076
|
-
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
5077
|
-
if (this.heldThisTurn) return;
|
|
5078
|
-
if (this.fabricationCut) return;
|
|
5079
|
-
const msg = ev.message;
|
|
5080
|
-
this.reflexBuf += msg;
|
|
5081
|
-
const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
|
|
5082
|
-
if (m) {
|
|
5083
|
-
this.fabricationCut = true;
|
|
5084
|
-
log9.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
|
|
5085
|
-
const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
|
|
5086
|
-
if (!safe) return;
|
|
5087
|
-
if (safe.trim()) this.spokeThisTurn = true;
|
|
5088
|
-
host.notify?.({ ...ev, message: safe });
|
|
5089
|
-
return;
|
|
5090
|
-
}
|
|
5091
|
-
const held = this.reflexBuf.length - this.reflexForwarded;
|
|
5092
|
-
const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
|
|
5093
|
-
const upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
|
|
5094
|
-
const out = this.reflexBuf.slice(this.reflexForwarded, upto);
|
|
5095
|
-
this.reflexForwarded = upto;
|
|
5096
|
-
if (!out) return;
|
|
5097
|
-
if (out.trim()) this.spokeThisTurn = true;
|
|
5098
|
-
host.notify?.({ ...ev, message: out });
|
|
5099
|
-
return;
|
|
5100
|
-
}
|
|
5101
|
-
host.notify?.(ev);
|
|
5102
|
-
}
|
|
5103
|
-
};
|
|
5104
|
-
this.voice = new Agent({
|
|
5105
|
-
ai: o.ai,
|
|
5106
|
-
fs: new MemFilesystem2(),
|
|
5107
|
-
model: o.reflexModel,
|
|
5108
|
-
stream: true,
|
|
5109
|
-
host: voiceHost,
|
|
5110
|
-
// The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
|
|
5111
|
-
// never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
|
|
5112
|
-
// it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
|
|
5113
|
-
askUserQuestion: false,
|
|
5114
|
-
systemPrompt: prompt,
|
|
5115
|
-
instructionFiles: false,
|
|
5116
|
-
maxSteps: 8,
|
|
5117
|
-
timeoutMs: 3e4,
|
|
5118
|
-
...o.reflexOptions,
|
|
5119
|
-
tools,
|
|
5120
|
-
// Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
|
|
5121
|
-
hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
|
|
5122
|
-
});
|
|
5123
|
-
}
|
|
5124
|
-
/** Resolve memory tools + inject index into voice system prompt (once). */
|
|
5125
|
-
async initMemory() {
|
|
5126
|
-
if (!this.memoryReady) return;
|
|
5127
|
-
const mem = await this.memoryReady;
|
|
5128
|
-
this.memoryReady = void 0;
|
|
5129
|
-
this.voice.options.tools.push(...mem.tools);
|
|
5130
|
-
if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
|
|
5131
|
-
}
|
|
5132
|
-
/** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
|
|
5133
|
-
* turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
|
|
5134
|
-
flushHeldReflexTail() {
|
|
5135
|
-
if (this.fabricationCut) return;
|
|
5136
|
-
const tail = this.reflexBuf.slice(this.reflexForwarded);
|
|
5137
|
-
this.reflexForwarded = this.reflexBuf.length;
|
|
5138
|
-
if (!tail) return;
|
|
5139
|
-
if (tail.trim()) this.spokeThisTurn = true;
|
|
5140
|
-
this.options.host?.notify?.({ kind: "text_delta", message: tail });
|
|
5141
|
-
}
|
|
5142
|
-
/** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
|
|
5143
|
-
resetTurn() {
|
|
5144
|
-
this.turnDispatched = false;
|
|
5145
|
-
this.turnBriefs.clear();
|
|
5146
|
-
this.spokeThisTurn = false;
|
|
5147
|
-
this.heldThisTurn = false;
|
|
5148
|
-
this.reflexBuf = "";
|
|
5149
|
-
this.reflexForwarded = 0;
|
|
5150
|
-
this.fabricationCut = false;
|
|
5151
|
-
this.turnFollowUp = false;
|
|
5152
|
-
this.voice.options.toolChoice = void 0;
|
|
5153
|
-
}
|
|
5154
|
-
/** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
|
|
5155
|
-
* now wait for the push" (CC's Task model). Block the temptations — TaskStatus polling and identical
|
|
5156
|
-
* re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
|
|
5157
|
-
* still allowed (parallel independent work). During a re-ack pass, block every tool. */
|
|
5158
|
-
dispatchGuard() {
|
|
5159
|
-
return {
|
|
5160
|
-
preToolUse: (call) => {
|
|
5161
|
-
if (this.nudging) return { block: true, reason: "Just say one short spoken acknowledgement \u2014 no tools this turn." };
|
|
5162
|
-
if (!this.turnDispatched) return;
|
|
5163
|
-
if (call.name === "TaskStatus")
|
|
5164
|
-
return { block: true, reason: "You just dispatched a task this turn \u2014 do NOT poll. Give one short spoken acknowledgement and end your turn; the result arrives later as a [task \u2026] event." };
|
|
5165
|
-
if ((call.name === "Act" || call.name === "Think") && this.turnBriefs.has(String(call.args?.brief ?? "")))
|
|
5166
|
-
return { block: true, reason: "You already dispatched this exact task \u2014 acknowledge briefly and end your turn." };
|
|
5167
|
-
}
|
|
5168
|
-
};
|
|
5169
|
-
}
|
|
5170
|
-
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
5171
|
-
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
5172
|
-
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
5173
|
-
* voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
|
|
5174
|
-
* Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
|
|
5175
|
-
get silentTurn() {
|
|
5176
|
-
return !!this.options.host && !this.spokeThisTurn && !this.heldThisTurn;
|
|
5177
|
-
}
|
|
5178
|
-
/** A turn that voiced nothing is dead air. Re-prompt the reflex ONCE so the LLM itself voices a short
|
|
5179
|
-
* line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
|
|
5180
|
-
* Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
|
|
5181
|
-
async ackIfSilent(fallback) {
|
|
5182
|
-
const dispatched = this.turnDispatched;
|
|
5183
|
-
this.nudging = true;
|
|
5184
|
-
try {
|
|
5185
|
-
await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
|
|
5186
|
-
} catch (e) {
|
|
5187
|
-
log9.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
|
|
5188
|
-
} finally {
|
|
5189
|
-
this.nudging = false;
|
|
5190
|
-
}
|
|
5191
|
-
if (!this.spokeThisTurn)
|
|
5192
|
-
this.options.host?.notify?.({ kind: "text_delta", message: fallback ?? (dispatched ? "Okay, on it." : "Sorry, could you say that again?") });
|
|
5193
|
-
}
|
|
5194
|
-
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns. */
|
|
5195
|
-
send(content) {
|
|
5196
|
-
return this.enqueue(async () => {
|
|
5197
|
-
await this.initMemory();
|
|
5198
|
-
this.resetTurn();
|
|
5199
|
-
const res = await this.voice.send(content);
|
|
5200
|
-
this.flushHeldReflexTail();
|
|
5201
|
-
if (this.silentTurn) await this.ackIfSilent();
|
|
5202
|
-
return res;
|
|
5203
|
-
});
|
|
5204
|
-
}
|
|
5205
|
-
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
5206
|
-
cancelTask(id) {
|
|
5207
|
-
const rec = this.tasks.get(id);
|
|
5208
|
-
if (!rec) return `No task '${id}'.`;
|
|
5209
|
-
if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
|
|
5210
|
-
rec.status = "cancelled";
|
|
5211
|
-
rec.controller.abort();
|
|
5212
|
-
return `Task ${rec.id} (${rec.label}) cancelled.`;
|
|
5213
|
-
}
|
|
5214
|
-
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
5215
|
-
* delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
|
|
5216
|
-
* tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
|
|
5217
|
-
* Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
|
|
5218
|
-
parkInFlightDeliveries() {
|
|
5219
|
-
const parked = [];
|
|
5220
|
-
for (const rec of this.tasks.values())
|
|
5221
|
-
if (rec.status === "running" && !rec.deliveryParked) {
|
|
5222
|
-
rec.deliveryParked = true;
|
|
5223
|
-
parked.push(rec.id);
|
|
5224
|
-
}
|
|
5225
|
-
return parked;
|
|
5119
|
+
/** Resolve when the engine is no longer speaking (immediate if already idle). */
|
|
5120
|
+
awaitIdle() {
|
|
5121
|
+
if (this.state !== "speaking" && this.state !== "thinking") return Promise.resolve();
|
|
5122
|
+
return new Promise((r) => this.idleWaiters.push(r));
|
|
5226
5123
|
}
|
|
5227
|
-
|
|
5228
|
-
|
|
5229
|
-
|
|
5230
|
-
|
|
5231
|
-
|
|
5232
|
-
|
|
5233
|
-
|
|
5234
|
-
|
|
5124
|
+
// --- speaking side (host-driven) ---
|
|
5125
|
+
/** open a spoken turn (idempotent — safe from both onUtterance and first-delta paths).
|
|
5126
|
+
* `ack` speaks the configured micro-ack as the context opener (utterance path only —
|
|
5127
|
+
* masks LLM TTFT; re-voice turns begun by their first delta skip it). */
|
|
5128
|
+
beginSpeech(ack = false) {
|
|
5129
|
+
if (this.speaking && this.ctxOpen) return;
|
|
5130
|
+
if (this.drainTimer) {
|
|
5131
|
+
this.clock.clearTimeout(this.drainTimer);
|
|
5132
|
+
this.drainTimer = null;
|
|
5235
5133
|
}
|
|
5134
|
+
this.interrupted = false;
|
|
5135
|
+
this.bcSupersede();
|
|
5136
|
+
this.resetOverlap(true);
|
|
5137
|
+
this.diag("tts_turn_begin", { ack, gapless: this.speaking });
|
|
5138
|
+
if (!this.speaking) this.player.markTurn();
|
|
5139
|
+
this.speaking = true;
|
|
5140
|
+
this.ctxOpen = true;
|
|
5141
|
+
this.spokeDeltas = false;
|
|
5142
|
+
this.reply = "";
|
|
5143
|
+
this.emo = this.options.emotions ? new EmotionStream(this.options.showEmotions) : null;
|
|
5144
|
+
this.echoWords = new Set(this.words(this.prevReply));
|
|
5145
|
+
this.tts.newContext();
|
|
5146
|
+
this.clearAckTimer("new_turn");
|
|
5147
|
+
if (ack && this.options.ackPhrase) {
|
|
5148
|
+
this.diag("ack_fired", { phrase: this.options.ackPhrase, adaptive: false });
|
|
5149
|
+
this.tts.speak(this.options.ackPhrase + " ", true);
|
|
5150
|
+
this.spokeDeltas = true;
|
|
5151
|
+
this.ackAt = this.clock.now();
|
|
5152
|
+
this.lastAck = this.options.ackPhrase;
|
|
5153
|
+
for (const w of this.words(this.options.ackPhrase)) this.echoWords.add(w);
|
|
5154
|
+
} else if (ack && this.options.adaptiveAckMs > 0) {
|
|
5155
|
+
this.ackTimer = this.clock.setTimeout(() => {
|
|
5156
|
+
this.ackTimer = null;
|
|
5157
|
+
if (!this.speaking || !this.ctxOpen || this.spokeDeltas || this.interrupted) return;
|
|
5158
|
+
const phrase = this.pickAck();
|
|
5159
|
+
this.diag("ack_fired", { phrase, adaptive: true, waitedMs: this.options.adaptiveAckMs });
|
|
5160
|
+
this.tts.speak(phrase + " ", true);
|
|
5161
|
+
this.spokeDeltas = true;
|
|
5162
|
+
this.ackAt = this.clock.now();
|
|
5163
|
+
this.lastAck = phrase;
|
|
5164
|
+
for (const w of this.words(phrase)) this.echoWords.add(w);
|
|
5165
|
+
this.setState("speaking");
|
|
5166
|
+
this.options.onAdaptiveAck();
|
|
5167
|
+
}, this.options.adaptiveAckMs);
|
|
5168
|
+
}
|
|
5169
|
+
if (!this.turnStartAt) this.turnStartAt = this.clock.now();
|
|
5170
|
+
this.setState("thinking");
|
|
5236
5171
|
}
|
|
5237
|
-
/**
|
|
5238
|
-
|
|
5239
|
-
|
|
5240
|
-
this.
|
|
5241
|
-
|
|
5242
|
-
|
|
5243
|
-
|
|
5244
|
-
|
|
5245
|
-
|
|
5246
|
-
|
|
5247
|
-
|
|
5248
|
-
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
5249
|
-
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
5250
|
-
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
5251
|
-
* text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
|
|
5252
|
-
queueRevoice(event, nonClean = false) {
|
|
5253
|
-
this.pendingEvents.push(event);
|
|
5254
|
-
if (nonClean) this.pendingNonClean = true;
|
|
5255
|
-
if (this.flushQueued) return;
|
|
5256
|
-
this.flushQueued = true;
|
|
5257
|
-
void this.enqueue(async () => {
|
|
5258
|
-
this.flushQueued = false;
|
|
5259
|
-
const events = this.pendingEvents.splice(0);
|
|
5260
|
-
const nonCleanTurn = this.pendingNonClean;
|
|
5261
|
-
this.pendingNonClean = false;
|
|
5262
|
-
if (!events.length) return;
|
|
5263
|
-
const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
|
|
5264
|
-
this.resetTurn();
|
|
5265
|
-
this.turnFollowUp = nonCleanTurn;
|
|
5266
|
-
await this.voice.send(events.join("\n"));
|
|
5267
|
-
this.flushHeldReflexTail();
|
|
5268
|
-
if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
|
|
5269
|
-
this.notify("revoice_done", "");
|
|
5270
|
-
});
|
|
5271
|
-
}
|
|
5272
|
-
/** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
|
|
5273
|
-
* Act briefs get a self-verify footer — the worker's report is trusted without review, so it
|
|
5274
|
-
* must check its own work before reporting (nearly free under prompt caching; measured honest:
|
|
5275
|
-
* it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
|
|
5276
|
-
buildBrief(brief, tier = "act", deliver = true) {
|
|
5277
|
-
const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
|
|
5278
|
-
const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
|
|
5279
|
-
const deliverContract = deliver ? "\n\n## DELIVER (spoken delivery)\nYou are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown." + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
|
|
5280
|
-
return (recent ? `${brief}
|
|
5281
|
-
|
|
5282
|
-
## Recent conversation (for context)
|
|
5283
|
-
${recent}` : brief) + verify + deliverContract;
|
|
5284
|
-
}
|
|
5285
|
-
/** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
|
|
5286
|
-
spawnWorker(id, label, briefText, tier, brief, followUp) {
|
|
5287
|
-
const o = this.options;
|
|
5288
|
-
const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
|
|
5289
|
-
const tierModel = tier === "think" ? o.thinkModel : o.actModel;
|
|
5290
|
-
const controller = new AbortController();
|
|
5291
|
-
const base = tierOpts?.hooks ?? o.actOptions?.hooks;
|
|
5292
|
-
const report = o.progressUpdates ? this.progressReporter(id) : void 0;
|
|
5293
|
-
const tail = [];
|
|
5294
|
-
const pushTail = (line) => {
|
|
5295
|
-
tail.push(line.slice(0, 200));
|
|
5296
|
-
if (tail.length > 120) tail.splice(0, tail.length - 120);
|
|
5297
|
-
};
|
|
5298
|
-
const hooks = {
|
|
5299
|
-
...base,
|
|
5300
|
-
preToolUse: async (call, meta) => {
|
|
5301
|
-
const d = await base?.preToolUse?.(call, meta);
|
|
5302
|
-
pushTail(`\u2699 ${describeCall(call)}`);
|
|
5303
|
-
report?.pre(call);
|
|
5304
|
-
return d;
|
|
5305
|
-
},
|
|
5306
|
-
postToolUse: async (call, result, meta) => {
|
|
5307
|
-
await base?.postToolUse?.(call, result, meta);
|
|
5308
|
-
const last = result?.trim().split("\n").filter(Boolean).pop();
|
|
5309
|
-
if (last) pushTail(` \u21B3 ${last}`);
|
|
5310
|
-
report?.post(call);
|
|
5311
|
-
},
|
|
5312
|
-
onToolOutput: (call, chunk, meta) => {
|
|
5313
|
-
base?.onToolOutput?.(call, chunk, meta);
|
|
5314
|
-
report?.output(chunk);
|
|
5315
|
-
}
|
|
5316
|
-
};
|
|
5317
|
-
const relayAsk = async (q2) => {
|
|
5318
|
-
const opts = q2.options?.length ? ` Options: ${q2.options.map((x) => x.label).join(", ")}.` : "";
|
|
5319
|
-
const a = await this.parkQuestion(id, `${q2.question}${opts}`);
|
|
5320
|
-
return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
|
|
5321
|
-
};
|
|
5322
|
-
const splitter = new SpokenSplitter();
|
|
5323
|
-
const speak = (seg) => {
|
|
5324
|
-
if (seg && !this.tasks.get(id)?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
|
|
5325
|
-
};
|
|
5326
|
-
const coalescer = new SentenceCoalescer();
|
|
5327
|
-
const feedSpoken = (s) => {
|
|
5328
|
-
const ready = coalescer.feed(s);
|
|
5329
|
-
if (ready) speak(ready);
|
|
5330
|
-
};
|
|
5331
|
-
const flushSpoken = () => speak(coalescer.flush());
|
|
5332
|
-
const askBridge = o.askRelay ? { ask: relayAsk } : o.host?.ask ? { ask: (q2) => o.host.ask(q2) } : {};
|
|
5333
|
-
const workerHost = {
|
|
5334
|
-
...askBridge,
|
|
5335
|
-
notify: (ev) => {
|
|
5336
|
-
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
5337
|
-
const { spoken, detail } = splitter.feed(ev.message);
|
|
5338
|
-
feedSpoken(spoken);
|
|
5339
|
-
if (detail.trim()) pushTail(detail.trim());
|
|
5340
|
-
return;
|
|
5341
|
-
}
|
|
5342
|
-
}
|
|
5343
|
-
};
|
|
5344
|
-
const agentOpts = {
|
|
5345
|
-
ai: o.ai,
|
|
5346
|
-
fs: o.fs,
|
|
5347
|
-
model: tierModel,
|
|
5348
|
-
...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
|
|
5349
|
-
...tierOpts,
|
|
5350
|
-
// Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
|
|
5351
|
-
// main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
|
|
5352
|
-
providerOptions: o.providerOptionsFor?.(tierModel),
|
|
5353
|
-
stream: true,
|
|
5354
|
-
// worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
|
|
5355
|
-
host: workerHost,
|
|
5356
|
-
// carries BOTH ask AND the <spoken>-splitting notify
|
|
5357
|
-
...hooks ? { hooks } : {},
|
|
5358
|
-
signal: controller.signal
|
|
5359
|
-
// shared with the checker so a cancel tears down both
|
|
5360
|
-
};
|
|
5361
|
-
const promise = new Agent(agentOpts).run(briefText).then((res) => {
|
|
5362
|
-
const { spoken, detail } = splitter.flush();
|
|
5363
|
-
feedSpoken(spoken);
|
|
5364
|
-
if (detail.trim()) pushTail(detail.trim());
|
|
5365
|
-
flushSpoken();
|
|
5366
|
-
return res;
|
|
5367
|
-
}).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err2) => this.onWorkerFailed(id, err2));
|
|
5368
|
-
this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
|
|
5369
|
-
if (this.tasks.size > this.options.maxTaskRecords)
|
|
5370
|
-
for (const [tid, rec] of this.tasks) {
|
|
5371
|
-
if (this.tasks.size <= this.options.maxTaskRecords) break;
|
|
5372
|
-
if (rec.status !== "running") this.tasks.delete(tid);
|
|
5373
|
-
}
|
|
5374
|
-
}
|
|
5375
|
-
/** Fresh-context check of a successful Act task: a NEW agent (same model/fs/tools, but NO shared
|
|
5376
|
-
* conversation context) re-reads the file state against the brief and fixes any gap. The fix lands
|
|
5377
|
-
* on the shared fs automatically (workers write fs directly, no overlay), so grading sees the
|
|
5378
|
-
* corrected state. Bounded to ONE pass. Off unless `verifyActTasks`; never runs for think/failed/
|
|
5379
|
-
* cancelled tasks. Usage is merged so /cost reflects the real (worker + checker) spend. */
|
|
5380
|
-
async maybeVerify(id, brief, res, tier, agentOpts, askBridge) {
|
|
5381
|
-
if (!this.options.verifyActTasks || tier !== "act" || res.finishReason !== "stop") return res;
|
|
5382
|
-
if (this.tasks.get(id)?.status === "cancelled") return res;
|
|
5383
|
-
const { stream: _stream, host: _host, ...restOpts } = agentOpts;
|
|
5384
|
-
const checkerOpts = {
|
|
5385
|
-
...restOpts,
|
|
5386
|
-
...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
|
|
5387
|
-
};
|
|
5388
|
-
const checkBrief = `${this.buildBrief(brief, tier, false)}
|
|
5389
|
-
|
|
5390
|
-
## VERIFY MODE
|
|
5391
|
-
Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
|
|
5392
|
-
this.notify("task_verify", `task ${id}: verifying`, { id });
|
|
5393
|
-
const cres = await new Agent(checkerOpts).run(checkBrief);
|
|
5394
|
-
if (cres.finishReason !== "stop") {
|
|
5395
|
-
log9.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
|
|
5396
|
-
this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
|
|
5172
|
+
/** Feed a spoken delta. Returns the on-screen echo text (emotion tags shown/hidden per config) so the
|
|
5173
|
+
* host renders the SAME stream that was parsed for TTS — no second, state-doubling parse. */
|
|
5174
|
+
speakDelta(text) {
|
|
5175
|
+
if (this.interrupted) return "";
|
|
5176
|
+
this.clearAckTimer("first_delta");
|
|
5177
|
+
if (!this.speaking || !this.ctxOpen) this.beginSpeech();
|
|
5178
|
+
let { speech, display, prose } = this.emo ? this.emo.feed(text) : { speech: text, display: text, prose: text };
|
|
5179
|
+
if (this.reply && /[.!?…]$/.test(this.reply) && /^[A-Z]/.test(prose)) {
|
|
5180
|
+
speech = " " + speech;
|
|
5181
|
+
display = " " + display;
|
|
5182
|
+
prose = " " + prose;
|
|
5397
5183
|
}
|
|
5398
|
-
|
|
5399
|
-
|
|
5400
|
-
|
|
5401
|
-
|
|
5402
|
-
|
|
5403
|
-
|
|
5404
|
-
|
|
5405
|
-
|
|
5406
|
-
usage: res.usage && cres.usage ? {
|
|
5407
|
-
promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
|
|
5408
|
-
completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
|
|
5409
|
-
totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
|
|
5410
|
-
cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
|
|
5411
|
-
cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
|
|
5412
|
-
} : res.usage ?? cres.usage
|
|
5413
|
-
};
|
|
5184
|
+
this.reply += prose;
|
|
5185
|
+
if (prose.trim()) this.repliedSinceDispatch = true;
|
|
5186
|
+
for (const w of this.words(this.reply)) this.echoWords.add(w);
|
|
5187
|
+
this.tts.speak(forSpeech(speech), true);
|
|
5188
|
+
if (!this.spokeDeltas && this.turnStartAt) log9.debug(`ttft: ${Math.round(this.clock.now() - this.turnStartAt)}ms`);
|
|
5189
|
+
this.spokeDeltas = true;
|
|
5190
|
+
this.setState("speaking");
|
|
5191
|
+
return display;
|
|
5414
5192
|
}
|
|
5415
|
-
/**
|
|
5416
|
-
|
|
5417
|
-
|
|
5418
|
-
|
|
5419
|
-
|
|
5420
|
-
|
|
5421
|
-
|
|
5422
|
-
|
|
5423
|
-
|
|
5424
|
-
|
|
5425
|
-
|
|
5426
|
-
|
|
5427
|
-
}
|
|
5428
|
-
|
|
5429
|
-
|
|
5430
|
-
|
|
5431
|
-
|
|
5432
|
-
|
|
5433
|
-
|
|
5434
|
-
|
|
5435
|
-
|
|
5436
|
-
|
|
5437
|
-
const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
|
|
5438
|
-
emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
|
|
5439
|
-
}, Math.max(this.options.progressIntervalMs, 250));
|
|
5440
|
-
timer.unref?.();
|
|
5441
|
-
return {
|
|
5442
|
-
pre: (call) => {
|
|
5443
|
-
inflight = { call, at: Date.now(), tail: "" };
|
|
5444
|
-
},
|
|
5445
|
-
output: (chunk) => {
|
|
5446
|
-
if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
|
|
5447
|
-
},
|
|
5448
|
-
// digest only — NEVER re-voices directly
|
|
5449
|
-
post: (call) => {
|
|
5450
|
-
steps++;
|
|
5451
|
-
inflight = null;
|
|
5452
|
-
const rec = due();
|
|
5453
|
-
if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
|
|
5193
|
+
/** close the spoken turn (idempotent); stays audible until ALL audio arrived AND playback drains */
|
|
5194
|
+
endSpeech() {
|
|
5195
|
+
this.interrupted = false;
|
|
5196
|
+
this.clearAckTimer("turn_end");
|
|
5197
|
+
if (!this.speaking) return;
|
|
5198
|
+
this.diag("tts_turn_end", { spoke: this.spokeDeltas, replyChars: this.reply.length });
|
|
5199
|
+
this.ctxOpen = false;
|
|
5200
|
+
if (this.emo) {
|
|
5201
|
+
const t = this.emo.flush();
|
|
5202
|
+
this.emo = null;
|
|
5203
|
+
if (t.prose) this.reply += t.prose;
|
|
5204
|
+
if (t.speech) this.tts.speak(forSpeech(t.speech), true);
|
|
5205
|
+
}
|
|
5206
|
+
if (this.reply) this.prevReply = this.reply;
|
|
5207
|
+
const settle = () => {
|
|
5208
|
+
if (this.ctxOpen) {
|
|
5209
|
+
this.drainTimer = null;
|
|
5210
|
+
return;
|
|
5211
|
+
}
|
|
5212
|
+
if (this.pausedAt) {
|
|
5213
|
+
this.drainTimer = this.clock.setTimeout(settle, 250);
|
|
5214
|
+
return;
|
|
5454
5215
|
}
|
|
5216
|
+
this.drainTimer = null;
|
|
5217
|
+
this.speaking = false;
|
|
5218
|
+
if (this.turnStartAt) log9.debug(`turn: ${Math.round(this.clock.now() - this.turnStartAt)}ms (incl. playback)`);
|
|
5219
|
+
this.echoUntil = this.clock.now() + 2500;
|
|
5220
|
+
if (!this.usingAec) this.stt.reset();
|
|
5221
|
+
this.setState("listening");
|
|
5222
|
+
if (this.uttQueue.length) this.pumpQueue();
|
|
5223
|
+
};
|
|
5224
|
+
const drainThenSettle = () => {
|
|
5225
|
+
if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
|
|
5226
|
+
this.drainTimer = this.clock.setTimeout(settle, this.player.drainMs() + 300);
|
|
5455
5227
|
};
|
|
5228
|
+
if (this.spokeDeltas) {
|
|
5229
|
+
this.tts.onDone = drainThenSettle;
|
|
5230
|
+
this.tts.end();
|
|
5231
|
+
if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
|
|
5232
|
+
this.drainTimer = this.clock.setTimeout(drainThenSettle, 15e3);
|
|
5233
|
+
} else drainThenSettle();
|
|
5456
5234
|
}
|
|
5457
|
-
/**
|
|
5458
|
-
*
|
|
5459
|
-
|
|
5460
|
-
|
|
5461
|
-
|
|
5462
|
-
|
|
5463
|
-
const finish = (answer) => {
|
|
5464
|
-
if (settled) return;
|
|
5465
|
-
settled = true;
|
|
5466
|
-
clearTimeout(timer);
|
|
5467
|
-
this.pendingAsks.delete(askId);
|
|
5468
|
-
resolve4(answer);
|
|
5469
|
-
};
|
|
5470
|
-
const timer = setTimeout(() => {
|
|
5471
|
-
this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
|
|
5472
|
-
finish("");
|
|
5473
|
-
}, this.options.askTimeoutMs);
|
|
5474
|
-
this.pendingAsks.set(askId, { question, resolve: finish });
|
|
5475
|
-
this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
|
|
5476
|
-
this.queueRevoice(`[task ${askId} asks] ${question}
|
|
5477
|
-
(Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
|
|
5478
|
-
});
|
|
5235
|
+
/** text of the reply cut by the last barge-in — consumed by the host to tell the model what
|
|
5236
|
+
* the user did NOT hear. Cleared on read. */
|
|
5237
|
+
takeInterruptedReply() {
|
|
5238
|
+
const r = this.lastInterrupted;
|
|
5239
|
+
this.lastInterrupted = null;
|
|
5240
|
+
return r;
|
|
5479
5241
|
}
|
|
5480
|
-
/**
|
|
5481
|
-
|
|
5482
|
-
this.
|
|
5242
|
+
/** Speak a short filler phrase without starting a model turn (stays in listening mode after). */
|
|
5243
|
+
speakFiller(text) {
|
|
5244
|
+
if (!text || this.speaking) return;
|
|
5245
|
+
const replied = this.repliedSinceDispatch;
|
|
5246
|
+
this.beginSpeech();
|
|
5247
|
+
this.speakDelta(text);
|
|
5248
|
+
this.endSpeech();
|
|
5249
|
+
this.repliedSinceDispatch = replied;
|
|
5483
5250
|
}
|
|
5484
|
-
/**
|
|
5485
|
-
*
|
|
5486
|
-
*
|
|
5487
|
-
|
|
5488
|
-
|
|
5489
|
-
|
|
5490
|
-
|
|
5491
|
-
|
|
5492
|
-
* • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
|
|
5493
|
-
* exists AND this task wasn't already a follow-up (one hop max). Wires the dead
|
|
5494
|
-
* "Reserve Think for a problem Act already FAILED at" promise.
|
|
5495
|
-
* • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
|
|
5496
|
-
* • ask → ask the user ONE concrete question if genuinely blocked.
|
|
5497
|
-
*
|
|
5498
|
-
* Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
|
|
5499
|
-
* failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
|
|
5500
|
-
integrationPrompt(rec, outcome, body, finishReason) {
|
|
5501
|
-
const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
|
|
5502
|
-
const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
|
|
5503
|
-
const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
|
|
5504
|
-
const hasThink = this.options.thinkModel !== false;
|
|
5505
|
-
const options = [];
|
|
5506
|
-
if (!rec.followUp && canEscalate && hasThink)
|
|
5507
|
-
options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
|
|
5508
|
-
if (!rec.followUp && canEscalate)
|
|
5509
|
-
options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
|
|
5510
|
-
options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
|
|
5511
|
-
options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
|
|
5512
|
-
const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
|
|
5513
|
-
const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
|
|
5514
|
-
return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
|
|
5251
|
+
/** Enqueue a COMPLETE worker utterance (already-split spoken text) onto the central speech queue.
|
|
5252
|
+
* If nothing is currently speaking it plays immediately; otherwise it queues and plays after the
|
|
5253
|
+
* current utterance fully ends (settle → pumpQueue) — never spliced into an open reflex utterance. */
|
|
5254
|
+
enqueueUtterance(text) {
|
|
5255
|
+
if (!text || !/[\p{L}\p{N}]/u.test(text)) return;
|
|
5256
|
+
this.diag("enqueue_utterance", { chars: text.length, queued: this.uttQueue.length, speaking: this.speaking });
|
|
5257
|
+
this.uttQueue.push(text);
|
|
5258
|
+
if (!this.speaking) this.pumpQueue();
|
|
5515
5259
|
}
|
|
5516
|
-
|
|
5517
|
-
|
|
5518
|
-
|
|
5519
|
-
if (
|
|
5520
|
-
|
|
5521
|
-
|
|
5522
|
-
|
|
5523
|
-
|
|
5524
|
-
|
|
5525
|
-
const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
|
|
5526
|
-
return this.failTask(rec, msg);
|
|
5527
|
-
}
|
|
5528
|
-
rec.status = "done";
|
|
5529
|
-
rec.result = res.text;
|
|
5530
|
-
const incomplete = res.finishReason !== "stop";
|
|
5531
|
-
log9.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
|
|
5532
|
-
this.notify("task_done", `task ${id} (${rec.label}) completed`, {
|
|
5533
|
-
id,
|
|
5534
|
-
text: res.text,
|
|
5535
|
-
usage: res.usage,
|
|
5536
|
-
usageEstimated: res.usageEstimated,
|
|
5537
|
-
finishReason: res.finishReason,
|
|
5538
|
-
steps: res.steps,
|
|
5539
|
-
toolCalls: res.messages.filter((m) => m.role === "tool").length
|
|
5540
|
-
});
|
|
5541
|
-
if (incomplete) {
|
|
5542
|
-
return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
|
|
5543
|
-
}
|
|
5544
|
-
const tail = rec.splitter?.flush();
|
|
5545
|
-
if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
|
|
5546
|
-
if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
|
|
5547
|
-
if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
|
|
5548
|
-
this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
|
|
5260
|
+
/** Play the next queued worker utterance as its own one-shot turn (begin → delta → end). Drives the
|
|
5261
|
+
* next one from the settle completion (endSpeech), so utterances serialize without overlap. */
|
|
5262
|
+
pumpQueue() {
|
|
5263
|
+
if (this.speaking) return;
|
|
5264
|
+
const text = this.uttQueue.shift();
|
|
5265
|
+
if (text == null) return;
|
|
5266
|
+
this.beginSpeech();
|
|
5267
|
+
this.speakDelta(text);
|
|
5268
|
+
this.endSpeech();
|
|
5549
5269
|
}
|
|
5550
|
-
|
|
5551
|
-
|
|
5270
|
+
/** Short varied adaptive acks (adaptiveAckMs) — two shape pools picked by the dispatched
|
|
5271
|
+
* utterance (question → thinking-ish, otherwise neutral/on-it), with anti-repetition (never one
|
|
5272
|
+
* of the last 4 used). A 3-phrase round-robin sounded synthetic live ("started with 'hmm' too
|
|
5273
|
+
* many times"). All phrases are sub-second and semantically safe for their shape.
|
|
5274
|
+
* No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
|
|
5275
|
+
static ACKS_NEUTRAL = ["Mm-hm.", "One sec.", "Right.", "Sec.", "On it.", "Sure, moment.", "Uh, one moment.", "Let me see."];
|
|
5276
|
+
static ACKS_QUESTION = ["Hmm.", "Let me think.", "Hm, let me see.", "Good question.", "Mm, one sec.", "Let's see."];
|
|
5277
|
+
recentAcks = [];
|
|
5278
|
+
pickAck() {
|
|
5279
|
+
const pool = this.lastDispatchWasQuestion ? _VoiceEngine.ACKS_QUESTION : _VoiceEngine.ACKS_NEUTRAL;
|
|
5280
|
+
const fresh = pool.filter((p) => !this.recentAcks.includes(p));
|
|
5281
|
+
const phrase = (fresh.length ? fresh : pool)[Math.floor(Math.random() * (fresh.length ? fresh.length : pool.length))];
|
|
5282
|
+
this.recentAcks.push(phrase);
|
|
5283
|
+
if (this.recentAcks.length > 4) this.recentAcks.shift();
|
|
5284
|
+
return phrase;
|
|
5285
|
+
}
|
|
5286
|
+
lastDispatchWasQuestion = false;
|
|
5287
|
+
clearAckTimer(reason) {
|
|
5288
|
+
if (!this.ackTimer) return;
|
|
5289
|
+
this.clock.clearTimeout(this.ackTimer);
|
|
5290
|
+
this.ackTimer = null;
|
|
5291
|
+
if (reason) this.diag("ack_cancelled", { reason });
|
|
5292
|
+
}
|
|
5293
|
+
/** Cancel a pending adaptive ack without touching the turn — hosts call this when the turn turns out
|
|
5294
|
+
* to be a Hold (intentionally silent; an ack would read as the start of an answer). */
|
|
5295
|
+
cancelPendingAck() {
|
|
5296
|
+
this.clearAckTimer("hold");
|
|
5552
5297
|
}
|
|
5553
|
-
|
|
5554
|
-
|
|
5555
|
-
|
|
5556
|
-
|
|
5557
|
-
|
|
5558
|
-
this.
|
|
5559
|
-
|
|
5298
|
+
/** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
|
|
5299
|
+
interrupt() {
|
|
5300
|
+
this.clearAckTimer("interrupt");
|
|
5301
|
+
if (this.uttQueue.length) log9.info(`barge-in dropped ${this.uttQueue.length} queued worker utterance(s)`);
|
|
5302
|
+
const droppedQueued = this.uttQueue.length;
|
|
5303
|
+
this.uttQueue = [];
|
|
5304
|
+
if (!this.speaking && !this.drainTimer) return;
|
|
5305
|
+
this.diag("interrupt", { droppedQueued, ctxOpen: this.ctxOpen, playedMs: Math.round(Math.max(0, this.player.playedMs())) });
|
|
5306
|
+
if (this.drainTimer) {
|
|
5307
|
+
this.clock.clearTimeout(this.drainTimer);
|
|
5308
|
+
this.drainTimer = null;
|
|
5309
|
+
}
|
|
5310
|
+
this.resetOverlap(false);
|
|
5311
|
+
this.lastResumeAt = 0;
|
|
5312
|
+
const heardChars = Math.round(Math.max(0, this.player.playedMs()) / 1e3 * 15);
|
|
5313
|
+
if (this.reply) this.lastInterrupted = { full: this.reply, heard: this.reply.slice(0, heardChars) };
|
|
5314
|
+
this.speaking = false;
|
|
5315
|
+
this.ctxOpen = false;
|
|
5316
|
+
this.interrupted = true;
|
|
5317
|
+
this.suspectUntil = 0;
|
|
5318
|
+
this.echoUntil = this.clock.now() + Math.max(2500, this.player.drainMs() + 3e3);
|
|
5319
|
+
this.tts.cancel();
|
|
5320
|
+
this.player.kill();
|
|
5321
|
+
if (!this.usingAec) this.stt.reset();
|
|
5322
|
+
if (this.reply) this.prevReply = this.reply;
|
|
5323
|
+
this.setState("listening");
|
|
5560
5324
|
}
|
|
5561
|
-
|
|
5562
|
-
|
|
5563
|
-
|
|
5564
|
-
|
|
5565
|
-
|
|
5566
|
-
|
|
5567
|
-
this.
|
|
5568
|
-
|
|
5569
|
-
|
|
5570
|
-
|
|
5571
|
-
|
|
5325
|
+
stop() {
|
|
5326
|
+
this.uttQueue = [];
|
|
5327
|
+
this.clearAckTimer();
|
|
5328
|
+
if (this.resumeTimer) this.clock.clearTimeout(this.resumeTimer);
|
|
5329
|
+
if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
|
|
5330
|
+
if (this.drainTimer) this.clock.clearTimeout(this.drainTimer);
|
|
5331
|
+
if (this.specTimer) this.clock.clearTimeout(this.specTimer);
|
|
5332
|
+
this.bcSupersede();
|
|
5333
|
+
this.stt.stop();
|
|
5334
|
+
this.player.kill();
|
|
5335
|
+
this.tts.close();
|
|
5336
|
+
this.setState("idle");
|
|
5572
5337
|
}
|
|
5573
|
-
|
|
5574
|
-
|
|
5575
|
-
|
|
5576
|
-
async dispatch(brief, tier = "act", label, followUp = false) {
|
|
5577
|
-
if (tier === "think" && this.options.thinkModel === false) tier = "act";
|
|
5578
|
-
if (followUp) this.autoEscalations++;
|
|
5579
|
-
const id = `t${++this.seq}`;
|
|
5580
|
-
const lbl = label ?? tier;
|
|
5581
|
-
await this.options.onTaskStart?.(id, lbl);
|
|
5582
|
-
this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
|
|
5583
|
-
this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
|
|
5584
|
-
return id;
|
|
5338
|
+
// --- listening side (STT-driven) ---
|
|
5339
|
+
words(s) {
|
|
5340
|
+
return s.toLowerCase().replace(/[^a-z0-9\s]/g, "").split(/\s+/).filter((w) => w.length >= 2);
|
|
5585
5341
|
}
|
|
5586
|
-
|
|
5587
|
-
return
|
|
5588
|
-
name: "Act",
|
|
5589
|
-
description: 'Escalate real work (reading/editing files, searching, running tasks, building) to a standard background worker. Returns immediately with a task id; the result arrives later as a "[task <id> completed]" event. Provide a clear, self-contained `brief` (the worker does not hear the live conversation).',
|
|
5590
|
-
parameters: {
|
|
5591
|
-
type: "object",
|
|
5592
|
-
required: ["brief"],
|
|
5593
|
-
properties: {
|
|
5594
|
-
brief: { type: "string", description: "full, self-contained instructions for the worker" },
|
|
5595
|
-
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
5596
|
-
}
|
|
5597
|
-
},
|
|
5598
|
-
run: async ({ brief, label }) => {
|
|
5599
|
-
this.turnDispatched = true;
|
|
5600
|
-
this.turnBriefs.add(String(brief ?? ""));
|
|
5601
|
-
this.voice.options.toolChoice = "none";
|
|
5602
|
-
const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
|
|
5603
|
-
return `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
5604
|
-
}
|
|
5605
|
-
};
|
|
5342
|
+
novelWords(text) {
|
|
5343
|
+
return this.words(text).filter((w) => !this.echoWords.has(w));
|
|
5606
5344
|
}
|
|
5607
|
-
|
|
5608
|
-
return
|
|
5609
|
-
name: "Think",
|
|
5610
|
-
description: "Escalate to a premium deep-reasoning agent for complex analysis, architecture decisions, hard debugging, or planning. Same async pattern as Act \u2014 returns a task id. Use when the problem needs careful thought before (or instead of) action. Do not use Think for simple tasks \u2014 Act is cheaper and faster.",
|
|
5611
|
-
parameters: {
|
|
5612
|
-
type: "object",
|
|
5613
|
-
required: ["brief"],
|
|
5614
|
-
properties: {
|
|
5615
|
-
brief: { type: "string", description: "the question or problem to reason about deeply" },
|
|
5616
|
-
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
5617
|
-
}
|
|
5618
|
-
},
|
|
5619
|
-
run: async ({ brief, label }) => {
|
|
5620
|
-
this.turnDispatched = true;
|
|
5621
|
-
this.turnBriefs.add(String(brief ?? ""));
|
|
5622
|
-
this.voice.options.toolChoice = "none";
|
|
5623
|
-
const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
|
|
5624
|
-
return `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
5625
|
-
}
|
|
5626
|
-
};
|
|
5345
|
+
echoActive() {
|
|
5346
|
+
return this.speaking || this.clock.now() < this.echoUntil;
|
|
5627
5347
|
}
|
|
5628
|
-
|
|
5629
|
-
|
|
5630
|
-
|
|
5631
|
-
|
|
5632
|
-
|
|
5633
|
-
|
|
5634
|
-
|
|
5635
|
-
|
|
5636
|
-
|
|
5637
|
-
}
|
|
5638
|
-
};
|
|
5348
|
+
/** Genuine user speech vs our own bleed (AEC tier): novel words must DOMINATE, not merely exist.
|
|
5349
|
+
* Degraded AEC + an STT mis-hearing manufactures a single novel word out of pure echo (a name or
|
|
5350
|
+
* rare word in our own reply comes back transcribed slightly differently — 1 novel / N words).
|
|
5351
|
+
* A real interjection is mostly novel ("stop", "wait what") — short utterances pass on ratio,
|
|
5352
|
+
* longer ones on count. */
|
|
5353
|
+
genuine(text) {
|
|
5354
|
+
const total = this.words(text).length;
|
|
5355
|
+
const novel = this.novelWords(text).length;
|
|
5356
|
+
return novel > 0 && novel / Math.max(1, total) > 0.5;
|
|
5639
5357
|
}
|
|
5640
|
-
|
|
5641
|
-
|
|
5642
|
-
|
|
5643
|
-
|
|
5644
|
-
|
|
5645
|
-
|
|
5646
|
-
|
|
5647
|
-
name: "QuickLook",
|
|
5648
|
-
description: `Instant read-only lookup \u2014 one of: ${kinds.join(", ")}. For trivial facts only; anything needing search, commands, or reasoning goes through Act.`,
|
|
5649
|
-
parameters: {
|
|
5650
|
-
type: "object",
|
|
5651
|
-
required: ["what"],
|
|
5652
|
-
properties: {
|
|
5653
|
-
what: { type: "string", enum: kinds, description: "what to look up" },
|
|
5654
|
-
path: { type: "string", description: "for ls/file: the path to look at" }
|
|
5358
|
+
handlePartial(text) {
|
|
5359
|
+
if (this.speaking) {
|
|
5360
|
+
if (!this.options.bargeIn) return;
|
|
5361
|
+
if (this.clock.now() < this.bargeGraceUntil) {
|
|
5362
|
+
if (this.lastGraceDiag !== this.bargeGraceUntil) {
|
|
5363
|
+
this.lastGraceDiag = this.bargeGraceUntil;
|
|
5364
|
+
this.diag("grace_suppress", { text: text.slice(0, 60) });
|
|
5655
5365
|
}
|
|
5656
|
-
|
|
5657
|
-
|
|
5658
|
-
|
|
5659
|
-
|
|
5660
|
-
|
|
5661
|
-
|
|
5662
|
-
|
|
5663
|
-
|
|
5664
|
-
|
|
5665
|
-
|
|
5666
|
-
|
|
5667
|
-
|
|
5668
|
-
|
|
5669
|
-
|
|
5670
|
-
|
|
5671
|
-
|
|
5672
|
-
|
|
5673
|
-
|
|
5674
|
-
|
|
5675
|
-
|
|
5676
|
-
|
|
5677
|
-
|
|
5678
|
-
return `Tools your background worker (Act) can actually use: ${names.join(", ")}. Read each name literally and match the request to a SPECIFIC tool; if none fits, you do NOT have that ability \u2014 say so honestly.` + webNote + mcpNote;
|
|
5679
|
-
}
|
|
5680
|
-
case "time":
|
|
5681
|
-
return (/* @__PURE__ */ new Date()).toString();
|
|
5682
|
-
case "branch": {
|
|
5683
|
-
if (!fs) return "unavailable (no filesystem)";
|
|
5684
|
-
const head = (await fs.readFile(".git/HEAD")).trim();
|
|
5685
|
-
return head.startsWith("ref: refs/heads/") ? `branch: ${head.slice("ref: refs/heads/".length)}` : `detached HEAD at ${head.slice(0, 12)}`;
|
|
5686
|
-
}
|
|
5687
|
-
case "ls": {
|
|
5688
|
-
if (!fs) return "unavailable (no filesystem)";
|
|
5689
|
-
const names = await fs.readDir(String(path ?? "."));
|
|
5690
|
-
return names.slice(0, 50).join("\n") + (names.length > 50 ? `
|
|
5691
|
-
\u2026 (+${names.length - 50} more)` : "");
|
|
5692
|
-
}
|
|
5693
|
-
case "file": {
|
|
5694
|
-
if (!fs) return "unavailable (no filesystem)";
|
|
5695
|
-
if (!path) return "file lookup needs a path";
|
|
5696
|
-
const text = await fs.readFile(String(path));
|
|
5697
|
-
return text.length > CAP2 ? text.slice(0, CAP2) + `
|
|
5698
|
-
\u2026 (truncated \u2014 ${text.length} chars total; Act for the full file)` : text;
|
|
5699
|
-
}
|
|
5700
|
-
default:
|
|
5701
|
-
return `unknown lookup '${what}'`;
|
|
5366
|
+
if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
|
|
5367
|
+
return;
|
|
5368
|
+
}
|
|
5369
|
+
if (this.overlapCapable) {
|
|
5370
|
+
const txt = text.trim();
|
|
5371
|
+
if (!txt || txt === this.lastOverlapPartial) return;
|
|
5372
|
+
this.lastOverlapPartial = txt;
|
|
5373
|
+
if (!this.genuine(txt)) {
|
|
5374
|
+
if (this.pausedAt) this.armResume();
|
|
5375
|
+
return;
|
|
5376
|
+
}
|
|
5377
|
+
if (!this.pausedAt) {
|
|
5378
|
+
this.pausedAt = this.clock.now();
|
|
5379
|
+
const sinceResumeMs = this.lastResumeAt ? Math.round(this.clock.now() - this.lastResumeAt) : void 0;
|
|
5380
|
+
this.diag("overlap_pause", { text: txt.slice(0, 60), sinceResumeMs });
|
|
5381
|
+
this.player.pause();
|
|
5382
|
+
if (this.lastResumeAt && this.clock.now() - this.lastResumeAt < this.options.overlapRepauseCedeMs) {
|
|
5383
|
+
const phase = this.ctxOpen ? "speaking" : "drain";
|
|
5384
|
+
this.diag("barge_in", { phase, trigger: "overlap_repause", sinceResumeMs });
|
|
5385
|
+
this.interrupt();
|
|
5386
|
+
this.options.onBargeIn(phase);
|
|
5387
|
+
return;
|
|
5702
5388
|
}
|
|
5703
|
-
} catch (e) {
|
|
5704
|
-
return `lookup failed: ${e?.message ?? e}`;
|
|
5705
5389
|
}
|
|
5390
|
+
if (this.words(txt).length >= 2) {
|
|
5391
|
+
const phase = this.ctxOpen ? "speaking" : "drain";
|
|
5392
|
+
this.diag("barge_in", { phase, trigger: "overlap_cede", novel: this.novelWords(txt).length, total: this.words(txt).length });
|
|
5393
|
+
this.interrupt();
|
|
5394
|
+
this.options.onBargeIn(phase);
|
|
5395
|
+
return;
|
|
5396
|
+
}
|
|
5397
|
+
this.armResume();
|
|
5398
|
+
return;
|
|
5706
5399
|
}
|
|
5707
|
-
|
|
5400
|
+
const barge = this.usingAec ? this.genuine(text) : this.novelWords(text).length >= (this.suspectUntil ? 1 : 2);
|
|
5401
|
+
if (barge) {
|
|
5402
|
+
const phase = this.ctxOpen ? "speaking" : "drain";
|
|
5403
|
+
this.diag("barge_in", { phase, trigger: this.usingAec ? "genuine" : "heuristic_novel", novel: this.novelWords(text).length, total: this.words(text).length });
|
|
5404
|
+
this.interrupt();
|
|
5405
|
+
this.options.onBargeIn(phase);
|
|
5406
|
+
}
|
|
5407
|
+
return;
|
|
5408
|
+
}
|
|
5409
|
+
if (this.pendingUtt && text.trim()) {
|
|
5410
|
+
if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
|
|
5411
|
+
this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), Math.max(800, this.options.utteranceMergeMs));
|
|
5412
|
+
}
|
|
5413
|
+
this.trackSpeculation(text);
|
|
5414
|
+
this.trackBackchannel(text);
|
|
5415
|
+
if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
|
|
5708
5416
|
}
|
|
5709
|
-
|
|
5710
|
-
|
|
5711
|
-
|
|
5712
|
-
|
|
5713
|
-
|
|
5714
|
-
|
|
5715
|
-
|
|
5716
|
-
|
|
5717
|
-
|
|
5718
|
-
|
|
5719
|
-
|
|
5720
|
-
|
|
5721
|
-
|
|
5722
|
-
|
|
5417
|
+
/** Speculative-reflex trigger (listening side only — the speaking branch returns before this): a
|
|
5418
|
+
* partial that hasn't CHANGED for speculativeMs and has ≥ speculativeMinWords is stable → fire
|
|
5419
|
+
* onSpeculate once. If the user then keeps talking well past the speculated text, onSpeculateAbort
|
|
5420
|
+
* tells the host to kill the held call early. The confirm/abort DECISION belongs to the host at
|
|
5421
|
+
* dispatch time (speculationConfirms against the real final) — flushUtterance only resets the
|
|
5422
|
+
* trigger for the next turn. Merge windows are untouched: no speculation while an endpointed
|
|
5423
|
+
* utterance is pending (the final would be the MERGED text, which the partial alone never matches). */
|
|
5424
|
+
trackSpeculation(text) {
|
|
5425
|
+
if (!this.options.speculativeMs) return;
|
|
5426
|
+
const t = text.trim();
|
|
5427
|
+
if (this.specText) {
|
|
5428
|
+
if (this.words(t).length > this.words(this.specText).length + 2) {
|
|
5429
|
+
this.specText = "";
|
|
5430
|
+
this.diag("speculate_abort", { reason: "partial_grew", partial: t.slice(0, 60) });
|
|
5431
|
+
this.options.onSpeculateAbort();
|
|
5723
5432
|
}
|
|
5433
|
+
return;
|
|
5434
|
+
}
|
|
5435
|
+
if (this.specSpent || this.pendingUtt || !t || t === this.specPartial) return;
|
|
5436
|
+
this.specPartial = t;
|
|
5437
|
+
if (this.specTimer) this.clock.clearTimeout(this.specTimer);
|
|
5438
|
+
this.specTimer = this.clock.setTimeout(() => {
|
|
5439
|
+
this.specTimer = null;
|
|
5440
|
+
if (this.speaking || this.specSpent || this.pendingUtt) return;
|
|
5441
|
+
if (this.words(this.specPartial).length < this.options.speculativeMinWords) return;
|
|
5442
|
+
const flat = this.specPartial.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
5443
|
+
if (flat && this.lastDispatchFlat && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS && (this.lastDispatchFlat.startsWith(flat) || flat.startsWith(this.lastDispatchFlat))) return;
|
|
5444
|
+
this.specSpent = true;
|
|
5445
|
+
this.specText = this.specPartial;
|
|
5446
|
+
log9.debug(`speculate: "${this.specText.slice(0, 60)}"`);
|
|
5447
|
+
this.diag("speculate", { text: this.specText });
|
|
5448
|
+
this.options.onSpeculate(this.specText);
|
|
5449
|
+
}, this.options.speculativeMs);
|
|
5450
|
+
}
|
|
5451
|
+
/** Backchannel blip pool — short, quiet, semantically inert. Chosen to be transcript-safe: if
|
|
5452
|
+
* imperfect AEC lets a blip reach Soniox MID-user-speech it lands inside their partial stream, so
|
|
5453
|
+
* every phrase is a word whose accidental presence barely hurts a transcript, and flushUtterance
|
|
5454
|
+
* strips an isolated echo of the exact phrase at a clause edge within 2s (stripBackchannelEcho).
|
|
5455
|
+
* 'Okay.'/'Right.' are fine here (unlike ACKS_NEUTRAL's no-Okay rule): a standalone user "Okay."
|
|
5456
|
+
* right after OUR blip is overwhelmingly the blip's echo — the squash guard eating it is the point. */
|
|
5457
|
+
static BACKCHANNELS = ["Mm-hm.", "Uh-huh.", "Right.", "Okay."];
|
|
5458
|
+
/** Backchannel trigger (listening side only — the speaking branch returns before this): a partial
|
|
5459
|
+
* that reached a clause boundary and then stayed UNCHANGED for backchannelMs (a micro-pause, still
|
|
5460
|
+
* BEFORE the silence endpoint) fires a blip — if long enough (≥ backchannelMinWords), predominantly
|
|
5461
|
+
* Latin, rate-limited (backchannelMinGapMs + max 2/turn), and no endpointed text is pending.
|
|
5462
|
+
* Touches nothing else: merge/endpoint/speculation timers and turn state are never affected. */
|
|
5463
|
+
trackBackchannel(text) {
|
|
5464
|
+
if (!this.options.backchannelMs) return;
|
|
5465
|
+
const t = text.trim();
|
|
5466
|
+
if (!t || t === this.bcPartial) return;
|
|
5467
|
+
this.bcPartial = t;
|
|
5468
|
+
if (this.bcTimer) this.clock.clearTimeout(this.bcTimer);
|
|
5469
|
+
this.bcTimer = this.clock.setTimeout(() => {
|
|
5470
|
+
this.bcTimer = null;
|
|
5471
|
+
const p = this.bcPartial;
|
|
5472
|
+
const veto = (reason) => this.diag("backchannel_suppress", { reason });
|
|
5473
|
+
if (this.speaking || this.pendingUtt || this.bcActive) return veto("busy");
|
|
5474
|
+
if (!this.usingAec || !this.options.bargeIn) return veto("half_duplex");
|
|
5475
|
+
if (this.bcCount >= 2 || this.lastBcAt && this.clock.now() - this.lastBcAt < this.options.backchannelMinGapMs) return veto("rate_limited");
|
|
5476
|
+
if (this.words(p).length < this.options.backchannelMinWords) return veto("short");
|
|
5477
|
+
if (!/[,.;:!?…]$/.test(p) && !this.looksIncomplete(p)) return veto("no_boundary");
|
|
5478
|
+
const letters = p.replace(/[^\p{L}]/gu, "");
|
|
5479
|
+
const latin = p.replace(/[^A-Za-z]/g, "");
|
|
5480
|
+
if (!letters || latin.length / letters.length < 0.7) return veto("non_latin");
|
|
5481
|
+
const flat = p.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
5482
|
+
if (flat && this.lastDispatchFlat && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS && (this.lastDispatchFlat.startsWith(flat) || flat.startsWith(this.lastDispatchFlat))) return;
|
|
5483
|
+
this.fireBackchannel();
|
|
5484
|
+
}, this.options.backchannelMs);
|
|
5485
|
+
}
|
|
5486
|
+
/** Speak one blip on a THROWAWAY TTS context. Zero floor-claim: no `speaking`, no state change, no
|
|
5487
|
+
* markTurn, no merge/endpoint/barge timers, no repliedSinceDispatch, no utterance queue. Echo
|
|
5488
|
+
* pre-seeding happens BEFORE any audio exists: the blip's words join echoWords (its mic echo is
|
|
5489
|
+
* never "novel"), the ack-squash guard is armed with the exact phrase (a standalone echo final is
|
|
5490
|
+
* swallowed), and the echo window extends so echo-shaped finals stay gated. */
|
|
5491
|
+
fireBackchannel() {
|
|
5492
|
+
const fresh = _VoiceEngine.BACKCHANNELS.filter((x) => !this.recentBc.includes(x));
|
|
5493
|
+
const pool = fresh.length ? fresh : _VoiceEngine.BACKCHANNELS;
|
|
5494
|
+
const phrase = pool[Math.floor(Math.random() * pool.length)];
|
|
5495
|
+
this.recentBc.push(phrase);
|
|
5496
|
+
if (this.recentBc.length > 2) this.recentBc.shift();
|
|
5497
|
+
const now4 = this.clock.now();
|
|
5498
|
+
for (const w of this.words(phrase)) this.echoWords.add(w);
|
|
5499
|
+
this.ackAt = now4;
|
|
5500
|
+
this.lastAck = phrase;
|
|
5501
|
+
this.lastBcAt = now4;
|
|
5502
|
+
this.lastBcPhrase = phrase;
|
|
5503
|
+
this.echoUntil = Math.max(this.echoUntil, now4 + 2500);
|
|
5504
|
+
this.bcCount++;
|
|
5505
|
+
this.bcActive = true;
|
|
5506
|
+
if (this.bcActiveTimer) this.clock.clearTimeout(this.bcActiveTimer);
|
|
5507
|
+
this.bcActiveTimer = this.clock.setTimeout(() => {
|
|
5508
|
+
this.bcActive = false;
|
|
5509
|
+
}, 3e3);
|
|
5510
|
+
this.tts.onDone = () => {
|
|
5511
|
+
this.bcActive = false;
|
|
5724
5512
|
};
|
|
5513
|
+
this.tts.newContext();
|
|
5514
|
+
this.tts.speak(phrase, false);
|
|
5515
|
+
log9.debug(`backchannel: "${phrase}"`);
|
|
5516
|
+
this.diag("backchannel", { phrase });
|
|
5517
|
+
this.options.onBackchannel(phrase);
|
|
5518
|
+
}
|
|
5519
|
+
/** A real turn (or shutdown) takes over mid-blip: close the audio bypass + stability timer. */
|
|
5520
|
+
bcSupersede() {
|
|
5521
|
+
this.bcActive = false;
|
|
5522
|
+
if (this.bcActiveTimer) {
|
|
5523
|
+
this.clock.clearTimeout(this.bcActiveTimer);
|
|
5524
|
+
this.bcActiveTimer = null;
|
|
5525
|
+
}
|
|
5526
|
+
if (this.bcTimer) {
|
|
5527
|
+
this.clock.clearTimeout(this.bcTimer);
|
|
5528
|
+
this.bcTimer = null;
|
|
5529
|
+
}
|
|
5530
|
+
}
|
|
5531
|
+
/** Strip the mic echo of the LAST blip from a dispatching final, conservatively: only the exact
|
|
5532
|
+
* phrase, as an isolated token at a clause edge (start/end of utterance or beside punctuation),
|
|
5533
|
+
* within 2s of the blip. "Right"/"okay" as genuine mid-sentence content words are never touched. */
|
|
5534
|
+
stripBackchannelEcho(text) {
|
|
5535
|
+
if (!this.lastBcPhrase || this.clock.now() - this.lastBcAt > 2e3) return text;
|
|
5536
|
+
const target = normWord(this.lastBcPhrase);
|
|
5537
|
+
if (!target) return text;
|
|
5538
|
+
const toks = text.split(/\s+/);
|
|
5539
|
+
const idx = toks.findIndex((tok, i) => {
|
|
5540
|
+
if (normWord(tok) !== target) return false;
|
|
5541
|
+
if (/\?/.test(tok)) return false;
|
|
5542
|
+
const edgeBefore = i === 0 || /[,.;:!?…]$/.test(toks[i - 1]);
|
|
5543
|
+
const edgeAfter = i === toks.length - 1 || /[,.;:!?…]$/.test(tok);
|
|
5544
|
+
return edgeBefore && edgeAfter;
|
|
5545
|
+
});
|
|
5546
|
+
if (idx < 0) return text;
|
|
5547
|
+
toks.splice(idx, 1);
|
|
5548
|
+
log9.verbose(`stripped backchannel echo "${this.lastBcPhrase}" from dispatching final`);
|
|
5549
|
+
return toks.join(" ");
|
|
5725
5550
|
}
|
|
5726
|
-
|
|
5727
|
-
|
|
5728
|
-
|
|
5729
|
-
|
|
5730
|
-
|
|
5731
|
-
|
|
5732
|
-
|
|
5733
|
-
|
|
5734
|
-
|
|
5735
|
-
|
|
5736
|
-
|
|
5737
|
-
|
|
5738
|
-
|
|
5739
|
-
|
|
5551
|
+
/** Merge a resumed utterance into the pending one, deduping any word-overlap. Soniox re-finalizes
|
|
5552
|
+
* overlapping audio when the silence-timer and the semantic `<end>` both endpoint a growing
|
|
5553
|
+
* utterance (or after a reconnect): the next "utterance" repeats the tail of the previous one, and
|
|
5554
|
+
* a naive `${prev} ${next}` produced the live duplication ("Um, I want to check if Um, I want to
|
|
5555
|
+
* check if…"). Find the longest suffix of `prev`'s words that prefixes `next` and drop it. */
|
|
5556
|
+
mergeUtterance(prev, next) {
|
|
5557
|
+
if (!prev) return next;
|
|
5558
|
+
if (!next) return prev;
|
|
5559
|
+
const pw = prev.split(/\s+/), nw = next.split(/\s+/);
|
|
5560
|
+
const norm2 = _VoiceEngine.normWord;
|
|
5561
|
+
if (pw.length >= 2 && pw.length <= nw.length) {
|
|
5562
|
+
let pfx = true;
|
|
5563
|
+
for (let i = 0; i < pw.length - 1; i++) if (norm2(pw[i]) !== norm2(nw[i])) {
|
|
5564
|
+
pfx = false;
|
|
5565
|
+
break;
|
|
5740
5566
|
}
|
|
5741
|
-
|
|
5567
|
+
if (pfx && norm2(nw[pw.length - 1]).startsWith(norm2(pw[pw.length - 1]))) return next;
|
|
5568
|
+
}
|
|
5569
|
+
const max = Math.min(pw.length, nw.length);
|
|
5570
|
+
for (let k = max; k > 0; k--) {
|
|
5571
|
+
let match = true;
|
|
5572
|
+
for (let i = 0; i < k; i++) if (norm2(pw[pw.length - k + i]) !== norm2(nw[i])) {
|
|
5573
|
+
match = false;
|
|
5574
|
+
break;
|
|
5575
|
+
}
|
|
5576
|
+
if (match) return [...pw, ...nw.slice(k)].join(" ");
|
|
5577
|
+
}
|
|
5578
|
+
return `${prev} ${next}`;
|
|
5742
5579
|
}
|
|
5743
|
-
|
|
5744
|
-
return
|
|
5745
|
-
|
|
5746
|
-
|
|
5747
|
-
|
|
5748
|
-
|
|
5749
|
-
|
|
5580
|
+
static normWord(w) {
|
|
5581
|
+
return normWord(w);
|
|
5582
|
+
}
|
|
5583
|
+
/** Soniox re-finalization of an ALREADY-DISPATCHED utterance, arriving PAST the merge window:
|
|
5584
|
+
* the new final is a strict word-prefix superset of the last dispatch (the last dispatched word
|
|
5585
|
+
* may be a char-prefix of the corresponding new word — a mid-word endpoint), and it lands within
|
|
5586
|
+
* DUP_FINAL_MS of the dispatch. Live: "Hi, please tell me a very short" dispatched, then the full
|
|
5587
|
+
* "…very short joke." re-finalized ~900ms later → dispatched TWICE (two replies, the second to a
|
|
5588
|
+
* question already being answered).
|
|
5589
|
+
* DESIGN — HYBRID at the caller (flushUtterance): this shape check identifies a re-finalization
|
|
5590
|
+
* (a human physically cannot re-speak a ≥3-word sentence plus extra words within 3s of the
|
|
5591
|
+
* previous dispatch; a genuine continuation arrives as NEW words, never as a superset), and
|
|
5592
|
+
* repliedSinceDispatch then picks the action:
|
|
5593
|
+
* • reply already streaming (the live trace: TTFT ~500ms < the ~900ms re-final) → DROP. True
|
|
5594
|
+
* "supersede" would mean aborting audible speech mid-word to re-answer nearly the same text —
|
|
5595
|
+
* worse UX — and needs turn-abort plumbing in every host (the lab bridge has none;
|
|
5596
|
+
* DuplexAgent.send is a non-cancelable queue, so a second send just stacks a SECOND full reply).
|
|
5597
|
+
* Soniox's premature endpoint fires at a prosodic boundary, so the loss is trailing word(s).
|
|
5598
|
+
* • NO reply yet → DISPATCH the fuller text. Live-verified necessity: the reflex Holds on the
|
|
5599
|
+
* truncated fragment ("Hi, please tell me a very short" → Hold), and dropping the re-final then
|
|
5600
|
+
* starves the conversation entirely — the fuller final IS the completion the Hold is waiting
|
|
5601
|
+
* for. A slow-but-answering reflex in this window degrades to today's two-reply behavior (rare
|
|
5602
|
+
* race), never worse. Fillers ("mhm") deliberately don't count as replies — see speakFiller. */
|
|
5603
|
+
refinalizes(next) {
|
|
5604
|
+
const pw = this.lastDispatchWords, nw = next.trim().split(/\s+/);
|
|
5605
|
+
const norm2 = _VoiceEngine.normWord;
|
|
5606
|
+
if (pw.length < 3 || nw.length < pw.length) return false;
|
|
5607
|
+
if (this.clock.now() - this.lastDispatchAt >= _VoiceEngine.DUP_FINAL_MS) return false;
|
|
5608
|
+
for (let i = 0; i < pw.length - 1; i++) if (norm2(pw[i]) !== norm2(nw[i])) return false;
|
|
5609
|
+
const pLast = norm2(pw[pw.length - 1]), nSame = norm2(nw[pw.length - 1]);
|
|
5610
|
+
if (!nSame.startsWith(pLast)) return false;
|
|
5611
|
+
return nw.length > pw.length || nSame.length > pLast.length;
|
|
5612
|
+
}
|
|
5613
|
+
static TRAIL_RE = /(?:^|\s)(?:and|but|or|so|to|the|a|an|of|in|for|with|that|if|uh|um|like|about|from|into|on|is|are|was|were|,)$/i;
|
|
5614
|
+
/** The utterance sounds like the user paused mid-thought (trailing conjunction/filler/comma). */
|
|
5615
|
+
looksIncomplete(text) {
|
|
5616
|
+
return _VoiceEngine.TRAIL_RE.test(text.trim());
|
|
5617
|
+
}
|
|
5618
|
+
handleUtterance(text) {
|
|
5619
|
+
if (this.speaking && (this.ctxOpen || this.pausedAt) && this.overlapCapable) {
|
|
5620
|
+
this.diag("utterance_drop", { reason: "overlap_noise", text: text.slice(0, 60) });
|
|
5621
|
+
this.stt.reset();
|
|
5622
|
+
return;
|
|
5623
|
+
}
|
|
5624
|
+
if (this.echoActive() && (!this.options.bargeIn || (this.usingAec ? !this.genuine(text) : this.novelWords(text).length < 2))) {
|
|
5625
|
+
this.diag("echo_swallow", { guard: "echo_gate", text: text.slice(0, 60) });
|
|
5626
|
+
this.stt.reset();
|
|
5627
|
+
return;
|
|
5628
|
+
}
|
|
5629
|
+
const squash = (t) => t.toLowerCase().replace(/[^a-z]/g, "").replace(/(.)\1+/g, "$1");
|
|
5630
|
+
if (this.ackAt && this.lastAck && this.clock.now() - this.ackAt < 6e3 && squash(text) === squash(this.lastAck)) {
|
|
5631
|
+
this.diag("echo_swallow", { guard: "ack_squash", text: text.slice(0, 60) });
|
|
5632
|
+
this.ackAt = 0;
|
|
5633
|
+
return;
|
|
5634
|
+
}
|
|
5635
|
+
if (this.pendingUtt) this.mergePath = "merged";
|
|
5636
|
+
this.pendingUtt = this.mergeUtterance(this.pendingUtt, text);
|
|
5637
|
+
if (this.pendingTimer) this.clock.clearTimeout(this.pendingTimer);
|
|
5638
|
+
if (this.options.incompleteMergeMs && this.looksIncomplete(this.pendingUtt)) {
|
|
5639
|
+
log9.verbose(`hold: incomplete utterance "${this.pendingUtt.slice(-40)}"`);
|
|
5640
|
+
this.diag("hold", { reason: "incomplete", tail: this.pendingUtt.slice(-40) });
|
|
5641
|
+
this.options.onHold();
|
|
5642
|
+
if (this.options.holdFiller && !this.speaking) this.speakFiller(this.options.holdFiller);
|
|
5643
|
+
this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), this.options.incompleteMergeMs);
|
|
5644
|
+
return;
|
|
5645
|
+
}
|
|
5646
|
+
if (!this.options.utteranceMergeMs || this.words(this.pendingUtt).length >= 4) return this.flushUtterance();
|
|
5647
|
+
this.pendingTimer = this.clock.setTimeout(() => this.flushUtterance(), this.options.utteranceMergeMs);
|
|
5648
|
+
}
|
|
5649
|
+
flushUtterance() {
|
|
5650
|
+
if (this.pendingTimer) {
|
|
5651
|
+
this.clock.clearTimeout(this.pendingTimer);
|
|
5652
|
+
this.pendingTimer = null;
|
|
5653
|
+
}
|
|
5654
|
+
if (this.specTimer) {
|
|
5655
|
+
this.clock.clearTimeout(this.specTimer);
|
|
5656
|
+
this.specTimer = null;
|
|
5657
|
+
}
|
|
5658
|
+
this.specSpent = false;
|
|
5659
|
+
this.specPartial = "";
|
|
5660
|
+
this.specText = "";
|
|
5661
|
+
if (this.bcTimer) {
|
|
5662
|
+
this.clock.clearTimeout(this.bcTimer);
|
|
5663
|
+
this.bcTimer = null;
|
|
5664
|
+
}
|
|
5665
|
+
this.bcCount = 0;
|
|
5666
|
+
this.bcPartial = "";
|
|
5667
|
+
const text = this.stripBackchannelEcho(this.pendingUtt);
|
|
5668
|
+
this.pendingUtt = "";
|
|
5669
|
+
const path = this.mergePath;
|
|
5670
|
+
this.mergePath = "direct";
|
|
5671
|
+
if (text) {
|
|
5672
|
+
const flat = text.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
5673
|
+
if (flat && flat === this.lastDispatchFlat && !this.repliedSinceDispatch && this.clock.now() - this.lastDispatchAt < _VoiceEngine.DUP_FINAL_MS) {
|
|
5674
|
+
log9.verbose(`dropped duplicate final "${text.slice(0, 40)}" (identical to the just-dispatched utterance, no reply between)`);
|
|
5675
|
+
this.diag("utterance_drop", { reason: "dup_final", text: text.slice(0, 60) });
|
|
5676
|
+
return;
|
|
5677
|
+
}
|
|
5678
|
+
const refinal = this.refinalizes(text);
|
|
5679
|
+
if (refinal && this.repliedSinceDispatch) {
|
|
5680
|
+
log9.verbose(`dropped re-finalization "${text.slice(0, 60)}" (superset of the just-dispatched utterance, reply already flowing)`);
|
|
5681
|
+
this.diag("utterance_drop", { reason: "refinalized", text: text.slice(0, 60) });
|
|
5682
|
+
this.lastDispatchFlat = flat;
|
|
5683
|
+
this.lastDispatchWords = text.trim().split(/\s+/);
|
|
5684
|
+
return;
|
|
5685
|
+
}
|
|
5686
|
+
this.lastDispatchFlat = flat;
|
|
5687
|
+
this.lastDispatchWords = text.trim().split(/\s+/);
|
|
5688
|
+
this.lastDispatchAt = this.clock.now();
|
|
5689
|
+
this.repliedSinceDispatch = false;
|
|
5690
|
+
this.lastDispatchWasQuestion = /\?\s*$/.test(text);
|
|
5691
|
+
this.turnStartAt = this.clock.now();
|
|
5692
|
+
this.bargeGraceUntil = this.clock.now() + this.options.bargeGraceMs;
|
|
5693
|
+
this.diag("dispatch", { text, path: refinal ? "refinalized" : path, question: this.lastDispatchWasQuestion });
|
|
5694
|
+
this.options.onUtterance(text);
|
|
5695
|
+
}
|
|
5696
|
+
}
|
|
5697
|
+
get overlapCapable() {
|
|
5698
|
+
return this.usingAec && this.options.overlapPause && !!this.player.pause && !!this.player.resume;
|
|
5699
|
+
}
|
|
5700
|
+
armResume() {
|
|
5701
|
+
if (this.resumeTimer) this.clock.clearTimeout(this.resumeTimer);
|
|
5702
|
+
this.resumeTimer = this.clock.setTimeout(() => {
|
|
5703
|
+
this.resumeTimer = null;
|
|
5704
|
+
if (!this.pausedAt) return;
|
|
5705
|
+
this.diag("overlap_resume", { heldMs: Math.round(this.clock.now() - this.pausedAt) });
|
|
5706
|
+
this.stt.reset();
|
|
5707
|
+
this.resetOverlap(true);
|
|
5708
|
+
}, this.options.overlapResumeMs);
|
|
5709
|
+
}
|
|
5710
|
+
resetOverlap(resume) {
|
|
5711
|
+
if (this.resumeTimer) {
|
|
5712
|
+
this.clock.clearTimeout(this.resumeTimer);
|
|
5713
|
+
this.resumeTimer = null;
|
|
5714
|
+
}
|
|
5715
|
+
if (this.pausedAt && resume) {
|
|
5716
|
+
this.player.resume?.();
|
|
5717
|
+
this.lastResumeAt = this.clock.now();
|
|
5718
|
+
}
|
|
5719
|
+
this.pausedAt = 0;
|
|
5720
|
+
this.lastOverlapPartial = "";
|
|
5721
|
+
this.gatePassTimes = [];
|
|
5722
|
+
}
|
|
5723
|
+
/** energy two-stage barge-in (heuristic tier only): spike over echo baseline → pause + confirm via STT */
|
|
5724
|
+
gatePassTimes = [];
|
|
5725
|
+
// recent gate-PASSING chunks (helper zeroes residue — nonzero = vetted)
|
|
5726
|
+
handleLevel(rms) {
|
|
5727
|
+
if (this.usingAec) {
|
|
5728
|
+
if (!this.options.overlapEnergyHold || !this.speaking || !this.overlapCapable || this.pausedAt || rms < 50) return;
|
|
5729
|
+
const t = this.clock.now();
|
|
5730
|
+
this.gatePassTimes = this.gatePassTimes.filter((x) => t - x < 350);
|
|
5731
|
+
this.gatePassTimes.push(t);
|
|
5732
|
+
if (this.gatePassTimes.length < 2) return;
|
|
5733
|
+
this.gatePassTimes = [];
|
|
5734
|
+
this.pausedAt = t;
|
|
5735
|
+
this.player.pause();
|
|
5736
|
+
this.armResume();
|
|
5737
|
+
return;
|
|
5738
|
+
}
|
|
5739
|
+
if (!this.speaking) {
|
|
5740
|
+
this.baseline = 0;
|
|
5741
|
+
this.hot = 0;
|
|
5742
|
+
return;
|
|
5743
|
+
}
|
|
5744
|
+
if (!this.baseline) {
|
|
5745
|
+
this.baseline = rms;
|
|
5746
|
+
return;
|
|
5747
|
+
}
|
|
5748
|
+
this.baseline = this.baseline * 0.9 + rms * 0.1;
|
|
5749
|
+
if (rms > Math.max(this.baseline * this.options.bargeRmsMult, this.options.bargeRmsFloor)) this.hot++;
|
|
5750
|
+
else this.hot = 0;
|
|
5751
|
+
if (this.hot >= 2 && !this.suspectUntil) {
|
|
5752
|
+
this.suspectUntil = this.clock.now() + 1300;
|
|
5753
|
+
this.clock.setTimeout(() => {
|
|
5754
|
+
this.suspectUntil = 0;
|
|
5755
|
+
}, 1350);
|
|
5756
|
+
}
|
|
5750
5757
|
}
|
|
5751
5758
|
};
|
|
5752
5759
|
|
|
5753
|
-
// src/
|
|
5754
|
-
|
|
5755
|
-
|
|
5756
|
-
|
|
5757
|
-
|
|
5758
|
-
|
|
5759
|
-
|
|
5760
|
-
|
|
5761
|
-
|
|
5762
|
-
if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
|
|
5763
|
-
images.push({ mimeType: c.mimeType, data: c.data });
|
|
5764
|
-
} else if (typeof c?.text === "string") {
|
|
5765
|
-
texts.push(c.text);
|
|
5766
|
-
} else {
|
|
5767
|
-
texts.push(JSON.stringify(c));
|
|
5768
|
-
}
|
|
5769
|
-
}
|
|
5770
|
-
const text = texts.join("\n");
|
|
5771
|
-
if (text || images.length) return { text, ...images.length ? { images } : {} };
|
|
5760
|
+
// src/voice/spokenSplitter.ts
|
|
5761
|
+
var OPEN = "<spoken>";
|
|
5762
|
+
var CLOSE = "</spoken>";
|
|
5763
|
+
var CLOSERS = `"')]}\xBB\u201D\u2019`;
|
|
5764
|
+
var hasSpeech = (s) => /[\p{L}\p{N}]/u.test(s);
|
|
5765
|
+
var SentenceCoalescer = class _SentenceCoalescer {
|
|
5766
|
+
buf = "";
|
|
5767
|
+
static isEnd(c) {
|
|
5768
|
+
return c === "\n" || c === "." || c === "!" || c === "?" || c === "\u2026";
|
|
5772
5769
|
}
|
|
5773
|
-
|
|
5774
|
-
|
|
5775
|
-
|
|
5776
|
-
|
|
5777
|
-
|
|
5778
|
-
|
|
5779
|
-
|
|
5780
|
-
|
|
5781
|
-
|
|
5782
|
-
|
|
5783
|
-
|
|
5784
|
-
|
|
5785
|
-
|
|
5786
|
-
|
|
5787
|
-
|
|
5788
|
-
}
|
|
5789
|
-
|
|
5790
|
-
|
|
5791
|
-
|
|
5792
|
-
|
|
5793
|
-
|
|
5794
|
-
|
|
5795
|
-
|
|
5796
|
-
|
|
5797
|
-
|
|
5798
|
-
|
|
5799
|
-
|
|
5800
|
-
|
|
5801
|
-
|
|
5802
|
-
|
|
5803
|
-
|
|
5804
|
-
|
|
5805
|
-
|
|
5806
|
-
|
|
5807
|
-
|
|
5808
|
-
|
|
5809
|
-
|
|
5810
|
-
|
|
5811
|
-
|
|
5812
|
-
|
|
5813
|
-
|
|
5814
|
-
|
|
5815
|
-
|
|
5816
|
-
properties: {
|
|
5817
|
-
name: { type: "string", description: "exact tool name from ToolSearch" },
|
|
5818
|
-
args: { type: "object", description: "arguments object for the tool (per its schema)" }
|
|
5770
|
+
feed(delta) {
|
|
5771
|
+
if (delta) this.buf += delta;
|
|
5772
|
+
let cut = -1;
|
|
5773
|
+
for (let i = 0; i < this.buf.length; i++) if (_SentenceCoalescer.isEnd(this.buf[i])) cut = i;
|
|
5774
|
+
if (cut < 0) return "";
|
|
5775
|
+
if (this.buf[cut] !== "\n") while (cut + 1 < this.buf.length && CLOSERS.includes(this.buf[cut + 1])) cut++;
|
|
5776
|
+
const ready = this.buf.slice(0, cut + 1).trim();
|
|
5777
|
+
this.buf = this.buf.slice(cut + 1);
|
|
5778
|
+
return hasSpeech(ready) ? ready : "";
|
|
5779
|
+
}
|
|
5780
|
+
flush() {
|
|
5781
|
+
const s = this.buf.trim();
|
|
5782
|
+
this.buf = "";
|
|
5783
|
+
return hasSpeech(s) ? s : "";
|
|
5784
|
+
}
|
|
5785
|
+
};
|
|
5786
|
+
var SpokenSplitter = class {
|
|
5787
|
+
buf = "";
|
|
5788
|
+
inSpoken = false;
|
|
5789
|
+
/** True once any spoken char has ever been emitted (drives the no-spoken fallback). */
|
|
5790
|
+
spokeAny = false;
|
|
5791
|
+
/** Feed a delta; returns the spoken/detail spans completed by this chunk (either may be ''). */
|
|
5792
|
+
feed(delta) {
|
|
5793
|
+
this.buf += delta;
|
|
5794
|
+
return this.drain(false);
|
|
5795
|
+
}
|
|
5796
|
+
/** Drain any buffered partial. A trailing `<…` that never completed a tag is emitted as detail. */
|
|
5797
|
+
flush() {
|
|
5798
|
+
return this.drain(true);
|
|
5799
|
+
}
|
|
5800
|
+
drain(final) {
|
|
5801
|
+
let spoken = "";
|
|
5802
|
+
let detail = "";
|
|
5803
|
+
while (this.buf.length) {
|
|
5804
|
+
const tag = this.inSpoken ? CLOSE : OPEN;
|
|
5805
|
+
const idx = this.buf.indexOf(tag);
|
|
5806
|
+
if (idx >= 0) {
|
|
5807
|
+
const text2 = this.buf.slice(0, idx);
|
|
5808
|
+
if (this.inSpoken) spoken += text2;
|
|
5809
|
+
else detail += text2;
|
|
5810
|
+
this.buf = this.buf.slice(idx + tag.length);
|
|
5811
|
+
this.inSpoken = !this.inSpoken;
|
|
5812
|
+
continue;
|
|
5819
5813
|
}
|
|
5820
|
-
|
|
5821
|
-
|
|
5822
|
-
const
|
|
5823
|
-
if (
|
|
5824
|
-
|
|
5825
|
-
|
|
5814
|
+
const lt = this.buf.lastIndexOf("<");
|
|
5815
|
+
const holdStart = lt >= 0 && tag.startsWith(this.buf.slice(lt)) ? lt : this.buf.length;
|
|
5816
|
+
const text = this.buf.slice(0, holdStart);
|
|
5817
|
+
if (this.inSpoken) spoken += text;
|
|
5818
|
+
else detail += text;
|
|
5819
|
+
this.buf = this.buf.slice(holdStart);
|
|
5820
|
+
break;
|
|
5826
5821
|
}
|
|
5827
|
-
|
|
5828
|
-
|
|
5829
|
-
|
|
5830
|
-
|
|
5831
|
-
const specs = [];
|
|
5832
|
-
const routes = /* @__PURE__ */ new Map();
|
|
5833
|
-
for (const m of servers) {
|
|
5834
|
-
for (const s of m.specs) {
|
|
5835
|
-
const base = `mcp__${m.name}__${s.name}`.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
|
|
5836
|
-
let display = base;
|
|
5837
|
-
for (let i = 2; routes.has(display); i++) display = `${base.slice(0, 128 - String(i).length - 1)}_${i}`;
|
|
5838
|
-
specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
|
|
5839
|
-
routes.set(display, { server: m.name, rawName: s.name });
|
|
5822
|
+
if (final && this.buf) {
|
|
5823
|
+
if (this.inSpoken) spoken += this.buf;
|
|
5824
|
+
else detail += this.buf;
|
|
5825
|
+
this.buf = "";
|
|
5840
5826
|
}
|
|
5827
|
+
if (spoken.trim()) this.spokeAny = true;
|
|
5828
|
+
return { spoken, detail };
|
|
5841
5829
|
}
|
|
5842
|
-
|
|
5843
|
-
}
|
|
5844
|
-
function searchOverCatalog(servers, specs, routes, resolve4, options) {
|
|
5845
|
-
const tools = specs.length ? makeMcpToolSearch(specs, (name, args) => {
|
|
5846
|
-
const r = routes.get(name);
|
|
5847
|
-
if (!r) throw new Error(`unknown MCP tool '${name}' \u2014 use ToolSearch to find valid names`);
|
|
5848
|
-
return resolve4(r.server, r.rawName, args ?? {});
|
|
5849
|
-
}, options) : [];
|
|
5850
|
-
return { tools, serverNames: servers, toolCount: specs.length };
|
|
5851
|
-
}
|
|
5852
|
-
function makeMcpToolSearchFromMounted(mounted, options) {
|
|
5853
|
-
const { specs, routes } = buildMcpCatalog(mounted);
|
|
5854
|
-
const byName = new Map(mounted.map((m) => [m.name, m]));
|
|
5855
|
-
return searchOverCatalog(mounted.map((m) => m.name), specs, routes, (server, rawName, args) => byName.get(server).client.callTool(rawName, args), options);
|
|
5856
|
-
}
|
|
5857
|
-
|
|
5858
|
-
// src/index.ts
|
|
5859
|
-
init_logging();
|
|
5830
|
+
};
|
|
5860
5831
|
|
|
5861
|
-
// src/
|
|
5862
|
-
|
|
5863
|
-
|
|
5864
|
-
|
|
5865
|
-
|
|
5866
|
-
|
|
5867
|
-
|
|
5868
|
-
|
|
5869
|
-
|
|
5870
|
-
|
|
5871
|
-
|
|
5872
|
-
|
|
5873
|
-
|
|
5874
|
-
|
|
5875
|
-
|
|
5876
|
-
|
|
5877
|
-
|
|
5878
|
-
/**
|
|
5879
|
-
|
|
5880
|
-
|
|
5881
|
-
|
|
5882
|
-
|
|
5883
|
-
|
|
5884
|
-
|
|
5885
|
-
/**
|
|
5886
|
-
|
|
5887
|
-
|
|
5888
|
-
|
|
5889
|
-
/**
|
|
5890
|
-
*
|
|
5891
|
-
|
|
5892
|
-
|
|
5893
|
-
|
|
5894
|
-
|
|
5895
|
-
|
|
5896
|
-
|
|
5897
|
-
|
|
5898
|
-
|
|
5899
|
-
*
|
|
5900
|
-
|
|
5901
|
-
|
|
5902
|
-
|
|
5903
|
-
|
|
5904
|
-
/**
|
|
5905
|
-
|
|
5906
|
-
|
|
5907
|
-
/**
|
|
5908
|
-
|
|
5909
|
-
|
|
5910
|
-
/**
|
|
5911
|
-
|
|
5912
|
-
|
|
5913
|
-
*
|
|
5914
|
-
*
|
|
5915
|
-
|
|
5916
|
-
|
|
5917
|
-
|
|
5918
|
-
|
|
5919
|
-
|
|
5920
|
-
|
|
5921
|
-
|
|
5922
|
-
*
|
|
5923
|
-
|
|
5924
|
-
|
|
5925
|
-
|
|
5926
|
-
|
|
5927
|
-
|
|
5928
|
-
|
|
5929
|
-
* mechanism-correct pause trigger; enable only if barge-in onset feels sluggish in a clean-AEC room. */
|
|
5930
|
-
overlapEnergyHold = false;
|
|
5931
|
-
/** Map inline `[emotion]` tags (emitted by the model, prompt-taught) into Cartesia inline emotion
|
|
5932
|
-
* tags in the spoken transcript (sonic-3 stitches the prosody). false = strip them silently. */
|
|
5933
|
-
emotions = true;
|
|
5934
|
-
/** Show the `[emotion]` tags in the on-screen echo (debug). false = hide (spoken-only). */
|
|
5935
|
-
showEmotions = false;
|
|
5832
|
+
// src/duplex.ts
|
|
5833
|
+
var log10 = forComponent("DuplexAgent");
|
|
5834
|
+
function describeCall(call) {
|
|
5835
|
+
const v = call.args && Object.values(call.args).find((x) => typeof x === "string" && x.trim());
|
|
5836
|
+
const hint = v ? ` (${String(v).replace(/\s+/g, " ").trim().slice(0, 48)})` : "";
|
|
5837
|
+
return `${call.name}${hint}`;
|
|
5838
|
+
}
|
|
5839
|
+
var DuplexAgentOptions = class {
|
|
5840
|
+
/** Any ai.libx.js AIClient — shared by all tiers (routed by model). */
|
|
5841
|
+
ai;
|
|
5842
|
+
/** The WORKER's filesystem (act + think). If omitted the worker keeps Agent's jailed-disk-at-cwd default. */
|
|
5843
|
+
fs;
|
|
5844
|
+
// The reflex IS the voice. 120b (not 20b) for channel discipline + instruction-following: the 20b
|
|
5845
|
+
// mislabels gpt-oss harmony channels under load, leaking raw analysis into the spoken `final` channel
|
|
5846
|
+
// (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
|
|
5847
|
+
reflexModel = "groq/openai/gpt-oss-120b";
|
|
5848
|
+
actModel = "anthropic/claude-sonnet-4-6";
|
|
5849
|
+
/** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
|
|
5850
|
+
thinkModel = "anthropic/claude-opus-4-8";
|
|
5851
|
+
/** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
|
|
5852
|
+
* provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
|
|
5853
|
+
* cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
|
|
5854
|
+
* leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
|
|
5855
|
+
providerOptionsFor;
|
|
5856
|
+
/** Escape hatches merged over the derived per-agent options. */
|
|
5857
|
+
reflexOptions;
|
|
5858
|
+
actOptions;
|
|
5859
|
+
thinkOptions;
|
|
5860
|
+
/** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
|
|
5861
|
+
* the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
|
|
5862
|
+
* pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
|
|
5863
|
+
* this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
|
|
5864
|
+
verifyActTasks = false;
|
|
5865
|
+
/** Receives the voice text_delta stream + task lifecycle events. */
|
|
5866
|
+
host;
|
|
5867
|
+
/** How many recent transcript messages are rendered into a worker's brief. */
|
|
5868
|
+
excerptTurns = 6;
|
|
5869
|
+
/** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
|
|
5870
|
+
* backchannels, impulsive first reactions before content (mimics real duplex conversation). */
|
|
5871
|
+
voiceStyle = "neutral";
|
|
5872
|
+
/** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
|
|
5873
|
+
* TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
|
|
5874
|
+
emotionTags = false;
|
|
5875
|
+
/** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
|
|
5876
|
+
* (post-spawn would race the worker's first edits). */
|
|
5877
|
+
onTaskStart;
|
|
5878
|
+
/** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
|
|
5879
|
+
* air. Off by default — each update costs a voice turn (LLM call + speech). */
|
|
5880
|
+
progressUpdates = false;
|
|
5881
|
+
/** Min ms between progress re-voices per task. */
|
|
5882
|
+
progressIntervalMs = 25e3;
|
|
5883
|
+
/** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
|
|
5884
|
+
* the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
|
|
5885
|
+
* voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
|
|
5886
|
+
askRelay = false;
|
|
5887
|
+
/** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
|
|
5888
|
+
askTimeoutMs = 12e4;
|
|
5889
|
+
/** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
|
|
5890
|
+
* bounding memory over a long-lived session. Running tasks are never evicted. */
|
|
5891
|
+
maxTaskRecords = 50;
|
|
5892
|
+
/** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
|
|
5893
|
+
* (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
|
|
5894
|
+
quickLook;
|
|
5895
|
+
/** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
|
|
5896
|
+
* tools directly (no delegation needed) and implicit capture guidance. */
|
|
5897
|
+
memoryDir;
|
|
5898
|
+
/** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
|
|
5899
|
+
memoryUserDir;
|
|
5936
5900
|
};
|
|
5937
|
-
var
|
|
5901
|
+
var RESERVED_EVENT_MARKER = /\[task\b[^\]\n]*\b(?:completed|failed|progress|asks)\b/i;
|
|
5902
|
+
var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
|
|
5903
|
+
var STAGE_DIRECTION_RE = /^\(\s*(?:(?:waiting|checking|searching|thinking|processing|loading|working|fetching|looking)\b[^)]*|[^)]*(?:\.\.\.|…)\s*)\)$/i;
|
|
5904
|
+
var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji. Never emit stage directions or parenthetical asides about your own process \u2014 nothing like "(waiting for the result...)" or "(checking)"; while work runs, either say it as plain speech or end your turn.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. DELEGATION RULE \u2014 decide for yourself, the user never has to push: if you cannot answer confidently from the conversation plus trivial well-known knowledge, do NOT refuse and do NOT guess \u2014 dispatch `Act` immediately with a clear brief and say you are checking. Anything needing CURRENT data (weather, news, prices, dates, sky/astronomy, "right now"), real computation, or verification is an automatic dispatch \u2014 never a refusal. The user should never need to say "search the web" or "think harder" to make you act; needing fresh or verified information IS the trigger. Refuse only what your worker genuinely cannot do (check `capabilities`), and say why. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 give at most a half-sentence aside ("still on it \u2014 running tests now") and end your turn. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, just say "still on it" and end the turn \u2014 never poll it again and again in a loop. Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
|
|
5905
|
+
var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
|
|
5906
|
+
var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
|
|
5907
|
+
var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
|
|
5908
|
+
var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
|
|
5909
|
+
var DuplexAgent = class _DuplexAgent {
|
|
5938
5910
|
options;
|
|
5939
|
-
|
|
5940
|
-
|
|
5941
|
-
|
|
5942
|
-
|
|
5943
|
-
|
|
5944
|
-
|
|
5945
|
-
|
|
5946
|
-
|
|
5947
|
-
|
|
5948
|
-
|
|
5949
|
-
|
|
5950
|
-
|
|
5951
|
-
|
|
5952
|
-
|
|
5953
|
-
|
|
5954
|
-
|
|
5955
|
-
|
|
5956
|
-
|
|
5957
|
-
|
|
5958
|
-
|
|
5959
|
-
|
|
5960
|
-
|
|
5961
|
-
//
|
|
5962
|
-
|
|
5963
|
-
//
|
|
5964
|
-
|
|
5965
|
-
|
|
5966
|
-
|
|
5967
|
-
|
|
5968
|
-
|
|
5969
|
-
//
|
|
5970
|
-
|
|
5971
|
-
|
|
5972
|
-
|
|
5973
|
-
|
|
5974
|
-
|
|
5975
|
-
|
|
5976
|
-
|
|
5977
|
-
|
|
5978
|
-
|
|
5979
|
-
|
|
5980
|
-
|
|
5981
|
-
|
|
5982
|
-
|
|
5983
|
-
|
|
5911
|
+
voice;
|
|
5912
|
+
tasks = /* @__PURE__ */ new Map();
|
|
5913
|
+
queue = Promise.resolve();
|
|
5914
|
+
seq = 0;
|
|
5915
|
+
pendingEvents = [];
|
|
5916
|
+
/** Out-of-band follow-up attribution for the events coalescing into the next flush turn: TRUE iff ≥1 of
|
|
5917
|
+
* the tasks being integrated was NON-CLEAN (early-stop/failure). Carried out-of-band on the enqueue call
|
|
5918
|
+
* by the caller that KNOWS the outcome — a plain boolean the MODEL CANNOT PERTURB. It is NOT scanned from
|
|
5919
|
+
* worker-authored event text (v1: an "Outcome:" substring over-stamped siblings) and NOT keyed on a brief
|
|
5920
|
+
* string the reflex re-authors (v2: a paraphrased escalation brief missed the Set → followUp:false →
|
|
5921
|
+
* RE-ENABLED unbounded auto-escalation, the dangerous runaway direction). See [[wrong-discriminator]] /
|
|
5922
|
+
* [[drive-real-reflex]] / [[fakeaiclient-blind-to-wire-format]]. */
|
|
5923
|
+
pendingNonClean = false;
|
|
5924
|
+
flushQueued = false;
|
|
5925
|
+
/** Per-voice-turn guards (reset by resetTurn at each turn's start). The reflex is a weak model:
|
|
5926
|
+
* left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
|
|
5927
|
+
* Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
|
|
5928
|
+
turnDispatched = false;
|
|
5929
|
+
// an Act/Think fired this turn
|
|
5930
|
+
spokeBeforeDispatch = false;
|
|
5931
|
+
// the reflex ALREADY acked before dispatching — the forced post-dispatch text step must stay silent (live: "Got it…" twice)
|
|
5932
|
+
turnBriefs = /* @__PURE__ */ new Set();
|
|
5933
|
+
// briefs dispatched this turn (detect identical re-dispatch)
|
|
5934
|
+
spokeThisTurn = false;
|
|
5935
|
+
// any non-empty text_delta streamed this turn
|
|
5936
|
+
externalSpeech = false;
|
|
5937
|
+
// host spoke on our behalf (adaptive micro-ack) — not reflex output
|
|
5938
|
+
heldThisTurn = false;
|
|
5939
|
+
// Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
|
|
5940
|
+
nudging = false;
|
|
5941
|
+
// re-ack pass in flight: block ALL tools, prevent recursion
|
|
5942
|
+
reflexBuf = "";
|
|
5943
|
+
// accumulated reflex text this turn (fabricated-event detection)
|
|
5944
|
+
reflexForwarded = 0;
|
|
5945
|
+
// chars of reflexBuf already forwarded to the host/TTS
|
|
5946
|
+
fabricationCut = false;
|
|
5947
|
+
// reflex emitted a reserved [task …] marker → suppress its tail
|
|
5948
|
+
/** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
|
|
5949
|
+
* carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
|
|
5950
|
+
* turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
|
|
5951
|
+
* escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
|
|
5952
|
+
* ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
|
|
5953
|
+
* means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
|
|
5954
|
+
* ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
|
|
5955
|
+
* flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
|
|
5956
|
+
* fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
|
|
5957
|
+
* ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
|
|
5958
|
+
turnFollowUp = false;
|
|
5959
|
+
/** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
|
|
5960
|
+
* the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
|
|
5961
|
+
autoEscalations = 0;
|
|
5962
|
+
static MAX_AUTO_ESCALATIONS = 8;
|
|
5963
|
+
/** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
|
|
5964
|
+
pendingAsks = /* @__PURE__ */ new Map();
|
|
5965
|
+
/** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
|
|
5966
|
+
* PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
|
|
5967
|
+
* confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
|
|
5968
|
+
* aborts it: output dropped, history rolled back, the final dispatches normally. */
|
|
5969
|
+
spec;
|
|
5970
|
+
/** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
|
|
5971
|
+
speculativeAbortedCalls = 0;
|
|
5972
|
+
/** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
|
|
5973
|
+
* intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
|
|
5974
|
+
* writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
|
|
5975
|
+
* instead (the endpointed final then dispatches normally and may use the tool for real). */
|
|
5976
|
+
static SPEC_SAFE_TOOLS = /* @__PURE__ */ new Set(["QuickLook", "TaskStatus", "Hold"]);
|
|
5977
|
+
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
5978
|
+
memoryReady;
|
|
5984
5979
|
constructor(options) {
|
|
5985
|
-
this.options = { ...new
|
|
5980
|
+
this.options = { ...new DuplexAgentOptions(), ...options };
|
|
5986
5981
|
const o = this.options;
|
|
5987
|
-
if (
|
|
5988
|
-
|
|
5989
|
-
|
|
5990
|
-
|
|
5991
|
-
|
|
5992
|
-
|
|
5993
|
-
|
|
5994
|
-
|
|
5982
|
+
if (o.memoryDir && o.fs) {
|
|
5983
|
+
this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
|
|
5984
|
+
}
|
|
5985
|
+
const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
|
|
5986
|
+
const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
|
|
5987
|
+
const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
|
|
5988
|
+
const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
|
|
5989
|
+
const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
|
|
5990
|
+
const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
|
|
5991
|
+
const mcpNames = [
|
|
5992
|
+
...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
|
|
5993
|
+
...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
|
|
5994
|
+
];
|
|
5995
|
+
const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
|
|
5996
|
+
const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
|
|
5997
|
+
Today's date: ${(/* @__PURE__ */ new Date()).toDateString()}.`;
|
|
5998
|
+
const tools = [
|
|
5999
|
+
...o.reflexOptions?.tools ?? [],
|
|
6000
|
+
this.actTool(),
|
|
6001
|
+
...o.thinkModel !== false ? [this.thinkTool()] : [],
|
|
6002
|
+
this.taskStatusTool(),
|
|
6003
|
+
this.cancelTaskTool(),
|
|
6004
|
+
this.quickLookTool(),
|
|
6005
|
+
this.answerTaskTool(),
|
|
6006
|
+
this.holdTool()
|
|
6007
|
+
];
|
|
6008
|
+
const host = o.host;
|
|
6009
|
+
const voiceHost = host && {
|
|
6010
|
+
ask: host.ask ? (q2) => host.ask(q2) : void 0,
|
|
6011
|
+
confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
|
|
6012
|
+
notify: (ev) => {
|
|
6013
|
+
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
6014
|
+
if (this.heldThisTurn) return;
|
|
6015
|
+
if (this.fabricationCut) return;
|
|
6016
|
+
if (this.turnDispatched && this.spokeBeforeDispatch) return;
|
|
6017
|
+
const msg = ev.message;
|
|
6018
|
+
this.reflexBuf += msg;
|
|
6019
|
+
this.scrubStageDirections();
|
|
6020
|
+
const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
|
|
6021
|
+
if (m) {
|
|
6022
|
+
this.fabricationCut = true;
|
|
6023
|
+
log10.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
|
|
6024
|
+
const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
|
|
6025
|
+
if (!safe) return;
|
|
6026
|
+
if (safe.trim()) this.spokeThisTurn = true;
|
|
6027
|
+
this.emitHost({ ...ev, message: safe });
|
|
6028
|
+
return;
|
|
6029
|
+
}
|
|
6030
|
+
const held = this.reflexBuf.length - this.reflexForwarded;
|
|
6031
|
+
const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
|
|
6032
|
+
let upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
|
|
6033
|
+
const paren = this.reflexBuf.lastIndexOf("(");
|
|
6034
|
+
if (paren >= this.reflexForwarded && !this.reflexBuf.includes(")", paren) && this.reflexBuf.length - paren <= 80)
|
|
6035
|
+
upto = Math.min(upto, paren);
|
|
6036
|
+
const out = this.reflexBuf.slice(this.reflexForwarded, upto);
|
|
6037
|
+
this.reflexForwarded = upto;
|
|
6038
|
+
if (!out) return;
|
|
6039
|
+
if (out.trim()) this.spokeThisTurn = true;
|
|
6040
|
+
this.emitHost({ ...ev, message: out });
|
|
6041
|
+
return;
|
|
6042
|
+
}
|
|
6043
|
+
host.notify?.(ev);
|
|
6044
|
+
}
|
|
5995
6045
|
};
|
|
5996
|
-
this.
|
|
5997
|
-
|
|
5998
|
-
|
|
5999
|
-
|
|
6000
|
-
|
|
6001
|
-
|
|
6046
|
+
this.voice = new Agent({
|
|
6047
|
+
ai: o.ai,
|
|
6048
|
+
fs: new MemFilesystem2(),
|
|
6049
|
+
model: o.reflexModel,
|
|
6050
|
+
stream: true,
|
|
6051
|
+
host: voiceHost,
|
|
6052
|
+
// The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
|
|
6053
|
+
// never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
|
|
6054
|
+
// it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
|
|
6055
|
+
askUserQuestion: false,
|
|
6056
|
+
systemPrompt: prompt,
|
|
6057
|
+
instructionFiles: false,
|
|
6058
|
+
maxSteps: 8,
|
|
6059
|
+
timeoutMs: 3e4,
|
|
6060
|
+
...o.reflexOptions,
|
|
6061
|
+
tools,
|
|
6062
|
+
// Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
|
|
6063
|
+
hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
|
|
6064
|
+
});
|
|
6065
|
+
}
|
|
6066
|
+
/** Resolve memory tools + inject index into voice system prompt (once). */
|
|
6067
|
+
async initMemory() {
|
|
6068
|
+
if (!this.memoryReady) return;
|
|
6069
|
+
const mem = await this.memoryReady;
|
|
6070
|
+
this.memoryReady = void 0;
|
|
6071
|
+
this.voice.options.tools.push(...mem.tools);
|
|
6072
|
+
if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
|
|
6073
|
+
}
|
|
6074
|
+
/** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
|
|
6075
|
+
* turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
|
|
6076
|
+
flushHeldReflexTail() {
|
|
6077
|
+
if (this.fabricationCut) return;
|
|
6078
|
+
this.scrubStageDirections();
|
|
6079
|
+
const tail = this.reflexBuf.slice(this.reflexForwarded);
|
|
6080
|
+
this.reflexForwarded = this.reflexBuf.length;
|
|
6081
|
+
if (!tail) return;
|
|
6082
|
+
if (tail.trim()) this.spokeThisTurn = true;
|
|
6083
|
+
this.emitHost({ kind: "text_delta", message: tail });
|
|
6002
6084
|
}
|
|
6003
|
-
|
|
6004
|
-
|
|
6085
|
+
/** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
|
|
6086
|
+
* Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
|
|
6087
|
+
* reflexForwarded would corrupt the forward offset. */
|
|
6088
|
+
scrubStageDirections() {
|
|
6089
|
+
if (this.reflexForwarded >= this.reflexBuf.length) return;
|
|
6090
|
+
const region = this.reflexBuf.slice(this.reflexForwarded);
|
|
6091
|
+
const scrubbed = region.replace(/\([^()]*\)/g, (s) => STAGE_DIRECTION_RE.test(s) ? "" : s);
|
|
6092
|
+
if (scrubbed !== region) this.reflexBuf = this.reflexBuf.slice(0, this.reflexForwarded) + scrubbed;
|
|
6005
6093
|
}
|
|
6006
|
-
/**
|
|
6007
|
-
|
|
6008
|
-
this.
|
|
6094
|
+
/** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
|
|
6095
|
+
resetTurn() {
|
|
6096
|
+
this.turnDispatched = false;
|
|
6097
|
+
this.spokeBeforeDispatch = false;
|
|
6098
|
+
this.turnBriefs.clear();
|
|
6099
|
+
this.spokeThisTurn = false;
|
|
6100
|
+
this.externalSpeech = false;
|
|
6101
|
+
this.heldThisTurn = false;
|
|
6102
|
+
this.reflexBuf = "";
|
|
6103
|
+
this.reflexForwarded = 0;
|
|
6104
|
+
this.fabricationCut = false;
|
|
6105
|
+
this.turnFollowUp = false;
|
|
6106
|
+
this.voice.options.toolChoice = void 0;
|
|
6009
6107
|
}
|
|
6010
|
-
/**
|
|
6011
|
-
|
|
6012
|
-
|
|
6108
|
+
/** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
|
|
6109
|
+
* now wait for the push" (CC's Task model). Block the temptations — TaskStatus polling and identical
|
|
6110
|
+
* re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
|
|
6111
|
+
* still allowed (parallel independent work). During a re-ack pass, block every tool. */
|
|
6112
|
+
dispatchGuard() {
|
|
6113
|
+
return {
|
|
6114
|
+
preToolUse: (call) => {
|
|
6115
|
+
if (this.spec?.state === "pending" && !_DuplexAgent.SPEC_SAFE_TOOLS.has(call.name)) {
|
|
6116
|
+
log10.verbose(`speculation aborted: reflex called ${call.name} on unconfirmed input`);
|
|
6117
|
+
this.abortSpeculation();
|
|
6118
|
+
return { block: true, reason: "Speculative turn aborted." };
|
|
6119
|
+
}
|
|
6120
|
+
if (this.nudging) return { block: true, reason: "Just say one short spoken acknowledgement \u2014 no tools this turn." };
|
|
6121
|
+
if (!this.turnDispatched) return;
|
|
6122
|
+
if (call.name === "TaskStatus")
|
|
6123
|
+
return { block: true, reason: "You just dispatched a task this turn \u2014 do NOT poll. Give one short spoken acknowledgement and end your turn; the result arrives later as a [task \u2026] event." };
|
|
6124
|
+
if ((call.name === "Act" || call.name === "Think") && this.turnBriefs.has(String(call.args?.brief ?? "")))
|
|
6125
|
+
return { block: true, reason: "You already dispatched this exact task \u2014 acknowledge briefly and end your turn." };
|
|
6126
|
+
}
|
|
6127
|
+
};
|
|
6013
6128
|
}
|
|
6014
|
-
|
|
6015
|
-
|
|
6016
|
-
|
|
6017
|
-
|
|
6018
|
-
|
|
6019
|
-
|
|
6020
|
-
for (const r of this.idleWaiters.splice(0)) r();
|
|
6021
|
-
}
|
|
6129
|
+
/** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
|
|
6130
|
+
* micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
|
|
6131
|
+
* inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
|
|
6132
|
+
* still dead air, so silentTurn ignores external speech unless work was dispatched. */
|
|
6133
|
+
noteExternalSpeech() {
|
|
6134
|
+
this.externalSpeech = true;
|
|
6022
6135
|
}
|
|
6023
|
-
/**
|
|
6024
|
-
|
|
6025
|
-
|
|
6026
|
-
|
|
6136
|
+
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
6137
|
+
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
6138
|
+
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
6139
|
+
* voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
|
|
6140
|
+
* Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
|
|
6141
|
+
get silentTurn() {
|
|
6142
|
+
const ackedByHost = this.externalSpeech && this.turnDispatched;
|
|
6143
|
+
return !!this.options.host && !this.spokeThisTurn && !ackedByHost && !this.heldThisTurn;
|
|
6027
6144
|
}
|
|
6028
|
-
|
|
6029
|
-
|
|
6030
|
-
*
|
|
6031
|
-
|
|
6032
|
-
|
|
6033
|
-
|
|
6034
|
-
|
|
6035
|
-
|
|
6036
|
-
|
|
6145
|
+
/** A turn that voiced nothing is dead air. Re-prompt the reflex ONCE so the LLM itself voices a short
|
|
6146
|
+
* line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
|
|
6147
|
+
* Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
|
|
6148
|
+
async ackIfSilent(fallback) {
|
|
6149
|
+
const dispatched = this.turnDispatched;
|
|
6150
|
+
this.nudging = true;
|
|
6151
|
+
try {
|
|
6152
|
+
await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
|
|
6153
|
+
} catch (e) {
|
|
6154
|
+
log10.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
|
|
6155
|
+
} finally {
|
|
6156
|
+
this.nudging = false;
|
|
6037
6157
|
}
|
|
6038
|
-
this.
|
|
6039
|
-
|
|
6040
|
-
|
|
6041
|
-
|
|
6042
|
-
|
|
6043
|
-
|
|
6044
|
-
|
|
6045
|
-
|
|
6046
|
-
|
|
6047
|
-
|
|
6048
|
-
|
|
6049
|
-
|
|
6050
|
-
|
|
6051
|
-
|
|
6158
|
+
if (!this.spokeThisTurn) {
|
|
6159
|
+
const pool = fallback ? [fallback] : dispatched ? _DuplexAgent.FALLBACK_ACKS : _DuplexAgent.FALLBACK_RETRY;
|
|
6160
|
+
const fresh = pool.filter((p) => p !== this.lastFallback);
|
|
6161
|
+
const line = (fresh.length ? fresh : pool)[Math.floor(Math.random() * (fresh.length || pool.length))];
|
|
6162
|
+
this.lastFallback = line;
|
|
6163
|
+
this.emitHost({ kind: "text_delta", message: line });
|
|
6164
|
+
}
|
|
6165
|
+
}
|
|
6166
|
+
/** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
|
|
6167
|
+
* hosts/tests key on it. */
|
|
6168
|
+
static FALLBACK_ACKS = ["Okay, on it.", "On it.", "Alright, working on it."];
|
|
6169
|
+
static FALLBACK_RETRY = ["Sorry, could you say that again?", "Hm, I missed that \u2014 could you say that again?"];
|
|
6170
|
+
lastFallback = "";
|
|
6171
|
+
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
|
|
6172
|
+
* If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
|
|
6173
|
+
* speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
|
|
6174
|
+
* host NOW (this is the latency win) and streaming continues live. Any other content aborts the
|
|
6175
|
+
* speculation first (rolled back silently) and runs a normal turn behind it. */
|
|
6176
|
+
send(content) {
|
|
6177
|
+
const spec = this.spec;
|
|
6178
|
+
if (spec?.state === "pending") {
|
|
6179
|
+
if (typeof content === "string" && speculationConfirms(spec.text, content)) {
|
|
6180
|
+
spec.state = "confirmed";
|
|
6181
|
+
spec.finalText = content;
|
|
6182
|
+
for (const ev of spec.buf.splice(0)) this.options.host?.notify?.(ev);
|
|
6183
|
+
spec.decide("confirm");
|
|
6184
|
+
this.notify("diag", "speculation_confirmed", { text: spec.text.slice(0, 80) });
|
|
6185
|
+
return spec.done;
|
|
6186
|
+
}
|
|
6187
|
+
this.abortSpeculation();
|
|
6052
6188
|
}
|
|
6053
|
-
|
|
6054
|
-
|
|
6189
|
+
return this.enqueue(async () => {
|
|
6190
|
+
await this.initMemory();
|
|
6191
|
+
this.resetTurn();
|
|
6192
|
+
const res = await this.voice.send(content);
|
|
6193
|
+
this.flushHeldReflexTail();
|
|
6194
|
+
if (this.silentTurn) await this.ackIfSilent();
|
|
6195
|
+
return res;
|
|
6196
|
+
});
|
|
6055
6197
|
}
|
|
6056
|
-
/**
|
|
6057
|
-
*
|
|
6058
|
-
|
|
6059
|
-
|
|
6060
|
-
|
|
6061
|
-
|
|
6062
|
-
this.
|
|
6063
|
-
|
|
6064
|
-
|
|
6065
|
-
|
|
6066
|
-
|
|
6067
|
-
|
|
6068
|
-
|
|
6198
|
+
/** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
|
|
6199
|
+
* output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
|
|
6200
|
+
* voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
|
|
6201
|
+
* into the transcript between the speculative messages and their rollback. No-op if a speculation
|
|
6202
|
+
* is already in flight. */
|
|
6203
|
+
speculate(text) {
|
|
6204
|
+
if (!text.trim() || this.spec) return;
|
|
6205
|
+
let decide;
|
|
6206
|
+
const decision = new Promise((r) => {
|
|
6207
|
+
decide = r;
|
|
6208
|
+
});
|
|
6209
|
+
const spec = {
|
|
6210
|
+
text,
|
|
6211
|
+
state: "pending",
|
|
6212
|
+
buf: [],
|
|
6213
|
+
ctl: new AbortController(),
|
|
6214
|
+
decide,
|
|
6215
|
+
decision,
|
|
6216
|
+
done: void 0
|
|
6217
|
+
};
|
|
6218
|
+
this.spec = spec;
|
|
6219
|
+
spec.done = this.enqueue(async () => {
|
|
6220
|
+
const empty = { text: "", steps: 0, finishReason: "aborted", messages: this.voice.transcript };
|
|
6221
|
+
if (spec.state === "aborted") {
|
|
6222
|
+
this.speculativeAbortedCalls++;
|
|
6223
|
+
if (this.spec === spec) this.spec = void 0;
|
|
6224
|
+
return empty;
|
|
6225
|
+
}
|
|
6226
|
+
await this.initMemory();
|
|
6227
|
+
this.resetTurn();
|
|
6228
|
+
const base = this.voice.transcript.length;
|
|
6229
|
+
const prevSignal = this.voice.options.signal;
|
|
6230
|
+
this.voice.options.signal = spec.ctl.signal;
|
|
6231
|
+
let res;
|
|
6232
|
+
try {
|
|
6233
|
+
res = await this.voice.send(spec.text);
|
|
6234
|
+
} catch (e) {
|
|
6235
|
+
log10.warn(`speculative turn failed: ${e instanceof Error ? e.message : e}`);
|
|
6236
|
+
} finally {
|
|
6237
|
+
this.voice.options.signal = prevSignal;
|
|
6238
|
+
}
|
|
6239
|
+
const timer = setTimeout(() => {
|
|
6240
|
+
spec.state = spec.state === "pending" ? "aborted" : spec.state;
|
|
6241
|
+
spec.decide("abort");
|
|
6242
|
+
}, 1e4);
|
|
6243
|
+
timer.unref?.();
|
|
6244
|
+
const d = await spec.decision;
|
|
6245
|
+
clearTimeout(timer);
|
|
6246
|
+
if (d === "abort") {
|
|
6247
|
+
if (this.voice.transcript.length > base) this.voice.transcript.length = base;
|
|
6248
|
+
this.speculativeAbortedCalls++;
|
|
6249
|
+
log10.verbose(`speculation aborted (${this.speculativeAbortedCalls} total): "${spec.text.slice(0, 50)}"`);
|
|
6250
|
+
if (this.spec === spec) this.spec = void 0;
|
|
6251
|
+
return res ?? empty;
|
|
6252
|
+
}
|
|
6253
|
+
for (let i = base; i < this.voice.transcript.length; i++) {
|
|
6254
|
+
const m = this.voice.transcript[i];
|
|
6255
|
+
if (m.role === "user" && contentText(m.content) === spec.text) {
|
|
6256
|
+
m.content = spec.finalText;
|
|
6257
|
+
break;
|
|
6258
|
+
}
|
|
6259
|
+
}
|
|
6260
|
+
this.spec = void 0;
|
|
6261
|
+
this.flushHeldReflexTail();
|
|
6262
|
+
if (this.silentTurn) await this.ackIfSilent();
|
|
6263
|
+
return res ?? empty;
|
|
6264
|
+
});
|
|
6069
6265
|
}
|
|
6070
|
-
/**
|
|
6071
|
-
|
|
6072
|
-
|
|
6073
|
-
|
|
6074
|
-
|
|
6075
|
-
if (
|
|
6076
|
-
|
|
6077
|
-
|
|
6078
|
-
|
|
6079
|
-
|
|
6266
|
+
/** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
|
|
6267
|
+
* Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
|
|
6268
|
+
* attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
|
|
6269
|
+
abortSpeculation() {
|
|
6270
|
+
const spec = this.spec;
|
|
6271
|
+
if (spec?.state !== "pending") return;
|
|
6272
|
+
spec.state = "aborted";
|
|
6273
|
+
spec.buf.length = 0;
|
|
6274
|
+
spec.ctl.abort();
|
|
6275
|
+
spec.decide("abort");
|
|
6276
|
+
this.notify("diag", "speculation_aborted", { text: spec.text.slice(0, 80) });
|
|
6277
|
+
}
|
|
6278
|
+
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
6279
|
+
cancelTask(id) {
|
|
6280
|
+
const rec = this.tasks.get(id);
|
|
6281
|
+
if (!rec) return `No task '${id}'.`;
|
|
6282
|
+
if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
|
|
6283
|
+
rec.status = "cancelled";
|
|
6284
|
+
rec.controller.abort();
|
|
6285
|
+
return `Task ${rec.id} (${rec.label}) cancelled.`;
|
|
6286
|
+
}
|
|
6287
|
+
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
6288
|
+
* delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
|
|
6289
|
+
* tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
|
|
6290
|
+
* Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
|
|
6291
|
+
parkInFlightDeliveries() {
|
|
6292
|
+
const parked = [];
|
|
6293
|
+
for (const rec of this.tasks.values())
|
|
6294
|
+
if (rec.status === "running" && !rec.deliveryParked) {
|
|
6295
|
+
rec.deliveryParked = true;
|
|
6296
|
+
parked.push(rec.id);
|
|
6297
|
+
}
|
|
6298
|
+
return parked;
|
|
6299
|
+
}
|
|
6300
|
+
/** Resolve when all queued voice turns AND all in-flight worker tasks have settled (tests, graceful shutdown). */
|
|
6301
|
+
async idle() {
|
|
6302
|
+
while (true) {
|
|
6303
|
+
const q2 = this.queue;
|
|
6304
|
+
await q2.catch(() => {
|
|
6305
|
+
});
|
|
6306
|
+
await Promise.all([...this.tasks.values()].map((t) => t.promise));
|
|
6307
|
+
if (this.queue === q2 && ![...this.tasks.values()].some((t) => t.status === "running")) return;
|
|
6080
6308
|
}
|
|
6081
|
-
|
|
6082
|
-
|
|
6083
|
-
|
|
6084
|
-
|
|
6309
|
+
}
|
|
6310
|
+
/** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
|
|
6311
|
+
enqueue(fn) {
|
|
6312
|
+
const run = this.queue.then(fn, fn);
|
|
6313
|
+
this.queue = run.then(() => {
|
|
6314
|
+
}, () => {
|
|
6315
|
+
});
|
|
6316
|
+
return run;
|
|
6317
|
+
}
|
|
6318
|
+
notify(kind, message, data) {
|
|
6319
|
+
this.emitHost({ kind, message, data });
|
|
6320
|
+
}
|
|
6321
|
+
/** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
|
|
6322
|
+
* hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
|
|
6323
|
+
* order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
|
|
6324
|
+
* which bypasses this via host.notify directly) passes through untouched. */
|
|
6325
|
+
emitHost(ev) {
|
|
6326
|
+
const spec = this.spec;
|
|
6327
|
+
if (spec && (ev.kind === "text_delta" || ev.kind === "hold_filler")) {
|
|
6328
|
+
if (spec.state === "pending") {
|
|
6329
|
+
spec.buf.push(ev);
|
|
6085
6330
|
return;
|
|
6086
6331
|
}
|
|
6087
|
-
if (
|
|
6088
|
-
|
|
6089
|
-
|
|
6332
|
+
if (spec.state === "aborted") return;
|
|
6333
|
+
}
|
|
6334
|
+
this.options.host?.notify?.(ev);
|
|
6335
|
+
}
|
|
6336
|
+
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
6337
|
+
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
6338
|
+
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
6339
|
+
* text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
|
|
6340
|
+
queueRevoice(event, nonClean = false) {
|
|
6341
|
+
this.pendingEvents.push(event);
|
|
6342
|
+
if (nonClean) this.pendingNonClean = true;
|
|
6343
|
+
if (this.flushQueued) return;
|
|
6344
|
+
this.flushQueued = true;
|
|
6345
|
+
void this.enqueue(async () => {
|
|
6346
|
+
this.flushQueued = false;
|
|
6347
|
+
const events = this.pendingEvents.splice(0);
|
|
6348
|
+
const nonCleanTurn = this.pendingNonClean;
|
|
6349
|
+
this.pendingNonClean = false;
|
|
6350
|
+
if (!events.length) return;
|
|
6351
|
+
const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
|
|
6352
|
+
this.resetTurn();
|
|
6353
|
+
this.turnFollowUp = nonCleanTurn;
|
|
6354
|
+
await this.voice.send(events.join("\n"));
|
|
6355
|
+
this.flushHeldReflexTail();
|
|
6356
|
+
if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
|
|
6357
|
+
this.notify("revoice_done", "");
|
|
6358
|
+
});
|
|
6359
|
+
}
|
|
6360
|
+
/** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
|
|
6361
|
+
* Act briefs get a self-verify footer — the worker's report is trusted without review, so it
|
|
6362
|
+
* must check its own work before reporting (nearly free under prompt caching; measured honest:
|
|
6363
|
+
* it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
|
|
6364
|
+
buildBrief(brief, tier = "act", deliver = true) {
|
|
6365
|
+
const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
|
|
6366
|
+
const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
|
|
6367
|
+
const deliverContract = deliver ? `
|
|
6368
|
+
|
|
6369
|
+
## DELIVER (spoken delivery)
|
|
6370
|
+
You are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown. NEVER enumerate in speech \u2014 no numbered or bulleted lists ("One. \u2026 Two. \u2026" is robotic). Deliver multiple items as flowing conversation with brief connective phrasing ("here's one\u2026", "and another\u2026", "oh, and\u2026"), pausing between items with sentence breaks, not numbers.` + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
|
|
6371
|
+
return (recent ? `${brief}
|
|
6372
|
+
|
|
6373
|
+
## Recent conversation (for context)
|
|
6374
|
+
${recent}` : brief) + verify + deliverContract;
|
|
6375
|
+
}
|
|
6376
|
+
/** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
|
|
6377
|
+
spawnWorker(id, label, briefText, tier, brief, followUp) {
|
|
6378
|
+
const o = this.options;
|
|
6379
|
+
const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
|
|
6380
|
+
const tierModel = tier === "think" ? o.thinkModel : o.actModel;
|
|
6381
|
+
const controller = new AbortController();
|
|
6382
|
+
const base = tierOpts?.hooks ?? o.actOptions?.hooks;
|
|
6383
|
+
const report = o.progressUpdates ? this.progressReporter(id) : void 0;
|
|
6384
|
+
const tail = [];
|
|
6385
|
+
const pushTail = (line) => {
|
|
6386
|
+
tail.push(line.slice(0, 200));
|
|
6387
|
+
if (tail.length > 120) tail.splice(0, tail.length - 120);
|
|
6388
|
+
};
|
|
6389
|
+
const hooks = {
|
|
6390
|
+
...base,
|
|
6391
|
+
preToolUse: async (call, meta) => {
|
|
6392
|
+
const d = await base?.preToolUse?.(call, meta);
|
|
6393
|
+
pushTail(`\u2699 ${describeCall(call)}`);
|
|
6394
|
+
report?.pre(call);
|
|
6395
|
+
return d;
|
|
6396
|
+
},
|
|
6397
|
+
postToolUse: async (call, result, meta) => {
|
|
6398
|
+
await base?.postToolUse?.(call, result, meta);
|
|
6399
|
+
const last = result?.trim().split("\n").filter(Boolean).pop();
|
|
6400
|
+
if (last) pushTail(` \u21B3 ${last}`);
|
|
6401
|
+
report?.post(call);
|
|
6402
|
+
},
|
|
6403
|
+
onToolOutput: (call, chunk, meta) => {
|
|
6404
|
+
base?.onToolOutput?.(call, chunk, meta);
|
|
6405
|
+
report?.output(chunk);
|
|
6406
|
+
}
|
|
6407
|
+
};
|
|
6408
|
+
const relayAsk = async (q2) => {
|
|
6409
|
+
const opts = q2.options?.length ? ` Options: ${q2.options.map((x) => x.label).join(", ")}.` : "";
|
|
6410
|
+
const a = await this.parkQuestion(id, `${q2.question}${opts}`);
|
|
6411
|
+
return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
|
|
6412
|
+
};
|
|
6413
|
+
const splitter = new SpokenSplitter();
|
|
6414
|
+
const speak = (seg) => {
|
|
6415
|
+
if (seg && !this.tasks.get(id)?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
|
|
6416
|
+
};
|
|
6417
|
+
const coalescer = new SentenceCoalescer();
|
|
6418
|
+
const feedSpoken = (s) => {
|
|
6419
|
+
const ready = coalescer.feed(s);
|
|
6420
|
+
if (ready) speak(ready);
|
|
6421
|
+
};
|
|
6422
|
+
const flushSpoken = () => speak(coalescer.flush());
|
|
6423
|
+
const askBridge = o.askRelay ? { ask: relayAsk } : o.host?.ask ? { ask: (q2) => o.host.ask(q2) } : {};
|
|
6424
|
+
const workerHost = {
|
|
6425
|
+
...askBridge,
|
|
6426
|
+
notify: (ev) => {
|
|
6427
|
+
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
6428
|
+
const { spoken, detail } = splitter.feed(ev.message);
|
|
6429
|
+
feedSpoken(spoken);
|
|
6430
|
+
if (detail.trim()) pushTail(detail.trim());
|
|
6431
|
+
return;
|
|
6432
|
+
}
|
|
6090
6433
|
}
|
|
6091
|
-
this.drainTimer = null;
|
|
6092
|
-
this.speaking = false;
|
|
6093
|
-
if (this.turnStartAt) log10.debug(`turn: ${Math.round(now() - this.turnStartAt)}ms (incl. playback)`);
|
|
6094
|
-
this.echoUntil = now() + 2500;
|
|
6095
|
-
if (!this.usingAec) this.stt.reset();
|
|
6096
|
-
this.setState("listening");
|
|
6097
|
-
if (this.uttQueue.length) this.pumpQueue();
|
|
6098
6434
|
};
|
|
6099
|
-
const
|
|
6100
|
-
|
|
6101
|
-
|
|
6435
|
+
const agentOpts = {
|
|
6436
|
+
ai: o.ai,
|
|
6437
|
+
fs: o.fs,
|
|
6438
|
+
model: tierModel,
|
|
6439
|
+
...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
|
|
6440
|
+
...tierOpts,
|
|
6441
|
+
// Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
|
|
6442
|
+
// main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
|
|
6443
|
+
providerOptions: o.providerOptionsFor?.(tierModel),
|
|
6444
|
+
stream: true,
|
|
6445
|
+
// worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
|
|
6446
|
+
host: workerHost,
|
|
6447
|
+
// carries BOTH ask AND the <spoken>-splitting notify
|
|
6448
|
+
...hooks ? { hooks } : {},
|
|
6449
|
+
signal: controller.signal
|
|
6450
|
+
// shared with the checker so a cancel tears down both
|
|
6102
6451
|
};
|
|
6103
|
-
|
|
6104
|
-
|
|
6105
|
-
|
|
6106
|
-
if (
|
|
6107
|
-
|
|
6108
|
-
|
|
6452
|
+
const promise = new Agent(agentOpts).run(briefText).then((res) => {
|
|
6453
|
+
const { spoken, detail } = splitter.flush();
|
|
6454
|
+
feedSpoken(spoken);
|
|
6455
|
+
if (detail.trim()) pushTail(detail.trim());
|
|
6456
|
+
flushSpoken();
|
|
6457
|
+
return res;
|
|
6458
|
+
}).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err2) => this.onWorkerFailed(id, err2));
|
|
6459
|
+
this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
|
|
6460
|
+
if (this.tasks.size > this.options.maxTaskRecords)
|
|
6461
|
+
for (const [tid, rec] of this.tasks) {
|
|
6462
|
+
if (this.tasks.size <= this.options.maxTaskRecords) break;
|
|
6463
|
+
if (rec.status !== "running") this.tasks.delete(tid);
|
|
6464
|
+
}
|
|
6109
6465
|
}
|
|
6110
|
-
/**
|
|
6111
|
-
* the
|
|
6112
|
-
|
|
6113
|
-
|
|
6114
|
-
|
|
6115
|
-
|
|
6466
|
+
/** Fresh-context check of a successful Act task: a NEW agent (same model/fs/tools, but NO shared
|
|
6467
|
+
* conversation context) re-reads the file state against the brief and fixes any gap. The fix lands
|
|
6468
|
+
* on the shared fs automatically (workers write fs directly, no overlay), so grading sees the
|
|
6469
|
+
* corrected state. Bounded to ONE pass. Off unless `verifyActTasks`; never runs for think/failed/
|
|
6470
|
+
* cancelled tasks. Usage is merged so /cost reflects the real (worker + checker) spend. */
|
|
6471
|
+
async maybeVerify(id, brief, res, tier, agentOpts, askBridge) {
|
|
6472
|
+
if (!this.options.verifyActTasks || tier !== "act" || res.finishReason !== "stop") return res;
|
|
6473
|
+
if (this.tasks.get(id)?.status === "cancelled") return res;
|
|
6474
|
+
const { stream: _stream, host: _host, ...restOpts } = agentOpts;
|
|
6475
|
+
const checkerOpts = {
|
|
6476
|
+
...restOpts,
|
|
6477
|
+
...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
|
|
6478
|
+
};
|
|
6479
|
+
const checkBrief = `${this.buildBrief(brief, tier, false)}
|
|
6480
|
+
|
|
6481
|
+
## VERIFY MODE
|
|
6482
|
+
Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
|
|
6483
|
+
this.notify("task_verify", `task ${id}: verifying`, { id });
|
|
6484
|
+
const cres = await new Agent(checkerOpts).run(checkBrief);
|
|
6485
|
+
if (cres.finishReason !== "stop") {
|
|
6486
|
+
log10.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
|
|
6487
|
+
this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
|
|
6488
|
+
}
|
|
6489
|
+
const sum = (a = 0, b = 0) => a + b;
|
|
6490
|
+
return {
|
|
6491
|
+
...res,
|
|
6492
|
+
steps: res.steps + cres.steps,
|
|
6493
|
+
// Merge the checker's messages so downstream tool-call/step accounting includes BOTH agents
|
|
6494
|
+
// (else a verified task's toolCalls would undercount vs its steps/usage).
|
|
6495
|
+
messages: [...res.messages, ...cres.messages],
|
|
6496
|
+
usageEstimated: res.usageEstimated || cres.usageEstimated,
|
|
6497
|
+
usage: res.usage && cres.usage ? {
|
|
6498
|
+
promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
|
|
6499
|
+
completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
|
|
6500
|
+
totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
|
|
6501
|
+
cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
|
|
6502
|
+
cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
|
|
6503
|
+
} : res.usage ?? cres.usage
|
|
6504
|
+
};
|
|
6116
6505
|
}
|
|
6117
|
-
/**
|
|
6118
|
-
|
|
6119
|
-
|
|
6120
|
-
|
|
6121
|
-
|
|
6122
|
-
|
|
6506
|
+
/** Throttled per-task progress: worker tool calls → at most one progress re-voice per interval.
|
|
6507
|
+
* Two sources, one throttle: completed steps (post) and a heartbeat for a SINGLE long tool call
|
|
6508
|
+
* (pre records the in-flight call; a self-cleaning timer narrates "still inside Bash — 70s").
|
|
6509
|
+
* Completion supersedes: nothing is emitted once the task has settled. */
|
|
6510
|
+
progressReporter(id) {
|
|
6511
|
+
let lastAt = Date.now();
|
|
6512
|
+
let steps = 0;
|
|
6513
|
+
let inflight = null;
|
|
6514
|
+
const due = () => {
|
|
6515
|
+
if (this.pendingAsks.size) return void 0;
|
|
6516
|
+
const rec = this.tasks.get(id);
|
|
6517
|
+
return rec && rec.status === "running" && Date.now() - lastAt >= this.options.progressIntervalMs ? rec : void 0;
|
|
6518
|
+
};
|
|
6519
|
+
const emit = (rec, line, call) => {
|
|
6520
|
+
lastAt = Date.now();
|
|
6521
|
+
this.notify("task_progress", `task ${id} (${rec.label}): ${line}`, { id, steps, call: call.name });
|
|
6522
|
+
this.queueRevoice(`[task ${id} progress] ${line}`);
|
|
6523
|
+
};
|
|
6524
|
+
const timer = setInterval(() => {
|
|
6525
|
+
const rec = this.tasks.get(id);
|
|
6526
|
+
if (!rec || rec.status !== "running") return clearInterval(timer);
|
|
6527
|
+
if (!inflight || !due()) return;
|
|
6528
|
+
const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
|
|
6529
|
+
emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
|
|
6530
|
+
}, Math.max(this.options.progressIntervalMs, 250));
|
|
6531
|
+
timer.unref?.();
|
|
6532
|
+
return {
|
|
6533
|
+
pre: (call) => {
|
|
6534
|
+
inflight = { call, at: Date.now(), tail: "" };
|
|
6535
|
+
},
|
|
6536
|
+
output: (chunk) => {
|
|
6537
|
+
if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
|
|
6538
|
+
},
|
|
6539
|
+
// digest only — NEVER re-voices directly
|
|
6540
|
+
post: (call) => {
|
|
6541
|
+
steps++;
|
|
6542
|
+
inflight = null;
|
|
6543
|
+
const rec = due();
|
|
6544
|
+
if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
|
|
6545
|
+
}
|
|
6546
|
+
};
|
|
6123
6547
|
}
|
|
6124
|
-
/**
|
|
6125
|
-
*
|
|
6126
|
-
*
|
|
6127
|
-
|
|
6128
|
-
|
|
6129
|
-
|
|
6130
|
-
|
|
6548
|
+
/** Park a question under `askId` (a task id, or any unique key for permission asks): re-voices
|
|
6549
|
+
* '[task <id> asks] …' and resolves with the user's answer via AnswerTask — or '' on timeout/
|
|
6550
|
+
* task settle (callers map '' to deny / best-judgment). Workers never block forever. */
|
|
6551
|
+
parkQuestion(askId, question) {
|
|
6552
|
+
return new Promise((resolve4) => {
|
|
6553
|
+
let settled = false;
|
|
6554
|
+
const finish = (answer) => {
|
|
6555
|
+
if (settled) return;
|
|
6556
|
+
settled = true;
|
|
6557
|
+
clearTimeout(timer);
|
|
6558
|
+
this.pendingAsks.delete(askId);
|
|
6559
|
+
resolve4(answer);
|
|
6560
|
+
};
|
|
6561
|
+
const timer = setTimeout(() => {
|
|
6562
|
+
this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
|
|
6563
|
+
finish("");
|
|
6564
|
+
}, this.options.askTimeoutMs);
|
|
6565
|
+
this.pendingAsks.set(askId, { question, resolve: finish });
|
|
6566
|
+
this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
|
|
6567
|
+
this.queueRevoice(`[task ${askId} asks] ${question}
|
|
6568
|
+
(Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
|
|
6569
|
+
});
|
|
6131
6570
|
}
|
|
6132
|
-
/**
|
|
6133
|
-
|
|
6134
|
-
|
|
6135
|
-
if (this.speaking) return;
|
|
6136
|
-
const text = this.uttQueue.shift();
|
|
6137
|
-
if (text == null) return;
|
|
6138
|
-
this.beginSpeech();
|
|
6139
|
-
this.speakDelta(text);
|
|
6140
|
-
this.endSpeech();
|
|
6571
|
+
/** Resolve any question a settling/cancelled task left parked (its answer can no longer matter). */
|
|
6572
|
+
dropAsk(id) {
|
|
6573
|
+
this.pendingAsks.get(id)?.resolve("");
|
|
6141
6574
|
}
|
|
6142
|
-
/**
|
|
6143
|
-
|
|
6144
|
-
|
|
6145
|
-
|
|
6146
|
-
|
|
6147
|
-
|
|
6148
|
-
|
|
6575
|
+
/** Build the INTEGRATION TURN prompt for a NON-CLEAN settled worker (early stop / failure). A clean
|
|
6576
|
+
* success never reaches here — it streams its own `<spoken>` delivery during the run. For a partial
|
|
6577
|
+
* or failed result the outcome re-enters the reflex as a decision (like a tool_result flowing back
|
|
6578
|
+
* into a normal agent loop): the reflex evaluates the outcome against the original intent and chooses
|
|
6579
|
+
* what to do next.
|
|
6580
|
+
*
|
|
6581
|
+
* Decision branches (the reflex acts on them with EXISTING tools — no new surface):
|
|
6582
|
+
* • accept → SPEAK the (partial) result plainly — don't dress a failure up as success.
|
|
6583
|
+
* • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
|
|
6584
|
+
* exists AND this task wasn't already a follow-up (one hop max). Wires the dead
|
|
6585
|
+
* "Reserve Think for a problem Act already FAILED at" promise.
|
|
6586
|
+
* • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
|
|
6587
|
+
* • ask → ask the user ONE concrete question if genuinely blocked.
|
|
6588
|
+
*
|
|
6589
|
+
* Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
|
|
6590
|
+
* failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
|
|
6591
|
+
integrationPrompt(rec, outcome, body, finishReason) {
|
|
6592
|
+
const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
|
|
6593
|
+
const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
|
|
6594
|
+
const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
|
|
6595
|
+
const hasThink = this.options.thinkModel !== false;
|
|
6596
|
+
const options = [];
|
|
6597
|
+
if (!rec.followUp && canEscalate && hasThink)
|
|
6598
|
+
options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
|
|
6599
|
+
if (!rec.followUp && canEscalate)
|
|
6600
|
+
options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
|
|
6601
|
+
options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
|
|
6602
|
+
options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
|
|
6603
|
+
const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
|
|
6604
|
+
const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
|
|
6605
|
+
return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
|
|
6606
|
+
}
|
|
6607
|
+
onWorkerSettled(id, res) {
|
|
6608
|
+
this.dropAsk(id);
|
|
6609
|
+
const rec = this.tasks.get(id);
|
|
6610
|
+
if (res.finishReason === "aborted" || rec.status === "cancelled") {
|
|
6611
|
+
rec.status = "cancelled";
|
|
6612
|
+
this.notify("task_cancelled", `task ${id} (${rec.label}) cancelled`);
|
|
6613
|
+
return;
|
|
6614
|
+
}
|
|
6615
|
+
if (res.finishReason === "error") {
|
|
6616
|
+
const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
|
|
6617
|
+
return this.failTask(rec, msg);
|
|
6618
|
+
}
|
|
6619
|
+
rec.status = "done";
|
|
6620
|
+
rec.result = res.text;
|
|
6621
|
+
const incomplete = res.finishReason !== "stop";
|
|
6622
|
+
log10.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
|
|
6623
|
+
this.notify("task_done", `task ${id} (${rec.label}) completed`, {
|
|
6624
|
+
id,
|
|
6625
|
+
text: res.text,
|
|
6626
|
+
usage: res.usage,
|
|
6627
|
+
usageEstimated: res.usageEstimated,
|
|
6628
|
+
finishReason: res.finishReason,
|
|
6629
|
+
steps: res.steps,
|
|
6630
|
+
toolCalls: res.messages.filter((m) => m.role === "tool").length
|
|
6631
|
+
});
|
|
6632
|
+
if (incomplete) {
|
|
6633
|
+
return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
|
|
6149
6634
|
}
|
|
6150
|
-
|
|
6151
|
-
this.
|
|
6152
|
-
|
|
6153
|
-
if (
|
|
6154
|
-
|
|
6155
|
-
this.ctxOpen = false;
|
|
6156
|
-
this.interrupted = true;
|
|
6157
|
-
this.suspectUntil = 0;
|
|
6158
|
-
this.echoUntil = now() + Math.max(2500, this.player.drainMs() + 3e3);
|
|
6159
|
-
this.tts.cancel();
|
|
6160
|
-
this.player.kill();
|
|
6161
|
-
if (!this.usingAec) this.stt.reset();
|
|
6162
|
-
if (this.reply) this.prevReply = this.reply;
|
|
6163
|
-
this.setState("listening");
|
|
6164
|
-
}
|
|
6165
|
-
stop() {
|
|
6166
|
-
this.uttQueue = [];
|
|
6167
|
-
if (this.resumeTimer) clearTimeout(this.resumeTimer);
|
|
6168
|
-
if (this.pendingTimer) clearTimeout(this.pendingTimer);
|
|
6169
|
-
if (this.drainTimer) clearTimeout(this.drainTimer);
|
|
6170
|
-
this.stt.stop();
|
|
6171
|
-
this.player.kill();
|
|
6172
|
-
this.tts.close();
|
|
6173
|
-
this.setState("idle");
|
|
6635
|
+
const tail = rec.splitter?.flush();
|
|
6636
|
+
if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
|
|
6637
|
+
if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
|
|
6638
|
+
if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
|
|
6639
|
+
this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
|
|
6174
6640
|
}
|
|
6175
|
-
|
|
6176
|
-
|
|
6177
|
-
return s.toLowerCase().replace(/[^a-z0-9\s]/g, "").split(/\s+/).filter((w) => w.length >= 2);
|
|
6641
|
+
onWorkerFailed(id, err2) {
|
|
6642
|
+
this.failTask(this.tasks.get(id), err2 instanceof Error ? err2.message : String(err2));
|
|
6178
6643
|
}
|
|
6179
|
-
|
|
6180
|
-
|
|
6644
|
+
failTask(rec, msg) {
|
|
6645
|
+
this.dropAsk(rec.id);
|
|
6646
|
+
rec.status = "error";
|
|
6647
|
+
rec.result = msg;
|
|
6648
|
+
log10.warn(`task ${rec.id} failed: ${msg}`);
|
|
6649
|
+
this.notify("task_error", `task ${rec.id} (${rec.label}) failed: ${msg}`);
|
|
6650
|
+
this.queueRevoice(this.integrationPrompt(rec, "error", msg, "error"), true);
|
|
6181
6651
|
}
|
|
6182
|
-
|
|
6183
|
-
|
|
6652
|
+
// --- voice tools (closures over this instance) ---
|
|
6653
|
+
/** Live-switch the think tier: `false` disables (removes the Think tool from the voice agent),
|
|
6654
|
+
* a model id enables (adds the tool if missing). The system-prompt THINK_SLOT text is frozen at
|
|
6655
|
+
* construction — the tool's own description carries the routing guidance, so a live enable works;
|
|
6656
|
+
* dispatch()'s think→act fallback covers any straggler calls after a live disable. */
|
|
6657
|
+
setThinkModel(model) {
|
|
6658
|
+
this.options.thinkModel = model;
|
|
6659
|
+
const tools = this.voice.options.tools;
|
|
6660
|
+
const i = tools.findIndex((t) => t.name === "Think");
|
|
6661
|
+
if (model === false && i >= 0) tools.splice(i, 1);
|
|
6662
|
+
else if (model !== false && i < 0) tools.push(this.thinkTool());
|
|
6184
6663
|
}
|
|
6185
|
-
/**
|
|
6186
|
-
*
|
|
6187
|
-
*
|
|
6188
|
-
|
|
6189
|
-
|
|
6190
|
-
|
|
6191
|
-
const
|
|
6192
|
-
const
|
|
6193
|
-
|
|
6664
|
+
/** User/programmatic spawn: the CLI's /act and /think commands. Returns the task id.
|
|
6665
|
+
* `followUp` marks an automatic escalation/re-delegation (set by the integration turn) so the new
|
|
6666
|
+
* task's own integration turn won't escalate again — capping auto-follow-ups to one hop. */
|
|
6667
|
+
async dispatch(brief, tier = "act", label, followUp = false) {
|
|
6668
|
+
if (tier === "think" && this.options.thinkModel === false) tier = "act";
|
|
6669
|
+
if (followUp) this.autoEscalations++;
|
|
6670
|
+
const id = `t${++this.seq}`;
|
|
6671
|
+
const lbl = label ?? tier;
|
|
6672
|
+
await this.options.onTaskStart?.(id, lbl);
|
|
6673
|
+
this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
|
|
6674
|
+
this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
|
|
6675
|
+
return id;
|
|
6194
6676
|
}
|
|
6195
|
-
|
|
6196
|
-
|
|
6197
|
-
|
|
6198
|
-
|
|
6199
|
-
|
|
6200
|
-
|
|
6201
|
-
|
|
6202
|
-
|
|
6203
|
-
|
|
6204
|
-
|
|
6205
|
-
this.lastOverlapPartial = txt;
|
|
6206
|
-
if (!this.genuine(txt)) {
|
|
6207
|
-
if (this.pausedAt) this.armResume();
|
|
6208
|
-
return;
|
|
6209
|
-
}
|
|
6210
|
-
if (!this.pausedAt) {
|
|
6211
|
-
this.pausedAt = now();
|
|
6212
|
-
this.player.pause();
|
|
6213
|
-
if (this.lastResumeAt && now() - this.lastResumeAt < this.options.overlapRepauseCedeMs) {
|
|
6214
|
-
this.interrupt();
|
|
6215
|
-
this.options.onBargeIn(this.ctxOpen ? "speaking" : "drain");
|
|
6216
|
-
return;
|
|
6217
|
-
}
|
|
6218
|
-
}
|
|
6219
|
-
if (this.words(txt).length >= 2) {
|
|
6220
|
-
const phase = this.ctxOpen ? "speaking" : "drain";
|
|
6221
|
-
this.interrupt();
|
|
6222
|
-
this.options.onBargeIn(phase);
|
|
6223
|
-
return;
|
|
6677
|
+
actTool() {
|
|
6678
|
+
return {
|
|
6679
|
+
name: "Act",
|
|
6680
|
+
description: 'Escalate real work (reading/editing files, searching, running tasks, building) to a standard background worker. Returns immediately with a task id; the result arrives later as a "[task <id> completed]" event. Provide a clear, self-contained `brief` (the worker does not hear the live conversation).',
|
|
6681
|
+
parameters: {
|
|
6682
|
+
type: "object",
|
|
6683
|
+
required: ["brief"],
|
|
6684
|
+
properties: {
|
|
6685
|
+
brief: { type: "string", description: "full, self-contained instructions for the worker" },
|
|
6686
|
+
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
6224
6687
|
}
|
|
6225
|
-
|
|
6226
|
-
|
|
6227
|
-
|
|
6228
|
-
|
|
6229
|
-
|
|
6230
|
-
|
|
6231
|
-
this.
|
|
6232
|
-
this.
|
|
6688
|
+
},
|
|
6689
|
+
run: async ({ brief, label }) => {
|
|
6690
|
+
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
6691
|
+
this.turnDispatched = true;
|
|
6692
|
+
this.turnBriefs.add(String(brief ?? ""));
|
|
6693
|
+
this.voice.options.toolChoice = "none";
|
|
6694
|
+
const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
|
|
6695
|
+
return this.spokeBeforeDispatch ? `Acting on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
6233
6696
|
}
|
|
6234
|
-
|
|
6235
|
-
}
|
|
6236
|
-
if (this.pendingUtt && text.trim()) {
|
|
6237
|
-
if (this.pendingTimer) clearTimeout(this.pendingTimer);
|
|
6238
|
-
this.pendingTimer = setTimeout(() => this.flushUtterance(), Math.max(800, this.options.utteranceMergeMs));
|
|
6239
|
-
}
|
|
6240
|
-
if (!this.echoActive() || (this.usingAec ? this.genuine(text) : this.novelWords(text).length >= 1)) this.options.onPartial(text);
|
|
6697
|
+
};
|
|
6241
6698
|
}
|
|
6242
|
-
|
|
6243
|
-
|
|
6244
|
-
|
|
6245
|
-
|
|
6246
|
-
|
|
6247
|
-
|
|
6248
|
-
|
|
6249
|
-
|
|
6250
|
-
|
|
6251
|
-
|
|
6252
|
-
|
|
6253
|
-
|
|
6254
|
-
|
|
6255
|
-
|
|
6256
|
-
|
|
6257
|
-
|
|
6699
|
+
thinkTool() {
|
|
6700
|
+
return {
|
|
6701
|
+
name: "Think",
|
|
6702
|
+
description: "Escalate to a premium deep-reasoning agent for complex analysis, architecture decisions, hard debugging, or planning. Same async pattern as Act \u2014 returns a task id. Use when the problem needs careful thought before (or instead of) action. Do not use Think for simple tasks \u2014 Act is cheaper and faster.",
|
|
6703
|
+
parameters: {
|
|
6704
|
+
type: "object",
|
|
6705
|
+
required: ["brief"],
|
|
6706
|
+
properties: {
|
|
6707
|
+
brief: { type: "string", description: "the question or problem to reason about deeply" },
|
|
6708
|
+
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
6709
|
+
}
|
|
6710
|
+
},
|
|
6711
|
+
run: async ({ brief, label }) => {
|
|
6712
|
+
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
6713
|
+
this.turnDispatched = true;
|
|
6714
|
+
this.turnBriefs.add(String(brief ?? ""));
|
|
6715
|
+
this.voice.options.toolChoice = "none";
|
|
6716
|
+
const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
|
|
6717
|
+
return this.spokeBeforeDispatch ? `Thinking on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
6258
6718
|
}
|
|
6259
|
-
|
|
6260
|
-
}
|
|
6261
|
-
return `${prev} ${next}`;
|
|
6719
|
+
};
|
|
6262
6720
|
}
|
|
6263
|
-
|
|
6264
|
-
|
|
6265
|
-
|
|
6266
|
-
|
|
6721
|
+
taskStatusTool() {
|
|
6722
|
+
return {
|
|
6723
|
+
name: "TaskStatus",
|
|
6724
|
+
description: "Status of background tasks. Pass `id` for one task, or omit it to list all.",
|
|
6725
|
+
parameters: { type: "object", properties: { id: { type: "string" } } },
|
|
6726
|
+
run: async ({ id }) => {
|
|
6727
|
+
const list = id ? [this.tasks.get(String(id))].filter(Boolean) : [...this.tasks.values()];
|
|
6728
|
+
if (!list.length) return id ? `No task '${id}'.` : "No background tasks.";
|
|
6729
|
+
return list.map((t) => `${t.id} (${t.label}): ${t.status}`).join("\n");
|
|
6730
|
+
}
|
|
6731
|
+
};
|
|
6267
6732
|
}
|
|
6268
|
-
|
|
6269
|
-
|
|
6270
|
-
|
|
6271
|
-
|
|
6272
|
-
|
|
6273
|
-
|
|
6274
|
-
|
|
6275
|
-
|
|
6276
|
-
|
|
6277
|
-
|
|
6278
|
-
|
|
6279
|
-
|
|
6280
|
-
|
|
6281
|
-
|
|
6282
|
-
|
|
6283
|
-
|
|
6284
|
-
|
|
6285
|
-
|
|
6286
|
-
|
|
6287
|
-
|
|
6288
|
-
|
|
6289
|
-
|
|
6290
|
-
|
|
6733
|
+
/** Sub-100ms read-only lookups the voice may do itself — everything else stays Act-only.
|
|
6734
|
+
* fs-only (no shell; the engine is VFS-abstracted): time, git branch (.git/HEAD read), ls, file
|
|
6735
|
+
* head. Output is hard-capped so a lookup can never bloat the skinny voice context. */
|
|
6736
|
+
quickLookTool() {
|
|
6737
|
+
const CAP2 = 2e3;
|
|
6738
|
+
const kinds = [.../* @__PURE__ */ new Set(["time", "branch", "ls", "file", "capabilities", ...Object.keys(this.options.quickLook ?? {})])];
|
|
6739
|
+
return {
|
|
6740
|
+
name: "QuickLook",
|
|
6741
|
+
description: `Instant read-only lookup \u2014 one of: ${kinds.join(", ")}. For trivial facts only; anything needing search, commands, or reasoning goes through Act.`,
|
|
6742
|
+
parameters: {
|
|
6743
|
+
type: "object",
|
|
6744
|
+
required: ["what"],
|
|
6745
|
+
properties: {
|
|
6746
|
+
what: { type: "string", enum: kinds, description: "what to look up" },
|
|
6747
|
+
path: { type: "string", description: "for ls/file: the path to look at" }
|
|
6748
|
+
}
|
|
6749
|
+
},
|
|
6750
|
+
run: async ({ what, path }) => {
|
|
6751
|
+
const fs = this.options.fs;
|
|
6752
|
+
try {
|
|
6753
|
+
const over = this.options.quickLook?.[String(what)];
|
|
6754
|
+
if (over) return await over(path ? String(path) : void 0);
|
|
6755
|
+
switch (String(what)) {
|
|
6756
|
+
case "capabilities": {
|
|
6757
|
+
const actTools = this.options.actOptions?.tools ?? [];
|
|
6758
|
+
const names = actTools.map((t) => t.name);
|
|
6759
|
+
const mcpServers = Object.keys(this.options.actOptions?.providerOptions?.mcpServers ?? {});
|
|
6760
|
+
const mcpNote = mcpServers.length ? ` Plus MCP servers your worker can use: ${mcpServers.join(", ")} (e.g. browser-bridge \u2192 drive a real browser: open tabs, navigate, click, screenshot).` : "";
|
|
6761
|
+
if (!names.length)
|
|
6762
|
+
return "Your worker uses Act's default local toolset (reading/editing files, running shell commands). No extra tools (e.g. web/internet) are configured; if a request is not a basic file or shell operation, assume you can't do it and say so." + mcpNote;
|
|
6763
|
+
const hasFetch = names.some((n) => /WebFetch/i.test(n));
|
|
6764
|
+
const hasBrowser = names.some((n) => /browser.*(navigate|click|page|type)/i.test(n));
|
|
6765
|
+
const hasSearch = names.some((n) => /(^|_)WebSearch$|search/i.test(n) && !/WebFetch|browser/i.test(n));
|
|
6766
|
+
const notes = [];
|
|
6767
|
+
if (hasFetch) notes.push("WebFetch retrieves ONE specific URL you are given \u2014 it is not a search engine.");
|
|
6768
|
+
if (hasBrowser) notes.push("The browser tools drive a real browser: you CAN open a site and, if needed, navigate to a search engine and search there \u2014 but it is manual and takes a moment, not an instant lookup.");
|
|
6769
|
+
else if (!hasSearch && hasFetch) notes.push('You have no general web-search tool, so for an instant "search the web" you can only fetch a URL they provide.');
|
|
6770
|
+
const webNote = notes.length ? " NOTE: " + notes.join(" ") : "";
|
|
6771
|
+
return `Tools your background worker (Act) can actually use: ${names.join(", ")}. Read each name literally and match the request to a SPECIFIC tool; if none fits, you do NOT have that ability \u2014 say so honestly.` + webNote + mcpNote;
|
|
6772
|
+
}
|
|
6773
|
+
case "time":
|
|
6774
|
+
return (/* @__PURE__ */ new Date()).toString();
|
|
6775
|
+
case "branch": {
|
|
6776
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
6777
|
+
const head = (await fs.readFile(".git/HEAD")).trim();
|
|
6778
|
+
return head.startsWith("ref: refs/heads/") ? `branch: ${head.slice("ref: refs/heads/".length)}` : `detached HEAD at ${head.slice(0, 12)}`;
|
|
6779
|
+
}
|
|
6780
|
+
case "ls": {
|
|
6781
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
6782
|
+
const p = String(path ?? ".");
|
|
6783
|
+
try {
|
|
6784
|
+
const names = await fs.readDir(p);
|
|
6785
|
+
return names.slice(0, 50).join("\n") + (names.length > 50 ? `
|
|
6786
|
+
\u2026 (+${names.length - 50} more)` : "");
|
|
6787
|
+
} catch {
|
|
6788
|
+
const names = await fs.readDir(".").catch(() => []);
|
|
6789
|
+
return `'${p}' not found here \u2014 you are likely already inside it. Current directory listing:
|
|
6790
|
+
` + names.slice(0, 50).join("\n") + (names.length > 50 ? `
|
|
6791
|
+
\u2026 (+${names.length - 50} more)` : "");
|
|
6792
|
+
}
|
|
6793
|
+
}
|
|
6794
|
+
case "file": {
|
|
6795
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
6796
|
+
if (!path) return "file lookup needs a path";
|
|
6797
|
+
try {
|
|
6798
|
+
const text = await fs.readFile(String(path));
|
|
6799
|
+
return text.length > CAP2 ? text.slice(0, CAP2) + `
|
|
6800
|
+
\u2026 (truncated \u2014 ${text.length} chars total; Act for the full file)` : text;
|
|
6801
|
+
} catch {
|
|
6802
|
+
const names = await fs.readDir(".").catch(() => []);
|
|
6803
|
+
return `'${path}' not found. Current directory contains:
|
|
6804
|
+
` + names.slice(0, 50).join("\n");
|
|
6805
|
+
}
|
|
6806
|
+
}
|
|
6807
|
+
default:
|
|
6808
|
+
return `unknown lookup '${what}'`;
|
|
6809
|
+
}
|
|
6810
|
+
} catch (e) {
|
|
6811
|
+
return `lookup failed: ${e?.message ?? e}`;
|
|
6812
|
+
}
|
|
6291
6813
|
}
|
|
6292
|
-
|
|
6293
|
-
return;
|
|
6294
|
-
}
|
|
6295
|
-
if (!this.options.utteranceMergeMs || this.words(this.pendingUtt).length >= 4) return this.flushUtterance();
|
|
6296
|
-
this.pendingTimer = setTimeout(() => this.flushUtterance(), this.options.utteranceMergeMs);
|
|
6814
|
+
};
|
|
6297
6815
|
}
|
|
6298
|
-
|
|
6299
|
-
|
|
6300
|
-
|
|
6301
|
-
|
|
6302
|
-
|
|
6303
|
-
|
|
6304
|
-
|
|
6305
|
-
|
|
6306
|
-
|
|
6307
|
-
|
|
6308
|
-
|
|
6309
|
-
|
|
6816
|
+
answerTaskTool() {
|
|
6817
|
+
return {
|
|
6818
|
+
name: "AnswerTask",
|
|
6819
|
+
description: `Relay the user's answer to a pending question from a background task (the "[task <id> asks]" events). Pass the id from the event and the user's answer.`,
|
|
6820
|
+
parameters: {
|
|
6821
|
+
type: "object",
|
|
6822
|
+
required: ["id", "answer"],
|
|
6823
|
+
properties: { id: { type: "string" }, answer: { type: "string", description: "the user's answer, verbatim or faithfully summarized" } }
|
|
6824
|
+
},
|
|
6825
|
+
run: async ({ id, answer }) => {
|
|
6826
|
+
const ask = this.pendingAsks.get(String(id));
|
|
6827
|
+
if (!ask) return `No pending question for '${id}' \u2014 it may have been answered already or timed out.`;
|
|
6828
|
+
ask.resolve(String(answer ?? ""));
|
|
6829
|
+
return `Answer relayed \u2014 task ${id} resumes.`;
|
|
6830
|
+
}
|
|
6831
|
+
};
|
|
6310
6832
|
}
|
|
6311
|
-
|
|
6312
|
-
return
|
|
6833
|
+
holdTool() {
|
|
6834
|
+
return {
|
|
6835
|
+
name: "Hold",
|
|
6836
|
+
description: 'The user seems mid-thought \u2014 hold the turn (stay listening) instead of answering. Optionally pass a short filler ("mhm", "go on") to speak while waiting. Use when the message sounds incomplete, trailing off, or like they paused to think.',
|
|
6837
|
+
parameters: {
|
|
6838
|
+
type: "object",
|
|
6839
|
+
properties: {
|
|
6840
|
+
filler: { type: "string", description: 'optional short filler to speak ("mhm", "go on", "mm-hm")' }
|
|
6841
|
+
}
|
|
6842
|
+
},
|
|
6843
|
+
run: async ({ filler }) => {
|
|
6844
|
+
this.heldThisTurn = true;
|
|
6845
|
+
this.notify("hold_filler", filler ? String(filler) : "");
|
|
6846
|
+
return "Holding \u2014 listening for the rest of the user's thought. Do not respond further this turn.";
|
|
6847
|
+
}
|
|
6848
|
+
};
|
|
6313
6849
|
}
|
|
6314
|
-
|
|
6315
|
-
|
|
6316
|
-
|
|
6317
|
-
|
|
6318
|
-
|
|
6319
|
-
this.
|
|
6320
|
-
|
|
6321
|
-
}, this.options.overlapResumeMs);
|
|
6850
|
+
cancelTaskTool() {
|
|
6851
|
+
return {
|
|
6852
|
+
name: "CancelTask",
|
|
6853
|
+
description: "Cancel a running background task by id.",
|
|
6854
|
+
parameters: { type: "object", required: ["id"], properties: { id: { type: "string" } } },
|
|
6855
|
+
run: async ({ id }) => this.cancelTask(String(id))
|
|
6856
|
+
};
|
|
6322
6857
|
}
|
|
6323
|
-
|
|
6324
|
-
|
|
6325
|
-
|
|
6326
|
-
|
|
6327
|
-
|
|
6328
|
-
|
|
6329
|
-
|
|
6330
|
-
|
|
6858
|
+
};
|
|
6859
|
+
|
|
6860
|
+
// src/mcp.ts
|
|
6861
|
+
function toResult(result) {
|
|
6862
|
+
if (result == null) return { text: "" };
|
|
6863
|
+
if (typeof result === "string") return { text: result };
|
|
6864
|
+
const content = result.content;
|
|
6865
|
+
if (Array.isArray(content)) {
|
|
6866
|
+
const texts = [];
|
|
6867
|
+
const images = [];
|
|
6868
|
+
for (const c of content) {
|
|
6869
|
+
if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
|
|
6870
|
+
images.push({ mimeType: c.mimeType, data: c.data });
|
|
6871
|
+
} else if (typeof c?.text === "string") {
|
|
6872
|
+
texts.push(c.text);
|
|
6873
|
+
} else {
|
|
6874
|
+
texts.push(JSON.stringify(c));
|
|
6875
|
+
}
|
|
6331
6876
|
}
|
|
6332
|
-
|
|
6333
|
-
|
|
6334
|
-
this.gatePassTimes = [];
|
|
6877
|
+
const text = texts.join("\n");
|
|
6878
|
+
if (text || images.length) return { text, ...images.length ? { images } : {} };
|
|
6335
6879
|
}
|
|
6336
|
-
|
|
6337
|
-
|
|
6338
|
-
|
|
6339
|
-
|
|
6340
|
-
|
|
6341
|
-
|
|
6342
|
-
|
|
6343
|
-
|
|
6344
|
-
|
|
6345
|
-
|
|
6346
|
-
this.gatePassTimes = [];
|
|
6347
|
-
this.pausedAt = t;
|
|
6348
|
-
this.player.pause();
|
|
6349
|
-
this.armResume();
|
|
6350
|
-
return;
|
|
6880
|
+
return { text: JSON.stringify(result) };
|
|
6881
|
+
}
|
|
6882
|
+
function mcpToolToAgentTool(spec, callTool, prefix = "mcp__") {
|
|
6883
|
+
return {
|
|
6884
|
+
name: `${prefix}${spec.name}`,
|
|
6885
|
+
description: spec.description ?? `MCP tool ${spec.name}`,
|
|
6886
|
+
parameters: spec.inputSchema ?? { type: "object", properties: {} },
|
|
6887
|
+
async run(args, _ctx) {
|
|
6888
|
+
const r = toResult(await callTool(spec.name, args ?? {}));
|
|
6889
|
+
return r.images?.length ? r : r.text;
|
|
6351
6890
|
}
|
|
6352
|
-
|
|
6353
|
-
|
|
6354
|
-
|
|
6355
|
-
|
|
6891
|
+
};
|
|
6892
|
+
}
|
|
6893
|
+
function mcpToolsToAgentTools(specs, callTool, prefix = "mcp__", filter) {
|
|
6894
|
+
return (filter ? specs.filter(filter) : specs).map((s) => mcpToolToAgentTool(s, callTool, prefix));
|
|
6895
|
+
}
|
|
6896
|
+
function describeSpec(s) {
|
|
6897
|
+
const schema = s.inputSchema ? `
|
|
6898
|
+
args: ${JSON.stringify(s.inputSchema)}` : "";
|
|
6899
|
+
return `${s.name} \u2014 ${s.description ?? "(no description)"}${schema}`;
|
|
6900
|
+
}
|
|
6901
|
+
function makeMcpToolSearch(specs, callTool, options = {}) {
|
|
6902
|
+
const maxResults = options.maxResults ?? 10;
|
|
6903
|
+
const byName = new Map(specs.map((s) => [s.name, s]));
|
|
6904
|
+
const catalogLine = `${specs.length} MCP tool(s) available \u2014 search by keyword, then call by exact name.`;
|
|
6905
|
+
const searchTool = {
|
|
6906
|
+
name: "ToolSearch",
|
|
6907
|
+
description: `Search the available MCP tools by keyword (${catalogLine}). Returns matching tool names + their argument schemas; call one with \`McpCall\`.`,
|
|
6908
|
+
parameters: { type: "object", required: ["query"], properties: { query: { type: "string", description: "keywords to match against tool name + description" } } },
|
|
6909
|
+
async run({ query }) {
|
|
6910
|
+
const q2 = String(query ?? "").trim();
|
|
6911
|
+
if (!q2) return catalogLine;
|
|
6912
|
+
const { kept } = topByRelevance(specs, q2, (s) => `${s.name} ${s.description ?? ""}`, maxResults);
|
|
6913
|
+
if (!kept.length) return `(no MCP tool matches "${q2}" \u2014 try broader keywords)`;
|
|
6914
|
+
return kept.map(describeSpec).join("\n");
|
|
6356
6915
|
}
|
|
6357
|
-
|
|
6358
|
-
|
|
6359
|
-
|
|
6916
|
+
};
|
|
6917
|
+
const callMcpTool = {
|
|
6918
|
+
name: "McpCall",
|
|
6919
|
+
description: "Call an MCP tool discovered via `ToolSearch`, by its exact name. Pass its arguments as `args`.",
|
|
6920
|
+
parameters: {
|
|
6921
|
+
type: "object",
|
|
6922
|
+
required: ["name"],
|
|
6923
|
+
properties: {
|
|
6924
|
+
name: { type: "string", description: "exact tool name from ToolSearch" },
|
|
6925
|
+
args: { type: "object", description: "arguments object for the tool (per its schema)" }
|
|
6926
|
+
}
|
|
6927
|
+
},
|
|
6928
|
+
async run({ name, args }) {
|
|
6929
|
+
const n = String(name ?? "");
|
|
6930
|
+
if (!byName.has(n)) return `Error: unknown MCP tool '${n}'. Use ToolSearch to find valid names.`;
|
|
6931
|
+
const r = toResult(await callTool(n, args ?? {}));
|
|
6932
|
+
return r.images?.length ? r : r.text;
|
|
6360
6933
|
}
|
|
6361
|
-
|
|
6362
|
-
|
|
6363
|
-
|
|
6364
|
-
|
|
6365
|
-
|
|
6366
|
-
|
|
6367
|
-
|
|
6368
|
-
|
|
6934
|
+
};
|
|
6935
|
+
return [searchTool, callMcpTool];
|
|
6936
|
+
}
|
|
6937
|
+
function buildMcpCatalog(servers) {
|
|
6938
|
+
const specs = [];
|
|
6939
|
+
const routes = /* @__PURE__ */ new Map();
|
|
6940
|
+
for (const m of servers) {
|
|
6941
|
+
for (const s of m.specs) {
|
|
6942
|
+
const base = `mcp__${m.name}__${s.name}`.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
|
|
6943
|
+
let display = base;
|
|
6944
|
+
for (let i = 2; routes.has(display); i++) display = `${base.slice(0, 128 - String(i).length - 1)}_${i}`;
|
|
6945
|
+
specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
|
|
6946
|
+
routes.set(display, { server: m.name, rawName: s.name });
|
|
6369
6947
|
}
|
|
6370
6948
|
}
|
|
6371
|
-
};
|
|
6949
|
+
return { specs, routes };
|
|
6950
|
+
}
|
|
6951
|
+
function searchOverCatalog(servers, specs, routes, resolve4, options) {
|
|
6952
|
+
const tools = specs.length ? makeMcpToolSearch(specs, (name, args) => {
|
|
6953
|
+
const r = routes.get(name);
|
|
6954
|
+
if (!r) throw new Error(`unknown MCP tool '${name}' \u2014 use ToolSearch to find valid names`);
|
|
6955
|
+
return resolve4(r.server, r.rawName, args ?? {});
|
|
6956
|
+
}, options) : [];
|
|
6957
|
+
return { tools, serverNames: servers, toolCount: specs.length };
|
|
6958
|
+
}
|
|
6959
|
+
function makeMcpToolSearchFromMounted(mounted, options) {
|
|
6960
|
+
const { specs, routes } = buildMcpCatalog(mounted);
|
|
6961
|
+
const byName = new Map(mounted.map((m) => [m.name, m]));
|
|
6962
|
+
return searchOverCatalog(mounted.map((m) => m.name), specs, routes, (server, rawName, args) => byName.get(server).client.callTool(rawName, args), options);
|
|
6963
|
+
}
|
|
6964
|
+
|
|
6965
|
+
// src/index.ts
|
|
6966
|
+
init_logging();
|
|
6372
6967
|
|
|
6373
6968
|
// src/voice/soniox.ts
|
|
6374
6969
|
init_logging();
|
|
@@ -6382,7 +6977,7 @@ async function resolveAuth(auth) {
|
|
|
6382
6977
|
|
|
6383
6978
|
// src/voice/soniox.ts
|
|
6384
6979
|
var log11 = forComponent("SonioxSTT");
|
|
6385
|
-
var
|
|
6980
|
+
var now = () => performance.now();
|
|
6386
6981
|
var SonioxSTTOptions = class {
|
|
6387
6982
|
auth = "";
|
|
6388
6983
|
source;
|
|
@@ -6411,6 +7006,20 @@ var SonioxSTT = class {
|
|
|
6411
7006
|
* loop). The host tears voice down instead of spinning forever. */
|
|
6412
7007
|
onFatal = () => {
|
|
6413
7008
|
};
|
|
7009
|
+
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
7010
|
+
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
7011
|
+
onDiag = () => {
|
|
7012
|
+
};
|
|
7013
|
+
diagOn = true;
|
|
7014
|
+
diag(kind, fields) {
|
|
7015
|
+
if (!this.diagOn) return;
|
|
7016
|
+
try {
|
|
7017
|
+
this.onDiag({ t: now(), kind, ...fields });
|
|
7018
|
+
} catch (e) {
|
|
7019
|
+
this.diagOn = false;
|
|
7020
|
+
log11.debug(`onDiag threw \u2014 STT diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
7021
|
+
}
|
|
7022
|
+
}
|
|
6414
7023
|
lastChunkAt = 0;
|
|
6415
7024
|
// timestamp of the most recent mic chunk (0 = none yet)
|
|
6416
7025
|
startedChunksAt = 0;
|
|
@@ -6451,8 +7060,12 @@ var SonioxSTT = class {
|
|
|
6451
7060
|
this.ws.onclose = (ev) => {
|
|
6452
7061
|
if (this.stopped) return;
|
|
6453
7062
|
log11.warn(`soniox ws closed (${ev.code} ${ev.reason || ""}) \u2014 reconnecting`);
|
|
7063
|
+
this.diag("stt_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: true });
|
|
6454
7064
|
this.reset();
|
|
6455
|
-
this.connectWs().catch((e) =>
|
|
7065
|
+
this.connectWs().catch((e) => {
|
|
7066
|
+
log11.error(`soniox reconnect failed: ${e.message}`);
|
|
7067
|
+
this.diag("stt_reconnect_failed", { message: e.message });
|
|
7068
|
+
});
|
|
6456
7069
|
};
|
|
6457
7070
|
}
|
|
6458
7071
|
async start() {
|
|
@@ -6461,20 +7074,21 @@ var SonioxSTT = class {
|
|
|
6461
7074
|
this.sourceStarted = true;
|
|
6462
7075
|
this.endpointTimer = setInterval(() => {
|
|
6463
7076
|
const combined = (this.finalText + this.partialText).trim();
|
|
6464
|
-
if (!combined ||
|
|
6465
|
-
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(
|
|
7077
|
+
if (!combined || now() - this.lastChangeAt < this.options.silenceEndpointMs) return;
|
|
7078
|
+
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192silence-endpoint, "${combined.slice(0, 60)}"`);
|
|
6466
7079
|
this.reset();
|
|
6467
|
-
this.onUtterance(combined,
|
|
7080
|
+
this.onUtterance(combined, now());
|
|
6468
7081
|
}, 120);
|
|
6469
7082
|
this.endpointTimer.unref?.();
|
|
6470
|
-
this.startedChunksAt =
|
|
7083
|
+
this.startedChunksAt = now();
|
|
6471
7084
|
const noAudioMs = this.options.noAudioTimeoutMs;
|
|
6472
7085
|
if (noAudioMs > 0) {
|
|
6473
7086
|
this.noAudioTimer = setInterval(() => {
|
|
6474
7087
|
if (this.stopped) return;
|
|
6475
7088
|
const ref = this.lastChunkAt || this.startedChunksAt;
|
|
6476
|
-
if (
|
|
7089
|
+
if (now() - ref > noAudioMs) {
|
|
6477
7090
|
log11.error(`stt: no mic audio for >${Math.round(noAudioMs / 1e3)}s \u2014 capture device stopped delivering`);
|
|
7091
|
+
this.diag("stt_watchdog_fatal", { noAudioMs });
|
|
6478
7092
|
this.onFatal("microphone stopped delivering audio (try a different input device, e.g. AirPods, or check System Settings \u2192 Sound \u2192 Input)");
|
|
6479
7093
|
this.stop();
|
|
6480
7094
|
}
|
|
@@ -6482,7 +7096,7 @@ var SonioxSTT = class {
|
|
|
6482
7096
|
this.noAudioTimer.unref?.();
|
|
6483
7097
|
}
|
|
6484
7098
|
await this.options.source.start((chunk) => {
|
|
6485
|
-
this.lastChunkAt =
|
|
7099
|
+
this.lastChunkAt = now();
|
|
6486
7100
|
let sum = 0;
|
|
6487
7101
|
const view = new DataView(chunk.buffer, chunk.byteOffset, chunk.byteLength);
|
|
6488
7102
|
for (let i = 0; i + 1 < chunk.byteLength; i += 2) {
|
|
@@ -6494,7 +7108,10 @@ var SonioxSTT = class {
|
|
|
6494
7108
|
});
|
|
6495
7109
|
}
|
|
6496
7110
|
handle(m) {
|
|
6497
|
-
if (m.error_message)
|
|
7111
|
+
if (m.error_message) {
|
|
7112
|
+
this.diag("stt_error", { message: String(m.error_message), code: m.error_code });
|
|
7113
|
+
return log11.error(`soniox: ${m.error_message}`);
|
|
7114
|
+
}
|
|
6498
7115
|
let endpoint = false;
|
|
6499
7116
|
for (const t of m.tokens ?? []) {
|
|
6500
7117
|
if (t.text === "<end>") endpoint = true;
|
|
@@ -6504,15 +7121,15 @@ var SonioxSTT = class {
|
|
|
6504
7121
|
const combined = this.finalText + this.partialText;
|
|
6505
7122
|
if (combined !== this.lastCombined) {
|
|
6506
7123
|
this.lastCombined = combined;
|
|
6507
|
-
this.lastChangeAt =
|
|
6508
|
-
if (!this.firstTokenAt && combined.trim()) this.firstTokenAt =
|
|
7124
|
+
this.lastChangeAt = now();
|
|
7125
|
+
if (!this.firstTokenAt && combined.trim()) this.firstTokenAt = now();
|
|
6509
7126
|
}
|
|
6510
7127
|
this.onPartial(combined);
|
|
6511
7128
|
if (endpoint && this.finalText.trim()) {
|
|
6512
7129
|
const utterance = this.finalText.trim();
|
|
6513
|
-
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(
|
|
7130
|
+
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192endpoint, "${utterance.slice(0, 60)}"`);
|
|
6514
7131
|
this.reset();
|
|
6515
|
-
this.onUtterance(utterance,
|
|
7132
|
+
this.onUtterance(utterance, now());
|
|
6516
7133
|
}
|
|
6517
7134
|
}
|
|
6518
7135
|
reset() {
|
|
@@ -6534,7 +7151,7 @@ var SonioxSTT = class {
|
|
|
6534
7151
|
// src/voice/cartesia.ts
|
|
6535
7152
|
init_logging();
|
|
6536
7153
|
var log12 = forComponent("CartesiaTTS");
|
|
6537
|
-
var
|
|
7154
|
+
var now2 = () => performance.now();
|
|
6538
7155
|
var CartesiaTTSOptions = class {
|
|
6539
7156
|
auth = "";
|
|
6540
7157
|
voiceId = "";
|
|
@@ -6551,6 +7168,20 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6551
7168
|
};
|
|
6552
7169
|
onDone = () => {
|
|
6553
7170
|
};
|
|
7171
|
+
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
7172
|
+
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
7173
|
+
onDiag = () => {
|
|
7174
|
+
};
|
|
7175
|
+
diagOn = true;
|
|
7176
|
+
diag(kind, fields) {
|
|
7177
|
+
if (!this.diagOn) return;
|
|
7178
|
+
try {
|
|
7179
|
+
this.onDiag({ t: now2(), kind, ...fields });
|
|
7180
|
+
} catch (e) {
|
|
7181
|
+
this.diagOn = false;
|
|
7182
|
+
log12.debug(`onDiag threw \u2014 TTS diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
7183
|
+
}
|
|
7184
|
+
}
|
|
6554
7185
|
firstAudioAt = 0;
|
|
6555
7186
|
/** Circuit breaker: consecutive error count + down flag. */
|
|
6556
7187
|
consecutiveErrors = 0;
|
|
@@ -6584,8 +7215,12 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6584
7215
|
});
|
|
6585
7216
|
this.ws.onclose = (ev) => {
|
|
6586
7217
|
log12.warn(`cartesia ws closed (${ev.code} ${ev.reason || ""})`);
|
|
7218
|
+
this.diag("tts_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: !this.closed });
|
|
6587
7219
|
if (!this.closed) {
|
|
6588
|
-
this.connecting = this.doConnect().catch((e) =>
|
|
7220
|
+
this.connecting = this.doConnect().catch((e) => {
|
|
7221
|
+
log12.error(`cartesia reconnect failed: ${e.message}`);
|
|
7222
|
+
this.diag("tts_reconnect_failed", { message: e.message });
|
|
7223
|
+
});
|
|
6589
7224
|
}
|
|
6590
7225
|
};
|
|
6591
7226
|
this.ws.onmessage = (ev) => {
|
|
@@ -6594,7 +7229,7 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6594
7229
|
if (m.type === "chunk" && m.data) {
|
|
6595
7230
|
this.consecutiveErrors = 0;
|
|
6596
7231
|
this.markRecovered();
|
|
6597
|
-
if (!this.firstAudioAt) this.firstAudioAt =
|
|
7232
|
+
if (!this.firstAudioAt) this.firstAudioAt = now2();
|
|
6598
7233
|
this.onAudio(base64ToBytes(m.data));
|
|
6599
7234
|
} else if (m.type === "done") {
|
|
6600
7235
|
this.consecutiveErrors = 0;
|
|
@@ -6603,15 +7238,17 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6603
7238
|
} else if (m.type === "error") {
|
|
6604
7239
|
if (/already been cancelled|does not exist/.test(m.message || "")) return;
|
|
6605
7240
|
this.consecutiveErrors++;
|
|
7241
|
+
this.diag("tts_error", { message: String(m.message || ""), code: m.status_code, contextId: m.context_id, consecutive: this.consecutiveErrors });
|
|
6606
7242
|
if (!this.down && this.consecutiveErrors >= _CartesiaTTS.CB_THRESHOLD) {
|
|
6607
7243
|
this.down = true;
|
|
6608
|
-
this.downAt =
|
|
7244
|
+
this.downAt = now2();
|
|
6609
7245
|
this.consecutiveOk = 0;
|
|
6610
7246
|
log12.warn(`TTS circuit breaker open \u2014 ${this.consecutiveErrors} consecutive errors, switching to text-only`);
|
|
7247
|
+
this.diag("tts_breaker_open", { errors: this.consecutiveErrors });
|
|
6611
7248
|
this.onDone();
|
|
6612
7249
|
this.startProbe();
|
|
6613
7250
|
} else if (!this.down) {
|
|
6614
|
-
log12.warn(`cartesia: ${JSON.stringify(m)}`);
|
|
7251
|
+
(/No valid transcripts/i.test(m.message || "") ? log12.debug : log12.warn)(`cartesia: ${JSON.stringify(m)}`);
|
|
6615
7252
|
}
|
|
6616
7253
|
}
|
|
6617
7254
|
};
|
|
@@ -6624,7 +7261,8 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6624
7261
|
this.down = false;
|
|
6625
7262
|
this.consecutiveOk = 0;
|
|
6626
7263
|
this.stopProbe();
|
|
6627
|
-
const downMs = this.downAt ?
|
|
7264
|
+
const downMs = this.downAt ? now2() - this.downAt : 0;
|
|
7265
|
+
this.diag("tts_breaker_close", { downMs: Math.round(downMs) });
|
|
6628
7266
|
(downMs < 2e3 ? log12.debug : log12.info)(`TTS recovered${downMs ? ` (down ${downMs}ms)` : ""}`);
|
|
6629
7267
|
}
|
|
6630
7268
|
/** Ensure the WS is open before sending — reconnects if idle-closed. */
|
|
@@ -6632,6 +7270,15 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6632
7270
|
if (this.connecting) await this.connecting;
|
|
6633
7271
|
if (this.ws?.readyState !== WebSocket.OPEN) await this.connect();
|
|
6634
7272
|
}
|
|
7273
|
+
/** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
|
|
7274
|
+
* a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
|
|
7275
|
+
* The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
|
|
7276
|
+
* newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
|
|
7277
|
+
warmup() {
|
|
7278
|
+
if (this.down) return;
|
|
7279
|
+
this.newContext();
|
|
7280
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
|
|
7281
|
+
}
|
|
6635
7282
|
newContext() {
|
|
6636
7283
|
this.ctxId = `ctx-${++this.ctxSeq}`;
|
|
6637
7284
|
this.firstAudioAt = 0;
|
|
@@ -6672,7 +7319,7 @@ var CartesiaTTS = class _CartesiaTTS {
|
|
|
6672
7319
|
}
|
|
6673
7320
|
this.consecutiveErrors = 0;
|
|
6674
7321
|
this.newContext();
|
|
6675
|
-
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame(".", false));
|
|
7322
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
|
|
6676
7323
|
}, _CartesiaTTS.CB_PROBE_MS);
|
|
6677
7324
|
this.probeTimer.unref?.();
|
|
6678
7325
|
}
|
|
@@ -7488,9 +8135,9 @@ Reference files in them by their mount path (the left side).`;
|
|
|
7488
8135
|
return po ? { providerOptions: po } : {};
|
|
7489
8136
|
})(),
|
|
7490
8137
|
...(() => {
|
|
7491
|
-
const
|
|
8138
|
+
const now4 = /* @__PURE__ */ new Date();
|
|
7492
8139
|
const platformNames = { darwin: "macOS", linux: "Linux", win32: "Windows" };
|
|
7493
|
-
const envNote = `Current date: ${
|
|
8140
|
+
const envNote = `Current date: ${now4.toLocaleDateString("en-CA")} ${now4.toLocaleTimeString("en-GB", { hour: "2-digit", minute: "2-digit" })} (${Intl.DateTimeFormat().resolvedOptions().timeZone})
|
|
7494
8141
|
Platform: ${platformNames[platform()] ?? platform()} ${arch()} (${release()})
|
|
7495
8142
|
User: ${userInfo().username}
|
|
7496
8143
|
Shell: ${process.env.SHELL ?? "unknown"}`;
|
|
@@ -7584,7 +8231,7 @@ import { homedir as homedir3 } from "os";
|
|
|
7584
8231
|
import { dirname as dirname3, join as join6 } from "path";
|
|
7585
8232
|
import { fileURLToPath } from "url";
|
|
7586
8233
|
var log17 = forComponent("VoiceIO");
|
|
7587
|
-
var
|
|
8234
|
+
var now3 = () => performance.now();
|
|
7588
8235
|
var Player = class {
|
|
7589
8236
|
proc = null;
|
|
7590
8237
|
bytesWritten = 0;
|
|
@@ -7605,19 +8252,19 @@ var Player = class {
|
|
|
7605
8252
|
}
|
|
7606
8253
|
write(chunk) {
|
|
7607
8254
|
if (!this.proc) this.markTurn();
|
|
7608
|
-
if (!this.startedAt) this.startedAt =
|
|
8255
|
+
if (!this.startedAt) this.startedAt = now3();
|
|
7609
8256
|
this.bytesWritten += chunk.length;
|
|
7610
8257
|
this.proc.stdin.write(chunk);
|
|
7611
8258
|
}
|
|
7612
8259
|
/** ms of audio actually played so far this turn */
|
|
7613
8260
|
playedMs() {
|
|
7614
|
-
return this.startedAt ?
|
|
8261
|
+
return this.startedAt ? now3() - this.startedAt : 0;
|
|
7615
8262
|
}
|
|
7616
8263
|
/** estimated ms until queued audio finishes playing */
|
|
7617
8264
|
drainMs() {
|
|
7618
8265
|
if (!this.startedAt) return 0;
|
|
7619
8266
|
const queuedMs = this.bytesWritten / (TTS_SAMPLE_RATE * 2) * 1e3;
|
|
7620
|
-
return Math.max(0, queuedMs - (
|
|
8267
|
+
return Math.max(0, queuedMs - (now3() - this.startedAt));
|
|
7621
8268
|
}
|
|
7622
8269
|
kill() {
|
|
7623
8270
|
this.proc?.kill("SIGKILL");
|
|
@@ -7864,17 +8511,17 @@ var AecDuplexAudio = class {
|
|
|
7864
8511
|
this.pausedAccum = 0;
|
|
7865
8512
|
}
|
|
7866
8513
|
write(chunk) {
|
|
7867
|
-
if (!this.startedAt) this.startedAt =
|
|
8514
|
+
if (!this.startedAt) this.startedAt = now3();
|
|
7868
8515
|
this.bytesWritten += chunk.length;
|
|
7869
8516
|
this.frame(chunk);
|
|
7870
8517
|
}
|
|
7871
8518
|
playedMs() {
|
|
7872
|
-
return this.startedAt ?
|
|
8519
|
+
return this.startedAt ? now3() - this.startedAt - this.pausedMs() : 0;
|
|
7873
8520
|
}
|
|
7874
8521
|
drainMs() {
|
|
7875
8522
|
if (!this.startedAt) return 0;
|
|
7876
8523
|
const queuedMs = this.bytesWritten / (TTS_SAMPLE_RATE * 2) * 1e3;
|
|
7877
|
-
return Math.max(0, queuedMs - (
|
|
8524
|
+
return Math.max(0, queuedMs - (now3() - this.startedAt - this.pausedMs()));
|
|
7878
8525
|
}
|
|
7879
8526
|
/** barge-in: silence NOW (in-band flush) — the capture side keeps running */
|
|
7880
8527
|
kill() {
|
|
@@ -7896,18 +8543,18 @@ var AecDuplexAudio = class {
|
|
|
7896
8543
|
}
|
|
7897
8544
|
pause() {
|
|
7898
8545
|
if (this.pausedSince) return;
|
|
7899
|
-
this.pausedSince =
|
|
8546
|
+
this.pausedSince = now3();
|
|
7900
8547
|
this.ctl(4294967295);
|
|
7901
8548
|
}
|
|
7902
8549
|
resume() {
|
|
7903
8550
|
if (!this.pausedSince) return;
|
|
7904
|
-
this.pausedAccum +=
|
|
8551
|
+
this.pausedAccum += now3() - this.pausedSince;
|
|
7905
8552
|
this.pausedSince = 0;
|
|
7906
8553
|
this.ctl(4294967294);
|
|
7907
8554
|
}
|
|
7908
8555
|
/** total paused time this turn — excluded from played/drain math (the tape held still) */
|
|
7909
8556
|
pausedMs() {
|
|
7910
|
-
return this.pausedAccum + (this.pausedSince ?
|
|
8557
|
+
return this.pausedAccum + (this.pausedSince ? now3() - this.pausedSince : 0);
|
|
7911
8558
|
}
|
|
7912
8559
|
/** TEST HARNESS: queue simulated user speech (s16le 16k mono) — the helper mixes it into the real
|
|
7913
8560
|
* mic stream pre-gate, so the full live pipeline runs while a script plays the human. */
|
|
@@ -7931,8 +8578,25 @@ var VoiceIOOptions = class extends VoiceEngineOptions {
|
|
|
7931
8578
|
cartesiaVoiceId = process.env.CARTESIA_VOICE_ID ?? "";
|
|
7932
8579
|
emotions = process.env.VOICE_EMOTIONS !== "0";
|
|
7933
8580
|
// Cartesia inline emotion tags (sonic-3)
|
|
8581
|
+
/** CLI voice default: mask a slow reflex TTFT with a one-shot varied micro-ack (a FIXED per-turn
|
|
8582
|
+
* ack was rejected as robotic — this speaks only when the reflex is actually slow). 600ms, not
|
|
8583
|
+
* 400: the ack is a SAFETY NET, not the voice — at 400 it fired often enough to sound synthetic
|
|
8584
|
+
* (live: "started with 'hmm' too many times"); the reflex itself is the natural instant response.
|
|
8585
|
+
* Engine default stays 0 (off); VOICE_ADAPTIVE_ACK_MS overrides, 0 disables. */
|
|
8586
|
+
adaptiveAckMs = Number(process.env.VOICE_ADAPTIVE_ACK_MS ?? 600);
|
|
7934
8587
|
showEmotions = process.env.VOICE_SHOW_EMOTIONS === "1";
|
|
7935
8588
|
// surface the tags in the on-screen echo (debug; opt-in)
|
|
8589
|
+
/** Speculative reflex start (reflex starts on a stable partial, output held until the final
|
|
8590
|
+
* confirms; see VoiceEngineOptions.speculativeMs). Default OFF: live A/B showed real speech
|
|
8591
|
+
* fires mostly on mid-phrase pauses → correct aborts that cost latency + billing; the win only
|
|
8592
|
+
* materializes when the stable partial IS the complete thought. Opt in with VOICE_SPECULATIVE=1
|
|
8593
|
+
* (250ms) or VOICE_SPECULATIVE_MS=<n> while gathering real-mic data. */
|
|
8594
|
+
speculativeMs = process.env.VOICE_SPECULATIVE === "1" ? Number(process.env.VOICE_SPECULATIVE_MS ?? 250) : Number(process.env.VOICE_SPECULATIVE_MS ?? 0);
|
|
8595
|
+
/** Agent-side backchanneling ("Mm-hm." blips at clause boundaries while the user speaks a long
|
|
8596
|
+
* turn; see VoiceEngineOptions.backchannelMs). Default OFF; opt in with VOICE_BACKCHANNEL=1
|
|
8597
|
+
* (250ms stability window) or VOICE_BACKCHANNEL_MS=<n>. 250, not 350+: live, Soniox emits a
|
|
8598
|
+
* semantic <end> within ~300-400ms of a clause pause — a longer window never fires (verified). */
|
|
8599
|
+
backchannelMs = process.env.VOICE_BACKCHANNEL === "1" ? Number(process.env.VOICE_BACKCHANNEL_MS ?? 250) : Number(process.env.VOICE_BACKCHANNEL_MS ?? 0);
|
|
7936
8600
|
};
|
|
7937
8601
|
var VoiceIO = class extends VoiceEngine {
|
|
7938
8602
|
duplexSource;
|
|
@@ -7952,6 +8616,8 @@ var VoiceIO = class extends VoiceEngine {
|
|
|
7952
8616
|
});
|
|
7953
8617
|
this.duplexSource = duplex;
|
|
7954
8618
|
if (duplex) duplex.onDegrade = () => this.setBargeIn(false);
|
|
8619
|
+
if (this.stt instanceof SonioxSTT) this.stt.onDiag = ({ t: _t, kind, ...rest }) => this.diag(kind, rest);
|
|
8620
|
+
if (this.tts instanceof CartesiaTTS) this.tts.onDiag = ({ t: _t, kind, ...rest }) => this.diag(kind, rest);
|
|
7955
8621
|
}
|
|
7956
8622
|
/** Host hook for an unrecoverable audio failure — mic permission denied (duplex source) or no mic
|
|
7957
8623
|
* audio at all (STT watchdog). Routed to whichever can detect it. */
|
|
@@ -8018,6 +8684,63 @@ function fakeVoiceParts(uttFile) {
|
|
|
8018
8684
|
return { stt, tts, player };
|
|
8019
8685
|
}
|
|
8020
8686
|
|
|
8687
|
+
// src/voice/diagSink.ts
|
|
8688
|
+
init_logging();
|
|
8689
|
+
import { appendFile, mkdirSync as mkdirSync6 } from "fs";
|
|
8690
|
+
import { dirname as dirname4 } from "path";
|
|
8691
|
+
var log18 = forComponent("VoiceDiag");
|
|
8692
|
+
var JsonlDiagSink = class {
|
|
8693
|
+
constructor(path) {
|
|
8694
|
+
this.path = path;
|
|
8695
|
+
}
|
|
8696
|
+
path;
|
|
8697
|
+
queue = [];
|
|
8698
|
+
writing = false;
|
|
8699
|
+
disabled = false;
|
|
8700
|
+
dirReady = false;
|
|
8701
|
+
/** Append one event as a JSON line (buffered; returns immediately). */
|
|
8702
|
+
push(ev) {
|
|
8703
|
+
if (this.disabled) return;
|
|
8704
|
+
try {
|
|
8705
|
+
this.queue.push(JSON.stringify(ev));
|
|
8706
|
+
} catch {
|
|
8707
|
+
return;
|
|
8708
|
+
}
|
|
8709
|
+
this.flush();
|
|
8710
|
+
}
|
|
8711
|
+
flush() {
|
|
8712
|
+
if (this.writing || this.disabled || !this.queue.length) return;
|
|
8713
|
+
if (!this.dirReady) {
|
|
8714
|
+
try {
|
|
8715
|
+
mkdirSync6(dirname4(this.path), { recursive: true });
|
|
8716
|
+
this.dirReady = true;
|
|
8717
|
+
} catch (e) {
|
|
8718
|
+
return this.disable(e);
|
|
8719
|
+
}
|
|
8720
|
+
}
|
|
8721
|
+
this.writing = true;
|
|
8722
|
+
const lines = this.queue.splice(0).join("\n") + "\n";
|
|
8723
|
+
appendFile(this.path, lines, (e) => {
|
|
8724
|
+
this.writing = false;
|
|
8725
|
+
if (e) return this.disable(e);
|
|
8726
|
+
this.flush();
|
|
8727
|
+
});
|
|
8728
|
+
}
|
|
8729
|
+
disable(e) {
|
|
8730
|
+
this.disabled = true;
|
|
8731
|
+
this.queue.length = 0;
|
|
8732
|
+
log18.debug(`voice diag sink disabled (${this.path}): ${e instanceof Error ? e.message : e}`);
|
|
8733
|
+
}
|
|
8734
|
+
/** True once a failure permanently disabled the sink (tests/inspection). */
|
|
8735
|
+
get isDisabled() {
|
|
8736
|
+
return this.disabled;
|
|
8737
|
+
}
|
|
8738
|
+
/** Resolve when all queued lines are on disk (or the sink disabled) — tests/graceful shutdown. */
|
|
8739
|
+
async settle() {
|
|
8740
|
+
while (!this.disabled && (this.writing || this.queue.length)) await new Promise((r) => setTimeout(r, 5));
|
|
8741
|
+
}
|
|
8742
|
+
};
|
|
8743
|
+
|
|
8021
8744
|
// cli/config.ts
|
|
8022
8745
|
import { homedir as homedir4 } from "os";
|
|
8023
8746
|
import { existsSync as existsSync5, readFileSync as readFileSync5 } from "fs";
|
|
@@ -8076,7 +8799,7 @@ async function loadConfig(cwd) {
|
|
|
8076
8799
|
|
|
8077
8800
|
// cli/hooks-config.ts
|
|
8078
8801
|
import { spawnSync as spawnSync3 } from "child_process";
|
|
8079
|
-
var
|
|
8802
|
+
var log19 = forComponent("hooks");
|
|
8080
8803
|
var escapeRegex = (s) => s.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
8081
8804
|
function ruleMatches(rule, toolName) {
|
|
8082
8805
|
if (!rule.tool || rule.tool === "*") return true;
|
|
@@ -8093,7 +8816,7 @@ function runCmd(rule, env) {
|
|
|
8093
8816
|
});
|
|
8094
8817
|
return { code: r.status ?? 1, out: ((r.stdout ?? "") + (r.stderr ?? "")).trim() };
|
|
8095
8818
|
} catch (e) {
|
|
8096
|
-
|
|
8819
|
+
log19.debug(`hook command failed: ${rule.command}`, e);
|
|
8097
8820
|
return { code: 1, out: String(e?.message ?? e) };
|
|
8098
8821
|
}
|
|
8099
8822
|
}
|
|
@@ -8197,10 +8920,10 @@ function formatDiff(ops, opts = {}) {
|
|
|
8197
8920
|
}
|
|
8198
8921
|
|
|
8199
8922
|
// cli/session.ts
|
|
8200
|
-
import { existsSync as existsSync6, mkdirSync as
|
|
8923
|
+
import { existsSync as existsSync6, mkdirSync as mkdirSync7, readFileSync as readFileSync6, writeFileSync as writeFileSync4, readdirSync, renameSync, symlinkSync as symlinkSync2, unlinkSync, readlinkSync } from "fs";
|
|
8201
8924
|
import { homedir as homedir5 } from "os";
|
|
8202
8925
|
import { join as join8 } from "path";
|
|
8203
|
-
var
|
|
8926
|
+
var log20 = forComponent("session");
|
|
8204
8927
|
var globalDir = () => join8(homedir5(), ".agent", "sessions");
|
|
8205
8928
|
var SessionStore = class {
|
|
8206
8929
|
dir;
|
|
@@ -8208,8 +8931,8 @@ var SessionStore = class {
|
|
|
8208
8931
|
this.dir = join8(cwd, ".agent", "sessions");
|
|
8209
8932
|
}
|
|
8210
8933
|
/** Sortable, human-readable id: `YYYYMMDD-HHMMSS-<folder>`. */
|
|
8211
|
-
newId(
|
|
8212
|
-
const d = new Date(
|
|
8934
|
+
newId(now4 = Date.now(), cwd) {
|
|
8935
|
+
const d = new Date(now4);
|
|
8213
8936
|
const p = (n, w = 2) => String(n).padStart(w, "0");
|
|
8214
8937
|
const slug2 = (cwd ?? process.cwd()).split("/").pop()?.replace(/[^A-Za-z0-9_-]/g, "") || "session";
|
|
8215
8938
|
let id = `${d.getFullYear()}${p(d.getMonth() + 1)}${p(d.getDate())}-${p(d.getHours())}${p(d.getMinutes())}${p(d.getSeconds())}-${slug2}`;
|
|
@@ -8230,14 +8953,14 @@ var SessionStore = class {
|
|
|
8230
8953
|
}
|
|
8231
8954
|
save(data) {
|
|
8232
8955
|
if (!this.safeId(data.meta.id)) throw new Error(`unsafe session id: ${data.meta.id}`);
|
|
8233
|
-
if (!existsSync6(this.dir))
|
|
8956
|
+
if (!existsSync6(this.dir)) mkdirSync7(this.dir, { recursive: true });
|
|
8234
8957
|
const path = join8(this.dir, `${data.meta.id}.json`);
|
|
8235
8958
|
const tmp = `${path}.${process.pid}.tmp`;
|
|
8236
8959
|
writeFileSync4(tmp, JSON.stringify(data));
|
|
8237
8960
|
renameSync(tmp, path);
|
|
8238
8961
|
try {
|
|
8239
8962
|
const gd = globalDir();
|
|
8240
|
-
if (!existsSync6(gd))
|
|
8963
|
+
if (!existsSync6(gd)) mkdirSync7(gd, { recursive: true });
|
|
8241
8964
|
const link2 = join8(gd, `${data.meta.id}.json`);
|
|
8242
8965
|
if (!existsSync6(link2)) symlinkSync2(path, link2);
|
|
8243
8966
|
} catch {
|
|
@@ -8245,7 +8968,7 @@ var SessionStore = class {
|
|
|
8245
8968
|
}
|
|
8246
8969
|
load(id) {
|
|
8247
8970
|
if (!this.safeId(id)) {
|
|
8248
|
-
|
|
8971
|
+
log20.debug(`rejecting unsafe session id: ${id}`);
|
|
8249
8972
|
return void 0;
|
|
8250
8973
|
}
|
|
8251
8974
|
const path = join8(this.dir, `${id}.json`);
|
|
@@ -8253,7 +8976,7 @@ var SessionStore = class {
|
|
|
8253
8976
|
try {
|
|
8254
8977
|
return JSON.parse(readFileSync6(path, "utf8"));
|
|
8255
8978
|
} catch (e) {
|
|
8256
|
-
|
|
8979
|
+
log20.debug(`unreadable session ${id} \u2014 ignoring`, e);
|
|
8257
8980
|
return void 0;
|
|
8258
8981
|
}
|
|
8259
8982
|
}
|
|
@@ -8266,7 +8989,7 @@ var SessionStore = class {
|
|
|
8266
8989
|
try {
|
|
8267
8990
|
metas.push(JSON.parse(readFileSync6(join8(this.dir, f), "utf8")).meta);
|
|
8268
8991
|
} catch (e) {
|
|
8269
|
-
|
|
8992
|
+
log20.debug(`skipping unreadable session file ${f}`, e);
|
|
8270
8993
|
}
|
|
8271
8994
|
}
|
|
8272
8995
|
return metas.sort((a, b) => b.updated - a.updated);
|
|
@@ -8389,14 +9112,14 @@ var CheckpointStack = class {
|
|
|
8389
9112
|
for (const f of this.frames) for (const [path, prior] of f.saved) if (!base.has(path)) base.set(path, prior);
|
|
8390
9113
|
const parts = [];
|
|
8391
9114
|
for (const [path, prior] of base) {
|
|
8392
|
-
let
|
|
9115
|
+
let now4 = null;
|
|
8393
9116
|
try {
|
|
8394
|
-
|
|
9117
|
+
now4 = await this.fs.readFile(path);
|
|
8395
9118
|
} catch {
|
|
8396
9119
|
}
|
|
8397
|
-
if ((prior ?? "") === (
|
|
8398
|
-
const ops = diffLines(prior ?? "",
|
|
8399
|
-
parts.push(`--- ${path}${prior == null ? " (new)" :
|
|
9120
|
+
if ((prior ?? "") === (now4 ?? "")) continue;
|
|
9121
|
+
const ops = diffLines(prior ?? "", now4 ?? "");
|
|
9122
|
+
parts.push(`--- ${path}${prior == null ? " (new)" : now4 == null ? " (deleted)" : ""}
|
|
8400
9123
|
${formatDiff(ops)}`);
|
|
8401
9124
|
}
|
|
8402
9125
|
return parts.join("\n");
|
|
@@ -8432,10 +9155,10 @@ ${formatDiff(ops)}`);
|
|
|
8432
9155
|
// cli/gitCheckpoints.ts
|
|
8433
9156
|
import { execFile as execFile3 } from "child_process";
|
|
8434
9157
|
import { promisify } from "util";
|
|
8435
|
-
import { writeFileSync as writeFileSync5, mkdirSync as
|
|
9158
|
+
import { writeFileSync as writeFileSync5, mkdirSync as mkdirSync8 } from "fs";
|
|
8436
9159
|
import { homedir as homedir6 } from "os";
|
|
8437
9160
|
import { join as join9, resolve as resolve2, sep as sep2, parse as parsePath } from "path";
|
|
8438
|
-
var
|
|
9161
|
+
var log21 = forComponent("checkpoints");
|
|
8439
9162
|
var exec = promisify(execFile3);
|
|
8440
9163
|
var DEFAULT_EXCLUDE = [".agent/", ".git/", "node_modules/", "dist/", "build/", ".next/", "target/", ".venv/", "__pycache__/", "*.log"];
|
|
8441
9164
|
function isBroadRoot(p) {
|
|
@@ -8474,14 +9197,14 @@ var ShadowRepo = class {
|
|
|
8474
9197
|
if (this.ready !== void 0) return this.ready;
|
|
8475
9198
|
try {
|
|
8476
9199
|
await exec(this.git, ["--version"]);
|
|
8477
|
-
|
|
9200
|
+
mkdirSync8(this.gitDir, { recursive: true });
|
|
8478
9201
|
const valid = await this.run("rev-parse", "--git-dir").then(() => true, () => false);
|
|
8479
9202
|
if (!valid) await this.run("init", "-q");
|
|
8480
|
-
|
|
9203
|
+
mkdirSync8(join9(this.gitDir, "info"), { recursive: true });
|
|
8481
9204
|
writeFileSync5(join9(this.gitDir, "info", "exclude"), this.exclude.join("\n") + "\n");
|
|
8482
9205
|
this.ready = true;
|
|
8483
9206
|
} catch (e) {
|
|
8484
|
-
|
|
9207
|
+
log21.debug(`git checkpoints unavailable for ${this.workTree}`, e);
|
|
8485
9208
|
this.ready = false;
|
|
8486
9209
|
}
|
|
8487
9210
|
return this.ready;
|
|
@@ -8492,7 +9215,7 @@ var ShadowRepo = class {
|
|
|
8492
9215
|
}
|
|
8493
9216
|
async commit(label, forced = []) {
|
|
8494
9217
|
await this.run("add", "-A");
|
|
8495
|
-
for (const p of forced) await this.run("add", "-f", "--", p).catch((e) =>
|
|
9218
|
+
for (const p of forced) await this.run("add", "-f", "--", p).catch((e) => log21.debug(`force-add failed: ${p}`, e));
|
|
8496
9219
|
await this.run("commit", "--allow-empty", "-q", "-m", label);
|
|
8497
9220
|
}
|
|
8498
9221
|
/** Inject the CURRENT (pre-edit) content of `paths` into the turn-open restore point by amending it.
|
|
@@ -8500,8 +9223,8 @@ var ShadowRepo = class {
|
|
|
8500
9223
|
* turn-boundary `add -A` would never have captured it. Amend (vs a new commit) keeps one restore
|
|
8501
9224
|
* point per turn, so the REPL's turn↔frame mapping stays intact. */
|
|
8502
9225
|
async amendForced(paths) {
|
|
8503
|
-
for (const p of paths) await this.run("add", "-f", "--", p).catch((e) =>
|
|
8504
|
-
await this.run("commit", "--amend", "--no-edit", "-q", "--allow-empty").catch((e) =>
|
|
9226
|
+
for (const p of paths) await this.run("add", "-f", "--", p).catch((e) => log21.debug(`force-capture failed: ${p}`, e));
|
|
9227
|
+
await this.run("commit", "--amend", "--no-edit", "-q", "--allow-empty").catch((e) => log21.debug("amend failed", e));
|
|
8505
9228
|
}
|
|
8506
9229
|
/** Commits on `ref`, oldest-first (canonical index space). */
|
|
8507
9230
|
async log(ref) {
|
|
@@ -8561,7 +9284,7 @@ var ShadowRepo = class {
|
|
|
8561
9284
|
await this.run("gc", "--auto", "-q").catch(() => {
|
|
8562
9285
|
});
|
|
8563
9286
|
} catch (e) {
|
|
8564
|
-
|
|
9287
|
+
log21.debug("checkpoint prune failed", e);
|
|
8565
9288
|
}
|
|
8566
9289
|
}
|
|
8567
9290
|
};
|
|
@@ -8629,7 +9352,7 @@ var GitCheckpoints = class {
|
|
|
8629
9352
|
use(sessionId) {
|
|
8630
9353
|
if (sessionId === this.session) return;
|
|
8631
9354
|
this.session = sessionId;
|
|
8632
|
-
if (this.started) for (const r of this.repos) void r.point(this.ref()).catch((e) =>
|
|
9355
|
+
if (this.started) for (const r of this.repos) void r.point(this.ref()).catch((e) => log21.debug("re-point failed", e));
|
|
8633
9356
|
}
|
|
8634
9357
|
async begin(label) {
|
|
8635
9358
|
if (!await this.start()) return;
|
|
@@ -8649,7 +9372,7 @@ var GitCheckpoints = class {
|
|
|
8649
9372
|
process.stderr.write(`\x1B[2m checkpoints: '${this.repos[i].root}' too large to snapshot in ${Math.round(this.options.opTimeoutMs / 1e3)}s \u2014 file rewind disabled here\x1B[0m
|
|
8650
9373
|
`);
|
|
8651
9374
|
} else {
|
|
8652
|
-
|
|
9375
|
+
log21.debug("checkpoint commit failed", e);
|
|
8653
9376
|
survivors.push(this.repos[i]);
|
|
8654
9377
|
survivorCaches.push(this.caches[i] ?? []);
|
|
8655
9378
|
}
|
|
@@ -8684,7 +9407,7 @@ var GitCheckpoints = class {
|
|
|
8684
9407
|
if (this.forced.has(abs)) continue;
|
|
8685
9408
|
this.forced.add(abs);
|
|
8686
9409
|
const i = this.repoIndexFor(abs);
|
|
8687
|
-
if (i >= 0) await this.repos[i].amendForced([abs]).catch((e) =>
|
|
9410
|
+
if (i >= 0) await this.repos[i].amendForced([abs]).catch((e) => log21.debug("amendForced failed", e));
|
|
8688
9411
|
}
|
|
8689
9412
|
}
|
|
8690
9413
|
};
|
|
@@ -8752,7 +9475,7 @@ var GitCheckpointsOptions = class {
|
|
|
8752
9475
|
};
|
|
8753
9476
|
|
|
8754
9477
|
// cli/permissions.ts
|
|
8755
|
-
import { writeFileSync as writeFileSync6, mkdirSync as
|
|
9478
|
+
import { writeFileSync as writeFileSync6, mkdirSync as mkdirSync9 } from "fs";
|
|
8756
9479
|
import { homedir as homedir7 } from "os";
|
|
8757
9480
|
import { join as join10 } from "path";
|
|
8758
9481
|
var RULE_RE = /^(\w+)(?:\((.+)\))?$/;
|
|
@@ -8795,7 +9518,7 @@ function persistRule(cwd, decision, ruleStr) {
|
|
|
8795
9518
|
const list = cur[decision] ??= [];
|
|
8796
9519
|
if (!list.includes(ruleStr)) list.push(ruleStr);
|
|
8797
9520
|
try {
|
|
8798
|
-
|
|
9521
|
+
mkdirSync9(join10(cwd, ".agent"), { recursive: true });
|
|
8799
9522
|
writeFileSync6(PERM_FILE(cwd), JSON.stringify(cur, null, 2) + "\n");
|
|
8800
9523
|
} catch {
|
|
8801
9524
|
}
|
|
@@ -8819,7 +9542,7 @@ function trustDir(cwd, file = TRUST_FILE) {
|
|
|
8819
9542
|
if (!Array.isArray(list)) list = [];
|
|
8820
9543
|
if (!list.includes(cwd)) list.push(cwd);
|
|
8821
9544
|
try {
|
|
8822
|
-
|
|
9545
|
+
mkdirSync9(join10(file, ".."), { recursive: true });
|
|
8823
9546
|
writeFileSync6(file, JSON.stringify(list, null, 2) + "\n");
|
|
8824
9547
|
} catch {
|
|
8825
9548
|
}
|
|
@@ -10400,10 +11123,10 @@ function readPlainLine() {
|
|
|
10400
11123
|
|
|
10401
11124
|
// cli/osScheduler.ts
|
|
10402
11125
|
import { spawnSync as spawnSync5 } from "child_process";
|
|
10403
|
-
import { writeFileSync as writeFileSync8, mkdirSync as
|
|
11126
|
+
import { writeFileSync as writeFileSync8, mkdirSync as mkdirSync10, readdirSync as readdirSync2, unlinkSync as unlinkSync3, chmodSync, existsSync as existsSync7 } from "fs";
|
|
10404
11127
|
import { homedir as homedir8 } from "os";
|
|
10405
11128
|
import { join as join12 } from "path";
|
|
10406
|
-
var
|
|
11129
|
+
var log22 = forComponent("os-sched");
|
|
10407
11130
|
var OsScheduler = class {
|
|
10408
11131
|
options;
|
|
10409
11132
|
constructor(options) {
|
|
@@ -10427,7 +11150,7 @@ var OsScheduler = class {
|
|
|
10427
11150
|
/** Register the job with the OS. Returns a human description of the mechanism used. Throws on failure. */
|
|
10428
11151
|
schedule(spec) {
|
|
10429
11152
|
if (!this.available()) throw new Error(`no OS scheduler on ${this.options.platform}`);
|
|
10430
|
-
|
|
11153
|
+
mkdirSync10(this.dir, { recursive: true });
|
|
10431
11154
|
const oneOff = "at" in spec.trigger;
|
|
10432
11155
|
const script = this.writeScript(spec, oneOff);
|
|
10433
11156
|
const mechanism = this.options.platform === "darwin" ? this.scheduleDarwin(spec, script) : this.scheduleLinux(spec, script, oneOff);
|
|
@@ -10465,7 +11188,7 @@ var OsScheduler = class {
|
|
|
10465
11188
|
}
|
|
10466
11189
|
}
|
|
10467
11190
|
} catch (e) {
|
|
10468
|
-
|
|
11191
|
+
log22.debug(`cancel ${id}`, e);
|
|
10469
11192
|
}
|
|
10470
11193
|
for (const f of [`${id}.json`, `${id}.sh`]) {
|
|
10471
11194
|
try {
|
|
@@ -10526,7 +11249,7 @@ ${trigger}
|
|
|
10526
11249
|
<key>RunAtLoad</key><false/>
|
|
10527
11250
|
</dict></plist>
|
|
10528
11251
|
`;
|
|
10529
|
-
|
|
11252
|
+
mkdirSync10(join12(this.options.home, "Library", "LaunchAgents"), { recursive: true });
|
|
10530
11253
|
writeFileSync8(this.plistPath(spec.id), plist);
|
|
10531
11254
|
this.run("launchctl", ["load", this.plistPath(spec.id)]);
|
|
10532
11255
|
return `launchd:${this.label(spec.id)}`;
|
|
@@ -10570,19 +11293,19 @@ var OsSchedulerOptions = class {
|
|
|
10570
11293
|
return `${r.stdout ?? ""}${r.stderr ?? ""}`;
|
|
10571
11294
|
};
|
|
10572
11295
|
};
|
|
10573
|
-
function routeTrigger(trigger, backendHint,
|
|
11296
|
+
function routeTrigger(trigger, backendHint, now4 = Date.now()) {
|
|
10574
11297
|
if (backendHint === "os") return "os";
|
|
10575
11298
|
if (backendHint === "session") return "session";
|
|
10576
|
-
return "at" in trigger && trigger.at -
|
|
11299
|
+
return "at" in trigger && trigger.at - now4 >= 30 * 6e4 ? "os" : "session";
|
|
10577
11300
|
}
|
|
10578
11301
|
|
|
10579
11302
|
// cli/remoteTrigger.ts
|
|
10580
11303
|
import { createServer as createServer2, createConnection } from "net";
|
|
10581
11304
|
import { spawn as spawn3 } from "child_process";
|
|
10582
|
-
import { existsSync as existsSync8, mkdirSync as
|
|
11305
|
+
import { existsSync as existsSync8, mkdirSync as mkdirSync11, unlinkSync as unlinkSync4, readdirSync as readdirSync3 } from "fs";
|
|
10583
11306
|
import { homedir as homedir9 } from "os";
|
|
10584
11307
|
import { join as join13 } from "path";
|
|
10585
|
-
var
|
|
11308
|
+
var log23 = forComponent("remote-trigger");
|
|
10586
11309
|
var TRIGGER_DIR = () => join13(homedir9(), ".agent", "triggers");
|
|
10587
11310
|
var sockPath = (sessionId, dir = TRIGGER_DIR()) => join13(dir, `${sessionId}.sock`);
|
|
10588
11311
|
var TriggerServer = class {
|
|
@@ -10596,7 +11319,7 @@ var TriggerServer = class {
|
|
|
10596
11319
|
start(sessionId, dir = TRIGGER_DIR()) {
|
|
10597
11320
|
this.stop();
|
|
10598
11321
|
try {
|
|
10599
|
-
|
|
11322
|
+
mkdirSync11(dir, { recursive: true, mode: 448 });
|
|
10600
11323
|
const p = sockPath(sessionId, dir);
|
|
10601
11324
|
try {
|
|
10602
11325
|
unlinkSync4(p);
|
|
@@ -10620,13 +11343,13 @@ var TriggerServer = class {
|
|
|
10620
11343
|
conn.end(JSON.stringify({ ok: false, error: String(e) }) + "\n");
|
|
10621
11344
|
}
|
|
10622
11345
|
});
|
|
10623
|
-
conn.on("error", (e) =>
|
|
11346
|
+
conn.on("error", (e) => log23.debug("trigger conn error", e));
|
|
10624
11347
|
});
|
|
10625
|
-
this.server.on("error", (e) =>
|
|
11348
|
+
this.server.on("error", (e) => log23.debug("trigger server error", e));
|
|
10626
11349
|
this.server.listen(p);
|
|
10627
11350
|
this.path = p;
|
|
10628
11351
|
} catch (e) {
|
|
10629
|
-
|
|
11352
|
+
log23.debug("trigger server unavailable", e);
|
|
10630
11353
|
}
|
|
10631
11354
|
}
|
|
10632
11355
|
/** Re-bind on /resume (the session id changed). */
|
|
@@ -10748,7 +11471,7 @@ var strike = C("9");
|
|
|
10748
11471
|
var link = (text, url) => useColor ? `\x1B]8;;${url}\x1B\\${cyan(text)}\x1B]8;;\x1B\\` : `${text} (${url})`;
|
|
10749
11472
|
var mdPalette = { bold, dim, cyan, italic, strike, green, link };
|
|
10750
11473
|
var err = (s) => process.stderr.write(s);
|
|
10751
|
-
var
|
|
11474
|
+
var log24 = forComponent("cli");
|
|
10752
11475
|
var VERSION = (() => {
|
|
10753
11476
|
try {
|
|
10754
11477
|
return JSON.parse(readFileSync8(new URL("../package.json", import.meta.url), "utf8")).version ?? "?";
|
|
@@ -11009,8 +11732,8 @@ function loadEnvFile(file) {
|
|
|
11009
11732
|
}
|
|
11010
11733
|
}
|
|
11011
11734
|
function loadInstallEnv() {
|
|
11012
|
-
let dir =
|
|
11013
|
-
for (let i = 0; i < 5 && !existsSync9(join14(dir, "package.json")); i++) dir =
|
|
11735
|
+
let dir = dirname5(import.meta.path);
|
|
11736
|
+
for (let i = 0; i < 5 && !existsSync9(join14(dir, "package.json")); i++) dir = dirname5(dir);
|
|
11014
11737
|
for (const name of [".env", ".env.local"]) {
|
|
11015
11738
|
loadEnvFile(join14(dir, name));
|
|
11016
11739
|
loadEnvFile(join14(homedir10(), ".agent", name));
|
|
@@ -11428,7 +12151,7 @@ function costOf(pricing, promptTokens = 0, completionTokens = 0, cacheCreationTo
|
|
|
11428
12151
|
function turnCost(model, usage) {
|
|
11429
12152
|
return costOf(getModelInfo(model)?.pricing, usage?.promptTokens ?? 0, usage?.completionTokens ?? 0, usage?.cacheCreationTokens ?? 0, usage?.cacheReadTokens ?? 0, model);
|
|
11430
12153
|
}
|
|
11431
|
-
async function evaluateGoal(ai, condition, transcript,
|
|
12154
|
+
async function evaluateGoal(ai, condition, transcript, log25) {
|
|
11432
12155
|
const recent = transcript.filter((m) => m.role === "assistant").slice(-8).map((m) => {
|
|
11433
12156
|
const text = typeof m.content === "string" ? m.content : m.content.filter((p) => p.type === "text").map((p) => p.text).join(" ");
|
|
11434
12157
|
return text.slice(0, 600);
|
|
@@ -11448,7 +12171,7 @@ ${recent}` }
|
|
|
11448
12171
|
const match = r.content.match(/\{[\s\S]*\}/);
|
|
11449
12172
|
if (match) return JSON.parse(match[0]);
|
|
11450
12173
|
} catch (e) {
|
|
11451
|
-
|
|
12174
|
+
log25(dim(` (goal evaluator error: ${e?.message ?? e})
|
|
11452
12175
|
`));
|
|
11453
12176
|
}
|
|
11454
12177
|
return { met: false, reason: "evaluation unclear" };
|
|
@@ -11663,7 +12386,7 @@ function mcpAgentTools(mounted, opts) {
|
|
|
11663
12386
|
return tools;
|
|
11664
12387
|
}
|
|
11665
12388
|
async function closeMcp(mounted) {
|
|
11666
|
-
await Promise.all(mounted.map((m) => m.client.close().catch((e) =>
|
|
12389
|
+
await Promise.all(mounted.map((m) => m.client.close().catch((e) => log24.debug("mcp close failed", e))));
|
|
11667
12390
|
}
|
|
11668
12391
|
var IMG_EXT = { ".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".gif": "image/gif", ".webp": "image/webp" };
|
|
11669
12392
|
function mentionRefs(line) {
|
|
@@ -11713,7 +12436,7 @@ async function expandMentions(fs, line) {
|
|
|
11713
12436
|
if (loaded.includes(ref) || missing.includes(ref)) continue;
|
|
11714
12437
|
if (ref.includes(":") && mcpMentionResolver) {
|
|
11715
12438
|
const body = await mcpMentionResolver(ref).catch((e) => {
|
|
11716
|
-
|
|
12439
|
+
log24.debug("mcp mention resolve failed", e);
|
|
11717
12440
|
return null;
|
|
11718
12441
|
});
|
|
11719
12442
|
if (body != null) {
|
|
@@ -11796,7 +12519,7 @@ async function runTurn(agent, store, session, task, cp, cwd = process.cwd(), sen
|
|
|
11796
12519
|
try {
|
|
11797
12520
|
store.save(session);
|
|
11798
12521
|
} catch (ex) {
|
|
11799
|
-
|
|
12522
|
+
log24.debug("mid-turn session flush failed", ex);
|
|
11800
12523
|
}
|
|
11801
12524
|
}
|
|
11802
12525
|
return origNotify(e);
|
|
@@ -11870,8 +12593,8 @@ function startSession(args, store, agent, cwd) {
|
|
|
11870
12593
|
if (data) {
|
|
11871
12594
|
agent.transcript = data.messages;
|
|
11872
12595
|
if (args.fork) {
|
|
11873
|
-
const
|
|
11874
|
-
const forked = { meta: { ...data.meta, id: args.sessionId ?? store.newId(
|
|
12596
|
+
const now5 = Date.now();
|
|
12597
|
+
const forked = { meta: { ...data.meta, id: args.sessionId ?? store.newId(now5, cwd), created: now5, updated: now5, turns: data.meta.turns }, messages: data.messages };
|
|
11875
12598
|
err(dim(` forked ${data.meta.id} \u2192 ${forked.meta.id} (${data.meta.turns} turns)
|
|
11876
12599
|
`));
|
|
11877
12600
|
if (!args.task) printHistory(data.messages);
|
|
@@ -11885,11 +12608,11 @@ function startSession(args, store, agent, cwd) {
|
|
|
11885
12608
|
err(yellow(` no session to resume \u2014 starting fresh
|
|
11886
12609
|
`));
|
|
11887
12610
|
}
|
|
11888
|
-
const
|
|
11889
|
-
const id = args.sessionId ?? store.newId(
|
|
12611
|
+
const now4 = Date.now();
|
|
12612
|
+
const id = args.sessionId ?? store.newId(now4, cwd);
|
|
11890
12613
|
if (!args.task) err(dim(` session ${id}
|
|
11891
12614
|
`));
|
|
11892
|
-
return { meta: { id, created:
|
|
12615
|
+
return { meta: { id, created: now4, updated: now4, cwd, model: agent.options.model, turns: 0, title: "" }, messages: [] };
|
|
11893
12616
|
}
|
|
11894
12617
|
var AGENTS_MD_TEMPLATE = `# ${"${name}"}
|
|
11895
12618
|
|
|
@@ -11926,7 +12649,7 @@ function persistSetting(cwd, key, value) {
|
|
|
11926
12649
|
const obj = existsSync9(path) ? JSON.parse(readFileSync8(path, "utf8")) : {};
|
|
11927
12650
|
if (obj[key] === value) return;
|
|
11928
12651
|
obj[key] = value;
|
|
11929
|
-
|
|
12652
|
+
mkdirSync12(dirname5(path), { recursive: true });
|
|
11930
12653
|
writeFileSync9(path, JSON.stringify(obj, null, 2) + "\n");
|
|
11931
12654
|
} catch (e) {
|
|
11932
12655
|
err(yellow(` \u26A0 couldn't persist ${key} to ${path} \u2014 ${e?.message ?? e}
|
|
@@ -11943,14 +12666,14 @@ var isCancelTeardown = (e) => {
|
|
|
11943
12666
|
function installCancelGuards(mounted) {
|
|
11944
12667
|
process.on("unhandledRejection", (e) => {
|
|
11945
12668
|
if (isCancelTeardown(e)) {
|
|
11946
|
-
|
|
12669
|
+
log24.debug("suppressed unhandledRejection (cursor stream cancel)", e);
|
|
11947
12670
|
return;
|
|
11948
12671
|
}
|
|
11949
|
-
|
|
12672
|
+
log24.error("unhandledRejection", e);
|
|
11950
12673
|
});
|
|
11951
12674
|
process.on("uncaughtException", (e) => {
|
|
11952
12675
|
if (isCancelTeardown(e)) {
|
|
11953
|
-
|
|
12676
|
+
log24.debug("suppressed uncaughtException (cursor stream cancel)", e);
|
|
11954
12677
|
return;
|
|
11955
12678
|
}
|
|
11956
12679
|
console.error(e);
|
|
@@ -12009,6 +12732,7 @@ async function repl(args, ai, cfg, cwd) {
|
|
|
12009
12732
|
const duplex = args.duplex;
|
|
12010
12733
|
let dx;
|
|
12011
12734
|
let voiceIO;
|
|
12735
|
+
let voiceDiag;
|
|
12012
12736
|
let voiceLineOpen = false;
|
|
12013
12737
|
const emotionsOn = process.env.VOICE_EMOTIONS !== "0";
|
|
12014
12738
|
let showEmotions = process.env.VOICE_SHOW_EMOTIONS === "1";
|
|
@@ -12081,6 +12805,10 @@ async function repl(args, ai, cfg, cwd) {
|
|
|
12081
12805
|
...base,
|
|
12082
12806
|
notify(e) {
|
|
12083
12807
|
if (voiceIO && (e.kind === "thinking_delta" || e.kind === "turn_start")) return;
|
|
12808
|
+
if (e.kind === "diag") {
|
|
12809
|
+
voiceDiag?.push({ t: performance.now(), kind: String(e.message), ...e.data ?? {} });
|
|
12810
|
+
return;
|
|
12811
|
+
}
|
|
12084
12812
|
if (e.kind === "speak_utterance") {
|
|
12085
12813
|
if (voiceIO) {
|
|
12086
12814
|
spinner.stop();
|
|
@@ -12111,7 +12839,8 @@ async function repl(args, ai, cfg, cwd) {
|
|
|
12111
12839
|
return;
|
|
12112
12840
|
}
|
|
12113
12841
|
if (e.kind === "hold_filler" && voiceIO) {
|
|
12114
|
-
voiceIO.
|
|
12842
|
+
voiceIO.cancelPendingAck();
|
|
12843
|
+
if (e.message) voiceIO.speakFiller(e.message);
|
|
12115
12844
|
return;
|
|
12116
12845
|
}
|
|
12117
12846
|
if (e.kind === "revoice_done") {
|
|
@@ -12319,7 +13048,7 @@ async function repl(args, ai, cfg, cwd) {
|
|
|
12319
13048
|
const grabClipboardAttachment = () => {
|
|
12320
13049
|
const dir = join14(tmpdir3(), "agentx-pasted");
|
|
12321
13050
|
try {
|
|
12322
|
-
|
|
13051
|
+
mkdirSync12(dir, { recursive: true });
|
|
12323
13052
|
} catch {
|
|
12324
13053
|
}
|
|
12325
13054
|
const img = grabClipboardImage(dir, String(Date.now()));
|
|
@@ -12430,10 +13159,10 @@ Added entries are loadable now via the Skill/SlashCommand tools; removed ones ar
|
|
|
12430
13159
|
const history = existsSync9(histPath) ? readFileSync8(histPath, "utf8").split("\n").filter(Boolean).reverse().slice(0, 500) : [];
|
|
12431
13160
|
const remember = (line) => {
|
|
12432
13161
|
try {
|
|
12433
|
-
|
|
13162
|
+
mkdirSync12(join14(cwd, ".agent"), { recursive: true });
|
|
12434
13163
|
appendFileSync(histPath, line + "\n");
|
|
12435
13164
|
} catch (e) {
|
|
12436
|
-
|
|
13165
|
+
log24.debug("history write failed", e);
|
|
12437
13166
|
}
|
|
12438
13167
|
};
|
|
12439
13168
|
const ago = (t) => {
|
|
@@ -12504,7 +13233,7 @@ Added entries are loadable now via the Skill/SlashCommand tools; removed ones ar
|
|
|
12504
13233
|
try {
|
|
12505
13234
|
store.save(session);
|
|
12506
13235
|
} catch (e) {
|
|
12507
|
-
|
|
13236
|
+
log24.debug("session save after rewind failed", e);
|
|
12508
13237
|
}
|
|
12509
13238
|
err(green(" \u27F2 jumped back") + dim(` \u2014 ${face.transcript.length} message(s) kept; edit + resend
|
|
12510
13239
|
`));
|
|
@@ -12538,7 +13267,7 @@ ${task}`;
|
|
|
12538
13267
|
bangContext.length = 0;
|
|
12539
13268
|
}
|
|
12540
13269
|
const delta = await refreshCatalogs().catch((e) => {
|
|
12541
|
-
|
|
13270
|
+
log24.debug("catalog refresh failed", e);
|
|
12542
13271
|
return "";
|
|
12543
13272
|
});
|
|
12544
13273
|
if (delta) {
|
|
@@ -12772,7 +13501,7 @@ ${task}`;
|
|
|
12772
13501
|
cfgFiles.length ? ok(`config: ${cfgFiles.join(", ")}`) : warn("no .agent/config.* found (project or ~) \u2014 running on defaults");
|
|
12773
13502
|
try {
|
|
12774
13503
|
const probe = `${cwd}/.agent/sessions/.doctor-probe`;
|
|
12775
|
-
|
|
13504
|
+
mkdirSync12(`${cwd}/.agent/sessions`, { recursive: true });
|
|
12776
13505
|
writeFileSync9(probe, "ok");
|
|
12777
13506
|
unlinkSync5(probe);
|
|
12778
13507
|
ok(`session store writable (${cwd}/.agent/sessions)`);
|
|
@@ -12801,7 +13530,7 @@ ${task}`;
|
|
|
12801
13530
|
desc: "rescan skills/commands dirs and rebuild the system prompt (one cache miss) \u2014 picks up entries created mid-session",
|
|
12802
13531
|
run: async () => {
|
|
12803
13532
|
await refreshCatalogs().catch((e) => {
|
|
12804
|
-
|
|
13533
|
+
log24.debug("catalog refresh failed", e);
|
|
12805
13534
|
});
|
|
12806
13535
|
face.reprepare();
|
|
12807
13536
|
err(green(` \u2713 reloaded \u2014 ${skills.length} skill(s), ${cmds.length} command(s); system prompt rebuilds on next message
|
|
@@ -13281,7 +14010,7 @@ ${task}`;
|
|
|
13281
14010
|
try {
|
|
13282
14011
|
for (const def of (await loadAgents(fs2, d)).agents) if (!seen.has(def.name)) seen.set(def.name, { def, from: d });
|
|
13283
14012
|
} catch (e) {
|
|
13284
|
-
|
|
14013
|
+
log24.debug(`loadAgents(${d}) failed`, e);
|
|
13285
14014
|
}
|
|
13286
14015
|
}
|
|
13287
14016
|
if (!seen.size) {
|
|
@@ -13369,7 +14098,7 @@ ${task}`;
|
|
|
13369
14098
|
}
|
|
13370
14099
|
if (idx >= 0) {
|
|
13371
14100
|
const old = mounted.splice(idx, 1)[0];
|
|
13372
|
-
await old.client.close().catch((e) =>
|
|
14101
|
+
await old.client.close().catch((e) => log24.debug("mcp close failed", e));
|
|
13373
14102
|
}
|
|
13374
14103
|
try {
|
|
13375
14104
|
const m = await mountMcpServer(name, conf);
|
|
@@ -13397,7 +14126,7 @@ ${task}`;
|
|
|
13397
14126
|
}
|
|
13398
14127
|
const m = mounted.splice(idx, 1)[0];
|
|
13399
14128
|
remountMcpTools();
|
|
13400
|
-
await m.client.close().catch((e) =>
|
|
14129
|
+
await m.client.close().catch((e) => log24.debug("mcp close failed", e));
|
|
13401
14130
|
err(dim(` removed "${name}"
|
|
13402
14131
|
`));
|
|
13403
14132
|
return;
|
|
@@ -13532,7 +14261,7 @@ ${task}`;
|
|
|
13532
14261
|
const name = a[0] ? extname(a[0]) ? a[0] : a[0] + ".md" : join14(".agent", "exports", `${session.meta.id}.md`);
|
|
13533
14262
|
const path = resolve3(cwd, name);
|
|
13534
14263
|
try {
|
|
13535
|
-
|
|
14264
|
+
mkdirSync12(dirname5(path), { recursive: true });
|
|
13536
14265
|
writeFileSync9(path, md);
|
|
13537
14266
|
err(green(` \u2713 exported \u2192 ${path}
|
|
13538
14267
|
`) + dim(` ${shown.length} message(s) \xB7 ${md.length} chars
|
|
@@ -13617,7 +14346,7 @@ ${task}`;
|
|
|
13617
14346
|
try {
|
|
13618
14347
|
return readdirSync4(join14(cwd, absDir.replace(/^\/+/, "")), { withFileTypes: true }).map((d) => ({ name: d.name, dir: d.isDirectory() }));
|
|
13619
14348
|
} catch (e) {
|
|
13620
|
-
|
|
14349
|
+
log24.debug("completion readdir failed", absDir, e);
|
|
13621
14350
|
return null;
|
|
13622
14351
|
}
|
|
13623
14352
|
};
|
|
@@ -13744,7 +14473,7 @@ ${out}
|
|
|
13744
14473
|
return;
|
|
13745
14474
|
}
|
|
13746
14475
|
await refreshCatalogs().catch((e) => {
|
|
13747
|
-
|
|
14476
|
+
log24.debug("catalog refresh failed", e);
|
|
13748
14477
|
});
|
|
13749
14478
|
const sk = skills.find((x) => x.name === name);
|
|
13750
14479
|
if (sk) {
|
|
@@ -13785,14 +14514,28 @@ ${out}
|
|
|
13785
14514
|
return false;
|
|
13786
14515
|
}
|
|
13787
14516
|
const fakeVoice = process.env.AGENTX_VOICE_FAKE ? fakeVoiceParts(process.env.AGENTX_VOICE_FAKE) : null;
|
|
14517
|
+
voiceDiag = new JsonlDiagSink(join14(store.dir, `${session.meta.id}.voice.jsonl`));
|
|
13788
14518
|
voiceIO = new VoiceIO({
|
|
13789
14519
|
...fakeVoice ?? {},
|
|
14520
|
+
onDiag: (ev) => voiceDiag?.push(ev),
|
|
13790
14521
|
emotions: emotionsOn,
|
|
13791
14522
|
showEmotions,
|
|
13792
14523
|
// local is authoritative (a /voice-emotions before mic-on still applies)
|
|
13793
|
-
// No ack phrase by default: a fixed "Mm-hm," every turn reads robotic,
|
|
13794
|
-
//
|
|
13795
|
-
//
|
|
14524
|
+
// No FIXED ack phrase by default: a fixed "Mm-hm," every turn reads robotic, and the
|
|
14525
|
+
// conversational register already opens with a natural reaction. Instead the ADAPTIVE ack
|
|
14526
|
+
// (VoiceIOOptions.adaptiveAckMs, default 600ms — a safety net, not the voice) speaks a short varied ack only when the reflex
|
|
14527
|
+
// is actually slow — cancelled by the first delta, a Hold, or a barge-in.
|
|
14528
|
+
// When it fires, tell the duplex the turn already spoke, so the dead-air repair (ackIfSilent)
|
|
14529
|
+
// doesn't stack a second ack on top.
|
|
14530
|
+
onAdaptiveAck: () => dx?.noteExternalSpeech(),
|
|
14531
|
+
// Speculative reflex start (VoiceIOOptions.speculativeMs; VOICE_SPECULATIVE=0 disables): a
|
|
14532
|
+
// stable partial starts the reflex EARLY with its output held; the final in onUtterance below
|
|
14533
|
+
// flows through dispatchLine → dx.send, which confirms (flush, ~500-850ms TTFT saved) or
|
|
14534
|
+
// aborts it. Command-shaped partials never speculate (they'd never reach dx.send).
|
|
14535
|
+
onSpeculate: (text) => {
|
|
14536
|
+
if (dx && !/^[!#/]/.test(text.trim())) dx.speculate(text);
|
|
14537
|
+
},
|
|
14538
|
+
onSpeculateAbort: () => dx?.abortSpeculation(),
|
|
13796
14539
|
onState: () => editorRef?.redrawNow(),
|
|
13797
14540
|
// Throttled: each redraw clears the screen below the prompt — a partial-per-token storm
|
|
13798
14541
|
// (fast speech, or echo bleed if AEC degrades) would continuously erase streamed text.
|
|
@@ -13819,6 +14562,7 @@ ${out}
|
|
|
13819
14562
|
voicePartial = "";
|
|
13820
14563
|
if (!text.trim()) return;
|
|
13821
14564
|
if (matchVoiceCommand(text) === "off") {
|
|
14565
|
+
dx?.abortSpeculation();
|
|
13822
14566
|
err(`\r\x1B[K ${bold(cyan("\u{1F3A4} \u203A"))} ${text}
|
|
13823
14567
|
`);
|
|
13824
14568
|
const v = voiceIO;
|
|
@@ -13833,6 +14577,7 @@ ${out}
|
|
|
13833
14577
|
const note = !cut || cut.full.length - cut.heard.length <= 40 ? "" : cut.heard.trim() ? `
|
|
13834
14578
|
[the user interrupted you mid-speech \u2014 they only heard up to: "\u2026${cut.heard.slice(-80)}". Work any unheard essentials into your reply naturally, only if still relevant.]` : `
|
|
13835
14579
|
[the user interrupted you before hearing any of your previous reply \u2014 none of it landed; do not assume they got it.]`;
|
|
14580
|
+
if (/^[!#/]/.test(text.trim()) || note || pendingImages.length) dx?.abortSpeculation();
|
|
13836
14581
|
if (!/^[!#/]/.test(text.trim())) voiceIO.beginSpeech(true);
|
|
13837
14582
|
err(`\r\x1B[K ${bold(cyan("\u{1F3A4} \u203A"))} ${text}
|
|
13838
14583
|
`);
|
|
@@ -13844,6 +14589,29 @@ ${out}
|
|
|
13844
14589
|
}).finally(() => editorRef?.redrawNow());
|
|
13845
14590
|
}
|
|
13846
14591
|
});
|
|
14592
|
+
{
|
|
14593
|
+
const vo = voiceIO.options;
|
|
14594
|
+
voiceDiag.push({
|
|
14595
|
+
t: performance.now(),
|
|
14596
|
+
kind: "session_start",
|
|
14597
|
+
sessionId: session.meta.id,
|
|
14598
|
+
reflexModel: dx?.options.reflexModel,
|
|
14599
|
+
actModel: dx?.options.actModel,
|
|
14600
|
+
thinkModel: dx?.options.thinkModel,
|
|
14601
|
+
engine: {
|
|
14602
|
+
adaptiveAckMs: vo.adaptiveAckMs,
|
|
14603
|
+
utteranceMergeMs: vo.utteranceMergeMs,
|
|
14604
|
+
incompleteMergeMs: vo.incompleteMergeMs,
|
|
14605
|
+
bargeGraceMs: vo.bargeGraceMs,
|
|
14606
|
+
bargeIn: vo.bargeIn,
|
|
14607
|
+
overlapPause: vo.overlapPause,
|
|
14608
|
+
overlapResumeMs: vo.overlapResumeMs,
|
|
14609
|
+
speculativeMs: vo.speculativeMs,
|
|
14610
|
+
backchannelMs: vo.backchannelMs,
|
|
14611
|
+
emotions: vo.emotions
|
|
14612
|
+
}
|
|
14613
|
+
});
|
|
14614
|
+
}
|
|
13847
14615
|
voiceIO.onFatal = (msg) => {
|
|
13848
14616
|
err(yellow(`
|
|
13849
14617
|
\u26A0 voice off \u2014 ${msg}
|
|
@@ -13895,6 +14663,7 @@ ${out}
|
|
|
13895
14663
|
}
|
|
13896
14664
|
if (duplex && process.stdin.isTTY) toggleVoice = async () => {
|
|
13897
14665
|
if (voiceIO) {
|
|
14666
|
+
dx?.abortSpeculation();
|
|
13898
14667
|
voiceIO.stop();
|
|
13899
14668
|
voiceIO = void 0;
|
|
13900
14669
|
voicePartial = "";
|
|
@@ -13905,7 +14674,7 @@ ${out}
|
|
|
13905
14674
|
await startVoice(false);
|
|
13906
14675
|
editorRef?.redrawNow();
|
|
13907
14676
|
};
|
|
13908
|
-
if (args.voice && duplex && process.stdin.isTTY) await startVoice(
|
|
14677
|
+
if (args.voice && duplex && process.stdin.isTTY) await startVoice(process.env.VOICE_GREETING !== "0");
|
|
13909
14678
|
let ctxWarned = 0;
|
|
13910
14679
|
while (true) {
|
|
13911
14680
|
if (pendingRewind) {
|