@livx.cc/agentx 0.99.13 → 0.99.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{Agent-DYMSg21f.d.ts → Agent-CsVuR2O0.d.ts} +10 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +1622 -1514
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +5 -551
- package/dist/index.js +1653 -1552
- package/dist/index.js.map +1 -1
- package/package.json +2 -1
package/dist/index.js
CHANGED
|
@@ -3280,7 +3280,34 @@ var AgentOptions = class {
|
|
|
3280
3280
|
* `'off'`/undefined = none; `'low'|'medium'|'high'` or a raw token budget. Mapped to the
|
|
3281
3281
|
* provider-specific request shape via {@link reasoningToChatFragment}; explicit `providerOptions` wins. */
|
|
3282
3282
|
reasoning;
|
|
3283
|
+
/** Seed a fresh `run()` with prior conversation turns (real role/content messages) so the model
|
|
3284
|
+
* sees history as structured turns instead of a flattened text blob. They are inserted AFTER the
|
|
3285
|
+
* stable system prompt and BEFORE the current user turn — so the cached system(+tools) prefix never
|
|
3286
|
+
* shifts. Turns are normalized before sending (empties dropped, adjacent same-role coalesced, and the
|
|
3287
|
+
* first tail turn forced to `user`) so the transcript is always a valid Anthropic alternation ending
|
|
3288
|
+
* at the current user prompt. Unset/empty ⇒ current behavior, byte-for-byte. */
|
|
3289
|
+
priorMessages;
|
|
3283
3290
|
};
|
|
3291
|
+
function seedTranscriptTail(prior, current) {
|
|
3292
|
+
const seq = (prior ?? []).filter((m) => contentText(m.content).trim().length > 0).map((m) => ({ role: m.role, content: m.content }));
|
|
3293
|
+
seq.push({ role: "user", content: current });
|
|
3294
|
+
const out = [];
|
|
3295
|
+
for (const m of seq) {
|
|
3296
|
+
const last = out[out.length - 1];
|
|
3297
|
+
if (last && last.role === m.role) {
|
|
3298
|
+
if (typeof last.content === "string" && typeof m.content === "string") {
|
|
3299
|
+
last.content = `${last.content}
|
|
3300
|
+
|
|
3301
|
+
${m.content}`;
|
|
3302
|
+
continue;
|
|
3303
|
+
}
|
|
3304
|
+
out.push({ role: m.role === "user" ? "assistant" : "user", content: "(continuing)" });
|
|
3305
|
+
}
|
|
3306
|
+
out.push({ role: m.role, content: m.content });
|
|
3307
|
+
}
|
|
3308
|
+
if (out.length && out[0].role !== "user") out.unshift({ role: "user", content: "(Earlier conversation)" });
|
|
3309
|
+
return out;
|
|
3310
|
+
}
|
|
3284
3311
|
var Agent = class _Agent {
|
|
3285
3312
|
options;
|
|
3286
3313
|
transcript = [];
|
|
@@ -3448,7 +3475,9 @@ var Agent = class _Agent {
|
|
|
3448
3475
|
const userContent = await this.applyPromptSubmit(task);
|
|
3449
3476
|
this.transcript = [
|
|
3450
3477
|
{ role: "system", content: systemPrompt + (startCtx ? "\n\n" + startCtx : "") },
|
|
3451
|
-
|
|
3478
|
+
// Prior turns (if any) sit between the stable system prefix and the live user turn — the cached
|
|
3479
|
+
// system(+tools) prefix is unaffected; normalization keeps the alternation valid.
|
|
3480
|
+
...seedTranscriptTail(this.options.priorMessages, userContent)
|
|
3452
3481
|
];
|
|
3453
3482
|
return this.runLoop();
|
|
3454
3483
|
}
|
|
@@ -4445,14 +4474,14 @@ function makeScheduleTools(scheduler, os) {
|
|
|
4445
4474
|
backend: { type: "string", enum: ["auto", "session", "os"], description: "Where the job lives (default auto)." }
|
|
4446
4475
|
}
|
|
4447
4476
|
},
|
|
4448
|
-
async run({ prompt, trigger, label, backend }) {
|
|
4477
|
+
async run({ prompt, trigger, label, backend: backend2 }) {
|
|
4449
4478
|
try {
|
|
4450
|
-
if (os && os.route(trigger,
|
|
4479
|
+
if (os && os.route(trigger, backend2) === "os") {
|
|
4451
4480
|
const id2 = `os-${Date.now().toString(36)}`;
|
|
4452
4481
|
const mechanism = os.schedule({ id: id2, prompt, sessionId: os.sessionId, cwd: os.cwd, trigger, label });
|
|
4453
4482
|
return `Scheduled ${id2}${label ? ` (${label})` : ""} on the OS scheduler (${mechanism}) \u2014 survives quitting; fires \`agentx --resume ${os.sessionId}\` headless.`;
|
|
4454
4483
|
}
|
|
4455
|
-
if (
|
|
4484
|
+
if (backend2 === "os") return "Error: no OS scheduler available on this platform \u2014 job not created (use the default in-session backend).";
|
|
4456
4485
|
const id = scheduler.add({ prompt, trigger, label });
|
|
4457
4486
|
const job = scheduler.get(id);
|
|
4458
4487
|
const next = scheduler.nextFire(job);
|
|
@@ -4744,9 +4773,37 @@ function digestRun(messages, maxChars) {
|
|
|
4744
4773
|
import { MemFilesystem as MemFilesystem2 } from "@livx.cc/wcli/core";
|
|
4745
4774
|
init_logging();
|
|
4746
4775
|
|
|
4747
|
-
//
|
|
4748
|
-
|
|
4749
|
-
|
|
4776
|
+
// node_modules/@bod.ee/voice/src/core/logging.ts
|
|
4777
|
+
function consoleForComponent(name) {
|
|
4778
|
+
const debugEnv = typeof process !== "undefined" && process.env?.DEBUG || typeof localStorage !== "undefined" && localStorage.getItem("DEBUG") || "";
|
|
4779
|
+
const gated = debugEnv === "*" || debugEnv.split(",").some((p) => p.trim() === name);
|
|
4780
|
+
const gate = (fn) => gated ? fn : () => {
|
|
4781
|
+
};
|
|
4782
|
+
const tag = `[${name}]`;
|
|
4783
|
+
return {
|
|
4784
|
+
debug: gate((...a) => console.debug(tag, ...a)),
|
|
4785
|
+
verbose: gate((...a) => console.debug(tag, ...a)),
|
|
4786
|
+
info: (...a) => console.info(tag, ...a),
|
|
4787
|
+
warn: (...a) => console.warn(tag, ...a),
|
|
4788
|
+
error: (...a) => console.error(tag, ...a)
|
|
4789
|
+
};
|
|
4790
|
+
}
|
|
4791
|
+
var backend = consoleForComponent;
|
|
4792
|
+
function configureLogging(forComponentImpl) {
|
|
4793
|
+
backend = forComponentImpl;
|
|
4794
|
+
}
|
|
4795
|
+
function forComponent2(name) {
|
|
4796
|
+
return {
|
|
4797
|
+
debug: (...a) => backend(name).debug(...a),
|
|
4798
|
+
verbose: (...a) => backend(name).verbose(...a),
|
|
4799
|
+
info: (...a) => backend(name).info(...a),
|
|
4800
|
+
warn: (...a) => backend(name).warn(...a),
|
|
4801
|
+
error: (...a) => backend(name).error(...a)
|
|
4802
|
+
};
|
|
4803
|
+
}
|
|
4804
|
+
|
|
4805
|
+
// node_modules/@bod.ee/voice/src/core/emotion.ts
|
|
4806
|
+
var log9 = forComponent2("Emotion");
|
|
4750
4807
|
var EMOTIONS = [
|
|
4751
4808
|
// primary (best results)
|
|
4752
4809
|
"neutral",
|
|
@@ -4922,9 +4979,8 @@ var EmotionStream = class {
|
|
|
4922
4979
|
}
|
|
4923
4980
|
};
|
|
4924
4981
|
|
|
4925
|
-
//
|
|
4926
|
-
|
|
4927
|
-
var log10 = forComponent("VoiceEngine");
|
|
4982
|
+
// node_modules/@bod.ee/voice/src/core/engine.ts
|
|
4983
|
+
var log10 = forComponent2("VoiceEngine");
|
|
4928
4984
|
var realClock = {
|
|
4929
4985
|
now: () => performance.now(),
|
|
4930
4986
|
setTimeout: (fn, ms) => setTimeout(fn, ms),
|
|
@@ -5108,6 +5164,13 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5108
5164
|
// the exact ack text last spoken (fixed OR adaptive) — the echo-leak guard matches it
|
|
5109
5165
|
ackTimer = null;
|
|
5110
5166
|
// one-shot adaptive-ack timer (armed at dispatch, cancelled on first delta)
|
|
5167
|
+
// The adaptive micro-ack ALREADY spoke this turn → the reflex's own leading ack ("Still on it.") is now
|
|
5168
|
+
// redundant (live: host said "On it." then the reflex said "Still on it."). While this is set, speakDelta
|
|
5169
|
+
// probes the reflex's opening sentence and DROPS it if it's purely an acknowledgement; real content passes
|
|
5170
|
+
// through. Reset per turn in beginSpeech()/interrupt().
|
|
5171
|
+
adaptiveAckFired = false;
|
|
5172
|
+
ackProbe = null;
|
|
5173
|
+
// buffered reflex prose while probing the leading ack (null = not probing)
|
|
5111
5174
|
bargeGraceUntil = 0;
|
|
5112
5175
|
// no barge-in until this time — the user's OWN trailing audio (after the
|
|
5113
5176
|
// utterance that JUST dispatched this turn) must not immediately re-interrupt the reply it requested.
|
|
@@ -5250,6 +5313,8 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5250
5313
|
this.speaking = true;
|
|
5251
5314
|
this.ctxOpen = true;
|
|
5252
5315
|
this.spokeDeltas = false;
|
|
5316
|
+
this.adaptiveAckFired = false;
|
|
5317
|
+
this.ackProbe = null;
|
|
5253
5318
|
this.reply = "";
|
|
5254
5319
|
this.revealText = "";
|
|
5255
5320
|
this.wordStarts = [];
|
|
@@ -5278,6 +5343,8 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5278
5343
|
this.lastAck = phrase;
|
|
5279
5344
|
for (const w of this.words(phrase)) this.echoWords.add(w);
|
|
5280
5345
|
this.setState("speaking");
|
|
5346
|
+
this.adaptiveAckFired = true;
|
|
5347
|
+
this.ackProbe = "";
|
|
5281
5348
|
this.options.onAdaptiveAck();
|
|
5282
5349
|
}, this.options.adaptiveAckMs);
|
|
5283
5350
|
}
|
|
@@ -5289,6 +5356,25 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5289
5356
|
speakDelta(text) {
|
|
5290
5357
|
if (this.interrupted) return "";
|
|
5291
5358
|
this.clearAckTimer("first_delta");
|
|
5359
|
+
if (this.adaptiveAckFired && this.ackProbe !== null) {
|
|
5360
|
+
this.ackProbe += text;
|
|
5361
|
+
const probe = this.ackProbe;
|
|
5362
|
+
const bare = probe.replace(/\[emotion[^\]]*\]/gi, "").replace(/^\s+/, "");
|
|
5363
|
+
const m = bare.match(/^([^.!?…\n]*[.!?…\n])([\s\S]*)$/);
|
|
5364
|
+
if (!m && bare.length < 80) return "";
|
|
5365
|
+
this.ackProbe = null;
|
|
5366
|
+
this.adaptiveAckFired = false;
|
|
5367
|
+
const lead = m ? m[1] : bare;
|
|
5368
|
+
const norm2 = lead.replace(/[^a-z ]/gi, " ").replace(/\s+/g, " ").trim().toLowerCase();
|
|
5369
|
+
if (_VoiceEngine.ACK_LEADING.test(norm2)) {
|
|
5370
|
+
const rest = m ? m[2] : "";
|
|
5371
|
+
this.diag("reflex_ack_suppressed", { lead: lead.trim() });
|
|
5372
|
+
if (!rest.trim()) return "";
|
|
5373
|
+
text = rest;
|
|
5374
|
+
} else {
|
|
5375
|
+
text = probe;
|
|
5376
|
+
}
|
|
5377
|
+
}
|
|
5292
5378
|
if (!this.speaking || !this.ctxOpen) this.beginSpeech();
|
|
5293
5379
|
let { speech, display, prose } = this.emo ? this.emo.feed(text) : { speech: text, display: text, prose: text };
|
|
5294
5380
|
if (this.reply && /[.!?…]$/.test(this.reply) && /^[A-Z]/.test(prose)) {
|
|
@@ -5347,6 +5433,14 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5347
5433
|
endSpeech() {
|
|
5348
5434
|
this.interrupted = false;
|
|
5349
5435
|
this.clearAckTimer("turn_end");
|
|
5436
|
+
const heldProbe = this.ackProbe;
|
|
5437
|
+
this.adaptiveAckFired = false;
|
|
5438
|
+
this.ackProbe = null;
|
|
5439
|
+
if (heldProbe) {
|
|
5440
|
+
const norm2 = heldProbe.replace(/\[emotion[^\]]*\]/gi, "").replace(/[^a-z ]/gi, " ").replace(/\s+/g, " ").trim().toLowerCase();
|
|
5441
|
+
if (_VoiceEngine.ACK_LEADING.test(norm2)) this.diag("reflex_ack_suppressed", { lead: heldProbe.trim() });
|
|
5442
|
+
else this.speakDelta(heldProbe);
|
|
5443
|
+
}
|
|
5350
5444
|
if (!this.speaking) return;
|
|
5351
5445
|
this.diag("tts_turn_end", { spoke: this.spokeDeltas, replyChars: this.reply.length });
|
|
5352
5446
|
this.ctxOpen = false;
|
|
@@ -5428,6 +5522,9 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5428
5522
|
* No 'Okay.': a genuine user "Okay." within the 6s squash window would be swallowed as ack echo. */
|
|
5429
5523
|
static ACKS_NEUTRAL = ["Mm-hm.", "One sec.", "Right.", "Sec.", "On it.", "Sure, moment.", "Uh, one moment.", "Let me see."];
|
|
5430
5524
|
static ACKS_QUESTION = ["Hmm.", "Let me think.", "Hm, let me see.", "Good question.", "Mm, one sec.", "Let's see."];
|
|
5525
|
+
/** A reflex opening sentence that is PURELY an acknowledgement (no content) — dropped when the adaptive
|
|
5526
|
+
* micro-ack already spoke one. Anchored end-to-end so only a whole ack-only clause matches. */
|
|
5527
|
+
static ACK_LEADING = /^(?:(?:okay|ok|alright|sure|yeah|yep|right|got it|on it|still|still on it|still working on it|working on it|checking|one sec|one moment|mm hm|mmhm|hmm|let me (?:see|check|think))[ ,]*)+$/;
|
|
5431
5528
|
recentAcks = [];
|
|
5432
5529
|
pickAck() {
|
|
5433
5530
|
const pool = this.lastDispatchWasQuestion ? _VoiceEngine.ACKS_QUESTION : _VoiceEngine.ACKS_NEUTRAL;
|
|
@@ -5452,6 +5549,8 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5452
5549
|
/** barge-in: stop audio NOW, cancel generation, reset for the user's utterance */
|
|
5453
5550
|
interrupt() {
|
|
5454
5551
|
this.clearAckTimer("interrupt");
|
|
5552
|
+
this.adaptiveAckFired = false;
|
|
5553
|
+
this.ackProbe = null;
|
|
5455
5554
|
if (this.uttQueue.length) log10.info(`barge-in dropped ${this.uttQueue.length} queued worker utterance(s)`);
|
|
5456
5555
|
const droppedQueued = this.uttQueue.length;
|
|
5457
5556
|
this.uttQueue = [];
|
|
@@ -5912,7 +6011,14 @@ var VoiceEngine = class _VoiceEngine {
|
|
|
5912
6011
|
}
|
|
5913
6012
|
};
|
|
5914
6013
|
|
|
5915
|
-
//
|
|
6014
|
+
// node_modules/@bod.ee/voice/src/core/types.ts
|
|
6015
|
+
var STT_SAMPLE_RATE = 16e3;
|
|
6016
|
+
var TTS_SAMPLE_RATE = 44100;
|
|
6017
|
+
async function resolveAuth(auth) {
|
|
6018
|
+
return typeof auth === "function" ? await auth() : auth;
|
|
6019
|
+
}
|
|
6020
|
+
|
|
6021
|
+
// node_modules/@bod.ee/voice/src/core/spokenSplitter.ts
|
|
5916
6022
|
var OPEN = "<spoken>";
|
|
5917
6023
|
var CLOSE = "</spoken>";
|
|
5918
6024
|
var CLOSERS = `"')]}\xBB\u201D\u2019`;
|
|
@@ -5984,1652 +6090,1647 @@ var SpokenSplitter = class {
|
|
|
5984
6090
|
}
|
|
5985
6091
|
};
|
|
5986
6092
|
|
|
5987
|
-
// src/
|
|
5988
|
-
var log11 =
|
|
5989
|
-
|
|
5990
|
-
|
|
5991
|
-
|
|
5992
|
-
|
|
5993
|
-
|
|
5994
|
-
|
|
5995
|
-
/**
|
|
5996
|
-
|
|
5997
|
-
|
|
5998
|
-
|
|
5999
|
-
|
|
6000
|
-
|
|
6001
|
-
// (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
|
|
6002
|
-
reflexModel = "groq/openai/gpt-oss-120b";
|
|
6003
|
-
actModel = "anthropic/claude-sonnet-4-6";
|
|
6004
|
-
/** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
|
|
6005
|
-
thinkModel = "anthropic/claude-opus-4-8";
|
|
6006
|
-
/** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
|
|
6007
|
-
* provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
|
|
6008
|
-
* cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
|
|
6009
|
-
* leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
|
|
6010
|
-
providerOptionsFor;
|
|
6011
|
-
/** Escape hatches merged over the derived per-agent options. */
|
|
6012
|
-
reflexOptions;
|
|
6013
|
-
actOptions;
|
|
6014
|
-
thinkOptions;
|
|
6015
|
-
/** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
|
|
6016
|
-
* the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
|
|
6017
|
-
* pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
|
|
6018
|
-
* this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
|
|
6019
|
-
verifyActTasks = false;
|
|
6020
|
-
/** Receives the voice text_delta stream + task lifecycle events. */
|
|
6021
|
-
host;
|
|
6022
|
-
/** How many recent transcript messages are rendered into a worker's brief. */
|
|
6023
|
-
excerptTurns = 6;
|
|
6024
|
-
/** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
|
|
6025
|
-
* backchannels, impulsive first reactions before content (mimics real duplex conversation). */
|
|
6026
|
-
voiceStyle = "neutral";
|
|
6027
|
-
/** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
|
|
6028
|
-
* TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
|
|
6029
|
-
emotionTags = false;
|
|
6030
|
-
/** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
|
|
6031
|
-
* (post-spawn would race the worker's first edits). */
|
|
6032
|
-
onTaskStart;
|
|
6033
|
-
/** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
|
|
6034
|
-
* air. Off by default — each update costs a voice turn (LLM call + speech). */
|
|
6035
|
-
progressUpdates = false;
|
|
6036
|
-
/** Min ms between progress re-voices per task. */
|
|
6037
|
-
progressIntervalMs = 25e3;
|
|
6038
|
-
/** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
|
|
6039
|
-
* the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
|
|
6040
|
-
* voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
|
|
6041
|
-
askRelay = false;
|
|
6042
|
-
/** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
|
|
6043
|
-
askTimeoutMs = 12e4;
|
|
6044
|
-
/** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
|
|
6045
|
-
* bounding memory over a long-lived session. Running tasks are never evicted. */
|
|
6046
|
-
maxTaskRecords = 50;
|
|
6047
|
-
/** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
|
|
6048
|
-
* (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
|
|
6049
|
-
quickLook;
|
|
6050
|
-
/** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
|
|
6051
|
-
* tools directly (no delegation needed) and implicit capture guidance. */
|
|
6052
|
-
memoryDir;
|
|
6053
|
-
/** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
|
|
6054
|
-
memoryUserDir;
|
|
6093
|
+
// node_modules/@bod.ee/voice/src/adapters/soniox.ts
|
|
6094
|
+
var log11 = forComponent2("SonioxSTT");
|
|
6095
|
+
var now = () => performance.now();
|
|
6096
|
+
var SonioxSTTOptions = class {
|
|
6097
|
+
auth = "";
|
|
6098
|
+
source;
|
|
6099
|
+
model = "stt-rt-preview";
|
|
6100
|
+
languageHints = ["en"];
|
|
6101
|
+
/** Client-side endpoint: finalized text + no new tokens for this long = utterance (don't wait for
|
|
6102
|
+
* Soniox's semantic <end>, which adds 0.5-1.5s — the difference between ping-pong and lag). */
|
|
6103
|
+
silenceEndpointMs = 500;
|
|
6104
|
+
/** No-audio watchdog: if the mic source stops delivering chunks for this long, capture is dead →
|
|
6105
|
+
* fire onFatal + stop (else Soniox idle-timeouts and reconnect-loops forever). 0 = disable. */
|
|
6106
|
+
noAudioTimeoutMs = 1e4;
|
|
6055
6107
|
};
|
|
6056
|
-
var
|
|
6057
|
-
var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
|
|
6058
|
-
var STAGE_DIRECTION_RE = /^\(\s*(?:(?:waiting|checking|searching|thinking|processing|loading|working|fetching|looking)\b[^)]*|[^)]*(?:\.\.\.|…)\s*)\)$/i;
|
|
6059
|
-
function nowLine() {
|
|
6060
|
-
return `Current date and time: ${(/* @__PURE__ */ new Date()).toLocaleString("en-US", { weekday: "long", year: "numeric", month: "long", day: "numeric", hour: "numeric", minute: "2-digit", timeZoneName: "short" })}`;
|
|
6061
|
-
}
|
|
6062
|
-
function isTrivialBarge(text) {
|
|
6063
|
-
const t = text.trim().toLowerCase().replace(/[.,!?…\s]+$/g, "").replace(/^[.,!?…\s]+/, "");
|
|
6064
|
-
if (!t) return false;
|
|
6065
|
-
if (/\b(stop|cancel|no|nope|nah|never\s?mind|nvm|forget it|drop it|don'?t|quiet|shut up|enough|hold on|wait|pause|hang on|actually)\b/.test(t)) return false;
|
|
6066
|
-
const core = t.replace(/^(?:oh|um+|uh+|er+|hmm+|ah+|well|so|okay|ok|yeah|yep)[\s,]+/, "").trim();
|
|
6067
|
-
return /^(?:oh|um+|uh+|er+|hmm+|ah+|sorry|oops|whoops|my bad|pardon|excuse me|apologies|go on|go ahead|continue|keep going|carry on|as you were|please continue|you were saying|sorry go on|sorry continue|go on then|go on please)$/.test(core || t);
|
|
6068
|
-
}
|
|
6069
|
-
var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji. Never emit stage directions or parenthetical asides about your own process \u2014 nothing like "(waiting for the result...)" or "(checking)"; while work runs, either say it as plain speech or end your turn.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. DELEGATION RULE \u2014 decide for yourself, the user never has to push: if you cannot answer confidently from the conversation plus trivial well-known knowledge, do NOT refuse and do NOT guess \u2014 dispatch `Act` immediately with a clear brief and say you are checking. Anything needing CURRENT data (weather, news, prices, dates, sky/astronomy, "right now"), real computation, or verification is an automatic dispatch \u2014 never a refusal. The user should never need to say "search the web" or "think harder" to make you act; needing fresh or verified information IS the trigger. Refuse only what your worker genuinely cannot do (check `capabilities`), and say why. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result. Exactly ONE short line: never stack a second acknowledgement, and never narrate your own presence or status while waiting ("I\'m here", "let me see", "still checking" right after you already acked). One clean ack then silence reads as competent; repeated check-ins read as nervous.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 on a genuinely LONG wait you MAY give one brief half-sentence aside, but only occasionally; silence while working is normal and fine (the user knows you are on it). Do NOT narrate every step or re-announce yourself. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, just say "still on it" and end the turn \u2014 never poll it again and again in a loop. Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
|
|
6070
|
-
var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
|
|
6071
|
-
var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
|
|
6072
|
-
var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
|
|
6073
|
-
var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
|
|
6074
|
-
var DuplexAgent = class _DuplexAgent {
|
|
6108
|
+
var SonioxSTT = class {
|
|
6075
6109
|
options;
|
|
6076
|
-
|
|
6077
|
-
|
|
6078
|
-
|
|
6079
|
-
|
|
6080
|
-
|
|
6081
|
-
|
|
6082
|
-
|
|
6083
|
-
|
|
6084
|
-
|
|
6085
|
-
|
|
6086
|
-
/**
|
|
6087
|
-
*
|
|
6088
|
-
|
|
6089
|
-
|
|
6090
|
-
|
|
6091
|
-
*
|
|
6092
|
-
|
|
6093
|
-
|
|
6094
|
-
|
|
6095
|
-
|
|
6096
|
-
|
|
6097
|
-
|
|
6098
|
-
|
|
6099
|
-
|
|
6100
|
-
|
|
6101
|
-
|
|
6102
|
-
// an Act/Think fired this turn
|
|
6103
|
-
spokeBeforeDispatch = false;
|
|
6104
|
-
// the reflex ALREADY acked before dispatching — the forced post-dispatch text step must stay silent (live: "Got it…" twice)
|
|
6105
|
-
turnBriefs = /* @__PURE__ */ new Set();
|
|
6106
|
-
// briefs dispatched this turn (detect identical re-dispatch)
|
|
6107
|
-
spokeThisTurn = false;
|
|
6108
|
-
// any non-empty text_delta streamed this turn
|
|
6109
|
-
externalSpeech = false;
|
|
6110
|
-
// host spoke on our behalf (adaptive micro-ack) — not reflex output
|
|
6111
|
-
heldThisTurn = false;
|
|
6112
|
-
// Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
|
|
6113
|
-
nudging = false;
|
|
6114
|
-
// re-ack pass in flight: block ALL tools, prevent recursion
|
|
6115
|
-
reflexBuf = "";
|
|
6116
|
-
// accumulated reflex text this turn (fabricated-event detection)
|
|
6117
|
-
reflexForwarded = 0;
|
|
6118
|
-
// chars of reflexBuf already forwarded to the host/TTS
|
|
6119
|
-
fabricationCut = false;
|
|
6120
|
-
// reflex emitted a reserved [task …] marker → suppress its tail
|
|
6121
|
-
/** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
|
|
6122
|
-
* carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
|
|
6123
|
-
* turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
|
|
6124
|
-
* escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
|
|
6125
|
-
* ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
|
|
6126
|
-
* means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
|
|
6127
|
-
* ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
|
|
6128
|
-
* flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
|
|
6129
|
-
* fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
|
|
6130
|
-
* ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
|
|
6131
|
-
turnFollowUp = false;
|
|
6132
|
-
/** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
|
|
6133
|
-
* the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
|
|
6134
|
-
autoEscalations = 0;
|
|
6135
|
-
static MAX_AUTO_ESCALATIONS = 8;
|
|
6136
|
-
/** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
|
|
6137
|
-
pendingAsks = /* @__PURE__ */ new Map();
|
|
6138
|
-
/** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
|
|
6139
|
-
* PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
|
|
6140
|
-
* confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
|
|
6141
|
-
* aborts it: output dropped, history rolled back, the final dispatches normally. */
|
|
6142
|
-
spec;
|
|
6143
|
-
/** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
|
|
6144
|
-
speculativeAbortedCalls = 0;
|
|
6145
|
-
/** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
|
|
6146
|
-
* intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
|
|
6147
|
-
* writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
|
|
6148
|
-
* instead (the endpointed final then dispatches normally and may use the tool for real). */
|
|
6149
|
-
static SPEC_SAFE_TOOLS = /* @__PURE__ */ new Set(["QuickLook", "TaskStatus", "Hold"]);
|
|
6150
|
-
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
6151
|
-
memoryReady;
|
|
6152
|
-
constructor(options) {
|
|
6153
|
-
this.options = { ...new DuplexAgentOptions(), ...options };
|
|
6154
|
-
const o = this.options;
|
|
6155
|
-
if (o.memoryDir && o.fs) {
|
|
6156
|
-
this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
|
|
6110
|
+
ws;
|
|
6111
|
+
stopped = false;
|
|
6112
|
+
sourceStarted = false;
|
|
6113
|
+
onPartial = () => {
|
|
6114
|
+
};
|
|
6115
|
+
onUtterance = () => {
|
|
6116
|
+
};
|
|
6117
|
+
/** mic energy (RMS) per chunk — drives the energy-based heuristic barge-in tier */
|
|
6118
|
+
onLevel = () => {
|
|
6119
|
+
};
|
|
6120
|
+
/** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
|
|
6121
|
+
* loop). The host tears voice down instead of spinning forever. */
|
|
6122
|
+
onFatal = () => {
|
|
6123
|
+
};
|
|
6124
|
+
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
6125
|
+
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
6126
|
+
onDiag = () => {
|
|
6127
|
+
};
|
|
6128
|
+
diagOn = true;
|
|
6129
|
+
diag(kind, fields) {
|
|
6130
|
+
if (!this.diagOn) return;
|
|
6131
|
+
try {
|
|
6132
|
+
this.onDiag({ t: now(), kind, ...fields });
|
|
6133
|
+
} catch (e) {
|
|
6134
|
+
this.diagOn = false;
|
|
6135
|
+
log11.debug(`onDiag threw \u2014 STT diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
6157
6136
|
}
|
|
6158
|
-
const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
|
|
6159
|
-
const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
|
|
6160
|
-
const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
|
|
6161
|
-
const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
|
|
6162
|
-
const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
|
|
6163
|
-
const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
|
|
6164
|
-
const mcpNames = [
|
|
6165
|
-
...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
|
|
6166
|
-
...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
|
|
6167
|
-
];
|
|
6168
|
-
const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
|
|
6169
|
-
const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
|
|
6170
|
-
${nowLine()}. Anchor every relative or time-sensitive reference \u2014 "today", "now", "current", "recent", "latest", "this year" \u2014 to THIS moment. When the user asks about something recent (news, sports, events), they mean near this date, not a well-known past instance; brief your worker with the current year so it searches for what is happening NOW, not a famous older event.`;
|
|
6171
|
-
const tools = [
|
|
6172
|
-
...o.reflexOptions?.tools ?? [],
|
|
6173
|
-
this.actTool(),
|
|
6174
|
-
...o.thinkModel !== false ? [this.thinkTool()] : [],
|
|
6175
|
-
this.taskStatusTool(),
|
|
6176
|
-
this.cancelTaskTool(),
|
|
6177
|
-
this.quickLookTool(),
|
|
6178
|
-
this.answerTaskTool(),
|
|
6179
|
-
this.holdTool()
|
|
6180
|
-
];
|
|
6181
|
-
const host = o.host;
|
|
6182
|
-
const voiceHost = host && {
|
|
6183
|
-
ask: host.ask ? (q) => host.ask(q) : void 0,
|
|
6184
|
-
confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
|
|
6185
|
-
notify: (ev) => {
|
|
6186
|
-
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
6187
|
-
if (this.heldThisTurn) return;
|
|
6188
|
-
if (this.fabricationCut) return;
|
|
6189
|
-
if (this.turnDispatched && this.spokeBeforeDispatch) return;
|
|
6190
|
-
const msg = ev.message;
|
|
6191
|
-
this.reflexBuf += msg;
|
|
6192
|
-
this.scrubStageDirections();
|
|
6193
|
-
const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
|
|
6194
|
-
if (m) {
|
|
6195
|
-
this.fabricationCut = true;
|
|
6196
|
-
log11.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
|
|
6197
|
-
const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
|
|
6198
|
-
if (!safe) return;
|
|
6199
|
-
if (safe.trim()) this.spokeThisTurn = true;
|
|
6200
|
-
this.emitHost({ ...ev, message: safe });
|
|
6201
|
-
return;
|
|
6202
|
-
}
|
|
6203
|
-
const held = this.reflexBuf.length - this.reflexForwarded;
|
|
6204
|
-
const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
|
|
6205
|
-
let upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
|
|
6206
|
-
const paren = this.reflexBuf.lastIndexOf("(");
|
|
6207
|
-
if (paren >= this.reflexForwarded && !this.reflexBuf.includes(")", paren) && this.reflexBuf.length - paren <= 80)
|
|
6208
|
-
upto = Math.min(upto, paren);
|
|
6209
|
-
const out = this.reflexBuf.slice(this.reflexForwarded, upto);
|
|
6210
|
-
this.reflexForwarded = upto;
|
|
6211
|
-
if (!out) return;
|
|
6212
|
-
if (out.trim()) this.spokeThisTurn = true;
|
|
6213
|
-
this.emitHost({ ...ev, message: out });
|
|
6214
|
-
return;
|
|
6215
|
-
}
|
|
6216
|
-
host.notify?.(ev);
|
|
6217
|
-
}
|
|
6218
|
-
};
|
|
6219
|
-
this.voice = new Agent({
|
|
6220
|
-
ai: o.ai,
|
|
6221
|
-
fs: new MemFilesystem2(),
|
|
6222
|
-
model: o.reflexModel,
|
|
6223
|
-
stream: true,
|
|
6224
|
-
host: voiceHost,
|
|
6225
|
-
// The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
|
|
6226
|
-
// never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
|
|
6227
|
-
// it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
|
|
6228
|
-
askUserQuestion: false,
|
|
6229
|
-
systemPrompt: prompt,
|
|
6230
|
-
instructionFiles: false,
|
|
6231
|
-
maxSteps: 8,
|
|
6232
|
-
timeoutMs: 3e4,
|
|
6233
|
-
...o.reflexOptions,
|
|
6234
|
-
tools,
|
|
6235
|
-
// Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
|
|
6236
|
-
hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
|
|
6237
|
-
});
|
|
6238
|
-
}
|
|
6239
|
-
/** Resolve memory tools + inject index into voice system prompt (once). */
|
|
6240
|
-
async initMemory() {
|
|
6241
|
-
if (!this.memoryReady) return;
|
|
6242
|
-
const mem = await this.memoryReady;
|
|
6243
|
-
this.memoryReady = void 0;
|
|
6244
|
-
this.voice.options.tools.push(...mem.tools);
|
|
6245
|
-
if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
|
|
6246
6137
|
}
|
|
6247
|
-
|
|
6248
|
-
|
|
6249
|
-
|
|
6250
|
-
|
|
6251
|
-
|
|
6252
|
-
|
|
6253
|
-
|
|
6254
|
-
|
|
6255
|
-
|
|
6256
|
-
|
|
6138
|
+
lastChunkAt = 0;
|
|
6139
|
+
// timestamp of the most recent mic chunk (0 = none yet)
|
|
6140
|
+
startedChunksAt = 0;
|
|
6141
|
+
// when capture started (grace before the first chunk)
|
|
6142
|
+
noAudioTimer = null;
|
|
6143
|
+
finalText = "";
|
|
6144
|
+
partialText = "";
|
|
6145
|
+
lastChangeAt = 0;
|
|
6146
|
+
lastCombined = "";
|
|
6147
|
+
endpointTimer = null;
|
|
6148
|
+
firstTokenAt = 0;
|
|
6149
|
+
// first speech token in current utterance
|
|
6150
|
+
constructor(options) {
|
|
6151
|
+
this.options = { ...new SonioxSTTOptions(), ...options };
|
|
6257
6152
|
}
|
|
6258
|
-
|
|
6259
|
-
|
|
6260
|
-
* reflexForwarded would corrupt the forward offset. */
|
|
6261
|
-
scrubStageDirections() {
|
|
6262
|
-
if (this.reflexForwarded >= this.reflexBuf.length) return;
|
|
6263
|
-
const region = this.reflexBuf.slice(this.reflexForwarded);
|
|
6264
|
-
const scrubbed = region.replace(/\([^()]*\)/g, (s) => STAGE_DIRECTION_RE.test(s) ? "" : s);
|
|
6265
|
-
if (scrubbed !== region) this.reflexBuf = this.reflexBuf.slice(0, this.reflexForwarded) + scrubbed;
|
|
6153
|
+
get usingAec() {
|
|
6154
|
+
return this.options.source?.aec ?? false;
|
|
6266
6155
|
}
|
|
6267
|
-
|
|
6268
|
-
|
|
6269
|
-
this.
|
|
6270
|
-
|
|
6271
|
-
|
|
6272
|
-
|
|
6273
|
-
|
|
6274
|
-
this.
|
|
6275
|
-
|
|
6276
|
-
|
|
6277
|
-
|
|
6278
|
-
|
|
6279
|
-
|
|
6156
|
+
async connectWs() {
|
|
6157
|
+
const apiKey = await resolveAuth(this.options.auth);
|
|
6158
|
+
this.ws = new WebSocket("wss://stt-rt.soniox.com/transcribe-websocket");
|
|
6159
|
+
await new Promise((res, rej) => {
|
|
6160
|
+
this.ws.onopen = () => res();
|
|
6161
|
+
this.ws.onerror = (e) => rej(new Error(`soniox ws: ${e.message || "connect failed"}`));
|
|
6162
|
+
});
|
|
6163
|
+
this.ws.send(
|
|
6164
|
+
JSON.stringify({
|
|
6165
|
+
api_key: apiKey,
|
|
6166
|
+
model: this.options.model,
|
|
6167
|
+
audio_format: "pcm_s16le",
|
|
6168
|
+
sample_rate: STT_SAMPLE_RATE,
|
|
6169
|
+
num_channels: 1,
|
|
6170
|
+
language_hints: this.options.languageHints,
|
|
6171
|
+
enable_endpoint_detection: true
|
|
6172
|
+
})
|
|
6173
|
+
);
|
|
6174
|
+
this.ws.onmessage = (ev) => this.handle(JSON.parse(String(ev.data)));
|
|
6175
|
+
this.ws.onclose = (ev) => {
|
|
6176
|
+
if (this.stopped) return;
|
|
6177
|
+
log11.warn(`soniox ws closed (${ev.code} ${ev.reason || ""}) \u2014 reconnecting`);
|
|
6178
|
+
this.diag("stt_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: true });
|
|
6179
|
+
this.reset();
|
|
6180
|
+
this.connectWs().catch((e) => {
|
|
6181
|
+
log11.error(`soniox reconnect failed: ${e.message}`);
|
|
6182
|
+
this.diag("stt_reconnect_failed", { message: e.message });
|
|
6183
|
+
});
|
|
6184
|
+
};
|
|
6280
6185
|
}
|
|
6281
|
-
|
|
6282
|
-
|
|
6283
|
-
|
|
6284
|
-
|
|
6285
|
-
|
|
6286
|
-
|
|
6287
|
-
|
|
6288
|
-
|
|
6289
|
-
|
|
6290
|
-
|
|
6291
|
-
|
|
6186
|
+
async start() {
|
|
6187
|
+
await this.connectWs();
|
|
6188
|
+
if (this.sourceStarted) return;
|
|
6189
|
+
this.sourceStarted = true;
|
|
6190
|
+
this.endpointTimer = setInterval(() => {
|
|
6191
|
+
const combined = (this.finalText + this.partialText).trim();
|
|
6192
|
+
if (!combined || now() - this.lastChangeAt < this.options.silenceEndpointMs) return;
|
|
6193
|
+
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192silence-endpoint, "${combined.slice(0, 60)}"`);
|
|
6194
|
+
this.reset();
|
|
6195
|
+
this.onUtterance(combined, now());
|
|
6196
|
+
}, 120);
|
|
6197
|
+
this.endpointTimer.unref?.();
|
|
6198
|
+
this.startedChunksAt = now();
|
|
6199
|
+
const noAudioMs = this.options.noAudioTimeoutMs;
|
|
6200
|
+
if (noAudioMs > 0) {
|
|
6201
|
+
this.noAudioTimer = setInterval(() => {
|
|
6202
|
+
if (this.stopped) return;
|
|
6203
|
+
const ref = this.lastChunkAt || this.startedChunksAt;
|
|
6204
|
+
if (now() - ref > noAudioMs) {
|
|
6205
|
+
log11.error(`stt: no mic audio for >${Math.round(noAudioMs / 1e3)}s \u2014 capture device stopped delivering`);
|
|
6206
|
+
this.diag("stt_watchdog_fatal", { noAudioMs });
|
|
6207
|
+
this.onFatal("microphone stopped delivering audio (try a different input device, e.g. AirPods, or check System Settings \u2192 Sound \u2192 Input)");
|
|
6208
|
+
this.stop();
|
|
6292
6209
|
}
|
|
6293
|
-
|
|
6294
|
-
|
|
6295
|
-
|
|
6296
|
-
|
|
6297
|
-
|
|
6298
|
-
|
|
6210
|
+
}, Math.max(250, Math.min(2e3, noAudioMs / 4)));
|
|
6211
|
+
this.noAudioTimer.unref?.();
|
|
6212
|
+
}
|
|
6213
|
+
await this.options.source.start((chunk) => {
|
|
6214
|
+
this.lastChunkAt = now();
|
|
6215
|
+
let sum = 0;
|
|
6216
|
+
const view = new DataView(chunk.buffer, chunk.byteOffset, chunk.byteLength);
|
|
6217
|
+
for (let i = 0; i + 1 < chunk.byteLength; i += 2) {
|
|
6218
|
+
const v = view.getInt16(i, true);
|
|
6219
|
+
sum += v * v;
|
|
6299
6220
|
}
|
|
6300
|
-
|
|
6301
|
-
|
|
6302
|
-
|
|
6303
|
-
* micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
|
|
6304
|
-
* inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
|
|
6305
|
-
* still dead air, so silentTurn ignores external speech unless work was dispatched. */
|
|
6306
|
-
noteExternalSpeech() {
|
|
6307
|
-
this.externalSpeech = true;
|
|
6308
|
-
}
|
|
6309
|
-
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
6310
|
-
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
6311
|
-
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
6312
|
-
* voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
|
|
6313
|
-
* Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
|
|
6314
|
-
get silentTurn() {
|
|
6315
|
-
const ackedByHost = this.externalSpeech && this.turnDispatched;
|
|
6316
|
-
return !!this.options.host && !this.spokeThisTurn && !ackedByHost && !this.heldThisTurn;
|
|
6221
|
+
this.onLevel(Math.sqrt(sum / (chunk.byteLength / 2)));
|
|
6222
|
+
if (this.ws.readyState === WebSocket.OPEN) this.ws.send(chunk);
|
|
6223
|
+
});
|
|
6317
6224
|
}
|
|
6318
|
-
|
|
6319
|
-
|
|
6320
|
-
|
|
6321
|
-
|
|
6322
|
-
const dispatched = this.turnDispatched;
|
|
6323
|
-
this.nudging = true;
|
|
6324
|
-
try {
|
|
6325
|
-
await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
|
|
6326
|
-
} catch (e) {
|
|
6327
|
-
log11.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
|
|
6328
|
-
} finally {
|
|
6329
|
-
this.nudging = false;
|
|
6225
|
+
handle(m) {
|
|
6226
|
+
if (m.error_message) {
|
|
6227
|
+
this.diag("stt_error", { message: String(m.error_message), code: m.error_code });
|
|
6228
|
+
return log11.error(`soniox: ${m.error_message}`);
|
|
6330
6229
|
}
|
|
6331
|
-
|
|
6332
|
-
|
|
6333
|
-
|
|
6334
|
-
|
|
6335
|
-
this.lastFallback = line;
|
|
6336
|
-
this.emitHost({ kind: "text_delta", message: line });
|
|
6230
|
+
let endpoint = false;
|
|
6231
|
+
for (const t of m.tokens ?? []) {
|
|
6232
|
+
if (t.text === "<end>") endpoint = true;
|
|
6233
|
+
else if (t.is_final) this.finalText += t.text;
|
|
6337
6234
|
}
|
|
6338
|
-
|
|
6339
|
-
|
|
6340
|
-
|
|
6341
|
-
|
|
6342
|
-
|
|
6343
|
-
|
|
6344
|
-
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
|
|
6345
|
-
* If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
|
|
6346
|
-
* speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
|
|
6347
|
-
* host NOW (this is the latency win) and streaming continues live. Any other content aborts the
|
|
6348
|
-
* speculation first (rolled back silently) and runs a normal turn behind it. */
|
|
6349
|
-
send(content) {
|
|
6350
|
-
if (typeof content === "string" && isTrivialBarge(content) && this.hasParkedDelivery()) {
|
|
6351
|
-
if (this.spec) this.abortSpeculation();
|
|
6352
|
-
return this.enqueue(async () => {
|
|
6353
|
-
this.resetTurn();
|
|
6354
|
-
if (!this.redeliverParked()) this.awaitTrivialRedeliver = true;
|
|
6355
|
-
return { text: "", steps: 0, finishReason: "stop", messages: [] };
|
|
6356
|
-
});
|
|
6235
|
+
this.partialText = (m.tokens ?? []).filter((t) => !t.is_final && t.text !== "<end>").map((t) => t.text).join("");
|
|
6236
|
+
const combined = this.finalText + this.partialText;
|
|
6237
|
+
if (combined !== this.lastCombined) {
|
|
6238
|
+
this.lastCombined = combined;
|
|
6239
|
+
this.lastChangeAt = now();
|
|
6240
|
+
if (!this.firstTokenAt && combined.trim()) this.firstTokenAt = now();
|
|
6357
6241
|
}
|
|
6358
|
-
|
|
6359
|
-
if (
|
|
6360
|
-
|
|
6361
|
-
|
|
6362
|
-
|
|
6363
|
-
|
|
6364
|
-
spec.decide("confirm");
|
|
6365
|
-
this.notify("diag", "speculation_confirmed", { text: spec.text.slice(0, 80) });
|
|
6366
|
-
return spec.done;
|
|
6367
|
-
}
|
|
6368
|
-
this.abortSpeculation();
|
|
6242
|
+
this.onPartial(combined);
|
|
6243
|
+
if (endpoint && this.finalText.trim()) {
|
|
6244
|
+
const utterance = this.finalText.trim();
|
|
6245
|
+
if (this.firstTokenAt) log11.debug(`stt: ${Math.round(now() - this.firstTokenAt)}ms first-token\u2192endpoint, "${utterance.slice(0, 60)}"`);
|
|
6246
|
+
this.reset();
|
|
6247
|
+
this.onUtterance(utterance, now());
|
|
6369
6248
|
}
|
|
6370
|
-
return this.enqueue(async () => {
|
|
6371
|
-
await this.initMemory();
|
|
6372
|
-
this.resetTurn();
|
|
6373
|
-
this.awaitTrivialRedeliver = false;
|
|
6374
|
-
this.parkedRedeliver = [];
|
|
6375
|
-
this.suppressParkedResume = true;
|
|
6376
|
-
const res = await this.voice.send(content);
|
|
6377
|
-
this.flushHeldReflexTail();
|
|
6378
|
-
if (this.silentTurn) await this.ackIfSilent();
|
|
6379
|
-
return res;
|
|
6380
|
-
});
|
|
6381
|
-
}
|
|
6382
|
-
/** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
|
|
6383
|
-
* output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
|
|
6384
|
-
* voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
|
|
6385
|
-
* into the transcript between the speculative messages and their rollback. No-op if a speculation
|
|
6386
|
-
* is already in flight. */
|
|
6387
|
-
speculate(text) {
|
|
6388
|
-
if (!text.trim() || this.spec) return;
|
|
6389
|
-
let decide;
|
|
6390
|
-
const decision = new Promise((r) => {
|
|
6391
|
-
decide = r;
|
|
6392
|
-
});
|
|
6393
|
-
const spec = {
|
|
6394
|
-
text,
|
|
6395
|
-
state: "pending",
|
|
6396
|
-
buf: [],
|
|
6397
|
-
ctl: new AbortController(),
|
|
6398
|
-
decide,
|
|
6399
|
-
decision,
|
|
6400
|
-
done: void 0
|
|
6401
|
-
};
|
|
6402
|
-
this.spec = spec;
|
|
6403
|
-
spec.done = this.enqueue(async () => {
|
|
6404
|
-
const empty = { text: "", steps: 0, finishReason: "aborted", messages: this.voice.transcript };
|
|
6405
|
-
if (spec.state === "aborted") {
|
|
6406
|
-
this.speculativeAbortedCalls++;
|
|
6407
|
-
if (this.spec === spec) this.spec = void 0;
|
|
6408
|
-
return empty;
|
|
6409
|
-
}
|
|
6410
|
-
await this.initMemory();
|
|
6411
|
-
this.resetTurn();
|
|
6412
|
-
const base = this.voice.transcript.length;
|
|
6413
|
-
const prevSignal = this.voice.options.signal;
|
|
6414
|
-
this.voice.options.signal = spec.ctl.signal;
|
|
6415
|
-
let res;
|
|
6416
|
-
try {
|
|
6417
|
-
res = await this.voice.send(spec.text);
|
|
6418
|
-
} catch (e) {
|
|
6419
|
-
log11.warn(`speculative turn failed: ${e instanceof Error ? e.message : e}`);
|
|
6420
|
-
} finally {
|
|
6421
|
-
this.voice.options.signal = prevSignal;
|
|
6422
|
-
}
|
|
6423
|
-
const timer = setTimeout(() => {
|
|
6424
|
-
spec.state = spec.state === "pending" ? "aborted" : spec.state;
|
|
6425
|
-
spec.decide("abort");
|
|
6426
|
-
}, 1e4);
|
|
6427
|
-
timer.unref?.();
|
|
6428
|
-
const d = await spec.decision;
|
|
6429
|
-
clearTimeout(timer);
|
|
6430
|
-
if (d === "abort") {
|
|
6431
|
-
if (this.voice.transcript.length > base) this.voice.transcript.length = base;
|
|
6432
|
-
this.speculativeAbortedCalls++;
|
|
6433
|
-
log11.verbose(`speculation aborted (${this.speculativeAbortedCalls} total): "${spec.text.slice(0, 50)}"`);
|
|
6434
|
-
if (this.spec === spec) this.spec = void 0;
|
|
6435
|
-
return res ?? empty;
|
|
6436
|
-
}
|
|
6437
|
-
for (let i = base; i < this.voice.transcript.length; i++) {
|
|
6438
|
-
const m = this.voice.transcript[i];
|
|
6439
|
-
if (m.role === "user" && contentText(m.content) === spec.text) {
|
|
6440
|
-
m.content = spec.finalText;
|
|
6441
|
-
break;
|
|
6442
|
-
}
|
|
6443
|
-
}
|
|
6444
|
-
this.spec = void 0;
|
|
6445
|
-
this.flushHeldReflexTail();
|
|
6446
|
-
if (this.silentTurn) await this.ackIfSilent();
|
|
6447
|
-
return res ?? empty;
|
|
6448
|
-
});
|
|
6449
|
-
}
|
|
6450
|
-
/** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
|
|
6451
|
-
* Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
|
|
6452
|
-
* attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
|
|
6453
|
-
abortSpeculation() {
|
|
6454
|
-
const spec = this.spec;
|
|
6455
|
-
if (spec?.state !== "pending") return;
|
|
6456
|
-
spec.state = "aborted";
|
|
6457
|
-
spec.buf.length = 0;
|
|
6458
|
-
spec.ctl.abort();
|
|
6459
|
-
spec.decide("abort");
|
|
6460
|
-
this.notify("diag", "speculation_aborted", { text: spec.text.slice(0, 80) });
|
|
6461
|
-
}
|
|
6462
|
-
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
6463
|
-
cancelTask(id) {
|
|
6464
|
-
const rec = this.tasks.get(id);
|
|
6465
|
-
if (!rec) return `No task '${id}'.`;
|
|
6466
|
-
if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
|
|
6467
|
-
rec.status = "cancelled";
|
|
6468
|
-
rec.controller.abort();
|
|
6469
|
-
return `Task ${rec.id} (${rec.label}) cancelled.`;
|
|
6470
|
-
}
|
|
6471
|
-
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
6472
|
-
* delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
|
|
6473
|
-
* tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
|
|
6474
|
-
* Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
|
|
6475
|
-
parkInFlightDeliveries() {
|
|
6476
|
-
this.suppressParkedResume = false;
|
|
6477
|
-
const parked = [];
|
|
6478
|
-
for (const rec of this.tasks.values())
|
|
6479
|
-
if (rec.status === "running" && !rec.deliveryParked) {
|
|
6480
|
-
rec.deliveryParked = true;
|
|
6481
|
-
parked.push(rec.id);
|
|
6482
|
-
}
|
|
6483
|
-
return parked;
|
|
6484
6249
|
}
|
|
6485
|
-
|
|
6486
|
-
|
|
6487
|
-
|
|
6488
|
-
|
|
6250
|
+
reset() {
|
|
6251
|
+
this.finalText = "";
|
|
6252
|
+
this.partialText = "";
|
|
6253
|
+
this.lastCombined = "";
|
|
6254
|
+
this.firstTokenAt = 0;
|
|
6489
6255
|
}
|
|
6490
|
-
|
|
6491
|
-
|
|
6492
|
-
|
|
6493
|
-
if (
|
|
6494
|
-
|
|
6495
|
-
if (
|
|
6496
|
-
|
|
6497
|
-
this.notify("diag", "parked_redelivered", { chars: text.length });
|
|
6498
|
-
}
|
|
6499
|
-
return true;
|
|
6256
|
+
stop() {
|
|
6257
|
+
this.stopped = true;
|
|
6258
|
+
if (this.endpointTimer) clearInterval(this.endpointTimer);
|
|
6259
|
+
if (this.noAudioTimer) clearInterval(this.noAudioTimer);
|
|
6260
|
+
this.options.source?.stop();
|
|
6261
|
+
if (this.ws) this.ws.onclose = null;
|
|
6262
|
+
this.ws?.close();
|
|
6500
6263
|
}
|
|
6501
|
-
|
|
6502
|
-
|
|
6503
|
-
|
|
6504
|
-
|
|
6505
|
-
|
|
6506
|
-
|
|
6507
|
-
|
|
6508
|
-
|
|
6264
|
+
};
|
|
6265
|
+
|
|
6266
|
+
// node_modules/@bod.ee/voice/src/adapters/cartesia.ts
|
|
6267
|
+
var log12 = forComponent2("CartesiaTTS");
|
|
6268
|
+
var now2 = () => performance.now();
|
|
6269
|
+
var CartesiaTTSOptions = class {
|
|
6270
|
+
auth = "";
|
|
6271
|
+
voiceId = "";
|
|
6272
|
+
model = "sonic-3.5";
|
|
6273
|
+
/** 'apiKey' (server/CLI) → `api_key=` URL param; 'token' (browser, BE-minted) → `access_token=`. */
|
|
6274
|
+
authMode = "apiKey";
|
|
6275
|
+
};
|
|
6276
|
+
var CartesiaTTS = class _CartesiaTTS {
|
|
6277
|
+
options;
|
|
6278
|
+
ws;
|
|
6279
|
+
ctxSeq = 0;
|
|
6280
|
+
ctxId = "";
|
|
6281
|
+
onAudio = () => {
|
|
6282
|
+
};
|
|
6283
|
+
onDone = () => {
|
|
6284
|
+
};
|
|
6285
|
+
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
6286
|
+
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
6287
|
+
onTimestamps = () => {
|
|
6288
|
+
};
|
|
6289
|
+
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
6290
|
+
* when revealMode==='word'. */
|
|
6291
|
+
wantTimestamps = false;
|
|
6292
|
+
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
6293
|
+
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
6294
|
+
onDiag = () => {
|
|
6295
|
+
};
|
|
6296
|
+
diagOn = true;
|
|
6297
|
+
diag(kind, fields) {
|
|
6298
|
+
if (!this.diagOn) return;
|
|
6299
|
+
try {
|
|
6300
|
+
this.onDiag({ t: now2(), kind, ...fields });
|
|
6301
|
+
} catch (e) {
|
|
6302
|
+
this.diagOn = false;
|
|
6303
|
+
log12.debug(`onDiag threw \u2014 TTS diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
6509
6304
|
}
|
|
6510
6305
|
}
|
|
6511
|
-
|
|
6512
|
-
|
|
6513
|
-
|
|
6514
|
-
|
|
6515
|
-
|
|
6516
|
-
|
|
6517
|
-
|
|
6306
|
+
firstAudioAt = 0;
|
|
6307
|
+
/** Circuit breaker: consecutive error count + down flag. */
|
|
6308
|
+
consecutiveErrors = 0;
|
|
6309
|
+
consecutiveOk = 0;
|
|
6310
|
+
down = false;
|
|
6311
|
+
downAt = 0;
|
|
6312
|
+
probeTimer = null;
|
|
6313
|
+
static CB_THRESHOLD = 3;
|
|
6314
|
+
// open after 3 consecutive errors
|
|
6315
|
+
static CB_RECOVER_OK = 2;
|
|
6316
|
+
// close only after 2 consecutive good frames (no single-frame flap)
|
|
6317
|
+
static CB_PROBE_MS = 3e4;
|
|
6318
|
+
constructor(options) {
|
|
6319
|
+
this.options = { ...new CartesiaTTSOptions(), ...options };
|
|
6518
6320
|
}
|
|
6519
|
-
|
|
6520
|
-
|
|
6321
|
+
closed = false;
|
|
6322
|
+
connecting = null;
|
|
6323
|
+
async connect() {
|
|
6324
|
+
this.closed = false;
|
|
6325
|
+
this.connecting = this.doConnect();
|
|
6326
|
+
await this.connecting;
|
|
6327
|
+
this.connecting = null;
|
|
6521
6328
|
}
|
|
6522
|
-
|
|
6523
|
-
|
|
6524
|
-
|
|
6525
|
-
|
|
6526
|
-
|
|
6527
|
-
|
|
6528
|
-
|
|
6529
|
-
|
|
6530
|
-
|
|
6531
|
-
|
|
6329
|
+
async doConnect() {
|
|
6330
|
+
const key = await resolveAuth(this.options.auth);
|
|
6331
|
+
const param = this.options.authMode === "token" ? "access_token" : "api_key";
|
|
6332
|
+
this.ws = new WebSocket(`wss://api.cartesia.ai/tts/websocket?cartesia_version=2026-03-01&${param}=${key}`);
|
|
6333
|
+
await new Promise((res, rej) => {
|
|
6334
|
+
this.ws.onopen = () => res();
|
|
6335
|
+
this.ws.onerror = (e) => rej(new Error(`cartesia ws: ${e.message || "connect failed"}`));
|
|
6336
|
+
});
|
|
6337
|
+
this.ws.onclose = (ev) => {
|
|
6338
|
+
log12.warn(`cartesia ws closed (${ev.code} ${ev.reason || ""})`);
|
|
6339
|
+
this.diag("tts_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: !this.closed });
|
|
6340
|
+
if (!this.closed) {
|
|
6341
|
+
this.connecting = this.doConnect().catch((e) => {
|
|
6342
|
+
log12.error(`cartesia reconnect failed: ${e.message}`);
|
|
6343
|
+
this.diag("tts_reconnect_failed", { message: e.message });
|
|
6344
|
+
});
|
|
6532
6345
|
}
|
|
6533
|
-
if (spec.state === "aborted") return;
|
|
6534
|
-
}
|
|
6535
|
-
this.options.host?.notify?.(ev);
|
|
6536
|
-
}
|
|
6537
|
-
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
6538
|
-
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
6539
|
-
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
6540
|
-
* text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
|
|
6541
|
-
queueRevoice(event, nonClean = false) {
|
|
6542
|
-
this.pendingEvents.push(event);
|
|
6543
|
-
if (nonClean) this.pendingNonClean = true;
|
|
6544
|
-
if (this.flushQueued) return;
|
|
6545
|
-
this.flushQueued = true;
|
|
6546
|
-
void this.enqueue(async () => {
|
|
6547
|
-
this.flushQueued = false;
|
|
6548
|
-
const events = this.pendingEvents.splice(0);
|
|
6549
|
-
const nonCleanTurn = this.pendingNonClean;
|
|
6550
|
-
this.pendingNonClean = false;
|
|
6551
|
-
if (!events.length) return;
|
|
6552
|
-
const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
|
|
6553
|
-
this.resetTurn();
|
|
6554
|
-
this.turnFollowUp = nonCleanTurn;
|
|
6555
|
-
await this.voice.send(events.join("\n"));
|
|
6556
|
-
this.flushHeldReflexTail();
|
|
6557
|
-
if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
|
|
6558
|
-
this.notify("revoice_done", "");
|
|
6559
|
-
});
|
|
6560
|
-
}
|
|
6561
|
-
/** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
|
|
6562
|
-
* Act briefs get a self-verify footer — the worker's report is trusted without review, so it
|
|
6563
|
-
* must check its own work before reporting (nearly free under prompt caching; measured honest:
|
|
6564
|
-
* it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
|
|
6565
|
-
buildBrief(brief, tier = "act", deliver = true) {
|
|
6566
|
-
const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
|
|
6567
|
-
const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
|
|
6568
|
-
const deliverContract = deliver ? `
|
|
6569
|
-
|
|
6570
|
-
## DELIVER (spoken delivery)
|
|
6571
|
-
You are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown. NEVER enumerate in speech \u2014 no numbered or bulleted lists ("One. \u2026 Two. \u2026" is robotic). Deliver multiple items as flowing conversation with brief connective phrasing ("here's one\u2026", "and another\u2026", "oh, and\u2026"), pausing between items with sentence breaks, not numbers.` + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
|
|
6572
|
-
return `${nowLine()}.
|
|
6573
|
-
|
|
6574
|
-
` + (recent ? `${brief}
|
|
6575
|
-
|
|
6576
|
-
## Recent conversation (for context)
|
|
6577
|
-
${recent}` : brief) + verify + deliverContract;
|
|
6578
|
-
}
|
|
6579
|
-
/** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
|
|
6580
|
-
spawnWorker(id, label, briefText, tier, brief, followUp) {
|
|
6581
|
-
const o = this.options;
|
|
6582
|
-
const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
|
|
6583
|
-
const tierModel = tier === "think" ? o.thinkModel : o.actModel;
|
|
6584
|
-
const controller = new AbortController();
|
|
6585
|
-
const base = tierOpts?.hooks ?? o.actOptions?.hooks;
|
|
6586
|
-
const report = o.progressUpdates ? this.progressReporter(id) : void 0;
|
|
6587
|
-
const tail = [];
|
|
6588
|
-
const pushTail = (line) => {
|
|
6589
|
-
tail.push(line.slice(0, 200));
|
|
6590
|
-
if (tail.length > 120) tail.splice(0, tail.length - 120);
|
|
6591
|
-
};
|
|
6592
|
-
const hooks = {
|
|
6593
|
-
...base,
|
|
6594
|
-
preToolUse: async (call, meta) => {
|
|
6595
|
-
const d = await base?.preToolUse?.(call, meta);
|
|
6596
|
-
pushTail(`\u2699 ${describeCall(call)}`);
|
|
6597
|
-
report?.pre(call);
|
|
6598
|
-
return d;
|
|
6599
|
-
},
|
|
6600
|
-
postToolUse: async (call, result, meta) => {
|
|
6601
|
-
await base?.postToolUse?.(call, result, meta);
|
|
6602
|
-
const last = result?.trim().split("\n").filter(Boolean).pop();
|
|
6603
|
-
if (last) pushTail(` \u21B3 ${last}`);
|
|
6604
|
-
report?.post(call);
|
|
6605
|
-
},
|
|
6606
|
-
onToolOutput: (call, chunk, meta) => {
|
|
6607
|
-
base?.onToolOutput?.(call, chunk, meta);
|
|
6608
|
-
report?.output(chunk);
|
|
6609
|
-
}
|
|
6610
|
-
};
|
|
6611
|
-
const relayAsk = async (q) => {
|
|
6612
|
-
const opts = q.options?.length ? ` Options: ${q.options.map((x) => x.label).join(", ")}.` : "";
|
|
6613
|
-
const a = await this.parkQuestion(id, `${q.question}${opts}`);
|
|
6614
|
-
return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
|
|
6615
|
-
};
|
|
6616
|
-
const splitter = new SpokenSplitter();
|
|
6617
|
-
const speak = (seg) => {
|
|
6618
|
-
if (!seg) return;
|
|
6619
|
-
const r = this.tasks.get(id);
|
|
6620
|
-
if (r) r.spokenText = (r.spokenText ? r.spokenText + " " : "") + seg;
|
|
6621
|
-
if (!r?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
|
|
6622
|
-
};
|
|
6623
|
-
const coalescer = new SentenceCoalescer();
|
|
6624
|
-
const feedSpoken = (s) => {
|
|
6625
|
-
const ready = coalescer.feed(s);
|
|
6626
|
-
if (ready) speak(ready);
|
|
6627
6346
|
};
|
|
6628
|
-
|
|
6629
|
-
|
|
6630
|
-
|
|
6631
|
-
|
|
6632
|
-
|
|
6633
|
-
|
|
6634
|
-
|
|
6635
|
-
|
|
6636
|
-
|
|
6637
|
-
|
|
6347
|
+
this.ws.onmessage = (ev) => {
|
|
6348
|
+
const m = JSON.parse(String(ev.data));
|
|
6349
|
+
if (m.context_id && m.context_id !== this.ctxId) return;
|
|
6350
|
+
if (m.type === "chunk" && m.data) {
|
|
6351
|
+
this.consecutiveErrors = 0;
|
|
6352
|
+
this.markRecovered();
|
|
6353
|
+
if (!this.firstAudioAt) this.firstAudioAt = now2();
|
|
6354
|
+
this.onAudio(base64ToBytes(m.data));
|
|
6355
|
+
} else if (m.type === "done") {
|
|
6356
|
+
this.consecutiveErrors = 0;
|
|
6357
|
+
this.markRecovered();
|
|
6358
|
+
this.onDone();
|
|
6359
|
+
} else if (m.type === "timestamps" && m.word_timestamps) {
|
|
6360
|
+
const wt = m.word_timestamps;
|
|
6361
|
+
if (wt.words?.length && wt.start?.length) this.onTimestamps(wt.words, wt.start);
|
|
6362
|
+
} else if (m.type === "error") {
|
|
6363
|
+
if (/already been cancelled|does not exist/.test(m.message || "")) return;
|
|
6364
|
+
this.consecutiveErrors++;
|
|
6365
|
+
this.diag("tts_error", { message: String(m.message || ""), code: m.status_code, contextId: m.context_id, consecutive: this.consecutiveErrors });
|
|
6366
|
+
if (!this.down && this.consecutiveErrors >= _CartesiaTTS.CB_THRESHOLD) {
|
|
6367
|
+
this.down = true;
|
|
6368
|
+
this.downAt = now2();
|
|
6369
|
+
this.consecutiveOk = 0;
|
|
6370
|
+
log12.warn(`TTS circuit breaker open \u2014 ${this.consecutiveErrors} consecutive errors, switching to text-only`);
|
|
6371
|
+
this.diag("tts_breaker_open", { errors: this.consecutiveErrors });
|
|
6372
|
+
this.onDone();
|
|
6373
|
+
this.startProbe();
|
|
6374
|
+
} else if (!this.down) {
|
|
6375
|
+
(/No valid transcripts/i.test(m.message || "") ? log12.debug : log12.warn)(`cartesia: ${JSON.stringify(m)}`);
|
|
6638
6376
|
}
|
|
6639
6377
|
}
|
|
6640
6378
|
};
|
|
6641
|
-
const agentOpts = {
|
|
6642
|
-
ai: o.ai,
|
|
6643
|
-
fs: o.fs,
|
|
6644
|
-
model: tierModel,
|
|
6645
|
-
...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
|
|
6646
|
-
...tierOpts,
|
|
6647
|
-
// Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
|
|
6648
|
-
// main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
|
|
6649
|
-
providerOptions: o.providerOptionsFor?.(tierModel),
|
|
6650
|
-
stream: true,
|
|
6651
|
-
// worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
|
|
6652
|
-
host: workerHost,
|
|
6653
|
-
// carries BOTH ask AND the <spoken>-splitting notify
|
|
6654
|
-
...hooks ? { hooks } : {},
|
|
6655
|
-
signal: controller.signal
|
|
6656
|
-
// shared with the checker so a cancel tears down both
|
|
6657
|
-
};
|
|
6658
|
-
const promise = new Agent(agentOpts).run(briefText).then((res) => {
|
|
6659
|
-
const { spoken, detail } = splitter.flush();
|
|
6660
|
-
feedSpoken(spoken);
|
|
6661
|
-
if (detail.trim()) pushTail(detail.trim());
|
|
6662
|
-
flushSpoken();
|
|
6663
|
-
return res;
|
|
6664
|
-
}).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err) => this.onWorkerFailed(id, err));
|
|
6665
|
-
this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
|
|
6666
|
-
if (this.tasks.size > this.options.maxTaskRecords)
|
|
6667
|
-
for (const [tid, rec] of this.tasks) {
|
|
6668
|
-
if (this.tasks.size <= this.options.maxTaskRecords) break;
|
|
6669
|
-
if (rec.status !== "running") this.tasks.delete(tid);
|
|
6670
|
-
}
|
|
6671
6379
|
}
|
|
6672
|
-
/**
|
|
6673
|
-
*
|
|
6674
|
-
|
|
6675
|
-
|
|
6676
|
-
|
|
6677
|
-
|
|
6678
|
-
|
|
6679
|
-
|
|
6680
|
-
const
|
|
6681
|
-
|
|
6682
|
-
|
|
6683
|
-
...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
|
|
6684
|
-
};
|
|
6685
|
-
const checkBrief = `${this.buildBrief(brief, tier, false)}
|
|
6686
|
-
|
|
6687
|
-
## VERIFY MODE
|
|
6688
|
-
Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
|
|
6689
|
-
this.notify("task_verify", `task ${id}: verifying`, { id });
|
|
6690
|
-
const cres = await new Agent(checkerOpts).run(checkBrief);
|
|
6691
|
-
if (cres.finishReason !== "stop") {
|
|
6692
|
-
log11.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
|
|
6693
|
-
this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
|
|
6694
|
-
}
|
|
6695
|
-
const sum = (a = 0, b = 0) => a + b;
|
|
6696
|
-
return {
|
|
6697
|
-
...res,
|
|
6698
|
-
steps: res.steps + cres.steps,
|
|
6699
|
-
// Merge the checker's messages so downstream tool-call/step accounting includes BOTH agents
|
|
6700
|
-
// (else a verified task's toolCalls would undercount vs its steps/usage).
|
|
6701
|
-
messages: [...res.messages, ...cres.messages],
|
|
6702
|
-
usageEstimated: res.usageEstimated || cres.usageEstimated,
|
|
6703
|
-
usage: res.usage && cres.usage ? {
|
|
6704
|
-
promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
|
|
6705
|
-
completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
|
|
6706
|
-
totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
|
|
6707
|
-
cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
|
|
6708
|
-
cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
|
|
6709
|
-
} : res.usage ?? cres.usage
|
|
6710
|
-
};
|
|
6380
|
+
/** Close the breaker only after CB_RECOVER_OK consecutive good frames, so a single straggler chunk
|
|
6381
|
+
* after a 503 burst doesn't flap open→recover in <1s. A sub-2s down-window is a transient blip → debug. */
|
|
6382
|
+
markRecovered() {
|
|
6383
|
+
if (!this.down) return;
|
|
6384
|
+
if (++this.consecutiveOk < _CartesiaTTS.CB_RECOVER_OK) return;
|
|
6385
|
+
this.down = false;
|
|
6386
|
+
this.consecutiveOk = 0;
|
|
6387
|
+
this.stopProbe();
|
|
6388
|
+
const downMs = this.downAt ? now2() - this.downAt : 0;
|
|
6389
|
+
this.diag("tts_breaker_close", { downMs: Math.round(downMs) });
|
|
6390
|
+
(downMs < 2e3 ? log12.debug : log12.info)(`TTS recovered${downMs ? ` (down ${downMs}ms)` : ""}`);
|
|
6711
6391
|
}
|
|
6712
|
-
/**
|
|
6713
|
-
|
|
6714
|
-
|
|
6715
|
-
|
|
6716
|
-
progressReporter(id) {
|
|
6717
|
-
let lastAt = Date.now();
|
|
6718
|
-
let steps = 0;
|
|
6719
|
-
let inflight = null;
|
|
6720
|
-
const due = () => {
|
|
6721
|
-
if (this.pendingAsks.size) return void 0;
|
|
6722
|
-
const rec = this.tasks.get(id);
|
|
6723
|
-
return rec && rec.status === "running" && Date.now() - lastAt >= this.options.progressIntervalMs ? rec : void 0;
|
|
6724
|
-
};
|
|
6725
|
-
const emit = (rec, line, call) => {
|
|
6726
|
-
lastAt = Date.now();
|
|
6727
|
-
this.notify("task_progress", `task ${id} (${rec.label}): ${line}`, { id, steps, call: call.name });
|
|
6728
|
-
this.queueRevoice(`[task ${id} progress] ${line}`);
|
|
6729
|
-
};
|
|
6730
|
-
const timer = setInterval(() => {
|
|
6731
|
-
const rec = this.tasks.get(id);
|
|
6732
|
-
if (!rec || rec.status !== "running") return clearInterval(timer);
|
|
6733
|
-
if (!inflight || !due()) return;
|
|
6734
|
-
const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
|
|
6735
|
-
emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
|
|
6736
|
-
}, Math.max(this.options.progressIntervalMs, 250));
|
|
6737
|
-
timer.unref?.();
|
|
6738
|
-
return {
|
|
6739
|
-
pre: (call) => {
|
|
6740
|
-
inflight = { call, at: Date.now(), tail: "" };
|
|
6741
|
-
},
|
|
6742
|
-
output: (chunk) => {
|
|
6743
|
-
if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
|
|
6744
|
-
},
|
|
6745
|
-
// digest only — NEVER re-voices directly
|
|
6746
|
-
post: (call) => {
|
|
6747
|
-
steps++;
|
|
6748
|
-
inflight = null;
|
|
6749
|
-
const rec = due();
|
|
6750
|
-
if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
|
|
6751
|
-
}
|
|
6752
|
-
};
|
|
6392
|
+
/** Ensure the WS is open before sending — reconnects if idle-closed. */
|
|
6393
|
+
async ensureConnected() {
|
|
6394
|
+
if (this.connecting) await this.connecting;
|
|
6395
|
+
if (this.ws?.readyState !== WebSocket.OPEN) await this.connect();
|
|
6753
6396
|
}
|
|
6754
|
-
/**
|
|
6755
|
-
*
|
|
6756
|
-
*
|
|
6757
|
-
|
|
6758
|
-
|
|
6759
|
-
|
|
6760
|
-
|
|
6761
|
-
|
|
6762
|
-
settled = true;
|
|
6763
|
-
clearTimeout(timer);
|
|
6764
|
-
this.pendingAsks.delete(askId);
|
|
6765
|
-
resolve(answer);
|
|
6766
|
-
};
|
|
6767
|
-
const timer = setTimeout(() => {
|
|
6768
|
-
this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
|
|
6769
|
-
finish("");
|
|
6770
|
-
}, this.options.askTimeoutMs);
|
|
6771
|
-
this.pendingAsks.set(askId, { question, resolve: finish });
|
|
6772
|
-
this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
|
|
6773
|
-
this.queueRevoice(`[task ${askId} asks] ${question}
|
|
6774
|
-
(Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
|
|
6775
|
-
});
|
|
6776
|
-
}
|
|
6777
|
-
/** Resolve any question a settling/cancelled task left parked (its answer can no longer matter). */
|
|
6778
|
-
dropAsk(id) {
|
|
6779
|
-
this.pendingAsks.get(id)?.resolve("");
|
|
6397
|
+
/** Prime the synthesis pipeline on a throwaway context right after connect: the FIRST synthesis on
|
|
6398
|
+
* a fresh WS pays Cartesia's cold spin-up (~1s+ live) — absorb it at mic-on instead of turn one.
|
|
6399
|
+
* The returned audio is discarded (engine drops onAudio while not speaking; once a real turn calls
|
|
6400
|
+
* newContext() any stragglers are stale-context-filtered). Respects the circuit breaker. */
|
|
6401
|
+
warmup() {
|
|
6402
|
+
if (this.down) return;
|
|
6403
|
+
this.newContext();
|
|
6404
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
|
|
6780
6405
|
}
|
|
6781
|
-
|
|
6782
|
-
|
|
6783
|
-
|
|
6784
|
-
|
|
6785
|
-
* what to do next.
|
|
6786
|
-
*
|
|
6787
|
-
* Decision branches (the reflex acts on them with EXISTING tools — no new surface):
|
|
6788
|
-
* • accept → SPEAK the (partial) result plainly — don't dress a failure up as success.
|
|
6789
|
-
* • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
|
|
6790
|
-
* exists AND this task wasn't already a follow-up (one hop max). Wires the dead
|
|
6791
|
-
* "Reserve Think for a problem Act already FAILED at" promise.
|
|
6792
|
-
* • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
|
|
6793
|
-
* • ask → ask the user ONE concrete question if genuinely blocked.
|
|
6794
|
-
*
|
|
6795
|
-
* Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
|
|
6796
|
-
* failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
|
|
6797
|
-
integrationPrompt(rec, outcome, body, finishReason) {
|
|
6798
|
-
const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
|
|
6799
|
-
const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
|
|
6800
|
-
const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
|
|
6801
|
-
const hasThink = this.options.thinkModel !== false;
|
|
6802
|
-
const options = [];
|
|
6803
|
-
if (!rec.followUp && canEscalate && hasThink)
|
|
6804
|
-
options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
|
|
6805
|
-
if (!rec.followUp && canEscalate)
|
|
6806
|
-
options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
|
|
6807
|
-
options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
|
|
6808
|
-
options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
|
|
6809
|
-
const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
|
|
6810
|
-
const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
|
|
6811
|
-
return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
|
|
6406
|
+
newContext() {
|
|
6407
|
+
this.ctxId = `ctx-${++this.ctxSeq}`;
|
|
6408
|
+
this.firstAudioAt = 0;
|
|
6409
|
+
return this.ctxId;
|
|
6812
6410
|
}
|
|
6813
|
-
|
|
6814
|
-
|
|
6815
|
-
|
|
6816
|
-
|
|
6817
|
-
|
|
6818
|
-
|
|
6819
|
-
|
|
6820
|
-
|
|
6821
|
-
|
|
6822
|
-
const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
|
|
6823
|
-
return this.failTask(rec, msg);
|
|
6824
|
-
}
|
|
6825
|
-
rec.status = "done";
|
|
6826
|
-
rec.result = res.text;
|
|
6827
|
-
const incomplete = res.finishReason !== "stop";
|
|
6828
|
-
log11.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
|
|
6829
|
-
this.notify("task_done", `task ${id} (${rec.label}) completed`, {
|
|
6830
|
-
id,
|
|
6831
|
-
text: res.text,
|
|
6832
|
-
usage: res.usage,
|
|
6833
|
-
usageEstimated: res.usageEstimated,
|
|
6834
|
-
finishReason: res.finishReason,
|
|
6835
|
-
steps: res.steps,
|
|
6836
|
-
toolCalls: res.messages.filter((m) => m.role === "tool").length
|
|
6411
|
+
frame(transcript, cont) {
|
|
6412
|
+
return JSON.stringify({
|
|
6413
|
+
model_id: this.options.model,
|
|
6414
|
+
transcript,
|
|
6415
|
+
voice: { mode: "id", id: this.options.voiceId },
|
|
6416
|
+
output_format: { container: "raw", encoding: "pcm_s16le", sample_rate: TTS_SAMPLE_RATE },
|
|
6417
|
+
context_id: this.ctxId,
|
|
6418
|
+
continue: cont,
|
|
6419
|
+
...this.wantTimestamps ? { add_timestamps: true } : {}
|
|
6837
6420
|
});
|
|
6838
|
-
if (incomplete) {
|
|
6839
|
-
return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
|
|
6840
|
-
}
|
|
6841
|
-
const tail = rec.splitter?.flush();
|
|
6842
|
-
if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
|
|
6843
|
-
if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
|
|
6844
|
-
if (rec.deliveryParked && !this.suppressParkedResume) {
|
|
6845
|
-
const gist = (rec.spokenText ?? "").trim() || res.text.trim();
|
|
6846
|
-
if (gist) {
|
|
6847
|
-
if (this.awaitTrivialRedeliver) {
|
|
6848
|
-
this.awaitTrivialRedeliver = false;
|
|
6849
|
-
this.options.host?.notify?.({ kind: "speak_utterance", message: gist });
|
|
6850
|
-
this.notify("diag", "parked_redelivered", { chars: gist.length });
|
|
6851
|
-
} else this.parkedRedeliver.push(gist);
|
|
6852
|
-
}
|
|
6853
|
-
}
|
|
6854
|
-
if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
|
|
6855
|
-
this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
|
|
6856
|
-
}
|
|
6857
|
-
onWorkerFailed(id, err) {
|
|
6858
|
-
this.failTask(this.tasks.get(id), err instanceof Error ? err.message : String(err));
|
|
6859
6421
|
}
|
|
6860
|
-
|
|
6861
|
-
this.
|
|
6862
|
-
|
|
6863
|
-
|
|
6864
|
-
|
|
6865
|
-
this.notify("task_error", `task ${rec.id} (${rec.label}) failed: ${msg}`);
|
|
6866
|
-
this.queueRevoice(this.integrationPrompt(rec, "error", msg, "error"), true);
|
|
6422
|
+
speak(text, cont) {
|
|
6423
|
+
if (this.down) return;
|
|
6424
|
+
if (cont && !text) return;
|
|
6425
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame(text, cont));
|
|
6426
|
+
else void this.ensureConnected().then(() => this.ws?.readyState === WebSocket.OPEN && this.ws.send(this.frame(text, cont)));
|
|
6867
6427
|
}
|
|
6868
|
-
|
|
6869
|
-
|
|
6870
|
-
|
|
6871
|
-
|
|
6872
|
-
|
|
6873
|
-
|
|
6874
|
-
this.options.thinkModel = model;
|
|
6875
|
-
const tools = this.voice.options.tools;
|
|
6876
|
-
const i = tools.findIndex((t) => t.name === "Think");
|
|
6877
|
-
if (model === false && i >= 0) tools.splice(i, 1);
|
|
6878
|
-
else if (model !== false && i < 0) tools.push(this.thinkTool());
|
|
6428
|
+
end() {
|
|
6429
|
+
if (this.down) {
|
|
6430
|
+
this.onDone();
|
|
6431
|
+
return;
|
|
6432
|
+
}
|
|
6433
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("", false));
|
|
6879
6434
|
}
|
|
6880
|
-
|
|
6881
|
-
|
|
6882
|
-
* task's own integration turn won't escalate again — capping auto-follow-ups to one hop. */
|
|
6883
|
-
async dispatch(brief, tier = "act", label, followUp = false) {
|
|
6884
|
-
if (tier === "think" && this.options.thinkModel === false) tier = "act";
|
|
6885
|
-
if (followUp) this.autoEscalations++;
|
|
6886
|
-
const id = `t${++this.seq}`;
|
|
6887
|
-
const lbl = label ?? tier;
|
|
6888
|
-
await this.options.onTaskStart?.(id, lbl);
|
|
6889
|
-
this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
|
|
6890
|
-
this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
|
|
6891
|
-
return id;
|
|
6435
|
+
cancel() {
|
|
6436
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(JSON.stringify({ context_id: this.ctxId, cancel: true }));
|
|
6892
6437
|
}
|
|
6893
|
-
|
|
6894
|
-
return
|
|
6895
|
-
|
|
6896
|
-
|
|
6897
|
-
|
|
6898
|
-
|
|
6899
|
-
required: ["brief"],
|
|
6900
|
-
properties: {
|
|
6901
|
-
brief: { type: "string", description: "full, self-contained instructions for the worker" },
|
|
6902
|
-
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
6903
|
-
}
|
|
6904
|
-
},
|
|
6905
|
-
run: async ({ brief, label }) => {
|
|
6906
|
-
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
6907
|
-
this.turnDispatched = true;
|
|
6908
|
-
this.turnBriefs.add(String(brief ?? ""));
|
|
6909
|
-
this.voice.options.toolChoice = "none";
|
|
6910
|
-
const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
|
|
6911
|
-
return this.spokeBeforeDispatch ? `Acting on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
6438
|
+
startProbe() {
|
|
6439
|
+
if (this.probeTimer) return;
|
|
6440
|
+
this.probeTimer = setInterval(() => {
|
|
6441
|
+
if (!this.down) {
|
|
6442
|
+
this.stopProbe();
|
|
6443
|
+
return;
|
|
6912
6444
|
}
|
|
6913
|
-
|
|
6445
|
+
this.consecutiveErrors = 0;
|
|
6446
|
+
this.newContext();
|
|
6447
|
+
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
|
|
6448
|
+
}, _CartesiaTTS.CB_PROBE_MS);
|
|
6449
|
+
this.probeTimer.unref?.();
|
|
6914
6450
|
}
|
|
6915
|
-
|
|
6916
|
-
|
|
6917
|
-
|
|
6918
|
-
|
|
6919
|
-
|
|
6920
|
-
type: "object",
|
|
6921
|
-
required: ["brief"],
|
|
6922
|
-
properties: {
|
|
6923
|
-
brief: { type: "string", description: "the question or problem to reason about deeply" },
|
|
6924
|
-
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
6925
|
-
}
|
|
6926
|
-
},
|
|
6927
|
-
run: async ({ brief, label }) => {
|
|
6928
|
-
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
6929
|
-
this.turnDispatched = true;
|
|
6930
|
-
this.turnBriefs.add(String(brief ?? ""));
|
|
6931
|
-
this.voice.options.toolChoice = "none";
|
|
6932
|
-
const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
|
|
6933
|
-
return this.spokeBeforeDispatch ? `Thinking on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
6934
|
-
}
|
|
6935
|
-
};
|
|
6451
|
+
stopProbe() {
|
|
6452
|
+
if (this.probeTimer) {
|
|
6453
|
+
clearInterval(this.probeTimer);
|
|
6454
|
+
this.probeTimer = null;
|
|
6455
|
+
}
|
|
6936
6456
|
}
|
|
6937
|
-
|
|
6938
|
-
|
|
6939
|
-
|
|
6940
|
-
|
|
6941
|
-
|
|
6942
|
-
run: async ({ id }) => {
|
|
6943
|
-
const list = id ? [this.tasks.get(String(id))].filter(Boolean) : [...this.tasks.values()];
|
|
6944
|
-
if (!list.length) return id ? `No task '${id}'.` : "No background tasks.";
|
|
6945
|
-
return list.map((t) => `${t.id} (${t.label}): ${t.status}`).join("\n");
|
|
6946
|
-
}
|
|
6947
|
-
};
|
|
6457
|
+
close() {
|
|
6458
|
+
this.closed = true;
|
|
6459
|
+
this.stopProbe();
|
|
6460
|
+
if (this.ws) this.ws.onclose = null;
|
|
6461
|
+
this.ws?.close();
|
|
6948
6462
|
}
|
|
6949
|
-
|
|
6950
|
-
|
|
6951
|
-
|
|
6952
|
-
|
|
6953
|
-
|
|
6954
|
-
|
|
6955
|
-
|
|
6956
|
-
|
|
6957
|
-
|
|
6958
|
-
|
|
6959
|
-
|
|
6960
|
-
|
|
6961
|
-
|
|
6962
|
-
|
|
6963
|
-
|
|
6964
|
-
|
|
6965
|
-
|
|
6966
|
-
|
|
6967
|
-
|
|
6968
|
-
|
|
6969
|
-
|
|
6970
|
-
|
|
6971
|
-
|
|
6972
|
-
|
|
6973
|
-
|
|
6974
|
-
|
|
6975
|
-
|
|
6976
|
-
|
|
6977
|
-
|
|
6978
|
-
|
|
6979
|
-
|
|
6980
|
-
|
|
6981
|
-
|
|
6982
|
-
|
|
6983
|
-
|
|
6984
|
-
|
|
6985
|
-
|
|
6986
|
-
|
|
6987
|
-
|
|
6988
|
-
|
|
6989
|
-
|
|
6990
|
-
|
|
6991
|
-
|
|
6992
|
-
|
|
6993
|
-
|
|
6994
|
-
|
|
6995
|
-
|
|
6996
|
-
|
|
6997
|
-
|
|
6998
|
-
|
|
6999
|
-
|
|
7000
|
-
|
|
7001
|
-
|
|
7002
|
-
|
|
7003
|
-
|
|
7004
|
-
|
|
7005
|
-
|
|
7006
|
-
`
|
|
7007
|
-
|
|
7008
|
-
|
|
7009
|
-
|
|
7010
|
-
|
|
7011
|
-
|
|
7012
|
-
|
|
7013
|
-
|
|
7014
|
-
|
|
7015
|
-
|
|
7016
|
-
|
|
7017
|
-
|
|
7018
|
-
|
|
7019
|
-
|
|
7020
|
-
|
|
7021
|
-
|
|
7022
|
-
|
|
7023
|
-
|
|
7024
|
-
|
|
6463
|
+
};
|
|
6464
|
+
function base64ToBytes(b64) {
|
|
6465
|
+
if (typeof Buffer !== "undefined") return Buffer.from(b64, "base64");
|
|
6466
|
+
const bin = atob(b64);
|
|
6467
|
+
const out = new Uint8Array(bin.length);
|
|
6468
|
+
for (let i = 0; i < bin.length; i++) out[i] = bin.charCodeAt(i);
|
|
6469
|
+
return out;
|
|
6470
|
+
}
|
|
6471
|
+
|
|
6472
|
+
// src/voice/engine.ts
|
|
6473
|
+
init_logging();
|
|
6474
|
+
configureLogging((name) => {
|
|
6475
|
+
const l = forComponent(name);
|
|
6476
|
+
const at = (m) => (...a) => l[m]?.(...a);
|
|
6477
|
+
return { debug: at("debug"), verbose: at("verbose"), info: at("info"), warn: at("warn"), error: at("error") };
|
|
6478
|
+
});
|
|
6479
|
+
|
|
6480
|
+
// src/duplex.ts
|
|
6481
|
+
var log13 = forComponent("DuplexAgent");
|
|
6482
|
+
function describeCall(call) {
|
|
6483
|
+
const v = call.args && Object.values(call.args).find((x) => typeof x === "string" && x.trim());
|
|
6484
|
+
const hint = v ? ` (${String(v).replace(/\s+/g, " ").trim().slice(0, 48)})` : "";
|
|
6485
|
+
return `${call.name}${hint}`;
|
|
6486
|
+
}
|
|
6487
|
+
var DuplexAgentOptions = class {
|
|
6488
|
+
/** Any ai.libx.js AIClient — shared by all tiers (routed by model). */
|
|
6489
|
+
ai;
|
|
6490
|
+
/** The WORKER's filesystem (act + think). If omitted the worker keeps Agent's jailed-disk-at-cwd default. */
|
|
6491
|
+
fs;
|
|
6492
|
+
// The reflex IS the voice. 120b (not 20b) for channel discipline + instruction-following: the 20b
|
|
6493
|
+
// mislabels gpt-oss harmony channels under load, leaking raw analysis into the spoken `final` channel
|
|
6494
|
+
// (and misfiring Hold). 120b is the same price tier (~$0.15/$0.60) — the quality/cost trade is free.
|
|
6495
|
+
reflexModel = "groq/openai/gpt-oss-120b";
|
|
6496
|
+
actModel = "anthropic/claude-sonnet-4-6";
|
|
6497
|
+
/** Premium reasoning model. Set to `false` to disable the Think tier entirely. */
|
|
6498
|
+
thinkModel = "anthropic/claude-opus-4-8";
|
|
6499
|
+
/** Per-worker providerOptions, derived from the worker's actual model at spawn time (IoC — keeps duplex
|
|
6500
|
+
* provider-agnostic). Workers override the reflex/main model, so provider-specific options (e.g. cursor's
|
|
6501
|
+
* cwd/cursorSession) must be recomputed for the worker's model, never inherited from the main template —
|
|
6502
|
+
* leaking cursor options to an anthropic worker is a hard 400. Returns undefined → no providerOptions. */
|
|
6503
|
+
providerOptionsFor;
|
|
6504
|
+
/** Escape hatches merged over the derived per-agent options. */
|
|
6505
|
+
reflexOptions;
|
|
6506
|
+
actOptions;
|
|
6507
|
+
thinkOptions;
|
|
6508
|
+
/** Fresh-context check on each successful Act task: a NEW agent (no self-confirmation bias) re-reads
|
|
6509
|
+
* the file state against the brief and fixes any gap before the result is re-voiced. Bounded to one
|
|
6510
|
+
* pass; ~2x Act cost so default OFF. The self-verify FOOTER (same context) was measured ineffective —
|
|
6511
|
+
* this is the structural fix (see mind/10). Think tasks are pure reasoning, never checked. */
|
|
6512
|
+
verifyActTasks = false;
|
|
6513
|
+
/** Receives the voice text_delta stream + task lifecycle events. */
|
|
6514
|
+
host;
|
|
6515
|
+
/** How many recent transcript messages are rendered into a worker's brief. */
|
|
6516
|
+
excerptTurns = 6;
|
|
6517
|
+
/** Voice register: 'neutral' = clean spoken style; 'conversational' = human-like — fillers,
|
|
6518
|
+
* backchannels, impulsive first reactions before content (mimics real duplex conversation). */
|
|
6519
|
+
voiceStyle = "neutral";
|
|
6520
|
+
/** Teach the model to emit inline `[emotion]` tags for Cartesia emotion control. Only set when the
|
|
6521
|
+
* TTS actually speaks them — text-duplex (no TTS) would otherwise print literal tags. */
|
|
6522
|
+
emotionTags = false;
|
|
6523
|
+
/** Awaited BEFORE a worker spawns — open a per-task checkpoint frame, audit, etc.
|
|
6524
|
+
* (post-spawn would race the worker's first edits). */
|
|
6525
|
+
onTaskStart;
|
|
6526
|
+
/** Re-voice throttled worker progress asides ('[task t1 progress] …') so long tasks aren't dead
|
|
6527
|
+
* air. Off by default — each update costs a voice turn (LLM call + speech). */
|
|
6528
|
+
progressUpdates = false;
|
|
6529
|
+
/** Min ms between progress re-voices per task. */
|
|
6530
|
+
progressIntervalMs = 25e3;
|
|
6531
|
+
/** Relay worker questions (AskUserQuestion + permission asks via parkQuestion) through the VOICE:
|
|
6532
|
+
* the question re-voices as '[task <id> asks] …', the user answers conversationally, and the
|
|
6533
|
+
* voice model resolves it with the AnswerTask tool. Off → host.ask passthrough (text menus). */
|
|
6534
|
+
askRelay = false;
|
|
6535
|
+
/** Parked questions auto-resolve empty after this long (callers map '' to deny/best-judgment). */
|
|
6536
|
+
askTimeoutMs = 12e4;
|
|
6537
|
+
/** Max retained task records: oldest SETTLED tasks (and their activity tails) are evicted past this,
|
|
6538
|
+
* bounding memory over a long-lived session. Running tasks are never evicted. */
|
|
6539
|
+
maxTaskRecords = 50;
|
|
6540
|
+
/** Host overrides for QuickLook lookups (keyed by `what`). The engine's defaults go through the
|
|
6541
|
+
* (possibly jailed) fs — e.g. `.git/**` is deny-listed, so the CLI supplies 'branch' itself. */
|
|
6542
|
+
quickLook;
|
|
6543
|
+
/** Memory directory/directories on the WORKER fs. If set, the voice agent gets Remember + Recall
|
|
6544
|
+
* tools directly (no delegation needed) and implicit capture guidance. */
|
|
6545
|
+
memoryDir;
|
|
6546
|
+
/** User-scope memory dir for global facts (type=user/feedback). Forwarded to Remember's routing. */
|
|
6547
|
+
memoryUserDir;
|
|
6548
|
+
};
|
|
6549
|
+
var RESERVED_EVENT_MARKER = /\[task\b[^\]\n]*\b(?:completed|failed|progress|asks)\b/i;
|
|
6550
|
+
var RESERVED_EVENT_OPENER = /\[\s*task\b/i;
|
|
6551
|
+
var STAGE_DIRECTION_RE = /^\(\s*(?:(?:waiting|checking|searching|thinking|processing|loading|working|fetching|looking)\b[^)]*|[^)]*(?:\.\.\.|…)\s*)\)$/i;
|
|
6552
|
+
function nowLine() {
|
|
6553
|
+
return `Current date and time: ${(/* @__PURE__ */ new Date()).toLocaleString("en-US", { weekday: "long", year: "numeric", month: "long", day: "numeric", hour: "numeric", minute: "2-digit", timeZoneName: "short" })}`;
|
|
6554
|
+
}
|
|
6555
|
+
function isTrivialBarge(text) {
|
|
6556
|
+
const t = text.trim().toLowerCase().replace(/[.,!?…\s]+$/g, "").replace(/^[.,!?…\s]+/, "");
|
|
6557
|
+
if (!t) return false;
|
|
6558
|
+
if (/\b(stop|cancel|no|nope|nah|never\s?mind|nvm|forget it|drop it|don'?t|quiet|shut up|enough|hold on|wait|pause|hang on|actually)\b/.test(t)) return false;
|
|
6559
|
+
const core = t.replace(/^(?:oh|um+|uh+|er+|hmm+|ah+|well|so|okay|ok|yeah|yep)[\s,]+/, "").trim();
|
|
6560
|
+
return /^(?:oh|um+|uh+|er+|hmm+|ah+|sorry|oops|whoops|my bad|pardon|excuse me|apologies|go on|go ahead|continue|keep going|carry on|as you were|please continue|you were saying|sorry go on|sorry continue|go on then|go on please)$/.test(core || t);
|
|
6561
|
+
}
|
|
6562
|
+
var VOICE_SYSTEM_PROMPT = 'You are a spoken voice assistant \u2014 the user HEARS everything you say. Use short sentences. One idea per sentence. No markdown, no bullet lists, no code blocks, no headings, no emoji. Never emit stage directions or parenthetical asides about your own process \u2014 nothing like "(waiting for the result...)" or "(checking)"; while work runs, either say it as plain speech or end your turn.\nThis holds even when asked to "print", "list", "show", or "make a table" \u2014 there is no screen for the spoken channel. Speak it as flowing prose ("Tuesday is half a meter, Wednesday a bit less\u2026"), or if they truly need it on screen, route it to Act to render. Never emit dashes or pipes into speech.\nKeep turns SHORT \u2014 one to three sentences, then stop. Never lecture, enumerate cases, or add caveats unprompted. Conversation is a fast exchange: give the one thing asked, and let the user pull more if they want it.\nYou have three cognitive tiers \u2014 like a human brain:\n\u2022 YOU (reflex) \u2014 instant, lightweight. Handle greetings, simple questions, status checks, QuickLook.\n\u2022 `Act` \u2014 your hands. A background worker with its own configured tools and access to the user\'s environment (files and shell{{WORKER_WEB}}). Use for reading, editing, searching, running tasks, building \u2014 any real work.\n{{THINK_SLOT}}\nWhen you are unsure whether you can do or access something, do NOT assume and do NOT claim a capability you have not confirmed. To check what you can do, QuickLook `capabilities` (instant \u2014 it lists your worker\'s real tools) and answer from that. Never promise an ability that is not in your capabilities; if it is not there, tell the user plainly you can\'t. To actually DO real work, call `Act`. When the user mentions their project, folder, files, or environment ("this project", "the current folder", "my code"), call `Act` IMMEDIATELY \u2014 do not ask for paths or details the worker can discover itself. Never pretend to have done the work or invent results \u2014 the worker\'s report is your only source.\nYou cannot mute the microphone or stop voice capture yourself \u2014 no tool does it. If the user asks you to stop listening or turn the voice off, never claim you did: tell them to say exactly "voice off" (handled by the app directly), or type /voice.\nYou are NOT a knowledge base. For any question whose answer needs SPECIFIC verifiable facts you do not already have in hand \u2014 how to build/configure/implement something, exact API, library, entitlement, command or option names, current events, or particular numbers, dates, or names \u2014 do NOT answer from your own memory: you will confidently make things up (a fake API, a wrong entitlement, an event that did not happen). Route it to `Act`, which can search and verify, and speak only what its report says. DELEGATION RULE \u2014 decide for yourself, the user never has to push: if you cannot answer confidently from the conversation plus trivial well-known knowledge, do NOT refuse and do NOT guess \u2014 dispatch `Act` immediately with a clear brief and say you are checking. Anything needing CURRENT data (weather, news, prices, dates, sky/astronomy, "right now"), real computation, or verification is an automatic dispatch \u2014 never a refusal. The user should never need to say "search the web" or "think harder" to make you act; needing fresh or verified information IS the trigger. Refuse only what your worker genuinely cannot do (check `capabilities`), and say why. Answer inline ONLY for general conversation, chit-chat, and trivia you are sure of, or facts you can see via QuickLook. When elaborating on a completed task ("tell me more", "the gist"), stay strictly within what that result actually said \u2014 if the user asks for something the result did not cover, that is NEW information: dispatch `Act`, do not improvise.\nALWAYS react before you work: the FIRST thing in your turn is a brief spoken acknowledgement of what you heard and what you are about to do ("got it \u2014 opening that now", "sure, let me pull it up", "okay, checking"). NEVER call a tool (Act, Think, QuickLook) silently \u2014 the user must hear you react before you go quiet to work. After dispatching Act or Think, that same one short sentence IS your turn \u2014 end it and do not wait for the result. Exactly ONE short line: never stack a second acknowledgement, and never narrate your own presence or status while waiting ("I\'m here", "let me see", "still checking" right after you already acked). One clean ack then silence reads as competent; repeated check-ins read as nervous.\nA completed task speaks its OWN result to the user (the worker voices what matters as it finishes) \u2014 you do NOT re-voice clean task results. A FAILED or INCOMPLETE task still arrives as a "[task t1 failed] \u2026" event for you to handle. The completed result stays in YOUR context \u2014 it is yours to draw on. When the user follows up ("tell me more", "what else", "and?"), answer FROM that result first: you already have the detail, so elaborate on what you have. Do NOT spawn a fresh worker to re-search or re-gather what you were just handed. Re-dispatch ONLY when genuinely new information is needed \u2014 e.g. the user wants the full contents of a SPECIFIC source, which is one WebFetch of that URL, not a brand-new search. "[task t1 progress] \u2026" events are interim status, NOT results \u2014 on a genuinely LONG wait you MAY give one brief half-sentence aside, but only occasionally; silence while working is normal and fine (the user knows you are on it). Do NOT narrate every step or re-announce yourself. Never present progress as a finished result.\nCRITICAL: while a task is still running you have NO answer yet \u2014 never state a specific result of any kind (a number, size, count, name, path, or value). The real answer arrives ONLY in the "[task \u2026 completed]" event; inventing one meanwhile (a made-up disk size, commit count, etc.) is a serious error. Until then, only acknowledge and wait.\nNever read raw file paths, diffs, or code aloud verbatim.\nDo NOT end every turn with the same canned offer ("want a rundown?", "want the steps?"). Offer once at most; if the user pushes back, repeats themselves, or sounds unsatisfied ("you know what I mean?", "think deeper", "are you sure?"), do NOT re-offer the same thing \u2014 change approach: dispatch `Act`/`Think` to actually dig in, or ask one concrete clarifying question. Repeating a non-answer is worse than silence.\n"[task t1 asks] \u2026" events are QUESTIONS from a background task \u2014 relay to the user in your own words, short, then end your turn. When the user answers, call `AnswerTask` with that id and their answer. NEVER answer on the user\'s behalf for permissions or risky operations; if their reply is ambiguous, confirm first.\nIf the user\'s message sounds INCOMPLETE \u2014 trailing off mid-sentence, a fragment that needs more context ("and then we", "but the problem is"), hesitation fillers ("uh", "um") \u2014 call `Hold` instead of answering. This keeps listening for the rest of their thought. Only respond with substance when you have a complete question or request.\nDispatch discipline: send ONE self-contained task per request \u2014 a single worker with the full brief beats several workers with fragments (each worker starts fresh and re-discovers context). NEVER dispatch a worker just to read files or gather information \u2014 workers explore and discover context themselves; pass on what you already know and let one worker do the whole job. Split into parallel tasks only when the user asks for genuinely independent things. When a task completes, report its result and stop \u2014 do NOT dispatch follow-up work (verification, polish, extras) the user did not ask for, unless the report itself signals failure or doubt.\nDo not fire a second Act/Think for work already in flight, and NEVER spawn a second task to re-count, cross-check, or verify a result a worker already gave you \u2014 trust its answer; a single question gets ONE task. Call `TaskStatus` at most ONCE per turn; if a task is still running, give one brief status aside and end the turn \u2014 never poll it again and again in a loop. "Still on it" is ONLY for a repeat check on a task you already acknowledged earlier \u2014 NEVER as the ack when you first dispatch work ("still" is wrong when nothing was running yet; use a fresh "on it"/"got it, checking" instead). Use `CancelTask` when the user asks to stop something.\nPRIORITY: when the user says goodbye or wants to end/finish/wrap up the session ("ok bye", "that\'s all", "let\'s finish", "let\'s end", "goodnight", "exit", "wrap up"), call `ExitSession` IMMEDIATELY \u2014 do not act, do not check status, just exit.\nFor TRIVIAL instant lookups only \u2014 current time, git branch, listing a folder, peeking at a small file, or checking your own `capabilities`/tools \u2014 use `QuickLook` (instant, no task). Whenever the user asks what you can do or whether you have some ability, QuickLook `capabilities` and answer from that \u2014 never guess. Anything requiring searching, reasoning, running commands, or editing goes through `Act`.\n{{MEMORY_SLOT}}\nUser messages may arrive via speech-to-text and can carry transcription artifacts \u2014 odd words, cut-offs, homophones ("for you" vs "folder"). Read for INTENT, not surface text. If a message seems garbled, surprising, or only half-parses, do NOT guess an action or improvise content from it \u2014 briefly confirm what they meant ("did you mean\u2026?") and wait. A one-line confirm beats a confident wrong answer or an invented response to a request you did not actually understand.';
|
|
6563
|
+
var THINK_GUIDANCE = "\u2022 `Think` \u2014 your brain. A premium reasoning model, FAR more expensive than Act. Reserve it for open-ended architecture/design questions, or a problem Act already FAILED at. ALL implementation work \u2014 coding, refactoring, debugging, edge cases, tests \u2014 goes to Act; Act is highly capable. Never send the same work to both.";
|
|
6564
|
+
var THINK_DISABLED_GUIDANCE = "(Think tier is not available \u2014 use Act for all escalations.)";
|
|
6565
|
+
var VOICE_STYLE_CONVERSATIONAL = `Speak like a person in a live conversation, not an assistant reading a script. React first, then deliver: a quick impulsive beat ("oh nice", "hmm, hold on", "ah, got it") before the substance. Use contractions always. Vary sentence length \u2014 some very short. Light fillers and backchannels are fine ("mm-hm", "right", "let's see") but at most one per reply \u2014 never stack them. When you escalate to Act or Think, say it like a human would ("hang on, let me actually dig into that \u2014 gimme a minute") instead of announcing a task. When a result comes back, react to it like you just found out ("okay so \u2014 turns out\u2026"). Match the user's energy: a quick question gets a quick answer \u2014 a few words is a perfectly good turn. Prefer a short answer plus an offer ("want the details?") over covering everything. Never narrate your own mechanics (no "I will now act", no task ids out loud).`;
|
|
6566
|
+
var EMOTION_TAGS_GUIDANCE = `EMOTION: your voice is synthesized with emotion control. Prefix a sentence with an inline [emotion] tag, placed directly before the sentence it colors, to shape how it is spoken. Use it ONLY when the emotion genuinely fits the words (it amplifies real feeling, it cannot fake it) \u2014 do not tag every sentence; reserve it for moments that carry feeling, and vary which one you use. You may also drop [laughter] for a natural laugh. Available emotions: ${EMOTIONS.join(", ")}.`;
|
|
6567
|
+
var DuplexAgent = class _DuplexAgent {
|
|
6568
|
+
options;
|
|
6569
|
+
voice;
|
|
6570
|
+
tasks = /* @__PURE__ */ new Map();
|
|
6571
|
+
queue = Promise.resolve();
|
|
6572
|
+
seq = 0;
|
|
6573
|
+
pendingEvents = [];
|
|
6574
|
+
/** Spoken text of parked deliveries that have SETTLED, awaiting a trivial-barge resume decision on the
|
|
6575
|
+
* next turn (a substantive turn clears it — the result stays in the transcript, recoverable by asking). */
|
|
6576
|
+
parkedRedeliver = [];
|
|
6577
|
+
/** A trivial barge arrived while a parked delivery was STILL RUNNING → resume it the moment it settles. */
|
|
6578
|
+
awaitTrivialRedeliver = false;
|
|
6579
|
+
/** Set by a SUBSTANTIVE turn after a barge: the user moved on, so a parked delivery that settles LATER
|
|
6580
|
+
* must NOT arm a resume (else a much-later "go on" replays a stale result). Reset by a fresh barge. */
|
|
6581
|
+
suppressParkedResume = false;
|
|
6582
|
+
/** Out-of-band follow-up attribution for the events coalescing into the next flush turn: TRUE iff ≥1 of
|
|
6583
|
+
* the tasks being integrated was NON-CLEAN (early-stop/failure). Carried out-of-band on the enqueue call
|
|
6584
|
+
* by the caller that KNOWS the outcome — a plain boolean the MODEL CANNOT PERTURB. It is NOT scanned from
|
|
6585
|
+
* worker-authored event text (v1: an "Outcome:" substring over-stamped siblings) and NOT keyed on a brief
|
|
6586
|
+
* string the reflex re-authors (v2: a paraphrased escalation brief missed the Set → followUp:false →
|
|
6587
|
+
* RE-ENABLED unbounded auto-escalation, the dangerous runaway direction). See [[wrong-discriminator]] /
|
|
6588
|
+
* [[drive-real-reflex]] / [[fakeaiclient-blind-to-wire-format]]. */
|
|
6589
|
+
pendingNonClean = false;
|
|
6590
|
+
flushQueued = false;
|
|
6591
|
+
/** Per-voice-turn guards (reset by resetTurn at each turn's start). The reflex is a weak model:
|
|
6592
|
+
* left unguarded it polls TaskStatus after a dispatch and/or dispatches silently (dead air).
|
|
6593
|
+
* Like CC's Task tool, a dispatch is "said my piece, now wait for the push" — these enforce that. */
|
|
6594
|
+
turnDispatched = false;
|
|
6595
|
+
// an Act/Think fired this turn
|
|
6596
|
+
spokeBeforeDispatch = false;
|
|
6597
|
+
// the reflex ALREADY acked before dispatching — the forced post-dispatch text step must stay silent (live: "Got it…" twice)
|
|
6598
|
+
turnBriefs = /* @__PURE__ */ new Set();
|
|
6599
|
+
// briefs dispatched this turn (detect identical re-dispatch)
|
|
6600
|
+
spokeThisTurn = false;
|
|
6601
|
+
// any non-empty text_delta streamed this turn
|
|
6602
|
+
externalSpeech = false;
|
|
6603
|
+
// host spoke on our behalf (adaptive micro-ack) — not reflex output
|
|
6604
|
+
heldThisTurn = false;
|
|
6605
|
+
// Hold called this turn → turn is INTENTIONALLY silent (suppress reflex text + no dead-air ack)
|
|
6606
|
+
nudging = false;
|
|
6607
|
+
// re-ack pass in flight: block ALL tools, prevent recursion
|
|
6608
|
+
reflexBuf = "";
|
|
6609
|
+
// accumulated reflex text this turn (fabricated-event detection)
|
|
6610
|
+
reflexForwarded = 0;
|
|
6611
|
+
// chars of reflexBuf already forwarded to the host/TTS
|
|
6612
|
+
fabricationCut = false;
|
|
6613
|
+
// reflex emitted a reserved [task …] marker → suppress its tail
|
|
6614
|
+
/** TRUE for the duration of a re-voice turn that is integrating ≥1 NON-CLEAN task (turn-eligibility,
|
|
6615
|
+
* carried out-of-band — NOT derived from any worker/brief string). ANY Act/Think dispatched in such a
|
|
6616
|
+
* turn is stamped followUp:true. This GUARANTEES the dangerous direction is impossible: a genuine
|
|
6617
|
+
* escalation (even one with a paraphrased brief) ALWAYS lands in a non-clean integration turn, so it is
|
|
6618
|
+
* ALWAYS recognized as a follow-up and CANNOT re-escalate (one hop). The single-dispatch-per-turn guard
|
|
6619
|
+
* means at most one dispatch happens per flush, so realistically "the one dispatch IS the escalation".
|
|
6620
|
+
* ACCEPTED SAFE-DIRECTION ERROR: if the reflex instead dispatches FRESH unrelated work during a non-clean
|
|
6621
|
+
* flush (rare — and only possible when it batches multiple calls in one step, bypassing the guard), that
|
|
6622
|
+
* fresh task is over-stamped followUp:true and forgoes ONE future auto-escalation. That is SAFE (it only
|
|
6623
|
+
* ever REMOVES a future escalation, never adds one — no runaway) and is the correct side to err on. */
|
|
6624
|
+
turnFollowUp = false;
|
|
6625
|
+
/** Hard absolute backstop against runaway regardless of attribution: total automatic escalations across
|
|
6626
|
+
* the whole conversation. Once it hits MAX_AUTO_ESCALATIONS, no integration turn offers escalate/re-delegate. */
|
|
6627
|
+
autoEscalations = 0;
|
|
6628
|
+
static MAX_AUTO_ESCALATIONS = 8;
|
|
6629
|
+
/** Parked worker questions awaiting a (voice-relayed) user answer, keyed by ask id. */
|
|
6630
|
+
pendingAsks = /* @__PURE__ */ new Map();
|
|
6631
|
+
/** SPECULATIVE REFLEX (VoiceEngine.onSpeculate → speculate()): the reflex turn starts on a stable
|
|
6632
|
+
* PARTIAL transcript, its spoken output buffered — nothing reaches TTS until the endpointed final
|
|
6633
|
+
* confirms via send() (which flushes the buffer and adopts the call as THE turn). A divergent final
|
|
6634
|
+
* aborts it: output dropped, history rolled back, the final dispatches normally. */
|
|
6635
|
+
spec;
|
|
6636
|
+
/** Aborted speculative calls (double-billing visibility — each one paid for tokens never spoken). */
|
|
6637
|
+
speculativeAbortedCalls = 0;
|
|
6638
|
+
/** Tools the reflex may execute on UNCONFIRMED (speculated) input — read-only lookups and the
|
|
6639
|
+
* intentionally-silent Hold. Anything else (Act/Think/AnswerTask/CancelTask/ExitSession/memory
|
|
6640
|
+
* writes/…) is a side effect on input the user may not have said: the guard ABORTS the speculation
|
|
6641
|
+
* instead (the endpointed final then dispatches normally and may use the tool for real). */
|
|
6642
|
+
static SPEC_SAFE_TOOLS = /* @__PURE__ */ new Set(["QuickLook", "TaskStatus", "Hold"]);
|
|
6643
|
+
/** Lazily resolved memory tools (async loadMemory runs in initMemory). */
|
|
6644
|
+
memoryReady;
|
|
6645
|
+
constructor(options) {
|
|
6646
|
+
this.options = { ...new DuplexAgentOptions(), ...options };
|
|
6647
|
+
const o = this.options;
|
|
6648
|
+
if (o.memoryDir && o.fs) {
|
|
6649
|
+
this.memoryReady = loadMemory(o.fs, o.memoryDir, { maxWritesPerSession: 10, userDir: o.memoryUserDir });
|
|
6650
|
+
}
|
|
6651
|
+
const memSlot = o.memoryDir && o.fs ? VOICE_MEMORY_PROMPT : "NEVER claim to have stored, saved, or remembered something durably \u2014 you cannot. Anything the user wants persisted (their name, preferences, notes) must go through Act so a worker writes it to memory.";
|
|
6652
|
+
const thinkSlot = o.thinkModel !== false ? THINK_GUIDANCE : THINK_DISABLED_GUIDANCE;
|
|
6653
|
+
const workerToolNames = (o.actOptions?.tools ?? []).map((t) => t.name);
|
|
6654
|
+
const canSearch = workerToolNames.some((n) => /WebSearch/i.test(n));
|
|
6655
|
+
const canFetch = workerToolNames.some((n) => /WebFetch/i.test(n));
|
|
6656
|
+
const workerWeb = canSearch ? `, and it CAN search the web and read web pages \u2014 so when the user gives you something specific to look up ("search for X", "find me\u2026", "what's the latest on\u2026"), route it to Act. But a bare capability QUESTION like "can you search the web?" just gets a short spoken "yes, I can" \u2014 do NOT dispatch and NEVER invent a query the user did not give you` : canFetch ? ", and it can fetch a specific web page URL (but cannot search the web)" : "";
|
|
6657
|
+
const mcpNames = [
|
|
6658
|
+
...Object.keys(o.actOptions?.providerOptions?.mcpServers ?? {}),
|
|
6659
|
+
...new Set(workerToolNames.filter((n) => n.startsWith("mcp__")).map((n) => n.slice(5).split("__")[0]))
|
|
6660
|
+
];
|
|
6661
|
+
const workerMcp = mcpNames.length ? `, and it can use these MCP servers: ${[...new Set(mcpNames)].join(", ")}` + (mcpNames.some((n) => /browser/i.test(n)) ? ' \u2014 including driving a REAL browser (open tabs, navigate, click, screenshot), so answer "yes" if asked whether you can control/drive a browser and route an actual browse to Act' : "") : "";
|
|
6662
|
+
const prompt = VOICE_SYSTEM_PROMPT.replace("{{MEMORY_SLOT}}", memSlot).replace("{{THINK_SLOT}}", thinkSlot).replace("{{WORKER_WEB}}", workerWeb + workerMcp) + (o.voiceStyle === "conversational" ? "\n" + VOICE_STYLE_CONVERSATIONAL : "") + (o.emotionTags ? "\n" + EMOTION_TAGS_GUIDANCE : "") + `
|
|
6663
|
+
${nowLine()}. Anchor every relative or time-sensitive reference \u2014 "today", "now", "current", "recent", "latest", "this year" \u2014 to THIS moment. When the user asks about something recent (news, sports, events), they mean near this date, not a well-known past instance; brief your worker with the current year so it searches for what is happening NOW, not a famous older event.`;
|
|
6664
|
+
const tools = [
|
|
6665
|
+
...o.reflexOptions?.tools ?? [],
|
|
6666
|
+
this.actTool(),
|
|
6667
|
+
...o.thinkModel !== false ? [this.thinkTool()] : [],
|
|
6668
|
+
this.taskStatusTool(),
|
|
6669
|
+
this.cancelTaskTool(),
|
|
6670
|
+
this.quickLookTool(),
|
|
6671
|
+
this.answerTaskTool(),
|
|
6672
|
+
this.holdTool()
|
|
6673
|
+
];
|
|
6674
|
+
const host = o.host;
|
|
6675
|
+
const voiceHost = host && {
|
|
6676
|
+
ask: host.ask ? (q) => host.ask(q) : void 0,
|
|
6677
|
+
confirm: host.confirm ? (p, m) => host.confirm(p, m) : void 0,
|
|
6678
|
+
notify: (ev) => {
|
|
6679
|
+
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
6680
|
+
if (this.heldThisTurn) return;
|
|
6681
|
+
if (this.fabricationCut) return;
|
|
6682
|
+
if (this.turnDispatched && this.spokeBeforeDispatch) return;
|
|
6683
|
+
const msg = ev.message;
|
|
6684
|
+
this.reflexBuf += msg;
|
|
6685
|
+
this.scrubStageDirections();
|
|
6686
|
+
const m = this.reflexBuf.match(RESERVED_EVENT_MARKER) ?? this.reflexBuf.match(RESERVED_EVENT_OPENER);
|
|
6687
|
+
if (m) {
|
|
6688
|
+
this.fabricationCut = true;
|
|
6689
|
+
log13.warn(`reflex fabricated a [task \u2026] event in its spoken stream \u2014 cutting it (kept ${m.index} chars)`);
|
|
6690
|
+
const safe = this.reflexBuf.slice(this.reflexForwarded, m.index);
|
|
6691
|
+
if (!safe) return;
|
|
6692
|
+
if (safe.trim()) this.spokeThisTurn = true;
|
|
6693
|
+
this.emitHost({ ...ev, message: safe });
|
|
6694
|
+
return;
|
|
7025
6695
|
}
|
|
7026
|
-
|
|
7027
|
-
|
|
6696
|
+
const held = this.reflexBuf.length - this.reflexForwarded;
|
|
6697
|
+
const partial = held > 0 && /\[\s*t?a?s?k?$/i.test(this.reflexBuf.slice(-Math.min(held, 6)));
|
|
6698
|
+
let upto = partial ? this.reflexBuf.length - this.reflexBuf.slice(-6).match(/\[\s*t?a?s?k?$/i)[0].length : this.reflexBuf.length;
|
|
6699
|
+
const paren = this.reflexBuf.lastIndexOf("(");
|
|
6700
|
+
if (paren >= this.reflexForwarded && !this.reflexBuf.includes(")", paren) && this.reflexBuf.length - paren <= 80)
|
|
6701
|
+
upto = Math.min(upto, paren);
|
|
6702
|
+
const out = this.reflexBuf.slice(this.reflexForwarded, upto);
|
|
6703
|
+
this.reflexForwarded = upto;
|
|
6704
|
+
if (!out) return;
|
|
6705
|
+
if (out.trim()) this.spokeThisTurn = true;
|
|
6706
|
+
this.emitHost({ ...ev, message: out });
|
|
6707
|
+
return;
|
|
7028
6708
|
}
|
|
6709
|
+
host.notify?.(ev);
|
|
7029
6710
|
}
|
|
7030
6711
|
};
|
|
6712
|
+
this.voice = new Agent({
|
|
6713
|
+
ai: o.ai,
|
|
6714
|
+
fs: new MemFilesystem2(),
|
|
6715
|
+
model: o.reflexModel,
|
|
6716
|
+
stream: true,
|
|
6717
|
+
host: voiceHost,
|
|
6718
|
+
// The reflex IS the conversational channel — it confirms ambiguity inline ("did you mean…?"),
|
|
6719
|
+
// never via the blocking AskUserQuestion tool (Agent auto-adds it whenever a host is set). Left in,
|
|
6720
|
+
// it stalls a voice turn until the kill-switch. Worker questions still reach the user via parkQuestion.
|
|
6721
|
+
askUserQuestion: false,
|
|
6722
|
+
systemPrompt: prompt,
|
|
6723
|
+
instructionFiles: false,
|
|
6724
|
+
maxSteps: 8,
|
|
6725
|
+
timeoutMs: 3e4,
|
|
6726
|
+
...o.reflexOptions,
|
|
6727
|
+
tools,
|
|
6728
|
+
// Composed AFTER the spread so the dispatch guard can't be dropped by reflexOptions.
|
|
6729
|
+
hooks: composeHooks(this.dispatchGuard(), o.reflexOptions?.hooks)
|
|
6730
|
+
});
|
|
7031
6731
|
}
|
|
7032
|
-
|
|
7033
|
-
|
|
7034
|
-
|
|
7035
|
-
|
|
7036
|
-
|
|
7037
|
-
|
|
7038
|
-
|
|
7039
|
-
properties: { id: { type: "string" }, answer: { type: "string", description: "the user's answer, verbatim or faithfully summarized" } }
|
|
7040
|
-
},
|
|
7041
|
-
run: async ({ id, answer }) => {
|
|
7042
|
-
const ask = this.pendingAsks.get(String(id));
|
|
7043
|
-
if (!ask) return `No pending question for '${id}' \u2014 it may have been answered already or timed out.`;
|
|
7044
|
-
ask.resolve(String(answer ?? ""));
|
|
7045
|
-
return `Answer relayed \u2014 task ${id} resumes.`;
|
|
7046
|
-
}
|
|
7047
|
-
};
|
|
6732
|
+
/** Resolve memory tools + inject index into voice system prompt (once). */
|
|
6733
|
+
async initMemory() {
|
|
6734
|
+
if (!this.memoryReady) return;
|
|
6735
|
+
const mem = await this.memoryReady;
|
|
6736
|
+
this.memoryReady = void 0;
|
|
6737
|
+
this.voice.options.tools.push(...mem.tools);
|
|
6738
|
+
if (mem.index) this.voice.options.systemPrompt += "\n\n" + mem.index;
|
|
7048
6739
|
}
|
|
7049
|
-
|
|
6740
|
+
/** Flush any held-back trailing fragment (a possible `[task` opener that never completed) once the
|
|
6741
|
+
* turn's stream is done — so a legit message ending in "[t" isn't silently dropped. */
|
|
6742
|
+
flushHeldReflexTail() {
|
|
6743
|
+
if (this.fabricationCut) return;
|
|
6744
|
+
this.scrubStageDirections();
|
|
6745
|
+
const tail = this.reflexBuf.slice(this.reflexForwarded);
|
|
6746
|
+
this.reflexForwarded = this.reflexBuf.length;
|
|
6747
|
+
if (!tail) return;
|
|
6748
|
+
if (tail.trim()) this.spokeThisTurn = true;
|
|
6749
|
+
this.emitHost({ kind: "text_delta", message: tail });
|
|
6750
|
+
}
|
|
6751
|
+
/** Remove complete stage-direction parentheticals from the UNFORWARDED reflex text (STAGE_DIRECTION_RE).
|
|
6752
|
+
* Only the unforwarded region is touched — already-spoken audio can't be unsent, and splicing before
|
|
6753
|
+
* reflexForwarded would corrupt the forward offset. */
|
|
6754
|
+
scrubStageDirections() {
|
|
6755
|
+
if (this.reflexForwarded >= this.reflexBuf.length) return;
|
|
6756
|
+
const region = this.reflexBuf.slice(this.reflexForwarded);
|
|
6757
|
+
const scrubbed = region.replace(/\([^()]*\)/g, (s) => STAGE_DIRECTION_RE.test(s) ? "" : s);
|
|
6758
|
+
if (scrubbed !== region) this.reflexBuf = this.reflexBuf.slice(0, this.reflexForwarded) + scrubbed;
|
|
6759
|
+
}
|
|
6760
|
+
/** Clear the per-turn guards. Called at the head of every voice turn (user send + re-voice flush). */
|
|
6761
|
+
resetTurn() {
|
|
6762
|
+
this.turnDispatched = false;
|
|
6763
|
+
this.spokeBeforeDispatch = false;
|
|
6764
|
+
this.turnBriefs.clear();
|
|
6765
|
+
this.spokeThisTurn = false;
|
|
6766
|
+
this.externalSpeech = false;
|
|
6767
|
+
this.heldThisTurn = false;
|
|
6768
|
+
this.reflexBuf = "";
|
|
6769
|
+
this.reflexForwarded = 0;
|
|
6770
|
+
this.fabricationCut = false;
|
|
6771
|
+
this.turnFollowUp = false;
|
|
6772
|
+
this.voice.options.toolChoice = void 0;
|
|
6773
|
+
}
|
|
6774
|
+
/** preToolUse guard on the reflex: once it has dispatched this turn, a dispatch is "said my piece,
|
|
6775
|
+
* now wait for the push" (CC's Task model). Block the temptations — TaskStatus polling and identical
|
|
6776
|
+
* re-dispatch — so the only remaining move is to voice a short ack and end. A genuinely NEW Act is
|
|
6777
|
+
* still allowed (parallel independent work). During a re-ack pass, block every tool. */
|
|
6778
|
+
dispatchGuard() {
|
|
7050
6779
|
return {
|
|
7051
|
-
|
|
7052
|
-
|
|
7053
|
-
|
|
7054
|
-
|
|
7055
|
-
|
|
7056
|
-
filler: { type: "string", description: 'optional short filler to speak ("mhm", "go on", "mm-hm")' }
|
|
6780
|
+
preToolUse: (call) => {
|
|
6781
|
+
if (this.spec?.state === "pending" && !_DuplexAgent.SPEC_SAFE_TOOLS.has(call.name)) {
|
|
6782
|
+
log13.verbose(`speculation aborted: reflex called ${call.name} on unconfirmed input`);
|
|
6783
|
+
this.abortSpeculation();
|
|
6784
|
+
return { block: true, reason: "Speculative turn aborted." };
|
|
7057
6785
|
}
|
|
7058
|
-
|
|
7059
|
-
|
|
7060
|
-
|
|
7061
|
-
|
|
7062
|
-
|
|
6786
|
+
if (this.nudging) return { block: true, reason: "Just say one short spoken acknowledgement \u2014 no tools this turn." };
|
|
6787
|
+
if (!this.turnDispatched) return;
|
|
6788
|
+
if (call.name === "TaskStatus")
|
|
6789
|
+
return { block: true, reason: "You just dispatched a task this turn \u2014 do NOT poll. Give one short spoken acknowledgement and end your turn; the result arrives later as a [task \u2026] event." };
|
|
6790
|
+
if ((call.name === "Act" || call.name === "Think") && this.turnBriefs.has(String(call.args?.brief ?? "")))
|
|
6791
|
+
return { block: true, reason: "You already dispatched this exact task \u2014 acknowledge briefly and end your turn." };
|
|
7063
6792
|
}
|
|
7064
6793
|
};
|
|
7065
6794
|
}
|
|
7066
|
-
|
|
7067
|
-
|
|
7068
|
-
|
|
7069
|
-
|
|
7070
|
-
|
|
7071
|
-
|
|
7072
|
-
};
|
|
6795
|
+
/** The host spoke on this turn's behalf OUTSIDE the reflex stream (e.g. the voice engine's adaptive
|
|
6796
|
+
* micro-ack). For a DISPATCHED turn that's a sufficient ack (don't stack a second one); for an
|
|
6797
|
+
* inline turn whose reply came back empty it is NOT — "One sec." followed by permanent silence is
|
|
6798
|
+
* still dead air, so silentTurn ignores external speech unless work was dispatched. */
|
|
6799
|
+
noteExternalSpeech() {
|
|
6800
|
+
this.externalSpeech = true;
|
|
7073
6801
|
}
|
|
7074
|
-
|
|
7075
|
-
|
|
7076
|
-
|
|
7077
|
-
|
|
7078
|
-
|
|
7079
|
-
|
|
7080
|
-
|
|
7081
|
-
|
|
7082
|
-
const texts = [];
|
|
7083
|
-
const images = [];
|
|
7084
|
-
for (const c of content) {
|
|
7085
|
-
if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
|
|
7086
|
-
images.push({ mimeType: c.mimeType, data: c.data });
|
|
7087
|
-
} else if (typeof c?.text === "string") {
|
|
7088
|
-
texts.push(c.text);
|
|
7089
|
-
} else {
|
|
7090
|
-
texts.push(JSON.stringify(c));
|
|
7091
|
-
}
|
|
7092
|
-
}
|
|
7093
|
-
const text = texts.join("\n");
|
|
7094
|
-
if (text || images.length) return { text, ...images.length ? { images } : {} };
|
|
6802
|
+
/** True when the just-finished turn voiced NOTHING — dead air to repair. Two ways this happens:
|
|
6803
|
+
* (a) a dispatch with no spoken ack, and (b) an INLINE turn whose `final` channel came back empty —
|
|
6804
|
+
* gpt-oss harmony sometimes puts the whole reply in `analysis` (→ thinking_delta, suppressed in
|
|
6805
|
+
* voice) and emits an empty `final`, so no text_delta ever streams. Both ship silence; both repair.
|
|
6806
|
+
* Requires a host: without one there's no stream to detect speech on (and no one to speak to). */
|
|
6807
|
+
get silentTurn() {
|
|
6808
|
+
const ackedByHost = this.externalSpeech && this.turnDispatched;
|
|
6809
|
+
return !!this.options.host && !this.spokeThisTurn && !ackedByHost && !this.heldThisTurn;
|
|
7095
6810
|
}
|
|
7096
|
-
|
|
7097
|
-
|
|
7098
|
-
|
|
7099
|
-
|
|
7100
|
-
|
|
7101
|
-
|
|
7102
|
-
|
|
7103
|
-
|
|
7104
|
-
|
|
7105
|
-
|
|
6811
|
+
/** A turn that voiced nothing is dead air. Re-prompt the reflex ONCE so the LLM itself voices a short
|
|
6812
|
+
* line (no template). If it STILL says nothing, fall back to a minimal line so silence never ships.
|
|
6813
|
+
* Wording adapts to whether work was dispatched (an ack) or the inline reply was simply lost. */
|
|
6814
|
+
async ackIfSilent(fallback) {
|
|
6815
|
+
const dispatched = this.turnDispatched;
|
|
6816
|
+
this.nudging = true;
|
|
6817
|
+
try {
|
|
6818
|
+
await this.voice.send(fallback ? "[reminder] You said nothing to the user this turn. Tell them, in ONE short spoken sentence, what just happened \u2014 no tools." : dispatched ? "[reminder] You dispatched a task but said nothing to the user. Say ONE short spoken acknowledgement now \u2014 no tools." : "[reminder] You said nothing to the user this turn. Give your ONE short spoken reply now \u2014 no tools.");
|
|
6819
|
+
} catch (e) {
|
|
6820
|
+
log13.warn(`ack nudge failed: ${e instanceof Error ? e.message : e}`);
|
|
6821
|
+
} finally {
|
|
6822
|
+
this.nudging = false;
|
|
7106
6823
|
}
|
|
7107
|
-
|
|
7108
|
-
|
|
7109
|
-
|
|
7110
|
-
|
|
7111
|
-
|
|
7112
|
-
|
|
7113
|
-
const schema = s.inputSchema ? `
|
|
7114
|
-
args: ${JSON.stringify(s.inputSchema)}` : "";
|
|
7115
|
-
return `${s.name} \u2014 ${s.description ?? "(no description)"}${schema}`;
|
|
7116
|
-
}
|
|
7117
|
-
function makeMcpToolSearch(specs, callTool, options = {}) {
|
|
7118
|
-
const maxResults = options.maxResults ?? 10;
|
|
7119
|
-
const byName = new Map(specs.map((s) => [s.name, s]));
|
|
7120
|
-
const catalogLine = `${specs.length} MCP tool(s) available \u2014 search by keyword, then call by exact name.`;
|
|
7121
|
-
const searchTool = {
|
|
7122
|
-
name: "ToolSearch",
|
|
7123
|
-
description: `Search the available MCP tools by keyword (${catalogLine}). Returns matching tool names + their argument schemas; call one with \`McpCall\`.`,
|
|
7124
|
-
parameters: { type: "object", required: ["query"], properties: { query: { type: "string", description: "keywords to match against tool name + description" } } },
|
|
7125
|
-
async run({ query }) {
|
|
7126
|
-
const q = String(query ?? "").trim();
|
|
7127
|
-
if (!q) return catalogLine;
|
|
7128
|
-
const { kept } = topByRelevance(specs, q, (s) => `${s.name} ${s.description ?? ""}`, maxResults);
|
|
7129
|
-
if (!kept.length) return `(no MCP tool matches "${q}" \u2014 try broader keywords)`;
|
|
7130
|
-
return kept.map(describeSpec).join("\n");
|
|
6824
|
+
if (!this.spokeThisTurn) {
|
|
6825
|
+
const pool = fallback ? [fallback] : dispatched ? _DuplexAgent.FALLBACK_ACKS : _DuplexAgent.FALLBACK_RETRY;
|
|
6826
|
+
const fresh = pool.filter((p) => p !== this.lastFallback);
|
|
6827
|
+
const line = (fresh.length ? fresh : pool)[Math.floor(Math.random() * (fresh.length || pool.length))];
|
|
6828
|
+
this.lastFallback = line;
|
|
6829
|
+
this.emitHost({ kind: "text_delta", message: line });
|
|
7131
6830
|
}
|
|
7132
|
-
}
|
|
7133
|
-
|
|
7134
|
-
|
|
7135
|
-
|
|
7136
|
-
|
|
7137
|
-
|
|
7138
|
-
|
|
7139
|
-
|
|
7140
|
-
|
|
7141
|
-
|
|
7142
|
-
|
|
7143
|
-
|
|
7144
|
-
|
|
7145
|
-
|
|
7146
|
-
|
|
7147
|
-
|
|
7148
|
-
|
|
6831
|
+
}
|
|
6832
|
+
/** Dead-air fallback pools (see ackIfSilent). Both retry lines keep the 'say that again' phrase —
|
|
6833
|
+
* hosts/tests key on it. */
|
|
6834
|
+
static FALLBACK_ACKS = ["Okay, on it.", "On it.", "Alright, working on it."];
|
|
6835
|
+
static FALLBACK_RETRY = ["Sorry, could you say that again?", "Hm, I missed that \u2014 could you say that again?"];
|
|
6836
|
+
lastFallback = "";
|
|
6837
|
+
/** One user turn: the voice agent streams the reply (and may Act/Think). Serialized with re-voice turns.
|
|
6838
|
+
* If a speculative reflex call is pending and this content CONFIRMS it (the final matches the
|
|
6839
|
+
* speculated partial), the in-flight call is adopted as THE turn: its buffered output flushes to the
|
|
6840
|
+
* host NOW (this is the latency win) and streaming continues live. Any other content aborts the
|
|
6841
|
+
* speculation first (rolled back silently) and runs a normal turn behind it. */
|
|
6842
|
+
send(content) {
|
|
6843
|
+
if (typeof content === "string" && isTrivialBarge(content) && this.hasParkedDelivery()) {
|
|
6844
|
+
if (this.spec) this.abortSpeculation();
|
|
6845
|
+
return this.enqueue(async () => {
|
|
6846
|
+
this.resetTurn();
|
|
6847
|
+
if (!this.redeliverParked()) this.awaitTrivialRedeliver = true;
|
|
6848
|
+
return { text: "", steps: 0, finishReason: "stop", messages: [] };
|
|
6849
|
+
});
|
|
7149
6850
|
}
|
|
7150
|
-
|
|
7151
|
-
|
|
7152
|
-
|
|
7153
|
-
|
|
7154
|
-
|
|
7155
|
-
|
|
7156
|
-
|
|
7157
|
-
|
|
7158
|
-
|
|
7159
|
-
|
|
7160
|
-
|
|
7161
|
-
specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
|
|
7162
|
-
routes.set(display, { server: m.name, rawName: s.name });
|
|
6851
|
+
const spec = this.spec;
|
|
6852
|
+
if (spec?.state === "pending") {
|
|
6853
|
+
if (typeof content === "string" && speculationConfirms(spec.text, content)) {
|
|
6854
|
+
spec.state = "confirmed";
|
|
6855
|
+
spec.finalText = content;
|
|
6856
|
+
for (const ev of spec.buf.splice(0)) this.options.host?.notify?.(ev);
|
|
6857
|
+
spec.decide("confirm");
|
|
6858
|
+
this.notify("diag", "speculation_confirmed", { text: spec.text.slice(0, 80) });
|
|
6859
|
+
return spec.done;
|
|
6860
|
+
}
|
|
6861
|
+
this.abortSpeculation();
|
|
7163
6862
|
}
|
|
6863
|
+
return this.enqueue(async () => {
|
|
6864
|
+
await this.initMemory();
|
|
6865
|
+
this.resetTurn();
|
|
6866
|
+
this.awaitTrivialRedeliver = false;
|
|
6867
|
+
this.parkedRedeliver = [];
|
|
6868
|
+
this.suppressParkedResume = true;
|
|
6869
|
+
const res = await this.voice.send(content);
|
|
6870
|
+
this.flushHeldReflexTail();
|
|
6871
|
+
if (this.silentTurn) await this.ackIfSilent();
|
|
6872
|
+
return res;
|
|
6873
|
+
});
|
|
7164
6874
|
}
|
|
7165
|
-
|
|
7166
|
-
|
|
7167
|
-
|
|
7168
|
-
|
|
7169
|
-
|
|
7170
|
-
|
|
7171
|
-
|
|
7172
|
-
|
|
7173
|
-
|
|
7174
|
-
|
|
7175
|
-
|
|
7176
|
-
|
|
7177
|
-
|
|
7178
|
-
|
|
7179
|
-
|
|
7180
|
-
|
|
7181
|
-
|
|
7182
|
-
|
|
7183
|
-
|
|
7184
|
-
|
|
7185
|
-
|
|
7186
|
-
|
|
7187
|
-
|
|
7188
|
-
|
|
7189
|
-
|
|
6875
|
+
/** Start a HELD speculative reflex turn on a stable partial (VoiceEngine.onSpeculate). Its spoken
|
|
6876
|
+
* output buffers in emitHost; send() later confirms (flush + adopt) or aborts it. The turn holds the
|
|
6877
|
+
* voice mutex until that decision (bounded by a safety timeout), so no other turn can interleave
|
|
6878
|
+
* into the transcript between the speculative messages and their rollback. No-op if a speculation
|
|
6879
|
+
* is already in flight. */
|
|
6880
|
+
speculate(text) {
|
|
6881
|
+
if (!text.trim() || this.spec) return;
|
|
6882
|
+
let decide;
|
|
6883
|
+
const decision = new Promise((r) => {
|
|
6884
|
+
decide = r;
|
|
6885
|
+
});
|
|
6886
|
+
const spec = {
|
|
6887
|
+
text,
|
|
6888
|
+
state: "pending",
|
|
6889
|
+
buf: [],
|
|
6890
|
+
ctl: new AbortController(),
|
|
6891
|
+
decide,
|
|
6892
|
+
decision,
|
|
6893
|
+
done: void 0
|
|
6894
|
+
};
|
|
6895
|
+
this.spec = spec;
|
|
6896
|
+
spec.done = this.enqueue(async () => {
|
|
6897
|
+
const empty = { text: "", steps: 0, finishReason: "aborted", messages: this.voice.transcript };
|
|
6898
|
+
if (spec.state === "aborted") {
|
|
6899
|
+
this.speculativeAbortedCalls++;
|
|
6900
|
+
if (this.spec === spec) this.spec = void 0;
|
|
6901
|
+
return empty;
|
|
6902
|
+
}
|
|
6903
|
+
await this.initMemory();
|
|
6904
|
+
this.resetTurn();
|
|
6905
|
+
const base = this.voice.transcript.length;
|
|
6906
|
+
const prevSignal = this.voice.options.signal;
|
|
6907
|
+
this.voice.options.signal = spec.ctl.signal;
|
|
6908
|
+
let res;
|
|
6909
|
+
try {
|
|
6910
|
+
res = await this.voice.send(spec.text);
|
|
6911
|
+
} catch (e) {
|
|
6912
|
+
log13.warn(`speculative turn failed: ${e instanceof Error ? e.message : e}`);
|
|
6913
|
+
} finally {
|
|
6914
|
+
this.voice.options.signal = prevSignal;
|
|
6915
|
+
}
|
|
6916
|
+
const timer = setTimeout(() => {
|
|
6917
|
+
spec.state = spec.state === "pending" ? "aborted" : spec.state;
|
|
6918
|
+
spec.decide("abort");
|
|
6919
|
+
}, 1e4);
|
|
6920
|
+
timer.unref?.();
|
|
6921
|
+
const d = await spec.decision;
|
|
6922
|
+
clearTimeout(timer);
|
|
6923
|
+
if (d === "abort") {
|
|
6924
|
+
if (this.voice.transcript.length > base) this.voice.transcript.length = base;
|
|
6925
|
+
this.speculativeAbortedCalls++;
|
|
6926
|
+
log13.verbose(`speculation aborted (${this.speculativeAbortedCalls} total): "${spec.text.slice(0, 50)}"`);
|
|
6927
|
+
if (this.spec === spec) this.spec = void 0;
|
|
6928
|
+
return res ?? empty;
|
|
6929
|
+
}
|
|
6930
|
+
for (let i = base; i < this.voice.transcript.length; i++) {
|
|
6931
|
+
const m = this.voice.transcript[i];
|
|
6932
|
+
if (m.role === "user" && contentText(m.content) === spec.text) {
|
|
6933
|
+
m.content = spec.finalText;
|
|
6934
|
+
break;
|
|
6935
|
+
}
|
|
6936
|
+
}
|
|
6937
|
+
this.spec = void 0;
|
|
6938
|
+
this.flushHeldReflexTail();
|
|
6939
|
+
if (this.silentTurn) await this.ackIfSilent();
|
|
6940
|
+
return res ?? empty;
|
|
6941
|
+
});
|
|
7190
6942
|
}
|
|
7191
|
-
|
|
7192
|
-
|
|
7193
|
-
|
|
7194
|
-
|
|
7195
|
-
|
|
7196
|
-
|
|
7197
|
-
|
|
7198
|
-
|
|
7199
|
-
|
|
6943
|
+
/** Abort a pending speculation (idempotent): kill the in-flight call, drop its buffered output.
|
|
6944
|
+
* Called on divergence (send), a stale partial (VoiceEngine.onSpeculateAbort), a side-effect tool
|
|
6945
|
+
* attempt (dispatchGuard), or host paths that will never dispatch a final (commands, voice-off). */
|
|
6946
|
+
abortSpeculation() {
|
|
6947
|
+
const spec = this.spec;
|
|
6948
|
+
if (spec?.state !== "pending") return;
|
|
6949
|
+
spec.state = "aborted";
|
|
6950
|
+
spec.buf.length = 0;
|
|
6951
|
+
spec.ctl.abort();
|
|
6952
|
+
spec.decide("abort");
|
|
6953
|
+
this.notify("diag", "speculation_aborted", { text: spec.text.slice(0, 80) });
|
|
7200
6954
|
}
|
|
7201
|
-
|
|
7202
|
-
|
|
6955
|
+
/** Cancel a running background task — shared by the CancelTask tool and the CLI /tasks picker. */
|
|
6956
|
+
cancelTask(id) {
|
|
6957
|
+
const rec = this.tasks.get(id);
|
|
6958
|
+
if (!rec) return `No task '${id}'.`;
|
|
6959
|
+
if (rec.status !== "running") return `Task ${rec.id} is already ${rec.status}.`;
|
|
6960
|
+
rec.status = "cancelled";
|
|
6961
|
+
rec.controller.abort();
|
|
6962
|
+
return `Task ${rec.id} (${rec.label}) cancelled.`;
|
|
7203
6963
|
}
|
|
7204
|
-
|
|
7205
|
-
|
|
6964
|
+
/** Barge-in: the user took the floor while task(s) were running. Suppress those tasks' remaining SPOKEN
|
|
6965
|
+
* delivery so a superseded topic never talks over the new one (the debt-after-jokes regression). The
|
|
6966
|
+
* tasks keep running and still fold their result into the transcript — recoverable, just not spoken.
|
|
6967
|
+
* Returns the parked ids (for logging). Does NOT cancel: that's a deliberate reflex/user action. */
|
|
6968
|
+
parkInFlightDeliveries() {
|
|
6969
|
+
this.suppressParkedResume = false;
|
|
6970
|
+
const parked = [];
|
|
6971
|
+
for (const rec of this.tasks.values())
|
|
6972
|
+
if (rec.status === "running" && !rec.deliveryParked) {
|
|
6973
|
+
rec.deliveryParked = true;
|
|
6974
|
+
parked.push(rec.id);
|
|
6975
|
+
}
|
|
6976
|
+
return parked;
|
|
7206
6977
|
}
|
|
7207
|
-
|
|
7208
|
-
|
|
6978
|
+
/** True while there is a parked delivery to potentially resume: either one already settled (queued) or
|
|
6979
|
+
* one still running that was parked by a barge. */
|
|
6980
|
+
hasParkedDelivery() {
|
|
6981
|
+
return this.parkedRedeliver.length > 0 || [...this.tasks.values()].some((t) => t.status === "running" && t.deliveryParked);
|
|
7209
6982
|
}
|
|
7210
|
-
|
|
7211
|
-
|
|
7212
|
-
|
|
7213
|
-
|
|
7214
|
-
this.
|
|
7215
|
-
|
|
6983
|
+
/** Speak any settled-and-queued parked delivery now (trivial-barge resume). Returns false if nothing was
|
|
6984
|
+
* queued yet (the caller then arms awaitTrivialRedeliver so the task resumes the instant it settles). */
|
|
6985
|
+
redeliverParked() {
|
|
6986
|
+
if (!this.parkedRedeliver.length) return false;
|
|
6987
|
+
const text = this.parkedRedeliver.splice(0).join(" ").trim();
|
|
6988
|
+
if (text) {
|
|
6989
|
+
this.options.host?.notify?.({ kind: "speak_utterance", message: text });
|
|
6990
|
+
this.notify("diag", "parked_redelivered", { chars: text.length });
|
|
6991
|
+
}
|
|
6992
|
+
return true;
|
|
7216
6993
|
}
|
|
7217
|
-
|
|
7218
|
-
|
|
7219
|
-
|
|
7220
|
-
|
|
7221
|
-
|
|
7222
|
-
|
|
7223
|
-
|
|
7224
|
-
|
|
7225
|
-
|
|
6994
|
+
/** Resolve when all queued voice turns AND all in-flight worker tasks have settled (tests, graceful shutdown). */
|
|
6995
|
+
async idle() {
|
|
6996
|
+
while (true) {
|
|
6997
|
+
const q = this.queue;
|
|
6998
|
+
await q.catch(() => {
|
|
6999
|
+
});
|
|
7000
|
+
await Promise.all([...this.tasks.values()].map((t) => t.promise));
|
|
7001
|
+
if (this.queue === q && ![...this.tasks.values()].some((t) => t.status === "running")) return;
|
|
7002
|
+
}
|
|
7226
7003
|
}
|
|
7227
|
-
|
|
7228
|
-
|
|
7229
|
-
|
|
7004
|
+
/** Promise-chain mutex: turns run strictly one at a time; a failed turn doesn't poison the chain. */
|
|
7005
|
+
enqueue(fn) {
|
|
7006
|
+
const run = this.queue.then(fn, fn);
|
|
7007
|
+
this.queue = run.then(() => {
|
|
7008
|
+
}, () => {
|
|
7009
|
+
});
|
|
7010
|
+
return run;
|
|
7230
7011
|
}
|
|
7231
|
-
|
|
7232
|
-
this.
|
|
7012
|
+
notify(kind, message, data) {
|
|
7013
|
+
this.emitHost({ kind, message, data });
|
|
7233
7014
|
}
|
|
7234
|
-
|
|
7235
|
-
|
|
7015
|
+
/** Host-boundary emit for the reflex's spoken channel: during a PENDING speculation, text_delta and
|
|
7016
|
+
* hold_filler are BUFFERED (nothing may reach TTS on unconfirmed input); confirm flushes them in
|
|
7017
|
+
* order, abort drops them silently. Everything else (task_* lifecycle, worker speak_utterance —
|
|
7018
|
+
* which bypasses this via host.notify directly) passes through untouched. */
|
|
7019
|
+
emitHost(ev) {
|
|
7020
|
+
const spec = this.spec;
|
|
7021
|
+
if (spec && (ev.kind === "text_delta" || ev.kind === "hold_filler")) {
|
|
7022
|
+
if (spec.state === "pending") {
|
|
7023
|
+
spec.buf.push(ev);
|
|
7024
|
+
return;
|
|
7025
|
+
}
|
|
7026
|
+
if (spec.state === "aborted") return;
|
|
7027
|
+
}
|
|
7028
|
+
this.options.host?.notify?.(ev);
|
|
7236
7029
|
}
|
|
7237
|
-
|
|
7238
|
-
|
|
7239
|
-
|
|
7240
|
-
|
|
7030
|
+
/** Queue a `[task …]` event for re-voicing. Events arriving while the voice is busy coalesce into ONE turn.
|
|
7031
|
+
* `nonClean` (out-of-band boolean, set by the caller that KNOWS this event integrates a NON-CLEAN outcome)
|
|
7032
|
+
* marks the coalesced flush as a non-clean integration turn — turn-eligibility, never inferred from event
|
|
7033
|
+
* text and never keyed on a (re-authored) brief string. Any dispatch in such a turn is a follow-up. */
|
|
7034
|
+
queueRevoice(event, nonClean = false) {
|
|
7035
|
+
this.pendingEvents.push(event);
|
|
7036
|
+
if (nonClean) this.pendingNonClean = true;
|
|
7037
|
+
if (this.flushQueued) return;
|
|
7038
|
+
this.flushQueued = true;
|
|
7039
|
+
void this.enqueue(async () => {
|
|
7040
|
+
this.flushQueued = false;
|
|
7041
|
+
const events = this.pendingEvents.splice(0);
|
|
7042
|
+
const nonCleanTurn = this.pendingNonClean;
|
|
7043
|
+
this.pendingNonClean = false;
|
|
7044
|
+
if (!events.length) return;
|
|
7045
|
+
const failed = events.find((e) => /^\[task\b[^\]\n]*\bfailed\b/i.test(e));
|
|
7046
|
+
this.resetTurn();
|
|
7047
|
+
this.turnFollowUp = nonCleanTurn;
|
|
7048
|
+
await this.voice.send(events.join("\n"));
|
|
7049
|
+
this.flushHeldReflexTail();
|
|
7050
|
+
if (this.silentTurn) await this.ackIfSilent(failed ? "Sorry, that didn't work \u2014 the task failed." : void 0);
|
|
7051
|
+
this.notify("revoice_done", "");
|
|
7052
|
+
});
|
|
7053
|
+
}
|
|
7054
|
+
/** The worker's brief: the Act/Think args + a STATIC text snapshot of the recent conversation.
|
|
7055
|
+
* Act briefs get a self-verify footer — the worker's report is trusted without review, so it
|
|
7056
|
+
* must check its own work before reporting (nearly free under prompt caching; measured honest:
|
|
7057
|
+
* it does NOT fix one-shot logic bugs — see mind/10). Think tasks are pure reasoning — no footer. */
|
|
7058
|
+
buildBrief(brief, tier = "act", deliver = true) {
|
|
7059
|
+
const recent = this.voice.transcript.filter((m) => (m.role === "user" || m.role === "assistant") && contentText(m.content).trim()).slice(-this.options.excerptTurns).map((m) => `${m.role}: ${contentText(m.content)}`).join("\n");
|
|
7060
|
+
const verify = tier === "act" ? "\n\nBefore reporting done: re-read what you changed and check it against EVERY requirement above \u2014 fix any gap first. Your report is trusted without review." : "";
|
|
7061
|
+
const deliverContract = deliver ? `
|
|
7241
7062
|
|
|
7242
|
-
|
|
7243
|
-
|
|
7063
|
+
## DELIVER (spoken delivery)
|
|
7064
|
+
You are reporting back to a user who is LISTENING. Stream your work normally \u2014 your prose is the written work record and detail, and is NOT spoken. Wrap anything the user should HEAR in <spoken>\u2026</spoken> tags. LEAD WITH the actual content they asked for: if they asked for a specific piece of content \u2014 a value, a name, the actual lines, the writing itself \u2014 that content goes INSIDE the <spoken> tags, not a remark about it. Your FIRST <spoken> segment is substantive \u2014 never a greeting or an acknowledgement (the front-end has already acked; do not double-ack). Keep spoken text concise and natural for the ear: short sentences, no markdown. NEVER enumerate in speech \u2014 no numbered or bulleted lists ("One. \u2026 Two. \u2026" is robotic). Deliver multiple items as flowing conversation with brief connective phrasing ("here's one\u2026", "and another\u2026", "oh, and\u2026"), pausing between items with sentence breaks, not numbers.` + (this.options.emotionTags ? " Inside <spoken>, you may prefix a sentence with an inline [emotion] tag (e.g. [excited], [curious]) to color how it is voiced \u2014 only when it genuinely fits, and vary it; [laughter] gives a natural laugh." : "") : "";
|
|
7065
|
+
return `${nowLine()}.
|
|
7244
7066
|
|
|
7245
|
-
|
|
7246
|
-
var STT_SAMPLE_RATE = 16e3;
|
|
7247
|
-
var TTS_SAMPLE_RATE = 44100;
|
|
7248
|
-
async function resolveAuth(auth) {
|
|
7249
|
-
return typeof auth === "function" ? await auth() : auth;
|
|
7250
|
-
}
|
|
7067
|
+
` + (recent ? `${brief}
|
|
7251
7068
|
|
|
7252
|
-
|
|
7253
|
-
|
|
7254
|
-
var now = () => performance.now();
|
|
7255
|
-
var SonioxSTTOptions = class {
|
|
7256
|
-
auth = "";
|
|
7257
|
-
source;
|
|
7258
|
-
model = "stt-rt-preview";
|
|
7259
|
-
languageHints = ["en"];
|
|
7260
|
-
/** Client-side endpoint: finalized text + no new tokens for this long = utterance (don't wait for
|
|
7261
|
-
* Soniox's semantic <end>, which adds 0.5-1.5s — the difference between ping-pong and lag). */
|
|
7262
|
-
silenceEndpointMs = 500;
|
|
7263
|
-
/** No-audio watchdog: if the mic source stops delivering chunks for this long, capture is dead →
|
|
7264
|
-
* fire onFatal + stop (else Soniox idle-timeouts and reconnect-loops forever). 0 = disable. */
|
|
7265
|
-
noAudioTimeoutMs = 1e4;
|
|
7266
|
-
};
|
|
7267
|
-
var SonioxSTT = class {
|
|
7268
|
-
options;
|
|
7269
|
-
ws;
|
|
7270
|
-
stopped = false;
|
|
7271
|
-
sourceStarted = false;
|
|
7272
|
-
onPartial = () => {
|
|
7273
|
-
};
|
|
7274
|
-
onUtterance = () => {
|
|
7275
|
-
};
|
|
7276
|
-
/** mic energy (RMS) per chunk — drives the energy-based heuristic barge-in tier */
|
|
7277
|
-
onLevel = () => {
|
|
7278
|
-
};
|
|
7279
|
-
/** Unrecoverable: the mic source stopped delivering audio (Soniox starves → idle-timeout reconnect
|
|
7280
|
-
* loop). The host tears voice down instead of spinning forever. */
|
|
7281
|
-
onFatal = () => {
|
|
7282
|
-
};
|
|
7283
|
-
/** Diagnostic tap (provider errors, ws reconnects, no-audio watchdog). Fire-and-forget: a throwing
|
|
7284
|
-
* handler disables itself — never affects transcription. Wired by VoiceIO / the lab bridge. */
|
|
7285
|
-
onDiag = () => {
|
|
7286
|
-
};
|
|
7287
|
-
diagOn = true;
|
|
7288
|
-
diag(kind, fields) {
|
|
7289
|
-
if (!this.diagOn) return;
|
|
7290
|
-
try {
|
|
7291
|
-
this.onDiag({ t: now(), kind, ...fields });
|
|
7292
|
-
} catch (e) {
|
|
7293
|
-
this.diagOn = false;
|
|
7294
|
-
log12.debug(`onDiag threw \u2014 STT diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
7295
|
-
}
|
|
7296
|
-
}
|
|
7297
|
-
lastChunkAt = 0;
|
|
7298
|
-
// timestamp of the most recent mic chunk (0 = none yet)
|
|
7299
|
-
startedChunksAt = 0;
|
|
7300
|
-
// when capture started (grace before the first chunk)
|
|
7301
|
-
noAudioTimer = null;
|
|
7302
|
-
finalText = "";
|
|
7303
|
-
partialText = "";
|
|
7304
|
-
lastChangeAt = 0;
|
|
7305
|
-
lastCombined = "";
|
|
7306
|
-
endpointTimer = null;
|
|
7307
|
-
firstTokenAt = 0;
|
|
7308
|
-
// first speech token in current utterance
|
|
7309
|
-
constructor(options) {
|
|
7310
|
-
this.options = { ...new SonioxSTTOptions(), ...options };
|
|
7069
|
+
## Recent conversation (for context)
|
|
7070
|
+
${recent}` : brief) + verify + deliverContract;
|
|
7311
7071
|
}
|
|
7312
|
-
|
|
7313
|
-
|
|
7072
|
+
/** Spawn a detached worker for task `id`; its settlement notifies + enqueues the re-voice turn. */
|
|
7073
|
+
spawnWorker(id, label, briefText, tier, brief, followUp) {
|
|
7074
|
+
const o = this.options;
|
|
7075
|
+
const tierOpts = tier === "think" ? o.thinkOptions : o.actOptions;
|
|
7076
|
+
const tierModel = tier === "think" ? o.thinkModel : o.actModel;
|
|
7077
|
+
const controller = new AbortController();
|
|
7078
|
+
const base = tierOpts?.hooks ?? o.actOptions?.hooks;
|
|
7079
|
+
const report = o.progressUpdates ? this.progressReporter(id) : void 0;
|
|
7080
|
+
const tail = [];
|
|
7081
|
+
const pushTail = (line) => {
|
|
7082
|
+
tail.push(line.slice(0, 200));
|
|
7083
|
+
if (tail.length > 120) tail.splice(0, tail.length - 120);
|
|
7084
|
+
};
|
|
7085
|
+
const hooks = {
|
|
7086
|
+
...base,
|
|
7087
|
+
preToolUse: async (call, meta) => {
|
|
7088
|
+
const d = await base?.preToolUse?.(call, meta);
|
|
7089
|
+
pushTail(`\u2699 ${describeCall(call)}`);
|
|
7090
|
+
report?.pre(call);
|
|
7091
|
+
return d;
|
|
7092
|
+
},
|
|
7093
|
+
postToolUse: async (call, result, meta) => {
|
|
7094
|
+
await base?.postToolUse?.(call, result, meta);
|
|
7095
|
+
const last = result?.trim().split("\n").filter(Boolean).pop();
|
|
7096
|
+
if (last) pushTail(` \u21B3 ${last}`);
|
|
7097
|
+
report?.post(call);
|
|
7098
|
+
},
|
|
7099
|
+
onToolOutput: (call, chunk, meta) => {
|
|
7100
|
+
base?.onToolOutput?.(call, chunk, meta);
|
|
7101
|
+
report?.output(chunk);
|
|
7102
|
+
}
|
|
7103
|
+
};
|
|
7104
|
+
const relayAsk = async (q) => {
|
|
7105
|
+
const opts = q.options?.length ? ` Options: ${q.options.map((x) => x.label).join(", ")}.` : "";
|
|
7106
|
+
const a = await this.parkQuestion(id, `${q.question}${opts}`);
|
|
7107
|
+
return a || "(no answer from the user \u2014 use your best judgment and note the assumption)";
|
|
7108
|
+
};
|
|
7109
|
+
const splitter = new SpokenSplitter();
|
|
7110
|
+
const speak = (seg) => {
|
|
7111
|
+
if (!seg) return;
|
|
7112
|
+
const r = this.tasks.get(id);
|
|
7113
|
+
if (r) r.spokenText = (r.spokenText ? r.spokenText + " " : "") + seg;
|
|
7114
|
+
if (!r?.deliveryParked) o.host?.notify?.({ kind: "speak_utterance", message: seg });
|
|
7115
|
+
};
|
|
7116
|
+
const coalescer = new SentenceCoalescer();
|
|
7117
|
+
const feedSpoken = (s) => {
|
|
7118
|
+
const ready = coalescer.feed(s);
|
|
7119
|
+
if (ready) speak(ready);
|
|
7120
|
+
};
|
|
7121
|
+
const flushSpoken = () => speak(coalescer.flush());
|
|
7122
|
+
const askBridge = o.askRelay ? { ask: relayAsk } : o.host?.ask ? { ask: (q) => o.host.ask(q) } : {};
|
|
7123
|
+
const workerHost = {
|
|
7124
|
+
...askBridge,
|
|
7125
|
+
notify: (ev) => {
|
|
7126
|
+
if (ev?.kind === "text_delta" && typeof ev.message === "string") {
|
|
7127
|
+
const { spoken, detail } = splitter.feed(ev.message);
|
|
7128
|
+
feedSpoken(spoken);
|
|
7129
|
+
if (detail.trim()) pushTail(detail.trim());
|
|
7130
|
+
return;
|
|
7131
|
+
}
|
|
7132
|
+
}
|
|
7133
|
+
};
|
|
7134
|
+
const agentOpts = {
|
|
7135
|
+
ai: o.ai,
|
|
7136
|
+
fs: o.fs,
|
|
7137
|
+
model: tierModel,
|
|
7138
|
+
...tier === "think" ? { reasoning: tierOpts?.reasoning ?? "high" } : {},
|
|
7139
|
+
...tierOpts,
|
|
7140
|
+
// Recompute providerOptions for THIS worker's model (after tierOpts so it wins over any inherited
|
|
7141
|
+
// main-template value) — prevents cursor-only cwd/cursorSession leaking onto an anthropic worker.
|
|
7142
|
+
providerOptions: o.providerOptionsFor?.(tierModel),
|
|
7143
|
+
stream: true,
|
|
7144
|
+
// worker streams text_delta so the splitter can extract <spoken> live (after tierOpts: never overridden off)
|
|
7145
|
+
host: workerHost,
|
|
7146
|
+
// carries BOTH ask AND the <spoken>-splitting notify
|
|
7147
|
+
...hooks ? { hooks } : {},
|
|
7148
|
+
signal: controller.signal
|
|
7149
|
+
// shared with the checker so a cancel tears down both
|
|
7150
|
+
};
|
|
7151
|
+
const promise = new Agent(agentOpts).run(briefText).then((res) => {
|
|
7152
|
+
const { spoken, detail } = splitter.flush();
|
|
7153
|
+
feedSpoken(spoken);
|
|
7154
|
+
if (detail.trim()) pushTail(detail.trim());
|
|
7155
|
+
flushSpoken();
|
|
7156
|
+
return res;
|
|
7157
|
+
}).then((res) => this.maybeVerify(id, brief, res, tier, agentOpts, askBridge)).then((res) => this.onWorkerSettled(id, res)).catch((err) => this.onWorkerFailed(id, err));
|
|
7158
|
+
this.tasks.set(id, { id, label, status: "running", controller, promise, tail, brief, followUp, splitter });
|
|
7159
|
+
if (this.tasks.size > this.options.maxTaskRecords)
|
|
7160
|
+
for (const [tid, rec] of this.tasks) {
|
|
7161
|
+
if (this.tasks.size <= this.options.maxTaskRecords) break;
|
|
7162
|
+
if (rec.status !== "running") this.tasks.delete(tid);
|
|
7163
|
+
}
|
|
7314
7164
|
}
|
|
7315
|
-
|
|
7316
|
-
|
|
7317
|
-
|
|
7318
|
-
|
|
7319
|
-
|
|
7320
|
-
|
|
7321
|
-
|
|
7322
|
-
this.
|
|
7323
|
-
|
|
7324
|
-
|
|
7325
|
-
|
|
7326
|
-
|
|
7327
|
-
sample_rate: STT_SAMPLE_RATE,
|
|
7328
|
-
num_channels: 1,
|
|
7329
|
-
language_hints: this.options.languageHints,
|
|
7330
|
-
enable_endpoint_detection: true
|
|
7331
|
-
})
|
|
7332
|
-
);
|
|
7333
|
-
this.ws.onmessage = (ev) => this.handle(JSON.parse(String(ev.data)));
|
|
7334
|
-
this.ws.onclose = (ev) => {
|
|
7335
|
-
if (this.stopped) return;
|
|
7336
|
-
log12.warn(`soniox ws closed (${ev.code} ${ev.reason || ""}) \u2014 reconnecting`);
|
|
7337
|
-
this.diag("stt_ws_closed", { code: ev.code, reason: String(ev.reason || ""), reconnecting: true });
|
|
7338
|
-
this.reset();
|
|
7339
|
-
this.connectWs().catch((e) => {
|
|
7340
|
-
log12.error(`soniox reconnect failed: ${e.message}`);
|
|
7341
|
-
this.diag("stt_reconnect_failed", { message: e.message });
|
|
7342
|
-
});
|
|
7165
|
+
/** Fresh-context check of a successful Act task: a NEW agent (same model/fs/tools, but NO shared
|
|
7166
|
+
* conversation context) re-reads the file state against the brief and fixes any gap. The fix lands
|
|
7167
|
+
* on the shared fs automatically (workers write fs directly, no overlay), so grading sees the
|
|
7168
|
+
* corrected state. Bounded to ONE pass. Off unless `verifyActTasks`; never runs for think/failed/
|
|
7169
|
+
* cancelled tasks. Usage is merged so /cost reflects the real (worker + checker) spend. */
|
|
7170
|
+
async maybeVerify(id, brief, res, tier, agentOpts, askBridge) {
|
|
7171
|
+
if (!this.options.verifyActTasks || tier !== "act" || res.finishReason !== "stop") return res;
|
|
7172
|
+
if (this.tasks.get(id)?.status === "cancelled") return res;
|
|
7173
|
+
const { stream: _stream, host: _host, ...restOpts } = agentOpts;
|
|
7174
|
+
const checkerOpts = {
|
|
7175
|
+
...restOpts,
|
|
7176
|
+
...askBridge.ask ? { host: { ask: askBridge.ask } } : {}
|
|
7343
7177
|
};
|
|
7344
|
-
|
|
7345
|
-
|
|
7346
|
-
|
|
7347
|
-
|
|
7348
|
-
this.
|
|
7349
|
-
|
|
7350
|
-
|
|
7351
|
-
|
|
7352
|
-
|
|
7353
|
-
this.reset();
|
|
7354
|
-
this.onUtterance(combined, now());
|
|
7355
|
-
}, 120);
|
|
7356
|
-
this.endpointTimer.unref?.();
|
|
7357
|
-
this.startedChunksAt = now();
|
|
7358
|
-
const noAudioMs = this.options.noAudioTimeoutMs;
|
|
7359
|
-
if (noAudioMs > 0) {
|
|
7360
|
-
this.noAudioTimer = setInterval(() => {
|
|
7361
|
-
if (this.stopped) return;
|
|
7362
|
-
const ref = this.lastChunkAt || this.startedChunksAt;
|
|
7363
|
-
if (now() - ref > noAudioMs) {
|
|
7364
|
-
log12.error(`stt: no mic audio for >${Math.round(noAudioMs / 1e3)}s \u2014 capture device stopped delivering`);
|
|
7365
|
-
this.diag("stt_watchdog_fatal", { noAudioMs });
|
|
7366
|
-
this.onFatal("microphone stopped delivering audio (try a different input device, e.g. AirPods, or check System Settings \u2192 Sound \u2192 Input)");
|
|
7367
|
-
this.stop();
|
|
7368
|
-
}
|
|
7369
|
-
}, Math.max(250, Math.min(2e3, noAudioMs / 4)));
|
|
7370
|
-
this.noAudioTimer.unref?.();
|
|
7178
|
+
const checkBrief = `${this.buildBrief(brief, tier, false)}
|
|
7179
|
+
|
|
7180
|
+
## VERIFY MODE
|
|
7181
|
+
Another agent just implemented the above. Independently check the CURRENT state of the files against EVERY requirement. Fix any gap you find. If everything is already correct, make NO changes \u2014 do not refactor or improve \u2014 and report "verified".`;
|
|
7182
|
+
this.notify("task_verify", `task ${id}: verifying`, { id });
|
|
7183
|
+
const cres = await new Agent(checkerOpts).run(checkBrief);
|
|
7184
|
+
if (cres.finishReason !== "stop") {
|
|
7185
|
+
log13.warn(`task ${id}: verify inconclusive (${cres.finishReason})`);
|
|
7186
|
+
this.notify("task_verify", `task ${id}: verify inconclusive (${cres.finishReason})`, { id, finishReason: cres.finishReason });
|
|
7371
7187
|
}
|
|
7372
|
-
|
|
7373
|
-
|
|
7374
|
-
|
|
7375
|
-
|
|
7376
|
-
|
|
7377
|
-
|
|
7378
|
-
|
|
7188
|
+
const sum = (a = 0, b = 0) => a + b;
|
|
7189
|
+
return {
|
|
7190
|
+
...res,
|
|
7191
|
+
steps: res.steps + cres.steps,
|
|
7192
|
+
// Merge the checker's messages so downstream tool-call/step accounting includes BOTH agents
|
|
7193
|
+
// (else a verified task's toolCalls would undercount vs its steps/usage).
|
|
7194
|
+
messages: [...res.messages, ...cres.messages],
|
|
7195
|
+
usageEstimated: res.usageEstimated || cres.usageEstimated,
|
|
7196
|
+
usage: res.usage && cres.usage ? {
|
|
7197
|
+
promptTokens: sum(res.usage.promptTokens, cres.usage.promptTokens),
|
|
7198
|
+
completionTokens: sum(res.usage.completionTokens, cres.usage.completionTokens),
|
|
7199
|
+
totalTokens: sum(res.usage.totalTokens, cres.usage.totalTokens),
|
|
7200
|
+
cacheCreationTokens: sum(res.usage.cacheCreationTokens, cres.usage.cacheCreationTokens),
|
|
7201
|
+
cacheReadTokens: sum(res.usage.cacheReadTokens, cres.usage.cacheReadTokens)
|
|
7202
|
+
} : res.usage ?? cres.usage
|
|
7203
|
+
};
|
|
7204
|
+
}
|
|
7205
|
+
/** Throttled per-task progress: worker tool calls → at most one progress re-voice per interval.
|
|
7206
|
+
* Two sources, one throttle: completed steps (post) and a heartbeat for a SINGLE long tool call
|
|
7207
|
+
* (pre records the in-flight call; a self-cleaning timer narrates "still inside Bash — 70s").
|
|
7208
|
+
* Completion supersedes: nothing is emitted once the task has settled. */
|
|
7209
|
+
progressReporter(id) {
|
|
7210
|
+
let lastAt = Date.now();
|
|
7211
|
+
let steps = 0;
|
|
7212
|
+
let inflight = null;
|
|
7213
|
+
const due = () => {
|
|
7214
|
+
if (this.pendingAsks.size) return void 0;
|
|
7215
|
+
const rec = this.tasks.get(id);
|
|
7216
|
+
return rec && rec.status === "running" && Date.now() - lastAt >= this.options.progressIntervalMs ? rec : void 0;
|
|
7217
|
+
};
|
|
7218
|
+
const emit = (rec, line, call) => {
|
|
7219
|
+
lastAt = Date.now();
|
|
7220
|
+
this.notify("task_progress", `task ${id} (${rec.label}): ${line}`, { id, steps, call: call.name });
|
|
7221
|
+
this.queueRevoice(`[task ${id} progress] ${line}`);
|
|
7222
|
+
};
|
|
7223
|
+
const timer = setInterval(() => {
|
|
7224
|
+
const rec = this.tasks.get(id);
|
|
7225
|
+
if (!rec || rec.status !== "running") return clearInterval(timer);
|
|
7226
|
+
if (!inflight || !due()) return;
|
|
7227
|
+
const last = inflight.tail.trim().split("\n").filter(Boolean).pop()?.slice(-80);
|
|
7228
|
+
emit(rec, `still inside ${describeCall(inflight.call)} \u2014 ${Math.round((Date.now() - inflight.at) / 1e3)}s on this step${last ? `, last output: ${last}` : ""}`, inflight.call);
|
|
7229
|
+
}, Math.max(this.options.progressIntervalMs, 250));
|
|
7230
|
+
timer.unref?.();
|
|
7231
|
+
return {
|
|
7232
|
+
pre: (call) => {
|
|
7233
|
+
inflight = { call, at: Date.now(), tail: "" };
|
|
7234
|
+
},
|
|
7235
|
+
output: (chunk) => {
|
|
7236
|
+
if (inflight) inflight.tail = (inflight.tail + chunk).slice(-500);
|
|
7237
|
+
},
|
|
7238
|
+
// digest only — NEVER re-voices directly
|
|
7239
|
+
post: (call) => {
|
|
7240
|
+
steps++;
|
|
7241
|
+
inflight = null;
|
|
7242
|
+
const rec = due();
|
|
7243
|
+
if (rec) emit(rec, `still running \u2014 ${steps} steps so far, now: ${describeCall(call)}`, call);
|
|
7379
7244
|
}
|
|
7380
|
-
|
|
7381
|
-
|
|
7245
|
+
};
|
|
7246
|
+
}
|
|
7247
|
+
/** Park a question under `askId` (a task id, or any unique key for permission asks): re-voices
|
|
7248
|
+
* '[task <id> asks] …' and resolves with the user's answer via AnswerTask — or '' on timeout/
|
|
7249
|
+
* task settle (callers map '' to deny / best-judgment). Workers never block forever. */
|
|
7250
|
+
parkQuestion(askId, question) {
|
|
7251
|
+
return new Promise((resolve) => {
|
|
7252
|
+
let settled = false;
|
|
7253
|
+
const finish = (answer) => {
|
|
7254
|
+
if (settled) return;
|
|
7255
|
+
settled = true;
|
|
7256
|
+
clearTimeout(timer);
|
|
7257
|
+
this.pendingAsks.delete(askId);
|
|
7258
|
+
resolve(answer);
|
|
7259
|
+
};
|
|
7260
|
+
const timer = setTimeout(() => {
|
|
7261
|
+
this.notify("task_ask_timeout", `task ${askId}: question timed out \u2014 proceeding without an answer`);
|
|
7262
|
+
finish("");
|
|
7263
|
+
}, this.options.askTimeoutMs);
|
|
7264
|
+
this.pendingAsks.set(askId, { question, resolve: finish });
|
|
7265
|
+
this.notify("task_ask", `task ${askId} asks: ${question}`, { id: askId, question });
|
|
7266
|
+
this.queueRevoice(`[task ${askId} asks] ${question}
|
|
7267
|
+
(Relay this to the user in your own words. When they answer, call AnswerTask with id "${askId}" and their answer.)`);
|
|
7382
7268
|
});
|
|
7383
7269
|
}
|
|
7384
|
-
|
|
7385
|
-
|
|
7386
|
-
|
|
7387
|
-
|
|
7388
|
-
|
|
7389
|
-
|
|
7390
|
-
|
|
7391
|
-
|
|
7392
|
-
|
|
7270
|
+
/** Resolve any question a settling/cancelled task left parked (its answer can no longer matter). */
|
|
7271
|
+
dropAsk(id) {
|
|
7272
|
+
this.pendingAsks.get(id)?.resolve("");
|
|
7273
|
+
}
|
|
7274
|
+
/** Build the INTEGRATION TURN prompt for a NON-CLEAN settled worker (early stop / failure). A clean
|
|
7275
|
+
* success never reaches here — it streams its own `<spoken>` delivery during the run. For a partial
|
|
7276
|
+
* or failed result the outcome re-enters the reflex as a decision (like a tool_result flowing back
|
|
7277
|
+
* into a normal agent loop): the reflex evaluates the outcome against the original intent and chooses
|
|
7278
|
+
* what to do next.
|
|
7279
|
+
*
|
|
7280
|
+
* Decision branches (the reflex acts on them with EXISTING tools — no new surface):
|
|
7281
|
+
* • accept → SPEAK the (partial) result plainly — don't dress a failure up as success.
|
|
7282
|
+
* • escalate → call `Think` with the SAME brief — only when Act failed/stalled AND a Think tier
|
|
7283
|
+
* exists AND this task wasn't already a follow-up (one hop max). Wires the dead
|
|
7284
|
+
* "Reserve Think for a problem Act already FAILED at" promise.
|
|
7285
|
+
* • re-delegate→ call `Act` with a CORRECTED brief — for a recoverable error / partial result.
|
|
7286
|
+
* • ask → ask the user ONE concrete question if genuinely blocked.
|
|
7287
|
+
*
|
|
7288
|
+
* Keeps the `[task <id> completed]` / `[task <id> failed]` opener so existing coalescing + the
|
|
7289
|
+
* failed-revoice fallback still fire, and the per-event transcript markers stay intact. */
|
|
7290
|
+
integrationPrompt(rec, outcome, body, finishReason) {
|
|
7291
|
+
const opener = outcome === "error" ? `[task ${rec.id} failed]` : `[task ${rec.id} completed]`;
|
|
7292
|
+
const underCap = this.autoEscalations < _DuplexAgent.MAX_AUTO_ESCALATIONS;
|
|
7293
|
+
const canEscalate = (outcome === "error" || outcome === "incomplete") && underCap;
|
|
7294
|
+
const hasThink = this.options.thinkModel !== false;
|
|
7295
|
+
const options = [];
|
|
7296
|
+
if (!rec.followUp && canEscalate && hasThink)
|
|
7297
|
+
options.push("ESCALATE to the Think tier (call Think with the same brief) if this is a hard/architectural problem the Act worker stalled or failed on");
|
|
7298
|
+
if (!rec.followUp && canEscalate)
|
|
7299
|
+
options.push("RE-DELEGATE to Act with a corrected brief if the failure looks recoverable (a wrong path, a fixable mistake)");
|
|
7300
|
+
options.push("ASK the user one short, concrete question if you genuinely cannot proceed without their input");
|
|
7301
|
+
options.push("ACCEPT and tell the user plainly what happened (don't dress a failure up as success)");
|
|
7302
|
+
const decision = options.length > 1 ? ` You must decide what to do next \u2014 choose ONE: ${options.map((o, i) => `(${i + 1}) ${o}`).join("; ")}. Pick exactly one and act on it; do not voice this as a finished success.` : ` Tell the user plainly what happened \u2014 do not present this as a finished success.`;
|
|
7303
|
+
const state = outcome === "error" ? `the worker FAILED with: ${body}` : `the worker STOPPED EARLY (${finishReason}) \u2014 its result is PARTIAL, not a finished success: ${body}`;
|
|
7304
|
+
return `${opener} Original request: "${rec.brief}". Outcome: ${state}.${decision}`;
|
|
7305
|
+
}
|
|
7306
|
+
onWorkerSettled(id, res) {
|
|
7307
|
+
this.dropAsk(id);
|
|
7308
|
+
const rec = this.tasks.get(id);
|
|
7309
|
+
if (res.finishReason === "aborted" || rec.status === "cancelled") {
|
|
7310
|
+
rec.status = "cancelled";
|
|
7311
|
+
this.notify("task_cancelled", `task ${id} (${rec.label}) cancelled`);
|
|
7312
|
+
return;
|
|
7393
7313
|
}
|
|
7394
|
-
|
|
7395
|
-
|
|
7396
|
-
|
|
7397
|
-
this.lastCombined = combined;
|
|
7398
|
-
this.lastChangeAt = now();
|
|
7399
|
-
if (!this.firstTokenAt && combined.trim()) this.firstTokenAt = now();
|
|
7314
|
+
if (res.finishReason === "error") {
|
|
7315
|
+
const msg = res.error instanceof Error ? res.error.message : String(res.error ?? "unknown error");
|
|
7316
|
+
return this.failTask(rec, msg);
|
|
7400
7317
|
}
|
|
7401
|
-
|
|
7402
|
-
|
|
7403
|
-
|
|
7404
|
-
|
|
7405
|
-
|
|
7406
|
-
|
|
7318
|
+
rec.status = "done";
|
|
7319
|
+
rec.result = res.text;
|
|
7320
|
+
const incomplete = res.finishReason !== "stop";
|
|
7321
|
+
log13.verbose(`task ${id} done (${res.steps} steps${incomplete ? `, INCOMPLETE: ${res.finishReason}` : ""})`);
|
|
7322
|
+
this.notify("task_done", `task ${id} (${rec.label}) completed`, {
|
|
7323
|
+
id,
|
|
7324
|
+
text: res.text,
|
|
7325
|
+
usage: res.usage,
|
|
7326
|
+
usageEstimated: res.usageEstimated,
|
|
7327
|
+
finishReason: res.finishReason,
|
|
7328
|
+
steps: res.steps,
|
|
7329
|
+
toolCalls: res.messages.filter((m) => m.role === "tool").length
|
|
7330
|
+
});
|
|
7331
|
+
if (incomplete) {
|
|
7332
|
+
return this.queueRevoice(this.integrationPrompt(rec, "incomplete", res.text, res.finishReason), true);
|
|
7407
7333
|
}
|
|
7408
|
-
|
|
7409
|
-
|
|
7410
|
-
this.
|
|
7411
|
-
this.
|
|
7412
|
-
|
|
7413
|
-
|
|
7414
|
-
|
|
7415
|
-
|
|
7416
|
-
|
|
7417
|
-
|
|
7418
|
-
|
|
7419
|
-
|
|
7420
|
-
if (this.ws) this.ws.onclose = null;
|
|
7421
|
-
this.ws?.close();
|
|
7422
|
-
}
|
|
7423
|
-
};
|
|
7424
|
-
|
|
7425
|
-
// src/voice/cartesia.ts
|
|
7426
|
-
init_logging();
|
|
7427
|
-
var log13 = forComponent("CartesiaTTS");
|
|
7428
|
-
var now2 = () => performance.now();
|
|
7429
|
-
var CartesiaTTSOptions = class {
|
|
7430
|
-
auth = "";
|
|
7431
|
-
voiceId = "";
|
|
7432
|
-
model = "sonic-3.5";
|
|
7433
|
-
/** 'apiKey' (server/CLI) → `api_key=` URL param; 'token' (browser, BE-minted) → `access_token=`. */
|
|
7434
|
-
authMode = "apiKey";
|
|
7435
|
-
};
|
|
7436
|
-
var CartesiaTTS = class _CartesiaTTS {
|
|
7437
|
-
options;
|
|
7438
|
-
ws;
|
|
7439
|
-
ctxSeq = 0;
|
|
7440
|
-
ctxId = "";
|
|
7441
|
-
onAudio = () => {
|
|
7442
|
-
};
|
|
7443
|
-
onDone = () => {
|
|
7444
|
-
};
|
|
7445
|
-
/** Word-level timestamps for karaoke reveal: `start[i]` = seconds from turn-audio start (absolute
|
|
7446
|
-
* across continuation chunks). Fired only when a consumer opts into reveal (see `wantTimestamps`). */
|
|
7447
|
-
onTimestamps = () => {
|
|
7448
|
-
};
|
|
7449
|
-
/** Request `add_timestamps` from Cartesia. Off by default (saves bandwidth) — the engine sets it
|
|
7450
|
-
* when revealMode==='word'. */
|
|
7451
|
-
wantTimestamps = false;
|
|
7452
|
-
/** Diagnostic tap (provider errors, circuit-breaker open/close, ws reconnects). Fire-and-forget:
|
|
7453
|
-
* a throwing handler disables itself — never affects synthesis. Wired by VoiceIO / the lab bridge. */
|
|
7454
|
-
onDiag = () => {
|
|
7455
|
-
};
|
|
7456
|
-
diagOn = true;
|
|
7457
|
-
diag(kind, fields) {
|
|
7458
|
-
if (!this.diagOn) return;
|
|
7459
|
-
try {
|
|
7460
|
-
this.onDiag({ t: now2(), kind, ...fields });
|
|
7461
|
-
} catch (e) {
|
|
7462
|
-
this.diagOn = false;
|
|
7463
|
-
log13.debug(`onDiag threw \u2014 TTS diagnostics disabled: ${e instanceof Error ? e.message : e}`);
|
|
7334
|
+
const tail = rec.splitter?.flush();
|
|
7335
|
+
if (tail?.spoken && !rec.deliveryParked) this.options.host?.notify?.({ kind: "speak_utterance", message: tail.spoken });
|
|
7336
|
+
if (res.text.trim()) this.voice.transcript.push({ role: "assistant", content: res.text });
|
|
7337
|
+
if (rec.deliveryParked && !this.suppressParkedResume) {
|
|
7338
|
+
const gist = (rec.spokenText ?? "").trim() || res.text.trim();
|
|
7339
|
+
if (gist) {
|
|
7340
|
+
if (this.awaitTrivialRedeliver) {
|
|
7341
|
+
this.awaitTrivialRedeliver = false;
|
|
7342
|
+
this.options.host?.notify?.({ kind: "speak_utterance", message: gist });
|
|
7343
|
+
this.notify("diag", "parked_redelivered", { chars: gist.length });
|
|
7344
|
+
} else this.parkedRedeliver.push(gist);
|
|
7345
|
+
}
|
|
7464
7346
|
}
|
|
7347
|
+
if (!rec.splitter?.spokeAny && res.text.trim() && !rec.deliveryParked)
|
|
7348
|
+
this.options.host?.notify?.({ kind: "speak_utterance", message: res.text });
|
|
7465
7349
|
}
|
|
7466
|
-
|
|
7467
|
-
|
|
7468
|
-
consecutiveErrors = 0;
|
|
7469
|
-
consecutiveOk = 0;
|
|
7470
|
-
down = false;
|
|
7471
|
-
downAt = 0;
|
|
7472
|
-
probeTimer = null;
|
|
7473
|
-
static CB_THRESHOLD = 3;
|
|
7474
|
-
// open after 3 consecutive errors
|
|
7475
|
-
static CB_RECOVER_OK = 2;
|
|
7476
|
-
// close only after 2 consecutive good frames (no single-frame flap)
|
|
7477
|
-
static CB_PROBE_MS = 3e4;
|
|
7478
|
-
constructor(options) {
|
|
7479
|
-
this.options = { ...new CartesiaTTSOptions(), ...options };
|
|
7350
|
+
onWorkerFailed(id, err) {
|
|
7351
|
+
this.failTask(this.tasks.get(id), err instanceof Error ? err.message : String(err));
|
|
7480
7352
|
}
|
|
7481
|
-
|
|
7482
|
-
|
|
7483
|
-
|
|
7484
|
-
|
|
7485
|
-
|
|
7486
|
-
|
|
7487
|
-
this.
|
|
7353
|
+
failTask(rec, msg) {
|
|
7354
|
+
this.dropAsk(rec.id);
|
|
7355
|
+
rec.status = "error";
|
|
7356
|
+
rec.result = msg;
|
|
7357
|
+
log13.warn(`task ${rec.id} failed: ${msg}`);
|
|
7358
|
+
this.notify("task_error", `task ${rec.id} (${rec.label}) failed: ${msg}`);
|
|
7359
|
+
this.queueRevoice(this.integrationPrompt(rec, "error", msg, "error"), true);
|
|
7488
7360
|
}
|
|
7489
|
-
|
|
7490
|
-
|
|
7491
|
-
|
|
7492
|
-
|
|
7493
|
-
|
|
7494
|
-
|
|
7495
|
-
|
|
7496
|
-
|
|
7497
|
-
|
|
7498
|
-
|
|
7499
|
-
|
|
7500
|
-
|
|
7501
|
-
|
|
7502
|
-
|
|
7503
|
-
|
|
7504
|
-
|
|
7361
|
+
// --- voice tools (closures over this instance) ---
|
|
7362
|
+
/** Live-switch the think tier: `false` disables (removes the Think tool from the voice agent),
|
|
7363
|
+
* a model id enables (adds the tool if missing). The system-prompt THINK_SLOT text is frozen at
|
|
7364
|
+
* construction — the tool's own description carries the routing guidance, so a live enable works;
|
|
7365
|
+
* dispatch()'s think→act fallback covers any straggler calls after a live disable. */
|
|
7366
|
+
setThinkModel(model) {
|
|
7367
|
+
this.options.thinkModel = model;
|
|
7368
|
+
const tools = this.voice.options.tools;
|
|
7369
|
+
const i = tools.findIndex((t) => t.name === "Think");
|
|
7370
|
+
if (model === false && i >= 0) tools.splice(i, 1);
|
|
7371
|
+
else if (model !== false && i < 0) tools.push(this.thinkTool());
|
|
7372
|
+
}
|
|
7373
|
+
/** User/programmatic spawn: the CLI's /act and /think commands. Returns the task id.
|
|
7374
|
+
* `followUp` marks an automatic escalation/re-delegation (set by the integration turn) so the new
|
|
7375
|
+
* task's own integration turn won't escalate again — capping auto-follow-ups to one hop. */
|
|
7376
|
+
async dispatch(brief, tier = "act", label, followUp = false) {
|
|
7377
|
+
if (tier === "think" && this.options.thinkModel === false) tier = "act";
|
|
7378
|
+
if (followUp) this.autoEscalations++;
|
|
7379
|
+
const id = `t${++this.seq}`;
|
|
7380
|
+
const lbl = label ?? tier;
|
|
7381
|
+
await this.options.onTaskStart?.(id, lbl);
|
|
7382
|
+
this.spawnWorker(id, lbl, this.buildBrief(brief, tier), tier, brief, followUp);
|
|
7383
|
+
this.notify("task_started", `task ${id} (${lbl}) started`, { id, brief, tier });
|
|
7384
|
+
return id;
|
|
7385
|
+
}
|
|
7386
|
+
actTool() {
|
|
7387
|
+
return {
|
|
7388
|
+
name: "Act",
|
|
7389
|
+
description: 'Escalate real work (reading/editing files, searching, running tasks, building) to a standard background worker. Returns immediately with a task id; the result arrives later as a "[task <id> completed]" event. Provide a clear, self-contained `brief` (the worker does not hear the live conversation).',
|
|
7390
|
+
parameters: {
|
|
7391
|
+
type: "object",
|
|
7392
|
+
required: ["brief"],
|
|
7393
|
+
properties: {
|
|
7394
|
+
brief: { type: "string", description: "full, self-contained instructions for the worker" },
|
|
7395
|
+
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
7396
|
+
}
|
|
7397
|
+
},
|
|
7398
|
+
run: async ({ brief, label }) => {
|
|
7399
|
+
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
7400
|
+
this.turnDispatched = true;
|
|
7401
|
+
this.turnBriefs.add(String(brief ?? ""));
|
|
7402
|
+
this.voice.options.toolChoice = "none";
|
|
7403
|
+
const id = await this.dispatch(String(brief ?? ""), "act", label ? String(label) : void 0, this.turnFollowUp);
|
|
7404
|
+
return this.spokeBeforeDispatch ? `Acting on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Acting on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
7505
7405
|
}
|
|
7506
7406
|
};
|
|
7507
|
-
|
|
7508
|
-
|
|
7509
|
-
|
|
7510
|
-
|
|
7511
|
-
|
|
7512
|
-
|
|
7513
|
-
|
|
7514
|
-
|
|
7515
|
-
|
|
7516
|
-
|
|
7517
|
-
|
|
7518
|
-
|
|
7519
|
-
}
|
|
7520
|
-
|
|
7521
|
-
|
|
7522
|
-
|
|
7523
|
-
|
|
7524
|
-
this.
|
|
7525
|
-
this.
|
|
7526
|
-
|
|
7527
|
-
|
|
7528
|
-
|
|
7529
|
-
|
|
7530
|
-
|
|
7531
|
-
|
|
7532
|
-
|
|
7533
|
-
|
|
7534
|
-
|
|
7535
|
-
|
|
7407
|
+
}
|
|
7408
|
+
thinkTool() {
|
|
7409
|
+
return {
|
|
7410
|
+
name: "Think",
|
|
7411
|
+
description: "Escalate to a premium deep-reasoning agent for complex analysis, architecture decisions, hard debugging, or planning. Same async pattern as Act \u2014 returns a task id. Use when the problem needs careful thought before (or instead of) action. Do not use Think for simple tasks \u2014 Act is cheaper and faster.",
|
|
7412
|
+
parameters: {
|
|
7413
|
+
type: "object",
|
|
7414
|
+
required: ["brief"],
|
|
7415
|
+
properties: {
|
|
7416
|
+
brief: { type: "string", description: "the question or problem to reason about deeply" },
|
|
7417
|
+
label: { type: "string", description: "a short (2-4 word) label for the task" }
|
|
7418
|
+
}
|
|
7419
|
+
},
|
|
7420
|
+
run: async ({ brief, label }) => {
|
|
7421
|
+
this.spokeBeforeDispatch = this.spokeThisTurn;
|
|
7422
|
+
this.turnDispatched = true;
|
|
7423
|
+
this.turnBriefs.add(String(brief ?? ""));
|
|
7424
|
+
this.voice.options.toolChoice = "none";
|
|
7425
|
+
const id = await this.dispatch(String(brief ?? ""), "think", label ? String(label) : void 0, this.turnFollowUp);
|
|
7426
|
+
return this.spokeBeforeDispatch ? `Thinking on task ${id}. You already acknowledged \u2014 end your turn now with NO further text; the result will arrive as a [task ${id} completed] event.` : `Thinking on task ${id}. Acknowledge briefly; the result will arrive as a [task ${id} completed] event.`;
|
|
7427
|
+
}
|
|
7428
|
+
};
|
|
7429
|
+
}
|
|
7430
|
+
taskStatusTool() {
|
|
7431
|
+
return {
|
|
7432
|
+
name: "TaskStatus",
|
|
7433
|
+
description: "Status of background tasks. Pass `id` for one task, or omit it to list all.",
|
|
7434
|
+
parameters: { type: "object", properties: { id: { type: "string" } } },
|
|
7435
|
+
run: async ({ id }) => {
|
|
7436
|
+
const list = id ? [this.tasks.get(String(id))].filter(Boolean) : [...this.tasks.values()];
|
|
7437
|
+
if (!list.length) return id ? `No task '${id}'.` : "No background tasks.";
|
|
7438
|
+
return list.map((t) => `${t.id} (${t.label}): ${t.status}`).join("\n");
|
|
7439
|
+
}
|
|
7440
|
+
};
|
|
7441
|
+
}
|
|
7442
|
+
/** Sub-100ms read-only lookups the voice may do itself — everything else stays Act-only.
|
|
7443
|
+
* fs-only (no shell; the engine is VFS-abstracted): time, git branch (.git/HEAD read), ls, file
|
|
7444
|
+
* head. Output is hard-capped so a lookup can never bloat the skinny voice context. */
|
|
7445
|
+
quickLookTool() {
|
|
7446
|
+
const CAP = 2e3;
|
|
7447
|
+
const kinds = [.../* @__PURE__ */ new Set(["time", "branch", "ls", "file", "capabilities", ...Object.keys(this.options.quickLook ?? {})])];
|
|
7448
|
+
return {
|
|
7449
|
+
name: "QuickLook",
|
|
7450
|
+
description: `Instant read-only lookup \u2014 one of: ${kinds.join(", ")}. For trivial facts only; anything needing search, commands, or reasoning goes through Act.`,
|
|
7451
|
+
parameters: {
|
|
7452
|
+
type: "object",
|
|
7453
|
+
required: ["what"],
|
|
7454
|
+
properties: {
|
|
7455
|
+
what: { type: "string", enum: kinds, description: "what to look up" },
|
|
7456
|
+
path: { type: "string", description: "for ls/file: the path to look at" }
|
|
7457
|
+
}
|
|
7458
|
+
},
|
|
7459
|
+
run: async ({ what, path }) => {
|
|
7460
|
+
const fs = this.options.fs;
|
|
7461
|
+
try {
|
|
7462
|
+
const over = this.options.quickLook?.[String(what)];
|
|
7463
|
+
if (over) return await over(path ? String(path) : void 0);
|
|
7464
|
+
switch (String(what)) {
|
|
7465
|
+
case "capabilities": {
|
|
7466
|
+
const actTools = this.options.actOptions?.tools ?? [];
|
|
7467
|
+
const names = actTools.map((t) => t.name);
|
|
7468
|
+
const mcpServers = Object.keys(this.options.actOptions?.providerOptions?.mcpServers ?? {});
|
|
7469
|
+
const mcpNote = mcpServers.length ? ` Plus MCP servers your worker can use: ${mcpServers.join(", ")} (e.g. browser-bridge \u2192 drive a real browser: open tabs, navigate, click, screenshot).` : "";
|
|
7470
|
+
if (!names.length)
|
|
7471
|
+
return "Your worker uses Act's default local toolset (reading/editing files, running shell commands). No extra tools (e.g. web/internet) are configured; if a request is not a basic file or shell operation, assume you can't do it and say so." + mcpNote;
|
|
7472
|
+
const hasFetch = names.some((n) => /WebFetch/i.test(n));
|
|
7473
|
+
const hasBrowser = names.some((n) => /browser.*(navigate|click|page|type)/i.test(n));
|
|
7474
|
+
const hasSearch = names.some((n) => /(^|_)WebSearch$|search/i.test(n) && !/WebFetch|browser/i.test(n));
|
|
7475
|
+
const notes = [];
|
|
7476
|
+
if (hasFetch) notes.push("WebFetch retrieves ONE specific URL you are given \u2014 it is not a search engine.");
|
|
7477
|
+
if (hasBrowser) notes.push("The browser tools drive a real browser: you CAN open a site and, if needed, navigate to a search engine and search there \u2014 but it is manual and takes a moment, not an instant lookup.");
|
|
7478
|
+
else if (!hasSearch && hasFetch) notes.push('You have no general web-search tool, so for an instant "search the web" you can only fetch a URL they provide.');
|
|
7479
|
+
const webNote = notes.length ? " NOTE: " + notes.join(" ") : "";
|
|
7480
|
+
return `Tools your background worker (Act) can actually use: ${names.join(", ")}. Read each name literally and match the request to a SPECIFIC tool; if none fits, you do NOT have that ability \u2014 say so honestly.` + webNote + mcpNote;
|
|
7481
|
+
}
|
|
7482
|
+
case "time":
|
|
7483
|
+
return (/* @__PURE__ */ new Date()).toString();
|
|
7484
|
+
case "branch": {
|
|
7485
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
7486
|
+
const head = (await fs.readFile(".git/HEAD")).trim();
|
|
7487
|
+
return head.startsWith("ref: refs/heads/") ? `branch: ${head.slice("ref: refs/heads/".length)}` : `detached HEAD at ${head.slice(0, 12)}`;
|
|
7488
|
+
}
|
|
7489
|
+
case "ls": {
|
|
7490
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
7491
|
+
const p = String(path ?? ".");
|
|
7492
|
+
try {
|
|
7493
|
+
const names = await fs.readDir(p);
|
|
7494
|
+
return names.slice(0, 50).join("\n") + (names.length > 50 ? `
|
|
7495
|
+
\u2026 (+${names.length - 50} more)` : "");
|
|
7496
|
+
} catch {
|
|
7497
|
+
const names = await fs.readDir(".").catch(() => []);
|
|
7498
|
+
return `'${p}' not found here \u2014 you are likely already inside it. Current directory listing:
|
|
7499
|
+
` + names.slice(0, 50).join("\n") + (names.length > 50 ? `
|
|
7500
|
+
\u2026 (+${names.length - 50} more)` : "");
|
|
7501
|
+
}
|
|
7502
|
+
}
|
|
7503
|
+
case "file": {
|
|
7504
|
+
if (!fs) return "unavailable (no filesystem)";
|
|
7505
|
+
if (!path) return "file lookup needs a path";
|
|
7506
|
+
try {
|
|
7507
|
+
const text = await fs.readFile(String(path));
|
|
7508
|
+
return text.length > CAP ? text.slice(0, CAP) + `
|
|
7509
|
+
\u2026 (truncated \u2014 ${text.length} chars total; Act for the full file)` : text;
|
|
7510
|
+
} catch {
|
|
7511
|
+
const names = await fs.readDir(".").catch(() => []);
|
|
7512
|
+
return `'${path}' not found. Current directory contains:
|
|
7513
|
+
` + names.slice(0, 50).join("\n");
|
|
7514
|
+
}
|
|
7515
|
+
}
|
|
7516
|
+
default:
|
|
7517
|
+
return `unknown lookup '${what}'`;
|
|
7518
|
+
}
|
|
7519
|
+
} catch (e) {
|
|
7520
|
+
return `lookup failed: ${e?.message ?? e}`;
|
|
7521
|
+
}
|
|
7522
|
+
}
|
|
7523
|
+
};
|
|
7524
|
+
}
|
|
7525
|
+
answerTaskTool() {
|
|
7526
|
+
return {
|
|
7527
|
+
name: "AnswerTask",
|
|
7528
|
+
description: `Relay the user's answer to a pending question from a background task (the "[task <id> asks]" events). Pass the id from the event and the user's answer.`,
|
|
7529
|
+
parameters: {
|
|
7530
|
+
type: "object",
|
|
7531
|
+
required: ["id", "answer"],
|
|
7532
|
+
properties: { id: { type: "string" }, answer: { type: "string", description: "the user's answer, verbatim or faithfully summarized" } }
|
|
7533
|
+
},
|
|
7534
|
+
run: async ({ id, answer }) => {
|
|
7535
|
+
const ask = this.pendingAsks.get(String(id));
|
|
7536
|
+
if (!ask) return `No pending question for '${id}' \u2014 it may have been answered already or timed out.`;
|
|
7537
|
+
ask.resolve(String(answer ?? ""));
|
|
7538
|
+
return `Answer relayed \u2014 task ${id} resumes.`;
|
|
7539
|
+
}
|
|
7540
|
+
};
|
|
7541
|
+
}
|
|
7542
|
+
holdTool() {
|
|
7543
|
+
return {
|
|
7544
|
+
name: "Hold",
|
|
7545
|
+
description: 'The user seems mid-thought \u2014 hold the turn (stay listening) instead of answering. Optionally pass a short filler ("mhm", "go on") to speak while waiting. Use when the message sounds incomplete, trailing off, or like they paused to think.',
|
|
7546
|
+
parameters: {
|
|
7547
|
+
type: "object",
|
|
7548
|
+
properties: {
|
|
7549
|
+
filler: { type: "string", description: 'optional short filler to speak ("mhm", "go on", "mm-hm")' }
|
|
7536
7550
|
}
|
|
7551
|
+
},
|
|
7552
|
+
run: async ({ filler }) => {
|
|
7553
|
+
this.heldThisTurn = true;
|
|
7554
|
+
this.notify("hold_filler", filler ? String(filler) : "");
|
|
7555
|
+
return "Holding \u2014 listening for the rest of the user's thought. Do not respond further this turn.";
|
|
7537
7556
|
}
|
|
7538
7557
|
};
|
|
7539
7558
|
}
|
|
7540
|
-
|
|
7541
|
-
|
|
7542
|
-
|
|
7543
|
-
|
|
7544
|
-
|
|
7545
|
-
|
|
7546
|
-
|
|
7547
|
-
this.stopProbe();
|
|
7548
|
-
const downMs = this.downAt ? now2() - this.downAt : 0;
|
|
7549
|
-
this.diag("tts_breaker_close", { downMs: Math.round(downMs) });
|
|
7550
|
-
(downMs < 2e3 ? log13.debug : log13.info)(`TTS recovered${downMs ? ` (down ${downMs}ms)` : ""}`);
|
|
7559
|
+
cancelTaskTool() {
|
|
7560
|
+
return {
|
|
7561
|
+
name: "CancelTask",
|
|
7562
|
+
description: "Cancel a running background task by id.",
|
|
7563
|
+
parameters: { type: "object", required: ["id"], properties: { id: { type: "string" } } },
|
|
7564
|
+
run: async ({ id }) => this.cancelTask(String(id))
|
|
7565
|
+
};
|
|
7551
7566
|
}
|
|
7552
|
-
|
|
7553
|
-
|
|
7554
|
-
|
|
7555
|
-
|
|
7567
|
+
};
|
|
7568
|
+
|
|
7569
|
+
// src/mcp.ts
|
|
7570
|
+
function toResult(result) {
|
|
7571
|
+
if (result == null) return { text: "" };
|
|
7572
|
+
if (typeof result === "string") return { text: result };
|
|
7573
|
+
const content = result.content;
|
|
7574
|
+
if (Array.isArray(content)) {
|
|
7575
|
+
const texts = [];
|
|
7576
|
+
const images = [];
|
|
7577
|
+
for (const c of content) {
|
|
7578
|
+
if (c?.type === "image" && typeof c.data === "string" && c.mimeType) {
|
|
7579
|
+
images.push({ mimeType: c.mimeType, data: c.data });
|
|
7580
|
+
} else if (typeof c?.text === "string") {
|
|
7581
|
+
texts.push(c.text);
|
|
7582
|
+
} else {
|
|
7583
|
+
texts.push(JSON.stringify(c));
|
|
7584
|
+
}
|
|
7585
|
+
}
|
|
7586
|
+
const text = texts.join("\n");
|
|
7587
|
+
if (text || images.length) return { text, ...images.length ? { images } : {} };
|
|
7556
7588
|
}
|
|
7557
|
-
|
|
7558
|
-
|
|
7559
|
-
|
|
7560
|
-
|
|
7561
|
-
|
|
7562
|
-
|
|
7563
|
-
|
|
7564
|
-
|
|
7589
|
+
return { text: JSON.stringify(result) };
|
|
7590
|
+
}
|
|
7591
|
+
function mcpToolToAgentTool(spec, callTool, prefix = "mcp__") {
|
|
7592
|
+
return {
|
|
7593
|
+
name: `${prefix}${spec.name}`,
|
|
7594
|
+
description: spec.description ?? `MCP tool ${spec.name}`,
|
|
7595
|
+
parameters: spec.inputSchema ?? { type: "object", properties: {} },
|
|
7596
|
+
async run(args, _ctx) {
|
|
7597
|
+
const r = toResult(await callTool(spec.name, args ?? {}));
|
|
7598
|
+
return r.images?.length ? r : r.text;
|
|
7599
|
+
}
|
|
7600
|
+
};
|
|
7601
|
+
}
|
|
7602
|
+
function mcpToolsToAgentTools(specs, callTool, prefix = "mcp__", filter) {
|
|
7603
|
+
return (filter ? specs.filter(filter) : specs).map((s) => mcpToolToAgentTool(s, callTool, prefix));
|
|
7604
|
+
}
|
|
7605
|
+
function describeSpec(s) {
|
|
7606
|
+
const schema = s.inputSchema ? `
|
|
7607
|
+
args: ${JSON.stringify(s.inputSchema)}` : "";
|
|
7608
|
+
return `${s.name} \u2014 ${s.description ?? "(no description)"}${schema}`;
|
|
7609
|
+
}
|
|
7610
|
+
function makeMcpToolSearch(specs, callTool, options = {}) {
|
|
7611
|
+
const maxResults = options.maxResults ?? 10;
|
|
7612
|
+
const byName = new Map(specs.map((s) => [s.name, s]));
|
|
7613
|
+
const catalogLine = `${specs.length} MCP tool(s) available \u2014 search by keyword, then call by exact name.`;
|
|
7614
|
+
const searchTool = {
|
|
7615
|
+
name: "ToolSearch",
|
|
7616
|
+
description: `Search the available MCP tools by keyword (${catalogLine}). Returns matching tool names + their argument schemas; call one with \`McpCall\`.`,
|
|
7617
|
+
parameters: { type: "object", required: ["query"], properties: { query: { type: "string", description: "keywords to match against tool name + description" } } },
|
|
7618
|
+
async run({ query }) {
|
|
7619
|
+
const q = String(query ?? "").trim();
|
|
7620
|
+
if (!q) return catalogLine;
|
|
7621
|
+
const { kept } = topByRelevance(specs, q, (s) => `${s.name} ${s.description ?? ""}`, maxResults);
|
|
7622
|
+
if (!kept.length) return `(no MCP tool matches "${q}" \u2014 try broader keywords)`;
|
|
7623
|
+
return kept.map(describeSpec).join("\n");
|
|
7624
|
+
}
|
|
7625
|
+
};
|
|
7626
|
+
const callMcpTool = {
|
|
7627
|
+
name: "McpCall",
|
|
7628
|
+
description: "Call an MCP tool discovered via `ToolSearch`, by its exact name. Pass its arguments as `args`.",
|
|
7629
|
+
parameters: {
|
|
7630
|
+
type: "object",
|
|
7631
|
+
required: ["name"],
|
|
7632
|
+
properties: {
|
|
7633
|
+
name: { type: "string", description: "exact tool name from ToolSearch" },
|
|
7634
|
+
args: { type: "object", description: "arguments object for the tool (per its schema)" }
|
|
7635
|
+
}
|
|
7636
|
+
},
|
|
7637
|
+
async run({ name, args }) {
|
|
7638
|
+
const n = String(name ?? "");
|
|
7639
|
+
if (!byName.has(n)) return `Error: unknown MCP tool '${n}'. Use ToolSearch to find valid names.`;
|
|
7640
|
+
const r = toResult(await callTool(n, args ?? {}));
|
|
7641
|
+
return r.images?.length ? r : r.text;
|
|
7642
|
+
}
|
|
7643
|
+
};
|
|
7644
|
+
return [searchTool, callMcpTool];
|
|
7645
|
+
}
|
|
7646
|
+
function buildMcpCatalog(servers) {
|
|
7647
|
+
const specs = [];
|
|
7648
|
+
const routes = /* @__PURE__ */ new Map();
|
|
7649
|
+
for (const m of servers) {
|
|
7650
|
+
for (const s of m.specs) {
|
|
7651
|
+
const base = `mcp__${m.name}__${s.name}`.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
|
|
7652
|
+
let display = base;
|
|
7653
|
+
for (let i = 2; routes.has(display); i++) display = `${base.slice(0, 128 - String(i).length - 1)}_${i}`;
|
|
7654
|
+
specs.push({ name: display, description: s.description, inputSchema: s.inputSchema });
|
|
7655
|
+
routes.set(display, { server: m.name, rawName: s.name });
|
|
7656
|
+
}
|
|
7565
7657
|
}
|
|
7566
|
-
|
|
7567
|
-
|
|
7568
|
-
|
|
7569
|
-
|
|
7658
|
+
return { specs, routes };
|
|
7659
|
+
}
|
|
7660
|
+
function searchOverCatalog(servers, specs, routes, resolve, options) {
|
|
7661
|
+
const tools = specs.length ? makeMcpToolSearch(specs, (name, args) => {
|
|
7662
|
+
const r = routes.get(name);
|
|
7663
|
+
if (!r) throw new Error(`unknown MCP tool '${name}' \u2014 use ToolSearch to find valid names`);
|
|
7664
|
+
return resolve(r.server, r.rawName, args ?? {});
|
|
7665
|
+
}, options) : [];
|
|
7666
|
+
return { tools, serverNames: servers, toolCount: specs.length };
|
|
7667
|
+
}
|
|
7668
|
+
function makeMcpToolSearchFromMounted(mounted, options) {
|
|
7669
|
+
const { specs, routes } = buildMcpCatalog(mounted);
|
|
7670
|
+
const byName = new Map(mounted.map((m) => [m.name, m]));
|
|
7671
|
+
return searchOverCatalog(mounted.map((m) => m.name), specs, routes, (server, rawName, args) => byName.get(server).client.callTool(rawName, args), options);
|
|
7672
|
+
}
|
|
7673
|
+
function makeLazyMcpToolSearch(servers, resolve, options) {
|
|
7674
|
+
const { specs, routes } = buildMcpCatalog(servers);
|
|
7675
|
+
return searchOverCatalog(servers.map((s) => s.name), specs, routes, resolve, options);
|
|
7676
|
+
}
|
|
7677
|
+
|
|
7678
|
+
// src/hooks.ts
|
|
7679
|
+
var RecordingHooks = class {
|
|
7680
|
+
/** tool name -> reason; a matching preToolUse call is blocked with that reason. */
|
|
7681
|
+
constructor(blocks = {}) {
|
|
7682
|
+
this.blocks = blocks;
|
|
7570
7683
|
}
|
|
7571
|
-
|
|
7572
|
-
|
|
7573
|
-
|
|
7574
|
-
|
|
7575
|
-
|
|
7576
|
-
|
|
7577
|
-
|
|
7578
|
-
|
|
7579
|
-
|
|
7580
|
-
});
|
|
7684
|
+
blocks;
|
|
7685
|
+
pre = [];
|
|
7686
|
+
post = [];
|
|
7687
|
+
outputs = [];
|
|
7688
|
+
stops = [];
|
|
7689
|
+
preToolUse(call, meta) {
|
|
7690
|
+
this.pre.push({ call, meta });
|
|
7691
|
+
const reason = this.blocks[call.name];
|
|
7692
|
+
if (reason != null) return { block: true, reason };
|
|
7581
7693
|
}
|
|
7582
|
-
|
|
7583
|
-
|
|
7584
|
-
if (cont && !text) return;
|
|
7585
|
-
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame(text, cont));
|
|
7586
|
-
else void this.ensureConnected().then(() => this.ws?.readyState === WebSocket.OPEN && this.ws.send(this.frame(text, cont)));
|
|
7694
|
+
postToolUse(call, result, meta) {
|
|
7695
|
+
this.post.push({ call, result, meta });
|
|
7587
7696
|
}
|
|
7588
|
-
|
|
7589
|
-
|
|
7590
|
-
this.onDone();
|
|
7591
|
-
return;
|
|
7592
|
-
}
|
|
7593
|
-
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("", false));
|
|
7697
|
+
onToolOutput(call, chunk, meta) {
|
|
7698
|
+
this.outputs.push({ call, chunk, meta });
|
|
7594
7699
|
}
|
|
7595
|
-
|
|
7596
|
-
|
|
7700
|
+
onStop(finalText) {
|
|
7701
|
+
this.stops.push(finalText);
|
|
7597
7702
|
}
|
|
7598
|
-
|
|
7599
|
-
|
|
7600
|
-
|
|
7601
|
-
|
|
7602
|
-
|
|
7603
|
-
|
|
7604
|
-
}
|
|
7605
|
-
this.consecutiveErrors = 0;
|
|
7606
|
-
this.newContext();
|
|
7607
|
-
if (this.ws?.readyState === WebSocket.OPEN) this.ws.send(this.frame("Ok.", false));
|
|
7608
|
-
}, _CartesiaTTS.CB_PROBE_MS);
|
|
7609
|
-
this.probeTimer.unref?.();
|
|
7703
|
+
};
|
|
7704
|
+
var RecordingLifecycle = class {
|
|
7705
|
+
/** @param startContext injected at session start; @param rewrite maps a submitted prompt to a new one. */
|
|
7706
|
+
constructor(startContext, rewrite) {
|
|
7707
|
+
this.startContext = startContext;
|
|
7708
|
+
this.rewrite = rewrite;
|
|
7610
7709
|
}
|
|
7611
|
-
|
|
7612
|
-
|
|
7613
|
-
|
|
7614
|
-
|
|
7615
|
-
|
|
7710
|
+
startContext;
|
|
7711
|
+
rewrite;
|
|
7712
|
+
starts = 0;
|
|
7713
|
+
prompts = [];
|
|
7714
|
+
compactions = [];
|
|
7715
|
+
subagentStops = [];
|
|
7716
|
+
onSessionStart() {
|
|
7717
|
+
this.starts++;
|
|
7718
|
+
return this.startContext;
|
|
7616
7719
|
}
|
|
7617
|
-
|
|
7618
|
-
this.
|
|
7619
|
-
this.
|
|
7620
|
-
|
|
7621
|
-
|
|
7720
|
+
onUserPromptSubmit(text) {
|
|
7721
|
+
this.prompts.push(text);
|
|
7722
|
+
return this.rewrite?.(text);
|
|
7723
|
+
}
|
|
7724
|
+
onPreCompact(messages) {
|
|
7725
|
+
this.compactions.push(messages.length);
|
|
7726
|
+
}
|
|
7727
|
+
onSubagentStop(summary, info) {
|
|
7728
|
+
this.subagentStops.push({ summary, label: info?.label });
|
|
7622
7729
|
}
|
|
7623
7730
|
};
|
|
7624
|
-
function base64ToBytes(b64) {
|
|
7625
|
-
if (typeof Buffer !== "undefined") return Buffer.from(b64, "base64");
|
|
7626
|
-
const bin = atob(b64);
|
|
7627
|
-
const out = new Uint8Array(bin.length);
|
|
7628
|
-
for (let i = 0; i < bin.length; i++) out[i] = bin.charCodeAt(i);
|
|
7629
|
-
return out;
|
|
7630
|
-
}
|
|
7631
7731
|
|
|
7632
7732
|
// src/index.ts
|
|
7733
|
+
init_logging();
|
|
7633
7734
|
import { MemFilesystem as MemFilesystem3, IndexedDbFilesystem, CommandExecutor as CommandExecutor2, registerHeadlessCommands as registerHeadlessCommands2 } from "@livx.cc/wcli/core";
|
|
7634
7735
|
export {
|
|
7635
7736
|
Agent,
|