aegis-desktop 0.5.0 → 0.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -109,18 +109,38 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
109
109
  const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
110
110
 
111
111
  /**
112
- * Only ever raises a too-low budget for a DeepSeek reasoning model — never
113
- * lowers whatever the caller (renderer dropdown, or "adaptive" ceiling)
114
- * already asked for. Everything else (non-DeepSeek models, non-reasoning
115
- * DeepSeek ids like deepseek-chat) passes through untouched. Effort defaults
116
- * to 'high' since custom endpoints have no effort selector of their own
117
- * (that UI is aegis-class/autonomous-only) — matching aegiscodex-dev's own
118
- * default effort.
112
+ * The budget a DeepSeek reasoning model runs on, resolved from EXACTLY ONE
113
+ * authority per call.
114
+ *
115
+ * A caller-stated number IS the budget, and is returned verbatim. For the
116
+ * non-pooled classes the renderer's max-tokens dropdown is the only budget
117
+ * control on offer — updateBudgetControls hides the effort row for them — so
118
+ * silently raising that number to an effort rung is precisely what made the
119
+ * figure beside the dropdown untrustworthy. The old form was
120
+ * `Math.max(stated, EFFORT_TOKEN_BUDGET[eff])`, which could only ever raise a
121
+ * deliberate cap: a caller asking for 1024 ran on 32768, and the number the
122
+ * UI displayed was never the number the call used.
123
+ *
124
+ * The effort rung is the DEFAULT, consulted only when no number was stated at
125
+ * all (the pooled class, which the renderer sends `effort` for and which the
126
+ * server sizes itself). This is the same rule doubledBudget() follows for its
127
+ * truncation retry: a stated cap is never overridden, by a rung or an order of
128
+ * magnitude.
129
+ *
130
+ * Truncated and empty turns are handled where they belong — the doubled-budget
131
+ * retry plus emptyTurnError — rather than by inflating the caller's ceiling up
132
+ * front. Escalating on a demonstrated empty turn is strictly cheaper than
133
+ * pre-emptively granting the top rung to every reasoning call.
134
+ *
135
+ * Everything else (non-DeepSeek models, non-reasoning DeepSeek ids like
136
+ * deepseek-chat) passes through untouched.
119
137
  */
120
- function deepseekReasoningFloor(model, maxTokens, effort) {
138
+ function reasoningBudget(model, maxTokens, effort) {
121
139
  if (!DEEPSEEK_REASONING_MODEL_RE.test(String(model || ''))) return maxTokens;
140
+ const stated = Number(maxTokens);
141
+ if (Number.isFinite(stated) && stated > 0) return stated;
122
142
  const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
123
- return Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]);
143
+ return EFFORT_TOKEN_BUDGET[eff];
124
144
  }
125
145
 
126
146
  /** Relay model entries arrive as ids or objects; keep only real model ids. */
@@ -326,6 +346,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
326
346
  // In-memory only, on purpose — never persisted, so a restart (or
327
347
  // newChat()'s clearSessionApprovals) always starts from a clean gate.
328
348
  const sessionAllowlists = new Map(); // rootSessionId -> Set<toolName>
349
+ // Denials are remembered for the same reason allows are, mirroring it for
350
+ // the other answer. Without this, a "Deny" was forgotten the instant it was
351
+ // given: the model read `… the user denied the request`, re-planned, called
352
+ // the SAME tool again, and the gate raised a SECOND card — so one "no" cost
353
+ // the user a prompt per round for up to maxRounds (24 chat / 40 autonomous)
354
+ // rounds, each round re-sending the whole conversation to the provider. A
355
+ // gate that only remembers "yes" turns a single click into a retry storm;
356
+ // remembering "no" makes the first answer stick.
357
+ const sessionDenials = new Map(); // rootSessionId -> Set<toolName>
329
358
  const pendingApprovals = new Map(); // approvalId -> { resolve }
330
359
 
331
360
  function sessionAllows(rootId, name) {
@@ -338,10 +367,44 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
338
367
  sessionAllowlists.get(rootId).add(name);
339
368
  }
340
369
 
370
+ /** Has this conversation already refused this tool? Checked before the card
371
+ * is raised, so a repeat call is refused outright instead of re-prompting. */
372
+ function sessionDenies(rootId, name) {
373
+ const set = sessionDenials.get(rootId);
374
+ return Boolean(set && set.has(name));
375
+ }
376
+
377
+ function denyForSession(rootId, name) {
378
+ if (!sessionDenials.has(rootId)) sessionDenials.set(rootId, new Set());
379
+ sessionDenials.get(rootId).add(name);
380
+ }
381
+
382
+ /**
383
+ * The text a refused tool hands back to the model. The old one-line
384
+ * `"<tool> was not executed — the user denied the request."` read to a
385
+ * capable agent as a transient failure to route around: it would apologise,
386
+ * pick a different command that does the same thing, and call the gate
387
+ * again. This says the durable part out loud (the refusal covers the rest of
388
+ * the conversation, not just that call) and asks for the one response that
389
+ * actually helps — say what you need and stop, so the user can re-enable it.
390
+ */
391
+ function denialText(name) {
392
+ return (
393
+ `${name} was not executed — the user denied this tool for this conversation. ` +
394
+ `Do NOT retry it and do NOT attempt the same effect by another route ` +
395
+ `(another command, a writeFile instead of an edit, a subagent). ` +
396
+ `Stop calling tools and reply in plain text: say what you were trying to do, ` +
397
+ `what you need, and that the user can re-enable ${name} to let it proceed.`
398
+ );
399
+ }
400
+
341
401
  /** newChat() in the renderer calls this so a fresh conversation never
342
- * inherits a prior thread's blanket allows. */
402
+ * inherits a prior thread's blanket allows — or its refusals. A new chat is
403
+ * a new gate in both directions: leaving denials behind would silently
404
+ * refuse a tool in a thread where the user never said no. */
343
405
  function clearSessionApprovals(rootSessionId) {
344
406
  sessionAllowlists.delete(rootSessionId);
407
+ sessionDenials.delete(rootSessionId);
345
408
  return { ok: true };
346
409
  }
347
410
 
@@ -359,22 +422,28 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
359
422
 
360
423
  /**
361
424
  * Ask the renderer to approve one mutating tool call. Resolves 'once',
362
- * 'session' or 'deny'. Sent over `rootOnDelta` (see chat()) as an
363
- * `{ approval }` chunk so it rides the exact same streaming channel as
364
- * tool-activity chunks — no new IPC surface needed on the push side, only
365
- * on the reply side (respondApproval). Fails safe: no listener able to
366
- * ever answer (no onDelta, or the turn was aborted) resolves 'deny'
367
- * instead of hanging the tool round forever.
425
+ * 'session' or 'deny' (an explicit choice by the user) — or 'cancel' when
426
+ * nobody ever answered: no listener able to reply, or the turn was aborted.
427
+ * 'cancel' is kept apart from 'deny' on purpose. Both refuse the call, but
428
+ * only 'deny' is a decision the user made, so only 'deny' may be remembered
429
+ * as "this conversation said no" (see sessionDenials). Folding the two
430
+ * together — which is what the old fail-safe did, resolving 'deny' for an
431
+ * abort — would let a cancelled turn permanently refuse a tool the user
432
+ * never ruled on. Sent over `rootOnDelta` (see chat()) as an `{ approval }`
433
+ * chunk so it rides the exact same streaming channel as tool-activity
434
+ * chunks — no new IPC surface needed on the push side, only on the reply
435
+ * side (respondApproval). Fails safe: an unanswered request refuses the
436
+ * call instead of hanging the tool round forever.
368
437
  */
369
438
  function requestApproval(rootSessionId, rootOnDelta, signal, info) {
370
439
  return new Promise((resolve) => {
371
440
  if (signal && signal.aborted) {
372
- resolve('deny');
441
+ resolve('cancel');
373
442
  return;
374
443
  }
375
444
  const id = randomUUID();
376
445
  let settled = false;
377
- const onAbort = () => finish('deny');
446
+ const onAbort = () => finish('cancel');
378
447
  const finish = (decision) => {
379
448
  if (settled) return;
380
449
  settled = true;
@@ -385,7 +454,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
385
454
  if (signal) signal.addEventListener('abort', onAbort, { once: true });
386
455
  pendingApprovals.set(id, { resolve: finish });
387
456
  if (typeof rootOnDelta !== 'function') {
388
- finish('deny');
457
+ finish('cancel');
389
458
  return;
390
459
  }
391
460
  rootOnDelta({
@@ -420,6 +489,13 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
420
489
  if (!T.MUTATING_TOOLS.has(name)) return T.executeTool(name, args, toolCtx);
421
490
  if (!confirmModeEnabled()) return T.executeTool(name, args, toolCtx);
422
491
  if (sessionAllows(rootSessionId, name)) return T.executeTool(name, args, toolCtx);
492
+ // Already refused in this conversation: refuse again WITHOUT raising a
493
+ // second card. Before this, the model's retry after a denial re-prompted
494
+ // the user for the same tool — one "no" produced a card per round for up
495
+ // to maxRounds rounds, each one a billed provider call re-sending the
496
+ // whole conversation. Checked after the confirm-mode short-circuit so
497
+ // turning the gate off still overrides an earlier refusal.
498
+ if (sessionDenies(rootSessionId, name)) return { ok: false, error: denialText(name) };
423
499
 
424
500
  let preview = null;
425
501
  if (name === 'writeFile' || name === 'editFile') {
@@ -433,8 +509,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
433
509
  diff: preview && preview.diff,
434
510
  });
435
511
 
512
+ // Only a click is remembered. 'cancel' (aborted turn / nobody able to
513
+ // answer) refuses this call but must not write a durable "no" the user
514
+ // never gave.
436
515
  if (decision === 'deny') {
437
- return { ok: false, error: `${name} was not executed — the user denied the request.` };
516
+ denyForSession(rootSessionId, name);
517
+ return { ok: false, error: denialText(name) };
518
+ }
519
+ if (decision !== 'session' && decision !== 'once') {
520
+ return { ok: false, error: `${name} was not executed — the request was cancelled.` };
438
521
  }
439
522
  if (decision === 'session') allowForSession(rootSessionId, name);
440
523
 
@@ -545,7 +628,24 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
545
628
  messages: opts.messages,
546
629
  model: opts.model,
547
630
  mode: opts.mode,
548
- maxTokens: opts.maxTokens,
631
+ // Only a cap the caller STATED travels; an effort-derived one does not.
632
+ // The server derives its own budget from `effort` (aegis1 pass_budgets
633
+ // splits its ladder across workers + synthesis), so forwarding the
634
+ // derived number states one decision twice — and the copies had already
635
+ // drifted: 974adc5 doubled aegis1's ladder while this side stood still,
636
+ // leaving the client capping below the budget it displayed. The cap was
637
+ // never the one the renderer showed either (updateBudgetControls hides
638
+ // the max-tokens dropdown for the pooled class). Omitting the field is
639
+ // what tells aegis1 "no cap stated — let effort decide", the same
640
+ // contract aegiscodex-dev sends.
641
+ //
642
+ // A cap the caller DID state is a different thing, and dropping it was
643
+ // a bug: aegis1 reads a body max_tokens as a ceiling over its ladder,
644
+ // so omitting it does not bound the call — it grants the full top rung
645
+ // instead. A deliberate 4096 would have run at 32768, which is the same
646
+ // "only ever raise the caller's ceiling" failure the old
647
+ // Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]) had.
648
+ maxTokens: opts.statedMaxTokens,
549
649
  stream: opts.stream !== false,
550
650
  // The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
551
651
  // stream reports no token usage unless asked. Without this the Aegis
@@ -561,12 +661,53 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
561
661
  // AUTONOMOUS_IDLE_TIMEOUT_MS). Undefined elsewhere -> 60s default.
562
662
  idleTimeoutMs: opts.idleTimeoutMs,
563
663
  signal: opts.signal,
564
- // aegis_memory: automatic, no button — the server both reads prior
565
- // synced memory into context AND writes this turn back to it, the
566
- // same flag aegis-online sets. Matches aegiscodex-dev's own
567
- // cross-session memory (auto-indexed, no manual tagging).
664
+ // aegis_recall — the read half of aegis_memory, and the only half a
665
+ // client on this engine should send.
666
+ //
667
+ // aegis_memory is both halves: it injects the account's synced memory
668
+ // into context AND persists this turn back into it, charging sync
669
+ // quota for the write. That pairing is right for aegis-online, whose
670
+ // chat has nowhere else to live. It is wrong here, because this engine
671
+ // is shared by the GUI and the CLI (cli/src/deps.js loads this file):
672
+ // either would be billed for every "hey" and would fill the user's
673
+ // memory with greetings. Verified against aegis1 app.py: every
674
+ // write-back site (the note/upsert calls) is gated on aegis_memory,
675
+ // and aegis_memory implies aegis_recall there — so dropping the write
676
+ // half leaves /online unchanged and costs nothing on the read side.
677
+ //
678
+ // Writing is still available, explicitly: the aegis_memory_save tool
679
+ // (mcp/tools.js; the CLI exposes it as /memory). Recall on every turn,
680
+ // store only what the user asks for — which is what aegiscodex-dev's
681
+ // own cross-session memory does, and what this comment claimed to
682
+ // match while sending both halves.
683
+ //
684
+ // Recall was previously unreachable for a client that would not pay
685
+ // for it: services/tiered_recall.py only ran behind a flag that also
686
+ // bought a write, so the tiered path existed with no caller able to
687
+ // afford it. The split is what makes it reachable. The DEEP tier of the
688
+ // same read is a third flag with its own price — see `opts.recallDeep`
689
+ // below, which is off unless the session opted in.
568
690
  extra: {
569
- aegis_memory: true,
691
+ aegis_recall: true,
692
+ // The DEEP tier of that read — brain corrections plus the semantic
693
+ // answer cache — is not the same price, so it does not ride along.
694
+ // aegis1 app.py:8367 reads `aegis_recall_deep` (or the
695
+ // X-AEGIS-Recall-Deep header) and services/brain_memory.py
696
+ // find_cached_answer embeds the query: one provider embedding per
697
+ // turn, metered. The server deliberately implies it from
698
+ // `aegis_memory` and NOT from `aegis_recall`, so that a terminal
699
+ // client can buy the cheap read without the embedding.
700
+ //
701
+ // This client is that terminal client (the CLI loads this file via
702
+ // cli/src/deps.js), so it must not opt itself in: the flag is sent
703
+ // only when the SESSION asked for it — CLI `/memory-deep on`, a
704
+ // desktop payload with `recallDeep: true` — and it defaults false
705
+ // everywhere. It also travels only on the user's own turn
706
+ // (`opts.recallDeep` is cleared for every other dispatch below): a
707
+ // tool round, the doubled-budget retry and the write-up re-dispatch
708
+ // all re-send a context whose embedding the first round already
709
+ // bought, which would turn one embedding per turn into one per round.
710
+ ...(opts.recallDeep ? { aegis_recall_deep: true } : {}),
570
711
  session: opts.sessionId,
571
712
  // The fan-out is opt-in per dispatch. `brain` is sent EXPLICITLY
572
713
  // whenever this dispatch is not the autonomous one, because the
@@ -638,7 +779,16 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
638
779
  async function chat(payload, onDelta) {
639
780
  const cls = payload && payload.class;
640
781
  const model = payload && payload.model;
641
- const maxTokens = deepseekReasoningFloor(model, payload && payload.maxTokens, payload && payload.effort);
782
+ const maxTokens = reasoningBudget(model, payload && payload.maxTokens, payload && payload.effort);
783
+ // The caller's OWN number, kept apart from `maxTokens` above. That one
784
+ // collapses two different facts into a single value — "the caller stated
785
+ // 4096" and "effort implies 32768" — and the pooled path must treat them
786
+ // differently. A stated cap is a liability ceiling the server honours
787
+ // downward (aegis1 pass_budgets: total = min(ladder, max_tokens x passes));
788
+ // an effort-derived one is the server's own arithmetic stated twice, and
789
+ // sending it is how the two copies came to disagree. So the pooled call
790
+ // forwards only what the caller actually asked for.
791
+ const statedMaxTokens = Number(payload && payload.maxTokens) > 0 ? Number(payload.maxTokens) : undefined;
642
792
  // "Work autonomously" — routes this call through aegis1's pool_brain
643
793
  // worker fan-out (services/pool_brain.py: N reasoning workers + a
644
794
  // synthesis pass) instead of a single provider call. UI-gated to the
@@ -710,9 +860,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
710
860
  }
711
861
 
712
862
  const base = {
713
- cls, model, mode: payload && payload.mode, maxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
863
+ cls, model, mode: payload && payload.mode, maxTokens, statedMaxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
714
864
  effort: payload && payload.effort,
715
865
  workers: payload && payload.workers,
866
+ // Deep recall (`aegis_recall_deep`) is an explicit per-session opt-in
867
+ // and never a default: it costs one provider embedding per turn
868
+ // server-side (aegis1 services/brain_memory.py find_cached_answer), so
869
+ // a client that pays per turn must not turn it on for itself. Only a
870
+ // literal `true` from the caller counts — an absent or `undefined`
871
+ // field is off, which is what keeps every existing caller (the
872
+ // renderer's IPC payloads included) on the cheap read.
873
+ recallDeep: payload && payload.recallDeep === true,
716
874
  onReasoning,
717
875
  idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
718
876
  // A caller with no live streaming surface (a `--no-stream` CLI flag, a
@@ -741,6 +899,26 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
741
899
  let truncationRetried = false;
742
900
  let synthesisDone = false;
743
901
 
902
+ // Round cap — the bound aegiscodex-dev has had all along and this engine
903
+ // did not. Removing the old fixed cap (12) was right in spirit and wrong
904
+ // in effect: it left the turn with NO horizon, and on 2026-09-15 the
905
+ // question "can you check the plan" ran ~70 rounds, grew the context
906
+ // from 2,260 to 116,011 tokens, cost about EUR 2, and spent those rounds
907
+ // writing 400 lines of unrequested code into the source tree. Each round
908
+ // re-sends the whole conversation, so an unbounded loop gets more
909
+ // expensive the longer it runs.
910
+ //
911
+ // The numbers match aegiscodex-dev's (src/autonomous.js) so both clients
912
+ // behave the same: 24 rounds for a chat turn, 40 for an autonomous one.
913
+ // Env-overridable for a deliberately long job.
914
+ const maxRounds = (() => {
915
+ const name = autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS';
916
+ const raw = Number.parseInt(process.env[name] || '', 10);
917
+ if (Number.isFinite(raw) && raw > 0) return raw;
918
+ return autonomous ? 40 : 24;
919
+ })();
920
+ let round = 0;
921
+
744
922
  // Token accounting for the whole TURN, not just its last round. An
745
923
  // agentic turn makes one provider call per tool round, and returning only
746
924
  // the final round's `usage` (what this did) reported a fraction of what
@@ -789,7 +967,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
789
967
  };
790
968
 
791
969
  for (;;) {
792
- const opts = { ...base, system, messages: history, prompt, tools: toolSchemas };
970
+ // Stop and SAY so. A turn that reaches its horizon has usually done
971
+ // real work; ending silently would paint an empty answer over it,
972
+ // which is the same "(empty response)" failure the guards below exist
973
+ // to prevent.
974
+ if (round >= maxRounds) {
975
+ const note =
976
+ `[stopped at ${maxRounds} tool rounds` +
977
+ `${turnUsage.total_tokens ? `, ${turnUsage.total_tokens.toLocaleString()} tokens` : ''}` +
978
+ `. Ask again to continue, or raise ` +
979
+ `${autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS'}.]`;
980
+ if (rootOnDelta) rootOnDelta({ delta: `\n\n${note}` });
981
+ return withTurnUsage({
982
+ model: base.model,
983
+ choices: [{ message: { content: note }, finish_reason: 'length' }],
984
+ stoppedOnRounds: true,
985
+ });
986
+ }
987
+ round += 1;
988
+ // `round === 1` is the user's own ask, and the ONLY dispatch allowed to
989
+ // carry the deep-recall opt-in: the deep tier embeds the query once per
990
+ // dispatch, so leaving it on for an agentic turn would charge one
991
+ // embedding per tool round instead of one per turn (the retries below
992
+ // clear it explicitly, being re-dispatches inside round 1).
993
+ const opts = { ...base, system, messages: history, prompt, tools: toolSchemas, recallDeep: base.recallDeep && round === 1 };
793
994
  let res;
794
995
  try {
795
996
  res = await dispatch(cls, opts);
@@ -833,7 +1034,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
833
1034
  res = await dispatch(cls, {
834
1035
  ...opts,
835
1036
  singlePass: true,
1037
+ // A re-dispatch, not a new ask: the deep tier's embedding was
1038
+ // bought by round 1, and buying it again here would charge a
1039
+ // second one for the same context.
1040
+ recallDeep: false,
836
1041
  maxTokens: doubledBudget(opts.maxTokens),
1042
+ // Doubling applies to the pooled path only when the caller stated a
1043
+ // number. With none stated, the server's effort ladder IS the
1044
+ // budget, and sending doubledBudget's 8192 floor would *lower* it
1045
+ // (aegis1 reads max_tokens as a ceiling over the ladder) — a
1046
+ // "double the budget" retry that halves it at high effort.
1047
+ statedMaxTokens: opts.statedMaxTokens ? doubledBudget(opts.statedMaxTokens) : undefined,
837
1048
  });
838
1049
  addUsage(res);
839
1050
  }
@@ -858,6 +1069,9 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
858
1069
  res = await dispatch(cls, {
859
1070
  ...opts,
860
1071
  singlePass: true,
1072
+ // Same as the truncation retry above: this pass writes up findings
1073
+ // already in `history`, and a fresh embedding buys it nothing.
1074
+ recallDeep: false,
861
1075
  messages: history,
862
1076
  prompt: '',
863
1077
  tools: [],
@@ -31,8 +31,15 @@ const MAIN_CHAT_PROMPT =
31
31
  `- Use tools silently. A one-line reason is enough; do not narrate your plan as a story. ` +
32
32
  `- Never claim what a tool found or what a command returned before the tool actually runs. ` +
33
33
  ` Report only the results you really received. ` +
34
- `- Act, don't just inspect. After at most 2 rounds of reading or exploration, start making ` +
35
- ` changes with writeFile or editFile. Reconnaissance is not progress — implement, then verify.\n` +
34
+ `- Match the response to the request. A greeting, a question about something you already ` +
35
+ ` know, or a request for an opinion is answered in plain words with NO tools. Only reach ` +
36
+ ` for a tool when the answer genuinely depends on something in this repo or on this ` +
37
+ ` machine. "hey" is not a task.\n` +
38
+ `- WHEN THE USER HAS ASKED FOR WORK: act, don't just inspect. After at most 2 rounds of ` +
39
+ ` reading or exploration, start making changes with writeFile or editFile. Reconnaissance ` +
40
+ ` is not progress — implement, then verify. This rule is about HOW to carry out a task you ` +
41
+ ` were given; it is never a reason to invent one. Never write or modify a file the user ` +
42
+ ` did not ask you to touch.\n` +
36
43
  `- When you have what you need, stop using tools and give a concise, direct answer to the ` +
37
44
  ` user's question. Never end your turn with an intention like "Let me check…" or "I'll now…" ` +
38
45
  ` — that is not an answer. ` +
@@ -235,27 +235,69 @@ function openaiToAnthropicTool(tool) {
235
235
  };
236
236
  }
237
237
 
238
+ /**
239
+ * Idle budget for a direct-provider stream: the longest silence tolerated
240
+ * between two real SSE `data:` frames. Without it `reader.read()` below waits
241
+ * forever on a provider that holds the socket open and never answers — the
242
+ * host hangs with no error and no way out but force-quit. Measured between
243
+ * payloads, never reset by a keep-alive: DeepSeek answers a stalled request
244
+ * with ": keep-alive" comments and nothing else, indefinitely, so a watchdog
245
+ * that treats those as progress can never fire.
246
+ * Generous enough that a slow reasoning model mid-answer is untouched.
247
+ */
248
+ const SSE_IDLE_TIMEOUT_MS = 2 * 60_000;
249
+
238
250
  /** Read an SSE body, invoking onEvent(json) for each parsed `data:` payload. */
239
- async function readSSE(res, onEvent) {
251
+ async function readSSE(res, onEvent, { idleTimeoutMs = SSE_IDLE_TIMEOUT_MS } = {}) {
240
252
  const reader = res.body.getReader();
241
253
  const decoder = new TextDecoder();
242
254
  let buffer = '';
255
+ let lastPayloadAt = Date.now();
256
+ let keepAlives = 0;
257
+ const readWithIdleTimeout = async () => {
258
+ let timer;
259
+ const remaining = Math.max(0, idleTimeoutMs - (Date.now() - lastPayloadAt));
260
+ const timeout = new Promise((_, reject) => {
261
+ timer = setTimeout(() => {
262
+ reject(new Error(
263
+ keepAlives > 0
264
+ ? `stream stalled - only keep-alives for ${idleTimeoutMs / 1000}s`
265
+ : `stream stalled - no data for ${idleTimeoutMs / 1000}s`
266
+ ));
267
+ }, remaining);
268
+ });
269
+ try {
270
+ return await Promise.race([reader.read(), timeout]);
271
+ } finally {
272
+ clearTimeout(timer);
273
+ }
274
+ };
243
275
  for (;;) {
244
- const { done, value } = await reader.read();
276
+ let done, value;
277
+ try {
278
+ ({ done, value } = await readWithIdleTimeout());
279
+ } catch (err) {
280
+ reader.cancel().catch(() => {});
281
+ throw err;
282
+ }
245
283
  if (done) break;
246
284
  buffer += decoder.decode(value, { stream: true });
247
285
  const lines = buffer.split('\n');
248
286
  buffer = lines.pop(); // keep the trailing partial line
249
287
  for (const raw of lines) {
250
288
  const line = raw.trim();
251
- if (!line.startsWith('data:')) continue;
289
+ if (!line.startsWith('data:')) {
290
+ if (line.startsWith(':')) keepAlives++;
291
+ continue;
292
+ }
293
+ lastPayloadAt = Date.now(); // a real frame: the stream is still speaking
252
294
  const payload = line.slice(5).trim();
253
295
  if (!payload || payload === '[DONE]') continue;
254
296
  let json;
255
297
  try {
256
298
  json = JSON.parse(payload);
257
299
  } catch {
258
- continue; // keepalive / partial line
300
+ continue; // partial line
259
301
  }
260
302
  onEvent(json);
261
303
  }
@@ -263,7 +305,7 @@ async function readSSE(res, onEvent) {
263
305
  }
264
306
 
265
307
  /** POST with stream:true; streams SSE events or falls back to plain JSON. */
266
- async function requestStream({ url, headers, body, signal, onEvent }) {
308
+ async function requestStream({ url, headers, body, signal, onEvent, idleTimeoutMs }) {
267
309
  const res = await fetch(url, {
268
310
  method: 'POST',
269
311
  headers,
@@ -282,7 +324,7 @@ async function requestStream({ url, headers, body, signal, onEvent }) {
282
324
 
283
325
  const contentType = res.headers.get('content-type') || '';
284
326
  if (contentType.includes('text/event-stream')) {
285
- await readSSE(res, onEvent);
327
+ await readSSE(res, onEvent, { idleTimeoutMs });
286
328
  return;
287
329
  }
288
330
 
@@ -342,15 +384,28 @@ async function openaiCompatible({
342
384
  if (json.usage) usage = json.usage;
343
385
  const choice = json.choices && json.choices[0];
344
386
  if (choice && choice.finish_reason) finishReason = choice.finish_reason;
345
- const delta =
346
- (choice &&
347
- ((choice.delta && choice.delta.content) ||
348
- (choice.message && choice.message.content))) ||
349
- '';
350
- if (delta) {
351
- fullText += delta;
352
- if (onDelta) onDelta({ delta });
387
+ // `delta.content` is an INCREMENT; `message.content` is a SNAPSHOT of
388
+ // the whole message. Collapsing them with `||` made any stream that
389
+ // ends with a message snapshot append the entire answer a second time —
390
+ // the same duplication the shared client had. Emit only the unseen tail.
391
+ const inc = choice && choice.delta && choice.delta.content;
392
+ const snap = choice && choice.message && choice.message.content;
393
+ let add = '';
394
+ if (typeof inc === 'string' && inc) {
395
+ add = inc;
396
+ fullText += inc;
397
+ } else if (typeof snap === 'string' && snap) {
398
+ if (!fullText) {
399
+ add = snap;
400
+ fullText = snap;
401
+ } else if (snap.startsWith(fullText)) {
402
+ add = snap.slice(fullText.length);
403
+ fullText = snap;
404
+ }
405
+ // Disjoint from what was already shown: appending would duplicate, so
406
+ // keep what the caller has seen.
353
407
  }
408
+ if (add && onDelta) onDelta({ delta: add });
354
409
  // Streamed fragments (delta.tool_calls) and the whole-answer shape a
355
410
  // non-streaming fallback returns (message.tool_calls) both land here.
356
411
  const fragments =