aegis-desktop 0.5.0 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/engine.js +243 -29
- package/lib/local/prompt.js +9 -2
- package/lib/local/providers.js +69 -14
- package/lib/local/tools.js +24 -39
- package/main.js +137 -8
- package/package.json +1 -1
- package/preload.js +6 -0
- package/renderer/app.js +236 -15
- package/renderer/index.html +36 -0
- package/renderer/style.css +237 -4
- package/renderer/transcript-view.js +75 -1
- package/vendor/aegis.js +77 -22
- package/vendor/session-store.js +196 -100
- package/vendor/update.js +140 -0
package/lib/local/engine.js
CHANGED
|
@@ -109,18 +109,38 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
|
|
|
109
109
|
const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
|
|
110
110
|
|
|
111
111
|
/**
|
|
112
|
-
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
*
|
|
118
|
-
*
|
|
112
|
+
* The budget a DeepSeek reasoning model runs on, resolved from EXACTLY ONE
|
|
113
|
+
* authority per call.
|
|
114
|
+
*
|
|
115
|
+
* A caller-stated number IS the budget, and is returned verbatim. For the
|
|
116
|
+
* non-pooled classes the renderer's max-tokens dropdown is the only budget
|
|
117
|
+
* control on offer — updateBudgetControls hides the effort row for them — so
|
|
118
|
+
* silently raising that number to an effort rung is precisely what made the
|
|
119
|
+
* figure beside the dropdown untrustworthy. The old form was
|
|
120
|
+
* `Math.max(stated, EFFORT_TOKEN_BUDGET[eff])`, which could only ever raise a
|
|
121
|
+
* deliberate cap: a caller asking for 1024 ran on 32768, and the number the
|
|
122
|
+
* UI displayed was never the number the call used.
|
|
123
|
+
*
|
|
124
|
+
* The effort rung is the DEFAULT, consulted only when no number was stated at
|
|
125
|
+
* all (the pooled class, which the renderer sends `effort` for and which the
|
|
126
|
+
* server sizes itself). This is the same rule doubledBudget() follows for its
|
|
127
|
+
* truncation retry: a stated cap is never overridden, by a rung or an order of
|
|
128
|
+
* magnitude.
|
|
129
|
+
*
|
|
130
|
+
* Truncated and empty turns are handled where they belong — the doubled-budget
|
|
131
|
+
* retry plus emptyTurnError — rather than by inflating the caller's ceiling up
|
|
132
|
+
* front. Escalating on a demonstrated empty turn is strictly cheaper than
|
|
133
|
+
* pre-emptively granting the top rung to every reasoning call.
|
|
134
|
+
*
|
|
135
|
+
* Everything else (non-DeepSeek models, non-reasoning DeepSeek ids like
|
|
136
|
+
* deepseek-chat) passes through untouched.
|
|
119
137
|
*/
|
|
120
|
-
function
|
|
138
|
+
function reasoningBudget(model, maxTokens, effort) {
|
|
121
139
|
if (!DEEPSEEK_REASONING_MODEL_RE.test(String(model || ''))) return maxTokens;
|
|
140
|
+
const stated = Number(maxTokens);
|
|
141
|
+
if (Number.isFinite(stated) && stated > 0) return stated;
|
|
122
142
|
const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
|
|
123
|
-
return
|
|
143
|
+
return EFFORT_TOKEN_BUDGET[eff];
|
|
124
144
|
}
|
|
125
145
|
|
|
126
146
|
/** Relay model entries arrive as ids or objects; keep only real model ids. */
|
|
@@ -326,6 +346,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
326
346
|
// In-memory only, on purpose — never persisted, so a restart (or
|
|
327
347
|
// newChat()'s clearSessionApprovals) always starts from a clean gate.
|
|
328
348
|
const sessionAllowlists = new Map(); // rootSessionId -> Set<toolName>
|
|
349
|
+
// Denials are remembered for the same reason allows are, mirroring it for
|
|
350
|
+
// the other answer. Without this, a "Deny" was forgotten the instant it was
|
|
351
|
+
// given: the model read `… the user denied the request`, re-planned, called
|
|
352
|
+
// the SAME tool again, and the gate raised a SECOND card — so one "no" cost
|
|
353
|
+
// the user a prompt per round for up to maxRounds (24 chat / 40 autonomous)
|
|
354
|
+
// rounds, each round re-sending the whole conversation to the provider. A
|
|
355
|
+
// gate that only remembers "yes" turns a single click into a retry storm;
|
|
356
|
+
// remembering "no" makes the first answer stick.
|
|
357
|
+
const sessionDenials = new Map(); // rootSessionId -> Set<toolName>
|
|
329
358
|
const pendingApprovals = new Map(); // approvalId -> { resolve }
|
|
330
359
|
|
|
331
360
|
function sessionAllows(rootId, name) {
|
|
@@ -338,10 +367,44 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
338
367
|
sessionAllowlists.get(rootId).add(name);
|
|
339
368
|
}
|
|
340
369
|
|
|
370
|
+
/** Has this conversation already refused this tool? Checked before the card
|
|
371
|
+
* is raised, so a repeat call is refused outright instead of re-prompting. */
|
|
372
|
+
function sessionDenies(rootId, name) {
|
|
373
|
+
const set = sessionDenials.get(rootId);
|
|
374
|
+
return Boolean(set && set.has(name));
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
function denyForSession(rootId, name) {
|
|
378
|
+
if (!sessionDenials.has(rootId)) sessionDenials.set(rootId, new Set());
|
|
379
|
+
sessionDenials.get(rootId).add(name);
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
/**
|
|
383
|
+
* The text a refused tool hands back to the model. The old one-line
|
|
384
|
+
* `"<tool> was not executed — the user denied the request."` read to a
|
|
385
|
+
* capable agent as a transient failure to route around: it would apologise,
|
|
386
|
+
* pick a different command that does the same thing, and call the gate
|
|
387
|
+
* again. This says the durable part out loud (the refusal covers the rest of
|
|
388
|
+
* the conversation, not just that call) and asks for the one response that
|
|
389
|
+
* actually helps — say what you need and stop, so the user can re-enable it.
|
|
390
|
+
*/
|
|
391
|
+
function denialText(name) {
|
|
392
|
+
return (
|
|
393
|
+
`${name} was not executed — the user denied this tool for this conversation. ` +
|
|
394
|
+
`Do NOT retry it and do NOT attempt the same effect by another route ` +
|
|
395
|
+
`(another command, a writeFile instead of an edit, a subagent). ` +
|
|
396
|
+
`Stop calling tools and reply in plain text: say what you were trying to do, ` +
|
|
397
|
+
`what you need, and that the user can re-enable ${name} to let it proceed.`
|
|
398
|
+
);
|
|
399
|
+
}
|
|
400
|
+
|
|
341
401
|
/** newChat() in the renderer calls this so a fresh conversation never
|
|
342
|
-
* inherits a prior thread's blanket allows.
|
|
402
|
+
* inherits a prior thread's blanket allows — or its refusals. A new chat is
|
|
403
|
+
* a new gate in both directions: leaving denials behind would silently
|
|
404
|
+
* refuse a tool in a thread where the user never said no. */
|
|
343
405
|
function clearSessionApprovals(rootSessionId) {
|
|
344
406
|
sessionAllowlists.delete(rootSessionId);
|
|
407
|
+
sessionDenials.delete(rootSessionId);
|
|
345
408
|
return { ok: true };
|
|
346
409
|
}
|
|
347
410
|
|
|
@@ -359,22 +422,28 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
359
422
|
|
|
360
423
|
/**
|
|
361
424
|
* Ask the renderer to approve one mutating tool call. Resolves 'once',
|
|
362
|
-
* 'session' or 'deny'
|
|
363
|
-
*
|
|
364
|
-
*
|
|
365
|
-
*
|
|
366
|
-
*
|
|
367
|
-
*
|
|
425
|
+
* 'session' or 'deny' (an explicit choice by the user) — or 'cancel' when
|
|
426
|
+
* nobody ever answered: no listener able to reply, or the turn was aborted.
|
|
427
|
+
* 'cancel' is kept apart from 'deny' on purpose. Both refuse the call, but
|
|
428
|
+
* only 'deny' is a decision the user made, so only 'deny' may be remembered
|
|
429
|
+
* as "this conversation said no" (see sessionDenials). Folding the two
|
|
430
|
+
* together — which is what the old fail-safe did, resolving 'deny' for an
|
|
431
|
+
* abort — would let a cancelled turn permanently refuse a tool the user
|
|
432
|
+
* never ruled on. Sent over `rootOnDelta` (see chat()) as an `{ approval }`
|
|
433
|
+
* chunk so it rides the exact same streaming channel as tool-activity
|
|
434
|
+
* chunks — no new IPC surface needed on the push side, only on the reply
|
|
435
|
+
* side (respondApproval). Fails safe: an unanswered request refuses the
|
|
436
|
+
* call instead of hanging the tool round forever.
|
|
368
437
|
*/
|
|
369
438
|
function requestApproval(rootSessionId, rootOnDelta, signal, info) {
|
|
370
439
|
return new Promise((resolve) => {
|
|
371
440
|
if (signal && signal.aborted) {
|
|
372
|
-
resolve('
|
|
441
|
+
resolve('cancel');
|
|
373
442
|
return;
|
|
374
443
|
}
|
|
375
444
|
const id = randomUUID();
|
|
376
445
|
let settled = false;
|
|
377
|
-
const onAbort = () => finish('
|
|
446
|
+
const onAbort = () => finish('cancel');
|
|
378
447
|
const finish = (decision) => {
|
|
379
448
|
if (settled) return;
|
|
380
449
|
settled = true;
|
|
@@ -385,7 +454,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
385
454
|
if (signal) signal.addEventListener('abort', onAbort, { once: true });
|
|
386
455
|
pendingApprovals.set(id, { resolve: finish });
|
|
387
456
|
if (typeof rootOnDelta !== 'function') {
|
|
388
|
-
finish('
|
|
457
|
+
finish('cancel');
|
|
389
458
|
return;
|
|
390
459
|
}
|
|
391
460
|
rootOnDelta({
|
|
@@ -420,6 +489,13 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
420
489
|
if (!T.MUTATING_TOOLS.has(name)) return T.executeTool(name, args, toolCtx);
|
|
421
490
|
if (!confirmModeEnabled()) return T.executeTool(name, args, toolCtx);
|
|
422
491
|
if (sessionAllows(rootSessionId, name)) return T.executeTool(name, args, toolCtx);
|
|
492
|
+
// Already refused in this conversation: refuse again WITHOUT raising a
|
|
493
|
+
// second card. Before this, the model's retry after a denial re-prompted
|
|
494
|
+
// the user for the same tool — one "no" produced a card per round for up
|
|
495
|
+
// to maxRounds rounds, each one a billed provider call re-sending the
|
|
496
|
+
// whole conversation. Checked after the confirm-mode short-circuit so
|
|
497
|
+
// turning the gate off still overrides an earlier refusal.
|
|
498
|
+
if (sessionDenies(rootSessionId, name)) return { ok: false, error: denialText(name) };
|
|
423
499
|
|
|
424
500
|
let preview = null;
|
|
425
501
|
if (name === 'writeFile' || name === 'editFile') {
|
|
@@ -433,8 +509,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
433
509
|
diff: preview && preview.diff,
|
|
434
510
|
});
|
|
435
511
|
|
|
512
|
+
// Only a click is remembered. 'cancel' (aborted turn / nobody able to
|
|
513
|
+
// answer) refuses this call but must not write a durable "no" the user
|
|
514
|
+
// never gave.
|
|
436
515
|
if (decision === 'deny') {
|
|
437
|
-
|
|
516
|
+
denyForSession(rootSessionId, name);
|
|
517
|
+
return { ok: false, error: denialText(name) };
|
|
518
|
+
}
|
|
519
|
+
if (decision !== 'session' && decision !== 'once') {
|
|
520
|
+
return { ok: false, error: `${name} was not executed — the request was cancelled.` };
|
|
438
521
|
}
|
|
439
522
|
if (decision === 'session') allowForSession(rootSessionId, name);
|
|
440
523
|
|
|
@@ -545,7 +628,24 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
545
628
|
messages: opts.messages,
|
|
546
629
|
model: opts.model,
|
|
547
630
|
mode: opts.mode,
|
|
548
|
-
|
|
631
|
+
// Only a cap the caller STATED travels; an effort-derived one does not.
|
|
632
|
+
// The server derives its own budget from `effort` (aegis1 pass_budgets
|
|
633
|
+
// splits its ladder across workers + synthesis), so forwarding the
|
|
634
|
+
// derived number states one decision twice — and the copies had already
|
|
635
|
+
// drifted: 974adc5 doubled aegis1's ladder while this side stood still,
|
|
636
|
+
// leaving the client capping below the budget it displayed. The cap was
|
|
637
|
+
// never the one the renderer showed either (updateBudgetControls hides
|
|
638
|
+
// the max-tokens dropdown for the pooled class). Omitting the field is
|
|
639
|
+
// what tells aegis1 "no cap stated — let effort decide", the same
|
|
640
|
+
// contract aegiscodex-dev sends.
|
|
641
|
+
//
|
|
642
|
+
// A cap the caller DID state is a different thing, and dropping it was
|
|
643
|
+
// a bug: aegis1 reads a body max_tokens as a ceiling over its ladder,
|
|
644
|
+
// so omitting it does not bound the call — it grants the full top rung
|
|
645
|
+
// instead. A deliberate 4096 would have run at 32768, which is the same
|
|
646
|
+
// "only ever raise the caller's ceiling" failure the old
|
|
647
|
+
// Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]) had.
|
|
648
|
+
maxTokens: opts.statedMaxTokens,
|
|
549
649
|
stream: opts.stream !== false,
|
|
550
650
|
// The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
|
|
551
651
|
// stream reports no token usage unless asked. Without this the Aegis
|
|
@@ -561,12 +661,53 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
561
661
|
// AUTONOMOUS_IDLE_TIMEOUT_MS). Undefined elsewhere -> 60s default.
|
|
562
662
|
idleTimeoutMs: opts.idleTimeoutMs,
|
|
563
663
|
signal: opts.signal,
|
|
564
|
-
//
|
|
565
|
-
//
|
|
566
|
-
//
|
|
567
|
-
//
|
|
664
|
+
// aegis_recall — the read half of aegis_memory, and the only half a
|
|
665
|
+
// client on this engine should send.
|
|
666
|
+
//
|
|
667
|
+
// aegis_memory is both halves: it injects the account's synced memory
|
|
668
|
+
// into context AND persists this turn back into it, charging sync
|
|
669
|
+
// quota for the write. That pairing is right for aegis-online, whose
|
|
670
|
+
// chat has nowhere else to live. It is wrong here, because this engine
|
|
671
|
+
// is shared by the GUI and the CLI (cli/src/deps.js loads this file):
|
|
672
|
+
// either would be billed for every "hey" and would fill the user's
|
|
673
|
+
// memory with greetings. Verified against aegis1 app.py: every
|
|
674
|
+
// write-back site (the note/upsert calls) is gated on aegis_memory,
|
|
675
|
+
// and aegis_memory implies aegis_recall there — so dropping the write
|
|
676
|
+
// half leaves /online unchanged and costs nothing on the read side.
|
|
677
|
+
//
|
|
678
|
+
// Writing is still available, explicitly: the aegis_memory_save tool
|
|
679
|
+
// (mcp/tools.js; the CLI exposes it as /memory). Recall on every turn,
|
|
680
|
+
// store only what the user asks for — which is what aegiscodex-dev's
|
|
681
|
+
// own cross-session memory does, and what this comment claimed to
|
|
682
|
+
// match while sending both halves.
|
|
683
|
+
//
|
|
684
|
+
// Recall was previously unreachable for a client that would not pay
|
|
685
|
+
// for it: services/tiered_recall.py only ran behind a flag that also
|
|
686
|
+
// bought a write, so the tiered path existed with no caller able to
|
|
687
|
+
// afford it. The split is what makes it reachable. The DEEP tier of the
|
|
688
|
+
// same read is a third flag with its own price — see `opts.recallDeep`
|
|
689
|
+
// below, which is off unless the session opted in.
|
|
568
690
|
extra: {
|
|
569
|
-
|
|
691
|
+
aegis_recall: true,
|
|
692
|
+
// The DEEP tier of that read — brain corrections plus the semantic
|
|
693
|
+
// answer cache — is not the same price, so it does not ride along.
|
|
694
|
+
// aegis1 app.py:8367 reads `aegis_recall_deep` (or the
|
|
695
|
+
// X-AEGIS-Recall-Deep header) and services/brain_memory.py
|
|
696
|
+
// find_cached_answer embeds the query: one provider embedding per
|
|
697
|
+
// turn, metered. The server deliberately implies it from
|
|
698
|
+
// `aegis_memory` and NOT from `aegis_recall`, so that a terminal
|
|
699
|
+
// client can buy the cheap read without the embedding.
|
|
700
|
+
//
|
|
701
|
+
// This client is that terminal client (the CLI loads this file via
|
|
702
|
+
// cli/src/deps.js), so it must not opt itself in: the flag is sent
|
|
703
|
+
// only when the SESSION asked for it — CLI `/memory-deep on`, a
|
|
704
|
+
// desktop payload with `recallDeep: true` — and it defaults false
|
|
705
|
+
// everywhere. It also travels only on the user's own turn
|
|
706
|
+
// (`opts.recallDeep` is cleared for every other dispatch below): a
|
|
707
|
+
// tool round, the doubled-budget retry and the write-up re-dispatch
|
|
708
|
+
// all re-send a context whose embedding the first round already
|
|
709
|
+
// bought, which would turn one embedding per turn into one per round.
|
|
710
|
+
...(opts.recallDeep ? { aegis_recall_deep: true } : {}),
|
|
570
711
|
session: opts.sessionId,
|
|
571
712
|
// The fan-out is opt-in per dispatch. `brain` is sent EXPLICITLY
|
|
572
713
|
// whenever this dispatch is not the autonomous one, because the
|
|
@@ -638,7 +779,16 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
638
779
|
async function chat(payload, onDelta) {
|
|
639
780
|
const cls = payload && payload.class;
|
|
640
781
|
const model = payload && payload.model;
|
|
641
|
-
const maxTokens =
|
|
782
|
+
const maxTokens = reasoningBudget(model, payload && payload.maxTokens, payload && payload.effort);
|
|
783
|
+
// The caller's OWN number, kept apart from `maxTokens` above. That one
|
|
784
|
+
// collapses two different facts into a single value — "the caller stated
|
|
785
|
+
// 4096" and "effort implies 32768" — and the pooled path must treat them
|
|
786
|
+
// differently. A stated cap is a liability ceiling the server honours
|
|
787
|
+
// downward (aegis1 pass_budgets: total = min(ladder, max_tokens x passes));
|
|
788
|
+
// an effort-derived one is the server's own arithmetic stated twice, and
|
|
789
|
+
// sending it is how the two copies came to disagree. So the pooled call
|
|
790
|
+
// forwards only what the caller actually asked for.
|
|
791
|
+
const statedMaxTokens = Number(payload && payload.maxTokens) > 0 ? Number(payload.maxTokens) : undefined;
|
|
642
792
|
// "Work autonomously" — routes this call through aegis1's pool_brain
|
|
643
793
|
// worker fan-out (services/pool_brain.py: N reasoning workers + a
|
|
644
794
|
// synthesis pass) instead of a single provider call. UI-gated to the
|
|
@@ -710,9 +860,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
710
860
|
}
|
|
711
861
|
|
|
712
862
|
const base = {
|
|
713
|
-
cls, model, mode: payload && payload.mode, maxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
|
|
863
|
+
cls, model, mode: payload && payload.mode, maxTokens, statedMaxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
|
|
714
864
|
effort: payload && payload.effort,
|
|
715
865
|
workers: payload && payload.workers,
|
|
866
|
+
// Deep recall (`aegis_recall_deep`) is an explicit per-session opt-in
|
|
867
|
+
// and never a default: it costs one provider embedding per turn
|
|
868
|
+
// server-side (aegis1 services/brain_memory.py find_cached_answer), so
|
|
869
|
+
// a client that pays per turn must not turn it on for itself. Only a
|
|
870
|
+
// literal `true` from the caller counts — an absent or `undefined`
|
|
871
|
+
// field is off, which is what keeps every existing caller (the
|
|
872
|
+
// renderer's IPC payloads included) on the cheap read.
|
|
873
|
+
recallDeep: payload && payload.recallDeep === true,
|
|
716
874
|
onReasoning,
|
|
717
875
|
idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
|
|
718
876
|
// A caller with no live streaming surface (a `--no-stream` CLI flag, a
|
|
@@ -741,6 +899,26 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
741
899
|
let truncationRetried = false;
|
|
742
900
|
let synthesisDone = false;
|
|
743
901
|
|
|
902
|
+
// Round cap — the bound aegiscodex-dev has had all along and this engine
|
|
903
|
+
// did not. Removing the old fixed cap (12) was right in spirit and wrong
|
|
904
|
+
// in effect: it left the turn with NO horizon, and on 2026-09-15 the
|
|
905
|
+
// question "can you check the plan" ran ~70 rounds, grew the context
|
|
906
|
+
// from 2,260 to 116,011 tokens, cost about EUR 2, and spent those rounds
|
|
907
|
+
// writing 400 lines of unrequested code into the source tree. Each round
|
|
908
|
+
// re-sends the whole conversation, so an unbounded loop gets more
|
|
909
|
+
// expensive the longer it runs.
|
|
910
|
+
//
|
|
911
|
+
// The numbers match aegiscodex-dev's (src/autonomous.js) so both clients
|
|
912
|
+
// behave the same: 24 rounds for a chat turn, 40 for an autonomous one.
|
|
913
|
+
// Env-overridable for a deliberately long job.
|
|
914
|
+
const maxRounds = (() => {
|
|
915
|
+
const name = autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS';
|
|
916
|
+
const raw = Number.parseInt(process.env[name] || '', 10);
|
|
917
|
+
if (Number.isFinite(raw) && raw > 0) return raw;
|
|
918
|
+
return autonomous ? 40 : 24;
|
|
919
|
+
})();
|
|
920
|
+
let round = 0;
|
|
921
|
+
|
|
744
922
|
// Token accounting for the whole TURN, not just its last round. An
|
|
745
923
|
// agentic turn makes one provider call per tool round, and returning only
|
|
746
924
|
// the final round's `usage` (what this did) reported a fraction of what
|
|
@@ -789,7 +967,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
789
967
|
};
|
|
790
968
|
|
|
791
969
|
for (;;) {
|
|
792
|
-
|
|
970
|
+
// Stop and SAY so. A turn that reaches its horizon has usually done
|
|
971
|
+
// real work; ending silently would paint an empty answer over it,
|
|
972
|
+
// which is the same "(empty response)" failure the guards below exist
|
|
973
|
+
// to prevent.
|
|
974
|
+
if (round >= maxRounds) {
|
|
975
|
+
const note =
|
|
976
|
+
`[stopped at ${maxRounds} tool rounds` +
|
|
977
|
+
`${turnUsage.total_tokens ? `, ${turnUsage.total_tokens.toLocaleString()} tokens` : ''}` +
|
|
978
|
+
`. Ask again to continue, or raise ` +
|
|
979
|
+
`${autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS'}.]`;
|
|
980
|
+
if (rootOnDelta) rootOnDelta({ delta: `\n\n${note}` });
|
|
981
|
+
return withTurnUsage({
|
|
982
|
+
model: base.model,
|
|
983
|
+
choices: [{ message: { content: note }, finish_reason: 'length' }],
|
|
984
|
+
stoppedOnRounds: true,
|
|
985
|
+
});
|
|
986
|
+
}
|
|
987
|
+
round += 1;
|
|
988
|
+
// `round === 1` is the user's own ask, and the ONLY dispatch allowed to
|
|
989
|
+
// carry the deep-recall opt-in: the deep tier embeds the query once per
|
|
990
|
+
// dispatch, so leaving it on for an agentic turn would charge one
|
|
991
|
+
// embedding per tool round instead of one per turn (the retries below
|
|
992
|
+
// clear it explicitly, being re-dispatches inside round 1).
|
|
993
|
+
const opts = { ...base, system, messages: history, prompt, tools: toolSchemas, recallDeep: base.recallDeep && round === 1 };
|
|
793
994
|
let res;
|
|
794
995
|
try {
|
|
795
996
|
res = await dispatch(cls, opts);
|
|
@@ -833,7 +1034,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
833
1034
|
res = await dispatch(cls, {
|
|
834
1035
|
...opts,
|
|
835
1036
|
singlePass: true,
|
|
1037
|
+
// A re-dispatch, not a new ask: the deep tier's embedding was
|
|
1038
|
+
// bought by round 1, and buying it again here would charge a
|
|
1039
|
+
// second one for the same context.
|
|
1040
|
+
recallDeep: false,
|
|
836
1041
|
maxTokens: doubledBudget(opts.maxTokens),
|
|
1042
|
+
// Doubling applies to the pooled path only when the caller stated a
|
|
1043
|
+
// number. With none stated, the server's effort ladder IS the
|
|
1044
|
+
// budget, and sending doubledBudget's 8192 floor would *lower* it
|
|
1045
|
+
// (aegis1 reads max_tokens as a ceiling over the ladder) — a
|
|
1046
|
+
// "double the budget" retry that halves it at high effort.
|
|
1047
|
+
statedMaxTokens: opts.statedMaxTokens ? doubledBudget(opts.statedMaxTokens) : undefined,
|
|
837
1048
|
});
|
|
838
1049
|
addUsage(res);
|
|
839
1050
|
}
|
|
@@ -858,6 +1069,9 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
858
1069
|
res = await dispatch(cls, {
|
|
859
1070
|
...opts,
|
|
860
1071
|
singlePass: true,
|
|
1072
|
+
// Same as the truncation retry above: this pass writes up findings
|
|
1073
|
+
// already in `history`, and a fresh embedding buys it nothing.
|
|
1074
|
+
recallDeep: false,
|
|
861
1075
|
messages: history,
|
|
862
1076
|
prompt: '',
|
|
863
1077
|
tools: [],
|
package/lib/local/prompt.js
CHANGED
|
@@ -31,8 +31,15 @@ const MAIN_CHAT_PROMPT =
|
|
|
31
31
|
`- Use tools silently. A one-line reason is enough; do not narrate your plan as a story. ` +
|
|
32
32
|
`- Never claim what a tool found or what a command returned before the tool actually runs. ` +
|
|
33
33
|
` Report only the results you really received. ` +
|
|
34
|
-
`-
|
|
35
|
-
`
|
|
34
|
+
`- Match the response to the request. A greeting, a question about something you already ` +
|
|
35
|
+
` know, or a request for an opinion is answered in plain words with NO tools. Only reach ` +
|
|
36
|
+
` for a tool when the answer genuinely depends on something in this repo or on this ` +
|
|
37
|
+
` machine. "hey" is not a task.\n` +
|
|
38
|
+
`- WHEN THE USER HAS ASKED FOR WORK: act, don't just inspect. After at most 2 rounds of ` +
|
|
39
|
+
` reading or exploration, start making changes with writeFile or editFile. Reconnaissance ` +
|
|
40
|
+
` is not progress — implement, then verify. This rule is about HOW to carry out a task you ` +
|
|
41
|
+
` were given; it is never a reason to invent one. Never write or modify a file the user ` +
|
|
42
|
+
` did not ask you to touch.\n` +
|
|
36
43
|
`- When you have what you need, stop using tools and give a concise, direct answer to the ` +
|
|
37
44
|
` user's question. Never end your turn with an intention like "Let me check…" or "I'll now…" ` +
|
|
38
45
|
` — that is not an answer. ` +
|
package/lib/local/providers.js
CHANGED
|
@@ -235,27 +235,69 @@ function openaiToAnthropicTool(tool) {
|
|
|
235
235
|
};
|
|
236
236
|
}
|
|
237
237
|
|
|
238
|
+
/**
|
|
239
|
+
* Idle budget for a direct-provider stream: the longest silence tolerated
|
|
240
|
+
* between two real SSE `data:` frames. Without it `reader.read()` below waits
|
|
241
|
+
* forever on a provider that holds the socket open and never answers — the
|
|
242
|
+
* host hangs with no error and no way out but force-quit. Measured between
|
|
243
|
+
* payloads, never reset by a keep-alive: DeepSeek answers a stalled request
|
|
244
|
+
* with ": keep-alive" comments and nothing else, indefinitely, so a watchdog
|
|
245
|
+
* that treats those as progress can never fire.
|
|
246
|
+
* Generous enough that a slow reasoning model mid-answer is untouched.
|
|
247
|
+
*/
|
|
248
|
+
const SSE_IDLE_TIMEOUT_MS = 2 * 60_000;
|
|
249
|
+
|
|
238
250
|
/** Read an SSE body, invoking onEvent(json) for each parsed `data:` payload. */
|
|
239
|
-
async function readSSE(res, onEvent) {
|
|
251
|
+
async function readSSE(res, onEvent, { idleTimeoutMs = SSE_IDLE_TIMEOUT_MS } = {}) {
|
|
240
252
|
const reader = res.body.getReader();
|
|
241
253
|
const decoder = new TextDecoder();
|
|
242
254
|
let buffer = '';
|
|
255
|
+
let lastPayloadAt = Date.now();
|
|
256
|
+
let keepAlives = 0;
|
|
257
|
+
const readWithIdleTimeout = async () => {
|
|
258
|
+
let timer;
|
|
259
|
+
const remaining = Math.max(0, idleTimeoutMs - (Date.now() - lastPayloadAt));
|
|
260
|
+
const timeout = new Promise((_, reject) => {
|
|
261
|
+
timer = setTimeout(() => {
|
|
262
|
+
reject(new Error(
|
|
263
|
+
keepAlives > 0
|
|
264
|
+
? `stream stalled - only keep-alives for ${idleTimeoutMs / 1000}s`
|
|
265
|
+
: `stream stalled - no data for ${idleTimeoutMs / 1000}s`
|
|
266
|
+
));
|
|
267
|
+
}, remaining);
|
|
268
|
+
});
|
|
269
|
+
try {
|
|
270
|
+
return await Promise.race([reader.read(), timeout]);
|
|
271
|
+
} finally {
|
|
272
|
+
clearTimeout(timer);
|
|
273
|
+
}
|
|
274
|
+
};
|
|
243
275
|
for (;;) {
|
|
244
|
-
|
|
276
|
+
let done, value;
|
|
277
|
+
try {
|
|
278
|
+
({ done, value } = await readWithIdleTimeout());
|
|
279
|
+
} catch (err) {
|
|
280
|
+
reader.cancel().catch(() => {});
|
|
281
|
+
throw err;
|
|
282
|
+
}
|
|
245
283
|
if (done) break;
|
|
246
284
|
buffer += decoder.decode(value, { stream: true });
|
|
247
285
|
const lines = buffer.split('\n');
|
|
248
286
|
buffer = lines.pop(); // keep the trailing partial line
|
|
249
287
|
for (const raw of lines) {
|
|
250
288
|
const line = raw.trim();
|
|
251
|
-
if (!line.startsWith('data:'))
|
|
289
|
+
if (!line.startsWith('data:')) {
|
|
290
|
+
if (line.startsWith(':')) keepAlives++;
|
|
291
|
+
continue;
|
|
292
|
+
}
|
|
293
|
+
lastPayloadAt = Date.now(); // a real frame: the stream is still speaking
|
|
252
294
|
const payload = line.slice(5).trim();
|
|
253
295
|
if (!payload || payload === '[DONE]') continue;
|
|
254
296
|
let json;
|
|
255
297
|
try {
|
|
256
298
|
json = JSON.parse(payload);
|
|
257
299
|
} catch {
|
|
258
|
-
continue; //
|
|
300
|
+
continue; // partial line
|
|
259
301
|
}
|
|
260
302
|
onEvent(json);
|
|
261
303
|
}
|
|
@@ -263,7 +305,7 @@ async function readSSE(res, onEvent) {
|
|
|
263
305
|
}
|
|
264
306
|
|
|
265
307
|
/** POST with stream:true; streams SSE events or falls back to plain JSON. */
|
|
266
|
-
async function requestStream({ url, headers, body, signal, onEvent }) {
|
|
308
|
+
async function requestStream({ url, headers, body, signal, onEvent, idleTimeoutMs }) {
|
|
267
309
|
const res = await fetch(url, {
|
|
268
310
|
method: 'POST',
|
|
269
311
|
headers,
|
|
@@ -282,7 +324,7 @@ async function requestStream({ url, headers, body, signal, onEvent }) {
|
|
|
282
324
|
|
|
283
325
|
const contentType = res.headers.get('content-type') || '';
|
|
284
326
|
if (contentType.includes('text/event-stream')) {
|
|
285
|
-
await readSSE(res, onEvent);
|
|
327
|
+
await readSSE(res, onEvent, { idleTimeoutMs });
|
|
286
328
|
return;
|
|
287
329
|
}
|
|
288
330
|
|
|
@@ -342,15 +384,28 @@ async function openaiCompatible({
|
|
|
342
384
|
if (json.usage) usage = json.usage;
|
|
343
385
|
const choice = json.choices && json.choices[0];
|
|
344
386
|
if (choice && choice.finish_reason) finishReason = choice.finish_reason;
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
387
|
+
// `delta.content` is an INCREMENT; `message.content` is a SNAPSHOT of
|
|
388
|
+
// the whole message. Collapsing them with `||` made any stream that
|
|
389
|
+
// ends with a message snapshot append the entire answer a second time —
|
|
390
|
+
// the same duplication the shared client had. Emit only the unseen tail.
|
|
391
|
+
const inc = choice && choice.delta && choice.delta.content;
|
|
392
|
+
const snap = choice && choice.message && choice.message.content;
|
|
393
|
+
let add = '';
|
|
394
|
+
if (typeof inc === 'string' && inc) {
|
|
395
|
+
add = inc;
|
|
396
|
+
fullText += inc;
|
|
397
|
+
} else if (typeof snap === 'string' && snap) {
|
|
398
|
+
if (!fullText) {
|
|
399
|
+
add = snap;
|
|
400
|
+
fullText = snap;
|
|
401
|
+
} else if (snap.startsWith(fullText)) {
|
|
402
|
+
add = snap.slice(fullText.length);
|
|
403
|
+
fullText = snap;
|
|
404
|
+
}
|
|
405
|
+
// Disjoint from what was already shown: appending would duplicate, so
|
|
406
|
+
// keep what the caller has seen.
|
|
353
407
|
}
|
|
408
|
+
if (add && onDelta) onDelta({ delta: add });
|
|
354
409
|
// Streamed fragments (delta.tool_calls) and the whole-answer shape a
|
|
355
410
|
// non-streaming fallback returns (message.tool_calls) both land here.
|
|
356
411
|
const fragments =
|