aegis-desktop 0.5.3 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -47,6 +47,28 @@ const toolsModule = require('./tools.js');
47
47
  const promptModule = require('./prompt.js');
48
48
  const { ShellSession } = require('./shell.js');
49
49
  const { agentSystemPrompt, agentRoleLabel } = require('./agents.js');
50
+ // Cooperative working-tree sharing (see each module's header). The lock
51
+ // serialises two hosts that start a turn on one checkout close together; the
52
+ // guard is the load-bearing half — it refuses a tree-wide git operation while
53
+ // another session's uncommitted work is present, which is the operation that
54
+ // destroyed a peer's file for real. Both live beside the engine because the
55
+ // CLI vendors this directory wholesale (cli/scripts/predist.mjs), so the GUI
56
+ // and the terminal get one implementation rather than two that drift.
57
+ const { beginTurnGuard, recordWrite, blocksDestructive } = require('./turn-guard.js');
58
+ const { acquireWorktreeLock, releaseWorktreeLock } = require('./worktree-lock.js');
59
+
60
+ /**
61
+ * How long a turn waits for the working-tree lock before running anyway.
62
+ *
63
+ * Short on purpose. This engine serves interactive hosts — a desktop window
64
+ * and a terminal — where stalling for the lock's own 11-minute default would
65
+ * be a worse failure than contending: the user is watching a spinner. So the
66
+ * lock is best-effort mutual exclusion for turns that start near-simultaneously
67
+ * (the common overlap), and the turn guard covers everything after that. A
68
+ * non-interactive caller that can afford to queue should pass
69
+ * `payload.worktreeWaitMs` with the long DEFAULT_WORKTREE_WAIT_MS instead.
70
+ */
71
+ const WORKTREE_LOCK_WAIT_MS = 1500;
50
72
 
51
73
  /** Classes whose transport is a user-supplied endpoint + credential. */
52
74
  const CUSTOM_CLASSES = Object.freeze(['openai-compat', 'anthropic']);
@@ -70,22 +92,30 @@ const CLASSES = [
70
92
  /**
71
93
  * Mirrors aegiscodex-dev's src/backend.js DEEPSEEK_REASONING_MODEL_RE +
72
94
  * EFFORT_TOKEN_BUDGET verbatim. DeepSeek's reasoning models (deepseek-flash,
73
- * deepseek-v4-pro, the deprecated deepseek-reasoner, and the legacy
74
- * v4-flash/v4.1-flash aliases some configs still carry) spend part of
75
- * max_tokens on hidden chain-of-thought before ever emitting visible
76
- * content — DeepSeek counts reasoning tokens against the same budget as
77
- * content. At the renderer's 4k default (index.html's max-tokens select),
78
- * any non-trivial question can burn the whole budget reasoning and finish
79
- * with empty content: no error, no tool calls, just a turn that "completes"
80
- * with nothing to show for it (the empty-response bug). A user pointing the
81
- * Custom OpenAI-compatible class straight at DeepSeek's API hits exactly
82
- * this, so the request floors to the same effort budget aegiscodex-dev uses
83
- * for its own direct DeepSeek calls instead of shipping whatever the
84
- * dropdown happens to have selected.
95
+ * i.e. "Flash 4.1", deepseek-v4-pro, the deprecated deepseek-reasoner, and the
96
+ * legacy v4-flash/v4.1-flash aliases some configs still carry) spend part of
97
+ * the budget on hidden chain-of-thought before ever emitting visible content —
98
+ * DeepSeek counts reasoning tokens against the same budget as content. Under
99
+ * the renderer's removed 4k dropdown default, any non-trivial question could
100
+ * burn the whole budget reasoning and finish with empty content: no error, no
101
+ * tool calls, just a turn that "completes" with nothing to show for it (the
102
+ * empty-response bug). A user pointing the Custom OpenAI-compatible class
103
+ * straight at DeepSeek's API hits exactly this, so the request is sized by the
104
+ * same effort budget aegiscodex-dev uses for its own direct DeepSeek calls.
105
+ *
106
+ * desktop/renderer/budget.js carries the renderer's copy of these two
107
+ * constants plus budgetFor() below; test/budget.test.mjs requires both and
108
+ * asserts they agree, so the mirror cannot drift silently.
85
109
  */
86
110
  const DEEPSEEK_REASONING_MODEL_RE = /^deepseek-(v4(\.\d+)?-(flash|pro)|flash|pro|reasoner)$/;
87
111
  const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
88
112
 
113
+ /** The one class whose wire format REQUIRES a stated `max_tokens`: Anthropic's
114
+ * Messages API 400s without it, so that field is derived from the effort rung
115
+ * rather than invented by the transport (which is what a blanket
116
+ * `max_tokens: maxTokens || 4096` did — see providers.anthropicMessages). */
117
+ const REQUIRES_STATED_BUDGET = new Set(['anthropic']);
118
+
89
119
  /**
90
120
  * Idle-stream budget for a pooled brain call ("work autonomously"). The
91
121
  * generic watchdog in vendor/aegis.js kills a stream that goes 60s without a
@@ -109,38 +139,38 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
109
139
  const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
110
140
 
111
141
  /**
112
- * The budget a DeepSeek reasoning model runs on, resolved from EXACTLY ONE
113
- * authority per call.
114
- *
115
- * A caller-stated number IS the budget, and is returned verbatim. For the
116
- * non-pooled classes the renderer's max-tokens dropdown is the only budget
117
- * control on offer — updateBudgetControls hides the effort row for them — so
118
- * silently raising that number to an effort rung is precisely what made the
119
- * figure beside the dropdown untrustworthy. The old form was
120
- * `Math.max(stated, EFFORT_TOKEN_BUDGET[eff])`, which could only ever raise a
121
- * deliberate cap: a caller asking for 1024 ran on 32768, and the number the
122
- * UI displayed was never the number the call used.
142
+ * The budget a request travels with, resolved from EXACTLY ONE authority per
143
+ * call — the engine-side half of the rule desktop/renderer/budget.js mirrors
144
+ * and test/budget.test.mjs compares the two halves of, so neither can drift.
123
145
  *
124
- * The effort rung is the DEFAULT, consulted only when no number was stated at
125
- * all (the pooled class, which the renderer sends `effort` for and which the
126
- * server sizes itself). This is the same rule doubledBudget() follows for its
127
- * truncation retry: a stated cap is never overridden, by a rung or an order of
128
- * magnitude.
146
+ * 1. a caller-stated number IS the budget and is returned verbatim. It is a
147
+ * deliberate liability ceiling (aegis1 pass_budgets honours it downward:
148
+ * total = min(ladder, max_tokens x passes)) and no rung may raise it — the
149
+ * old `Math.max(stated, EFFORT_TOKEN_BUDGET[eff])` form did exactly that,
150
+ * so a caller asking for 1024 silently ran on 32768.
151
+ * 2. with nothing stated, a model that reasons against its own output budget
152
+ * (DeepSeek bills hidden chain-of-thought against the SAME budget as the
153
+ * answer) or a class whose wire format REQUIRES the field (Anthropic's
154
+ * Messages API) gets the Effort rung. This is why the renderer's
155
+ * max-tokens dropdown was removed rather than fixed: at its 4k default a
156
+ * reasoning model spent the entire budget thinking and finished empty —
157
+ * no error, no tool call, just a "completed" turn with nothing in it.
158
+ * 3. otherwise `undefined` — no `max_tokens` goes on the wire and the
159
+ * provider's own output limit governs. The transport used to fill this
160
+ * gap with an invented 4096 default, which is the defect in (2).
129
161
  *
130
- * Truncated and empty turns are handled where they belong — the doubled-budget
131
- * retry plus emptyTurnError — rather than by inflating the caller's ceiling up
132
- * front. Escalating on a demonstrated empty turn is strictly cheaper than
133
- * pre-emptively granting the top rung to every reasoning call.
134
- *
135
- * Everything else (non-DeepSeek models, non-reasoning DeepSeek ids like
136
- * deepseek-chat) passes through untouched.
162
+ * Truncated and empty turns are still handled where they belong — the
163
+ * doubled-budget retry plus emptyTurnError — rather than by inflating the
164
+ * caller's ceiling up front.
137
165
  */
138
- function reasoningBudget(model, maxTokens, effort) {
139
- if (!DEEPSEEK_REASONING_MODEL_RE.test(String(model || ''))) return maxTokens;
166
+ function reasoningBudget(cls, model, maxTokens, effort) {
140
167
  const stated = Number(maxTokens);
141
168
  if (Number.isFinite(stated) && stated > 0) return stated;
142
- const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
143
- return EFFORT_TOKEN_BUDGET[eff];
169
+ if (DEEPSEEK_REASONING_MODEL_RE.test(String(model || '')) || REQUIRES_STATED_BUDGET.has(cls)) {
170
+ const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
171
+ return EFFORT_TOKEN_BUDGET[eff];
172
+ }
173
+ return undefined;
144
174
  }
145
175
 
146
176
  /** Relay model entries arrive as ids or objects; keep only real model ids. */
@@ -279,6 +309,33 @@ function doubledBudget(maxTokens) {
279
309
  return n > 0 ? n * 2 : 8192;
280
310
  }
281
311
 
312
+ /**
313
+ * The cap the CALLER stated, or `undefined` meaning "none was stated".
314
+ *
315
+ * The two branches resolve one question — whose number is this? — and the
316
+ * first is the fix. An internal re-entry (runSubagent's nested chat()) passes
317
+ * `statedMaxTokens` explicitly, because the `maxTokens` it holds is the rung
318
+ * reasoningBudget() derived from `effort`: a number the server is about to
319
+ * derive for itself. Re-reading that value as the caller's own is what put a
320
+ * guessed 32768 on the wire as a *ceiling over* aegis1's ladder
321
+ * (pass_budgets: total = min(ladder, max_tokens x passes)) — so a
322
+ * high-effort subagent silently ran at 8192 and the ladder the UI advertised
323
+ * was not the budget the call used.
324
+ *
325
+ * A non-numeric or non-positive value means "nothing stated" — never a 0- or
326
+ * NaN-token ceiling.
327
+ */
328
+ function statedCapOf(payload) {
329
+ const asCap = (v) => {
330
+ const n = Number(v);
331
+ return Number.isFinite(n) && n > 0 ? n : undefined;
332
+ };
333
+ if (payload && Object.prototype.hasOwnProperty.call(payload, 'statedMaxTokens')) {
334
+ return asCap(payload.statedMaxTokens);
335
+ }
336
+ return asCap(payload && payload.maxTokens);
337
+ }
338
+
282
339
  /**
283
340
  * The follow-up shown to a model that ended its turn with neither text nor a
284
341
  * tool call. Sent as a plain user message (never as a tool result — there is
@@ -305,9 +362,9 @@ const EMPTY_TURN_NUDGE =
305
362
  function emptyTurnError({ cls, model, maxTokens, finishReason }) {
306
363
  const err = new Error(
307
364
  `The model returned no answer after ${cls}/${model} was asked to summarise its results ` +
308
- `(stop reason: ${finishReason || 'none'}, max_tokens: ${maxTokens}). The token budget ` +
309
- 'was most likely consumed before any visible text — raise the max-tokens setting, ' +
310
- 'or lower effort.'
365
+ `(stop reason: ${finishReason || 'none'}, max_tokens: ${maxTokens || 'unstated'}). The token budget ` +
366
+ 'was most likely consumed before any visible text — raise the Effort rung, which is ' +
367
+ 'what sizes this call, or ask for a smaller piece of work.'
311
368
  );
312
369
  err.status = 502;
313
370
  return err;
@@ -484,11 +541,44 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
484
541
  * runs straight through exactly like a session-allowed one, so "don't ask"
485
542
  * is one switch rather than a per-tool blanket allow in every conversation.
486
543
  */
544
+ /**
545
+ * The single place a tool executor is called from the chat loop, so the
546
+ * turn guard cannot be bypassed by reaching a different branch (there are
547
+ * four `return`s in gatedExecuteTool below; all of them come through here).
548
+ *
549
+ * exec is *checked* before running and writeFile/editFile are *recorded*
550
+ * after, and the asymmetry is the point: a refusal has to happen before the
551
+ * command runs, while "this path is mine" is only true once the write
552
+ * actually happened — recording a denied or preview-failed write would hide
553
+ * a genuinely foreign file behind a claim of ownership.
554
+ *
555
+ * A refusal is returned as an ordinary tool error, so the model reads it and
556
+ * adapts (name the paths, or don't do it) with no new IPC or renderer
557
+ * channel — the same reason it works identically in the GUI and the CLI.
558
+ */
559
+ async function guardedExecute(name, args, toolCtx) {
560
+ const guard = toolCtx && toolCtx.guard;
561
+ if (guard && name === 'exec') {
562
+ const verdict = blocksDestructive(guard, args && args.command);
563
+ if (!verdict.ok) return { ok: false, error: verdict.reason };
564
+ }
565
+ // Awaited, not merely returned: "this path is mine" is only true once the
566
+ // write actually landed, and `executeTool` is async — inspecting `.ok` on
567
+ // an un-awaited promise would read undefined and record every write,
568
+ // including the ones that failed. Ownership is load-bearing in the other
569
+ // direction too: a path wrongly claimed stops being reported as foreign.
570
+ const result = await T.executeTool(name, args, toolCtx);
571
+ if (guard && (name === 'writeFile' || name === 'editFile') && result && result.ok !== false) {
572
+ recordWrite(guard, args && args.path);
573
+ }
574
+ return result;
575
+ }
576
+
487
577
  async function gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal }) {
488
578
  const { name, args } = call;
489
- if (!T.MUTATING_TOOLS.has(name)) return T.executeTool(name, args, toolCtx);
490
- if (!confirmModeEnabled()) return T.executeTool(name, args, toolCtx);
491
- if (sessionAllows(rootSessionId, name)) return T.executeTool(name, args, toolCtx);
579
+ if (!T.MUTATING_TOOLS.has(name)) return guardedExecute(name, args, toolCtx);
580
+ if (!confirmModeEnabled()) return guardedExecute(name, args, toolCtx);
581
+ if (sessionAllows(rootSessionId, name)) return guardedExecute(name, args, toolCtx);
492
582
  // Already refused in this conversation: refuse again WITHOUT raising a
493
583
  // second card. Before this, the model's retry after a denial re-prompted
494
584
  // the user for the same tool — one "no" produced a card per round for up
@@ -521,8 +611,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
521
611
  }
522
612
  if (decision === 'session') allowForSession(rootSessionId, name);
523
613
 
524
- if (preview) return T.applyChecked(name, args, preview);
525
- return T.executeTool(name, args, toolCtx);
614
+ if (preview) {
615
+ const res = T.applyChecked(name, args, preview);
616
+ // Recorded on the same condition as guardedExecute's path: only a write
617
+ // that landed is this session's. (The previous form returned here
618
+ // unconditionally, making every line below it unreachable — the
619
+ // approved-write path recorded nothing at all.)
620
+ if (res && res.ok !== false && toolCtx && toolCtx.guard) {
621
+ recordWrite(toolCtx.guard, args && args.path);
622
+ }
623
+ return res;
624
+ }
625
+ return guardedExecute(name, args, toolCtx);
526
626
  }
527
627
 
528
628
  /**
@@ -634,8 +734,9 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
634
734
  // derived number states one decision twice — and the copies had already
635
735
  // drifted: 974adc5 doubled aegis1's ladder while this side stood still,
636
736
  // leaving the client capping below the budget it displayed. The cap was
637
- // never the one the renderer showed either (updateBudgetControls hides
638
- // the max-tokens dropdown for the pooled class). Omitting the field is
737
+ // never the one the renderer showed either (the Max tokens dropdown
738
+ // that used to claim a figure for the pooled class is gone — see
739
+ // desktop/renderer/budget.js). Omitting the field is
639
740
  // what tells aegis1 "no cap stated — let effort decide", the same
640
741
  // contract aegiscodex-dev sends.
641
742
  //
@@ -779,7 +880,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
779
880
  async function chat(payload, onDelta) {
780
881
  const cls = payload && payload.class;
781
882
  const model = payload && payload.model;
782
- const maxTokens = reasoningBudget(model, payload && payload.maxTokens, payload && payload.effort);
883
+ const maxTokens = reasoningBudget(cls, model, payload && payload.maxTokens, payload && payload.effort);
783
884
  // The caller's OWN number, kept apart from `maxTokens` above. That one
784
885
  // collapses two different facts into a single value — "the caller stated
785
886
  // 4096" and "effort implies 32768" — and the pooled path must treat them
@@ -788,7 +889,10 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
788
889
  // an effort-derived one is the server's own arithmetic stated twice, and
789
890
  // sending it is how the two copies came to disagree. So the pooled call
790
891
  // forwards only what the caller actually asked for.
791
- const statedMaxTokens = Number(payload && payload.maxTokens) > 0 ? Number(payload.maxTokens) : undefined;
892
+ // ...and it is read through statedCapOf(), so a nested chat() that states
893
+ // the key overrides its own derived `maxTokens` instead of being mistaken
894
+ // for a caller who asked for that number (see statedCapOf).
895
+ const statedMaxTokens = statedCapOf(payload);
792
896
  // "Work autonomously" — routes this call through aegis1's pool_brain
793
897
  // worker fan-out (services/pool_brain.py: N reasoning workers + a
794
898
  // synthesis pass) instead of a single provider call. UI-gated to the
@@ -1105,7 +1209,8 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1105
1209
  for (const call of calls) {
1106
1210
  const result = call.name === T.SUBAGENT_TOOL
1107
1211
  ? await runSubagent(call.args, {
1108
- cls, model, maxTokens, mode: payload && payload.mode, parentSignal: signal, depth, rootSessionId, rootOnDelta,
1212
+ cls, model, statedMaxTokens, effort: payload && payload.effort,
1213
+ mode: payload && payload.mode, parentSignal: signal, depth, rootSessionId, rootOnDelta,
1109
1214
  })
1110
1215
  : await gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal });
1111
1216
  // A subagent's spend rides back on its tool result (see runSubagent).
@@ -1135,7 +1240,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1135
1240
  */
1136
1241
  async function runSubagent(
1137
1242
  { description, subagent_type, prompt: subPrompt } = {},
1138
- { cls, model, maxTokens, mode, parentSignal, depth, rootSessionId, rootOnDelta } = {}
1243
+ { cls, model, statedMaxTokens, effort, mode, parentSignal, depth, rootSessionId, rootOnDelta } = {}
1139
1244
  ) {
1140
1245
  const task = String(subPrompt || description || '').trim();
1141
1246
  if (!task) return { ok: false, error: 'task requires a prompt' };
@@ -1159,8 +1264,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1159
1264
  // silently hanging behind this call's no-op onDelta below.
1160
1265
  const res = await chat(
1161
1266
  {
1162
- class: cls, model, maxTokens, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1,
1267
+ class: cls, model, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1,
1163
1268
  rootSessionId, rootOnDelta,
1269
+ // The parent's budget AUTHORITY is forwarded, not the number it
1270
+ // implies. `maxTokens` is deliberately absent: it is the rung this
1271
+ // turn derived from `effort`, and a re-entry that hands it back gets
1272
+ // read as the caller's own cap (statedCapOf). The rung itself rides
1273
+ // along as `effort`, which is what aegis1 sizes the fan-out and its
1274
+ // ladder from — so a subagent now runs at the effort the user chose
1275
+ // instead of at a number that only meant anything for this model id.
1276
+ statedMaxTokens,
1277
+ effort,
1164
1278
  },
1165
1279
  () => {}
1166
1280
  );
@@ -1199,4 +1313,4 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1199
1313
  };
1200
1314
  }
1201
1315
 
1202
- module.exports = { CLASSES, createLocalEngine, extractToolCalls, parseArgs };
1316
+ module.exports = { CLASSES, createLocalEngine, extractToolCalls, parseArgs, reasoningBudget };
@@ -0,0 +1,71 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Minimal git worktree introspection: where the repo root is, and what is
5
+ * dirty right now with a content hash per path.
6
+ *
7
+ * Ported from aegiscodex-dev/src/git-scope.js (ESM → CommonJS). Only the two
8
+ * functions `turn-guard.js` needs are carried over — `gitRoot` and
9
+ * `gitStatusSnapshot` (plus their `contentHash` helper). The source file's
10
+ * `foreignChanges` and `scopedCommit` belong to the autonomous queue's
11
+ * commit path, which the plugin does not host, so they are deliberately not
12
+ * vendored: a copy of code nothing calls is a copy that drifts unnoticed.
13
+ *
14
+ * The "dirty before AND byte-identical now" definition lives in
15
+ * `gitStatusSnapshot`'s hashes; `turn-guard.js` is the only consumer.
16
+ */
17
+
18
+ const fs = require('node:fs');
19
+ const path = require('node:path');
20
+ const crypto = require('node:crypto');
21
+ const { spawnSync } = require('node:child_process');
22
+
23
+ /** The repo root for `cwd`, or null if it isn't inside a git work tree. */
24
+ function gitRoot(cwd) {
25
+ const r = spawnSync('git', ['rev-parse', '--show-toplevel'], { cwd, encoding: 'utf8' });
26
+ return r.status === 0 ? r.stdout.trim() : null;
27
+ }
28
+
29
+ /**
30
+ * Hash a path's worktree content, so "already dirty before the task ran" can
31
+ * be told apart from "this task changed it". `absent` covers deletions;
32
+ * non-files (a submodule, a directory) compare by kind.
33
+ */
34
+ function contentHash(cwd, rel) {
35
+ try {
36
+ const st = fs.lstatSync(path.join(cwd, rel));
37
+ if (!st.isFile()) return st.isDirectory() ? 'nonfile:dir' : 'nonfile:other';
38
+ return crypto.createHash('sha1').update(fs.readFileSync(path.join(cwd, rel))).digest('hex');
39
+ } catch {
40
+ return 'absent';
41
+ }
42
+ }
43
+
44
+ /**
45
+ * Every path git reports as dirty right now, mapped to its content hash.
46
+ * `--no-renames` matters: in `-z` porcelain a rename emits the old name as a
47
+ * second NUL-terminated token, which we would otherwise hash as a change.
48
+ * Untracked files are included, since that is exactly how concurrent work
49
+ * usually arrives. Returns null if `cwd` is not a usable git repo.
50
+ */
51
+ function gitStatusSnapshot(cwd) {
52
+ // Run from the repo root: porcelain reports paths root-relative, so hashing
53
+ // must resolve them against the same directory.
54
+ const root = gitRoot(cwd);
55
+ if (!root) return null;
56
+ const r = spawnSync(
57
+ 'git',
58
+ ['status', '--porcelain=v1', '-z', '--untracked-files=all', '--no-renames'],
59
+ { cwd: root, encoding: 'utf8', maxBuffer: 64 * 1024 * 1024 },
60
+ );
61
+ if (r.status !== 0) return null;
62
+ const snap = new Map();
63
+ for (const rec of r.stdout.split('\0')) {
64
+ if (rec.length < 4) continue;
65
+ const p = rec.slice(3);
66
+ snap.set(p, contentHash(root, p));
67
+ }
68
+ return snap;
69
+ }
70
+
71
+ module.exports = { gitRoot, contentHash, gitStatusSnapshot };
@@ -49,7 +49,7 @@ async function chat({
49
49
  messages,
50
50
  system,
51
51
  prompt,
52
- maxTokens = 4096,
52
+ maxTokens,
53
53
  temperature,
54
54
  tools,
55
55
  toolChoice,
@@ -341,7 +341,7 @@ async function openaiCompatible({
341
341
  messages,
342
342
  system,
343
343
  prompt,
344
- maxTokens = 4096,
344
+ maxTokens,
345
345
  temperature,
346
346
  tools,
347
347
  toolChoice,
@@ -356,9 +356,20 @@ async function openaiCompatible({
356
356
  const body = {
357
357
  model: modelId,
358
358
  messages: openAIMessages(messages, system, prompt),
359
- max_tokens: maxTokens || 4096,
360
359
  stream: true,
361
360
  };
361
+ // `max_tokens` travels only when the caller actually stated one. Nothing is
362
+ // invented here — not even as a signature default (this parameter used to
363
+ // read `maxTokens = 4096`, so the field appeared even when engine.js passed
364
+ // undefined on purpose). The output length of a reasoning model is not
365
+ // predictable from the request: DeepSeek bills hidden chain-of-thought
366
+ // against the same budget, so a client-chosen 4096 never bounded the answer,
367
+ // it bounded the thinking and truncated turns that would otherwise have
368
+ // finished. engine.js's reasoningBudget() supplies a value when the /effort
369
+ // rung is the only lever the provider exposes; for everything else the
370
+ // provider's own default is the number it was tuned for, and omitting the
371
+ // key is how you ask for it.
372
+ if (Number(maxTokens) > 0) body.max_tokens = Number(maxTokens);
362
373
  if (temperature != null) body.temperature = temperature;
363
374
  // Only advertise tools when the caller passes a non-empty list — an empty
364
375
  // `tools: []` is a 400 on some gateways.
@@ -485,7 +496,7 @@ async function anthropicMessages({
485
496
  messages,
486
497
  system,
487
498
  prompt,
488
- maxTokens = 4096,
499
+ maxTokens,
489
500
  temperature,
490
501
  tools,
491
502
  toolChoice,
@@ -505,10 +516,24 @@ async function anthropicMessages({
505
516
 
506
517
  const body = {
507
518
  model: modelId,
508
- max_tokens: maxTokens || 4096,
509
519
  stream: true,
510
520
  messages: buildAnthropicMessages(messages, prompt),
511
521
  };
522
+ // State the field only when the caller actually gave a number — the same rule
523
+ // the OpenAI-compatible transport below follows, and for the same reason.
524
+ //
525
+ // There is no client-side floor or ceiling constant here, deliberately. The
526
+ // Messages API does require `max_tokens`, but a number this side can only
527
+ // guess at is exactly what used to truncate reasoning turns: the budget is
528
+ // spent on hidden chain-of-thought before the first visible token, so the
529
+ // request "completed" having answered nothing. The size of a call belongs to
530
+ // whoever owns the model — engine.js's reasoningBudget() (the /effort rung,
531
+ // which is what the caller is passing here) or the platform itself, which
532
+ // sizes the pooled class from `effort` server-side and reads a body
533
+ // max_tokens as a ceiling OVER that ladder. So an unstated budget goes out
534
+ // unstated and the platform's value applies; this transport does not
535
+ // substitute one of its own.
536
+ if (Number(maxTokens) > 0) body.max_tokens = Number(maxTokens);
512
537
  if (system) body.system = system;
513
538
  if (temperature != null) body.temperature = temperature;
514
539
  // Anthropic wants {name, description, input_schema} — an OpenAI-shaped list