@nexrall/code-core 1.4.65 → 1.4.67

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/README.md +2 -2
  2. package/dist/agent/agentTypes.d.ts +8 -4
  3. package/dist/agent/agentTypes.d.ts.map +1 -1
  4. package/dist/agent/agentTypes.js +20 -147
  5. package/dist/agent/askOnce.d.ts +12 -0
  6. package/dist/agent/askOnce.d.ts.map +1 -0
  7. package/dist/agent/askOnce.js +41 -0
  8. package/dist/agent/compaction.d.ts +244 -0
  9. package/dist/agent/compaction.d.ts.map +1 -0
  10. package/dist/agent/compaction.js +976 -0
  11. package/dist/agent/fileLocks.d.ts +34 -0
  12. package/dist/agent/fileLocks.d.ts.map +1 -0
  13. package/dist/agent/fileLocks.js +114 -0
  14. package/dist/agent/hooks.d.ts +324 -0
  15. package/dist/agent/hooks.d.ts.map +1 -0
  16. package/dist/agent/hooks.js +1228 -0
  17. package/dist/agent/iterationPolicy.d.ts +121 -0
  18. package/dist/agent/iterationPolicy.d.ts.map +1 -0
  19. package/dist/agent/iterationPolicy.js +297 -0
  20. package/dist/agent/lifecycleHost.d.ts +55 -0
  21. package/dist/agent/lifecycleHost.d.ts.map +1 -0
  22. package/dist/agent/lifecycleHost.js +294 -0
  23. package/dist/agent/loop.d.ts +11 -491
  24. package/dist/agent/loop.d.ts.map +1 -1
  25. package/dist/agent/loop.js +418 -3032
  26. package/dist/agent/memory.d.ts +11 -0
  27. package/dist/agent/memory.d.ts.map +1 -1
  28. package/dist/agent/memory.js +23 -3
  29. package/dist/agent/planMode.d.ts +14 -0
  30. package/dist/agent/planMode.d.ts.map +1 -1
  31. package/dist/agent/planMode.js +146 -17
  32. package/dist/agent/securityLint.js +2 -2
  33. package/dist/agent/sharedTasks.d.ts +7 -0
  34. package/dist/agent/sharedTasks.d.ts.map +1 -1
  35. package/dist/agent/sharedTasks.js +16 -0
  36. package/dist/agent/subAgentBudget.d.ts +65 -0
  37. package/dist/agent/subAgentBudget.d.ts.map +1 -0
  38. package/dist/agent/subAgentBudget.js +269 -0
  39. package/dist/agent/subTask.d.ts +6 -0
  40. package/dist/agent/subTask.d.ts.map +1 -0
  41. package/dist/agent/subTask.js +713 -0
  42. package/dist/agent/subTaskSupport.d.ts +156 -0
  43. package/dist/agent/subTaskSupport.d.ts.map +1 -0
  44. package/dist/agent/subTaskSupport.js +409 -0
  45. package/dist/agent/toolDescriptions.d.ts +3 -0
  46. package/dist/agent/toolDescriptions.d.ts.map +1 -0
  47. package/dist/agent/toolDescriptions.js +116 -0
  48. package/dist/api/client.d.ts +1 -1
  49. package/dist/api/client.d.ts.map +1 -1
  50. package/dist/api/client.js +17 -0
  51. package/dist/index.d.ts +3 -0
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +3 -0
  54. package/dist/mcp/client.d.ts +104 -0
  55. package/dist/mcp/client.d.ts.map +1 -1
  56. package/dist/mcp/client.js +136 -2
  57. package/dist/mcp/httpClient.d.ts +17 -1
  58. package/dist/mcp/httpClient.d.ts.map +1 -1
  59. package/dist/mcp/httpClient.js +120 -19
  60. package/dist/mcp/manager.d.ts +77 -2
  61. package/dist/mcp/manager.d.ts.map +1 -1
  62. package/dist/mcp/manager.js +275 -9
  63. package/dist/mcp/sseClient.d.ts +7 -1
  64. package/dist/mcp/sseClient.d.ts.map +1 -1
  65. package/dist/mcp/sseClient.js +45 -1
  66. package/dist/mcp/stats.d.ts +41 -0
  67. package/dist/mcp/stats.d.ts.map +1 -0
  68. package/dist/mcp/stats.js +108 -0
  69. package/dist/permissions/destructive.d.ts +11 -7
  70. package/dist/permissions/destructive.d.ts.map +1 -1
  71. package/dist/permissions/destructive.js +59 -8
  72. package/dist/permissions/destructiveTokens.d.ts +34 -0
  73. package/dist/permissions/destructiveTokens.d.ts.map +1 -0
  74. package/dist/permissions/destructiveTokens.js +475 -0
  75. package/dist/permissions/modePolicy.d.ts +12 -8
  76. package/dist/permissions/modePolicy.d.ts.map +1 -1
  77. package/dist/permissions/modePolicy.js +19 -12
  78. package/dist/permissions/rules.d.ts +4 -1
  79. package/dist/permissions/rules.d.ts.map +1 -1
  80. package/dist/permissions/rules.js +29 -0
  81. package/dist/types.d.ts +47 -3
  82. package/dist/types.d.ts.map +1 -1
  83. package/package.json +8 -7
@@ -0,0 +1,976 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.CompactionUnavailableError = exports.PRUNE_KEEP_RECENT = exports.VERIFY_CMD_RE = exports.WRITE_TOOL_NAMES = exports.MAX_BODY_BYTES = exports.COMPACT_MAX_FAILURES = exports.COMPACT_MIN_RECLAIM_BYTES = exports.COMPACT_KEEP_MIN = exports._lastApiCallEndedAt = exports.PRUNE_MIN_RECLAIM_BYTES = exports.MODEL_CONTEXT_TOKENS = void 0;
4
+ exports.contextWindowFor = contextWindowFor;
5
+ exports.compactionLimits = compactionLimits;
6
+ exports.compactionThresholds = compactionThresholds;
7
+ exports.pruneReclaimFloor = pruneReclaimFloor;
8
+ exports.estimateBodyBytes = estimateBodyBytes;
9
+ exports.resolveAutoCompact = resolveAutoCompact;
10
+ exports.resolveVerificationNudge = resolveVerificationNudge;
11
+ exports.allowsTestOnlyWrite = allowsTestOnlyWrite;
12
+ exports.findSafeCutIndex = findSafeCutIndex;
13
+ exports.persistRuntimeContext = persistRuntimeContext;
14
+ exports.transcriptOf = transcriptOf;
15
+ exports.createLedger = createLedger;
16
+ exports.ledgerRecord = ledgerRecord;
17
+ exports.ledgerSummary = ledgerSummary;
18
+ exports.pruneOldToolResults = pruneOldToolResults;
19
+ exports.makeCachedSummarizer = makeCachedSummarizer;
20
+ exports.autoCompactMessages = autoCompactMessages;
21
+ exports.maybeCompactMemory = maybeCompactMemory;
22
+ exports.estimateTokensRough = estimateTokensRough;
23
+ exports.compactMessagesForResume = compactMessagesForResume;
24
+ const types_1 = require("../types");
25
+ const testIntegrity_1 = require("./testIntegrity");
26
+ const flaky_1 = require("./flaky");
27
+ const client_1 = require("../api/client");
28
+ const rules_1 = require("../permissions/rules");
29
+ const memory_1 = require("./memory");
30
+ const modelCatalogue_1 = require("./modelCatalogue");
31
+ const safeSlice_1 = require("../util/safeSlice");
32
+ const hooks_1 = require("./hooks");
33
+ // ─── Auto-compact ─────────────────────────────────────────────────────────────
34
+ //
35
+ // When the conversation's prompt size approaches the model's context window,
36
+ // summarise the older portion automatically (Claude-Code style) instead of
37
+ // letting the request fail or forcing the user to run /compact by hand.
38
+ //
39
+ // Compaction only happens at a turn boundary (top of the loop, before the next
40
+ // streamChat) and only cuts at a "safe" user message — one with no tool_result
41
+ // blocks — so tool_use/tool_result pairing is never broken.
42
+ // Context window per model, keyed by BOTH the tier aliases and the real model
43
+ // ids the picker now sends.
44
+ //
45
+ // This table drives auto-compaction, so a wrong number is expensive in one
46
+ // direction and merely wasteful in the other: too small compacts early and
47
+ // summarises lossily; too LARGE means compaction never fires before the real
48
+ // ceiling and the turn dies on a provider 400 — mid-conversation, on exactly
49
+ // the long sessions compaction exists to protect.
50
+ //
51
+ // It previously held only turbo/pro/ultra, all at 1M, which was correct while
52
+ // every tier was a 1M-context Claude. With several providers selectable by
53
+ // name that assumption breaks hard: gpt-4o-mini is 128K, i.e. 8x smaller than
54
+ // the value this table would have guessed for it.
55
+ //
56
+ // The Anthropic figures mirror the backend registry
57
+ // (services/providers/modelRegistry.js). The OpenAI figures were MEASURED
58
+ // against the live API rather than read from docs — note that gpt-5.4 is
59
+ // 922_000, not a round 1M, and gpt-4.1 is 1_047_576.
60
+ exports.MODEL_CONTEXT_TOKENS = {
61
+ // Tier aliases — still sent by older clients, saved settings and sub-agent
62
+ // frontmatter, so they must keep resolving.
63
+ turbo: 1000000,
64
+ pro: 1000000,
65
+ ultra: 1000000,
66
+ fast: 200000,
67
+ // Anthropic, by real model id.
68
+ 'claude-sonnet-5-5': 1000000,
69
+ 'claude-opus-5-5': 1000000,
70
+ 'claude-fable-5-1': 1000000,
71
+ 'claude-haiku-4-5-20251001': 200000,
72
+ // OpenAI, by real model id (measured).
73
+ 'gpt-5.4': 922000,
74
+ 'gpt-5.4-mini': 272000,
75
+ 'gpt-4.1': 1047576,
76
+ 'gpt-4o-mini': 128000,
77
+ // GPT-5.6 family: MEASURED 2026-09-03 via a 400 (same technique as
78
+ // gpt-5.4's 922_000 above) — and it is the EXACT SAME 922,000-token
79
+ // ceiling on all three sizes. This CONTRADICTS the publicly documented
80
+ // figure (openai.com/index/gpt-5-6 + OpenRouter's model card both
81
+ // advertise 1,050,000) — see backend services/providers/modelRegistry.js's
82
+ // own comment on these rows for the measured 400 body. Guessing 1.05M here
83
+ // would fire auto-compaction ~12% past the real wall.
84
+ 'gpt-5.6-sol': 922000,
85
+ 'gpt-5.6-terra': 922000,
86
+ 'gpt-5.6-luna': 922000,
87
+ // DeepSeek, by real model id (documented — see backend/services/providers/
88
+ // modelRegistry.js's own TODO(unverified-by-400): DeepSeek accepts an
89
+ // oversized max_completion_tokens without rejecting it, so there was no 400
90
+ // to measure the ceiling from the way the OpenAI rows above were).
91
+ 'deepseek-flash': 1048576,
92
+ // Qwen (DashScope), by real model id (documented max input, same caveat).
93
+ 'qwen3.7-max': 991800,
94
+ // Z.ai (GLM), by real model id. contextWindow is documented (Z.ai/
95
+ // Cloudflare Workers AI model cards, both list 1,048,576) rather than
96
+ // measured — a ~400K-token request was ACCEPTED (200), not rejected, so
97
+ // there was no 400 to read a real ceiling out of. maxOutputTokens IS
98
+ // measured: `max_tokens: 999999` was rejected with the ceiling in the
99
+ // error body (backend services/providers/modelRegistry.js's glm-5.3 row
100
+ // has the full verification notes).
101
+ 'glm-5.3': 1048576,
102
+ };
103
+ /**
104
+ * Context window (tokens) for a model alias or real model id.
105
+ *
106
+ * Checks the LIVE catalogue (GET /api/code/models, modelCatalogue.ts) first —
107
+ * populated once per process by whichever client fetched it (CLI at session
108
+ * start, VS Code via _postModelCatalogue, desktop via its main-process
109
+ * fetch) — falling back to this hard-coded table when no live data exists yet
110
+ * (offline, older backend, or the catalogue simply hasn't been fetched by
111
+ * this call site). This is what lets a model added to the backend registry
112
+ * (services/providers/modelRegistry.js) get the CORRECT context window here
113
+ * even before this table is updated by hand for a new nexrall-code release —
114
+ * exactly the class of bug GPT-5.6's 922K-vs-1.05M mismatch was (see
115
+ * modelRegistry.js's own comment on that row).
116
+ *
117
+ * The static-table fallback is deliberately the SMALLEST window in the table
118
+ * rather than the largest. An unknown model is most likely a newly added one
119
+ * this client build predates, and guessing high is the failure that cannot be
120
+ * recovered from: the turn hits a provider 400 with no chance to compact.
121
+ * Guessing low only costs an earlier, lossy compaction — annoying, not broken.
122
+ */
123
+ function contextWindowFor(model) {
124
+ const fallback = exports.MODEL_CONTEXT_TOKENS[model ?? 'turbo'] ?? 128000;
125
+ return (0, modelCatalogue_1.liveContextWindowFor)(model, fallback);
126
+ }
127
+ // ── Compaction thresholds (cost control) ─────────────────────────────────────
128
+ // Two independent triggers, deliberately at DIFFERENT levels:
129
+ //
130
+ // • PRUNE threshold (cheap, lossy-but-structure-preserving, NO model call):
131
+ // fires EARLY. Every turn a large history is resent, cache-read alone
132
+ // (0.10× input) is still billed on the whole prefix — on a 700K-token
133
+ // session that is real money accruing per turn long before the 1M wall.
134
+ // Anthropic's own server-side compaction defaults its trigger to 150K
135
+ // input tokens (docs: compact_20260112 default trigger 150000). We mirror
136
+ // that intent: start shedding already-consumed tool_result bulk at ~120K
137
+ // tokens (see compactionLimits) so the per-turn cache-read bill stops growing,
138
+ // WITHOUT paying for a summariser model call and WITHOUT dropping any turn
139
+ // (pruneOldToolResults keeps every tool_use/tool_result pair intact).
140
+ //
141
+ // • SUMMARISE threshold (a model call, lossy: drops whole turns): Claude Code's
142
+ // ~167K line (see compactionLimits). Later than prune, because summarise-of-summarise is the main cause of an
143
+ // agent "forgetting" earlier work. Only when cheap pruning can't keep the
144
+ // prompt under this line do we fall through to summarisation.
145
+ //
146
+ // Both are overridable via env for power users / tests.
147
+ function envFraction(name, fallback) {
148
+ const v = Number(process.env[name]);
149
+ return Number.isFinite(v) && v > 0 && v < 1 ? v : fallback;
150
+ }
151
+ // Where auto-compaction fires, in TOKENS — Claude Code's rule, not a fraction of a 1M window.
152
+ //
153
+ // Claude Code summarises at `window − min(maxOutput, 20K) − 13K buffer`, on a 200K window
154
+ // (~167K tokens). We used to wait for 80% of a 1M window (~800K): every request of a long
155
+ // run re-read that whole prefix from cache, so a session cost ~5× more per request than
156
+ // the same work in Claude Code long before anything was shed. The window used for this is
157
+ // capped (default 200K, like Claude Code) — the model's real 1M window still bounds what
158
+ // the backend will accept; this only decides when WE tidy up.
159
+ //
160
+ // NEXRALL_COMPACT_WINDOW / settings.json "autoCompactWindow": the cap (tokens). Set it
161
+ // to e.g. 1000000 to use the whole window before compacting.
162
+ // NEXRALL_PRUNE_THRESHOLD / NEXRALL_COMPACT_THRESHOLD: fractions of that capped window.
163
+ const DEFAULT_COMPACT_WINDOW = 200000;
164
+ const COMPACT_OUTPUT_RESERVE = 20000; // Claude Code: min(model max output, 20K)
165
+ const COMPACT_BUFFER_TOKENS = 13000; // Claude Code's autocompact buffer
166
+ const DEFAULT_PRUNE_FRACTION = 0.6; // cheap lossless prune ~120K, before the ~167K summarise
167
+ function compactWindowCap(settingsRaw) {
168
+ const fromEnv = Number(process.env.NEXRALL_COMPACT_WINDOW);
169
+ if (Number.isFinite(fromEnv) && fromEnv >= 50000)
170
+ return fromEnv;
171
+ const fromSettings = Number(settingsRaw?.autoCompactWindow);
172
+ if (Number.isFinite(fromSettings) && fromSettings >= 50000)
173
+ return fromSettings;
174
+ return DEFAULT_COMPACT_WINDOW;
175
+ }
176
+ /** Token counts at which auto-prune and auto-compact (summarise) fire for a model window. */
177
+ function compactionLimits(contextWindow, settingsRaw) {
178
+ const eff = Math.min(contextWindow, compactWindowCap(settingsRaw));
179
+ const compactFrac = envFraction('NEXRALL_COMPACT_THRESHOLD', 0);
180
+ const pruneFrac = envFraction('NEXRALL_PRUNE_THRESHOLD', 0);
181
+ const compact = compactFrac
182
+ ? eff * compactFrac
183
+ : Math.max(eff * 0.5, eff - Math.min(COMPACT_OUTPUT_RESERVE, eff * 0.1) - COMPACT_BUFFER_TOKENS);
184
+ const prune = Math.min(pruneFrac ? eff * pruneFrac : eff * DEFAULT_PRUNE_FRACTION, compact * 0.9);
185
+ return { prune: Math.floor(prune), compact: Math.floor(compact) };
186
+ }
187
+ /** Auto-prune / auto-compact thresholds as fractions of the context window — for UI display (e.g. `/context`). */
188
+ function compactionThresholds(contextWindow = 1000000, settingsRaw) {
189
+ const { prune, compact } = compactionLimits(contextWindow, settingsRaw);
190
+ return { prune: prune / contextWindow, compact: compact / contextWindow };
191
+ }
192
+ // Only bother pruning if it reclaims a meaningful amount — a tiny prune busts
193
+ // the message-level prompt cache (the pruned prefix changes) for little gain,
194
+ // so we require at least this many bytes reclaimed before accepting a prune.
195
+ // Sized for the ~120K-token prune line: at 256 KB a prune could rarely reclaim enough to
196
+ // qualify, so sessions skipped straight to the lossy summariser.
197
+ exports.PRUNE_MIN_RECLAIM_BYTES = 96 * 1024; // 96 KB
198
+ // Cache-aware prune floor. The 96 KB floor exists ONLY to avoid busting a WARM prompt
199
+ // cache for a small gain. Once the session has been idle past the provider cache TTL
200
+ // (Anthropic/OpenAI: 5 min), the whole prefix is re-written on the next request anyway,
201
+ // so a prune at that moment costs nothing extra — accept a much smaller reclaim then.
202
+ // Keyed by sessionId (same scheme as _subAgentBudgets) because runAgentLoop runs once
203
+ // per user turn and the idle gap that matters is BETWEEN turns.
204
+ const PRUNE_MIN_RECLAIM_BYTES_COLD = 16 * 1024; // 16 KB
205
+ const CACHE_COLD_AFTER_MS = 5 * 60000;
206
+ exports._lastApiCallEndedAt = new Map();
207
+ function pruneReclaimFloor(lastCallEndedAt, now = Date.now()) {
208
+ return lastCallEndedAt !== undefined && now - lastCallEndedAt >= CACHE_COLD_AFTER_MS
209
+ ? PRUNE_MIN_RECLAIM_BYTES_COLD
210
+ : exports.PRUNE_MIN_RECLAIM_BYTES;
211
+ }
212
+ exports.COMPACT_KEEP_MIN = 6; // always keep at least the last N messages verbatim
213
+ /**
214
+ * Bytes a compaction must reclaim to count as productive.
215
+ *
216
+ * Deliberately much smaller than PRUNE_MIN_RECLAIM_BYTES: a prune declines when the
217
+ * gain isn't worth busting the prompt cache, whereas by the time we are summarising
218
+ * we are already committed to rewriting the prefix — the only question is whether the
219
+ * summariser is making ANY headway. 32 KB is small enough that a genuinely useful
220
+ * compaction always clears it, large enough that shuffling a few bytes doesn't.
221
+ */
222
+ exports.COMPACT_MIN_RECLAIM_BYTES = 32 * 1024; // 32 KB
223
+ /**
224
+ * Consecutive non-productive compaction attempts before auto-compaction is switched
225
+ * off for the rest of the run.
226
+ *
227
+ * 3 rather than 1 because the failure is often transient — a summariser stream that
228
+ * blipped will usually succeed on the next turn, and giving up instantly would lose
229
+ * the safety net for a whole long session over one network hiccup. 3 also bounds the
230
+ * wasted spend: at most three summariser calls, not hundreds.
231
+ */
232
+ exports.COMPACT_MAX_FAILURES = 3;
233
+ // Byte-level safety net, independent of the token estimate.
234
+ //
235
+ // Tool-heavy sessions on large codebases accumulate many tool_result blocks
236
+ // (read_file / bash / search output). The token count can still look "under
237
+ // budget" while the SERIALISED body has grown to tens of MB — the char↔token
238
+ // ratio for JSON/code/logs is highly variable, so a token threshold alone does
239
+ // NOT bound the request body size. The backend rejects bodies over its limit
240
+ // (413), which the token-based compactor never anticipates because:
241
+ // • it reacts to lastPromptTokens from the PREVIOUS turn's usage event, so on
242
+ // a freshly-resumed (already-large) session it is 0 and never fires, and
243
+ // • 80% × 1M tokens of tool_result can be 25–45 MB — far past any body limit.
244
+ // This guard measures the ACTUAL body bytes before each send and forces a
245
+ // compaction whenever it crosses the threshold, regardless of the token count.
246
+ // Kept comfortably under the server's 25 MB /api/code limit.
247
+ exports.MAX_BODY_BYTES = 8 * 1024 * 1024; // 8 MB
248
+ /** Approximate serialised request-body size (bytes) for the messages array. */
249
+ function estimateBodyBytes(messages) {
250
+ try {
251
+ return Buffer.byteLength(JSON.stringify(messages), 'utf-8');
252
+ }
253
+ catch {
254
+ return 0; // circular/unserialisable — don't block on the estimate
255
+ }
256
+ }
257
+ function resolveAutoCompact(fromOptions, rawSettings) {
258
+ if (typeof fromOptions === 'boolean')
259
+ return fromOptions;
260
+ const env = (process.env.NEXRALL_AUTO_COMPACT ?? '').toLowerCase();
261
+ if (env === '0' || env === 'false' || env === 'off')
262
+ return false;
263
+ if (env === '1' || env === 'true' || env === 'on')
264
+ return true;
265
+ const s = rawSettings?.autoCompact;
266
+ if (typeof s === 'boolean')
267
+ return s;
268
+ return true;
269
+ }
270
+ /** Opt-out for the one-shot verification nudge (GAP D). Defaults to on. */
271
+ function resolveVerificationNudge(rawSettings) {
272
+ const env = (process.env.NEXRALL_VERIFY_NUDGE ?? '').toLowerCase();
273
+ if (env === '0' || env === 'false' || env === 'off')
274
+ return false;
275
+ if (env === '1' || env === 'true' || env === 'on')
276
+ return true;
277
+ const s = rawSettings?.verifyNudge;
278
+ if (typeof s === 'boolean')
279
+ return s;
280
+ return true;
281
+ }
282
+ /** Tools that mutate the filesystem — used by the verification nudge (GAP D). */
283
+ exports.WRITE_TOOL_NAMES = new Set(['write_file', 'edit_file', 'multi_edit', 'delete_file', 'move_file', 'copy_file', 'notebook_edit']);
284
+ /**
285
+ * May an agent restricted to `testFilesOnly` perform this tool call?
286
+ *
287
+ * A tool allowlist is all-or-nothing per tool: granting `edit_file` grants it for
288
+ * every path in the repo. A user-defined test-writer agent needs write access to produce
289
+ * tests, but must NOT be able to "fix" production source so a failing test goes
290
+ * green — the single most common way a test-writing agent destroys the signal it
291
+ * was asked to create. Its prompt says so; this makes it a refusal rather than a
292
+ * request.
293
+ *
294
+ * Pure + exported so the rules are testable directly, without running a real
295
+ * sub-agent.
296
+ *
297
+ * KNOWN LIMIT, stated rather than hidden: this gates the file TOOLS, not `bash`.
298
+ * A determined model could still write source via `bash: echo ... > src/x.ts`.
299
+ * Closing that means parsing shell redirection, which is not reliably doable — so
300
+ * this is a strong guardrail against the realistic failure mode, not a sandbox.
301
+ * Real isolation is the sandbox config (tools/sandbox.ts), a separate mechanism.
302
+ */
303
+ function allowsTestOnlyWrite(tool, input) {
304
+ // Non-write tools are unaffected: reading, searching and running tests are all
305
+ // essential to writing a test.
306
+ //
307
+ // WRITE_TOOL_NAMES deliberately excludes `create_directory`: isTestFile matches
308
+ // FILE paths, so a legitimate `create_directory('test/helpers')` would be
309
+ // refused and the agent could not scaffold the tree it needs — while an empty
310
+ // directory cannot damage production code, and files placed in it are still
311
+ // checked individually.
312
+ if (!exports.WRITE_TOOL_NAMES.has(tool))
313
+ return true;
314
+ // EVERY path the call could affect must be a test file, not just `path`:
315
+ // move_file takes {source, dest} and copy_file {source, destination}, so
316
+ // checking `path` alone would let `move_file src/index.ts -> /tmp/x` through and
317
+ // remove production code by relocating it.
318
+ //
319
+ // `source` is skipped for notebook_edit specifically, where it is the CELL
320
+ // CONTENT rather than a path — treating a blob of code as a path would refuse
321
+ // every legitimate notebook edit.
322
+ const pathKeys = tool === 'notebook_edit'
323
+ ? ['path']
324
+ : ['path', 'source', 'dest', 'destination'];
325
+ const candidates = pathKeys
326
+ .map((k) => input?.[k])
327
+ .filter((v) => typeof v === 'string' && v.length > 0);
328
+ // An unrecognised write shape (no path-like argument at all) is refused rather
329
+ // than allowed through, so a future tool cannot silently become a hole here.
330
+ if (candidates.length === 0)
331
+ return false;
332
+ return candidates.every((p) => (0, testIntegrity_1.isTestFile)(p));
333
+ }
334
+ /** Heuristic: does a bash command look like it's running tests/build/lint/typecheck? (GAP D) */
335
+ exports.VERIFY_CMD_RE = /\b(npm|yarn|pnpm)\s+(run\s+)?(test|build|lint|typecheck|tsc)\b|\bpytest\b|\bgo\s+(test|vet|build)\b|\btsc\b|\beslint\b|\bcargo\s+(test|build|check)\b/i;
336
+ /**
337
+ * Find the latest index ≤ maxIdx where history can be cut safely.
338
+ *
339
+ * A safe cut point is a **turn boundary**: an assistant message (which always
340
+ * begins a fresh turn after a user message). Cutting there guarantees that
341
+ * `messages[cut..]` starts with an assistant whose `tool_use` blocks are all
342
+ * answered by `tool_result`s that remain in the kept slice — so we never orphan
343
+ * a tool_result (which the API rejects). We deliberately allow cutting across
344
+ * tool_result-bearing user messages: the OLD implementation only cut at a
345
+ * *non*-tool_result user message, which never exists inside a single long
346
+ * agentic run (every user turn is a tool_result), so compaction was a no-op
347
+ * exactly when a long task needs it most.
348
+ */
349
+ function findSafeCutIndex(messages, maxIdx) {
350
+ for (let i = Math.min(maxIdx, messages.length - 1); i >= 2; i--) {
351
+ if (messages[i].role === 'assistant')
352
+ return i;
353
+ }
354
+ return -1;
355
+ }
356
+ // Hard ceiling on the transcript we hand to the summariser. Per-block truncation
357
+ // alone does NOT bound the total: a very long run has thousands of blocks, so the
358
+ // concatenated transcript can itself exceed the summariser call's context window →
359
+ // the summarise request 400s → autoCompactMessages returns false → NO compaction
360
+ // happens exactly when the session is largest (the context-wall failure mode).
361
+ // ~600K chars ≈ 150K tokens, well under a 1M window even with prompt overhead.
362
+ const MAX_TRANSCRIPT_CHARS = 600000;
363
+ /**
364
+ * Render messages to a plain-text transcript for the summariser (tool noise
365
+ * truncated per-block AND the whole transcript hard-capped). When the transcript
366
+ * would exceed MAX_TRANSCRIPT_CHARS we keep the HEAD (original task + early
367
+ * decisions) and the TAIL (most-recent, highest-signal context) and drop the
368
+ * middle — a middle-out elision that preserves both "what we set out to do" and
369
+ * "where we are now", which is what the continuation summary needs most.
370
+ */
371
+ /**
372
+ * Appends the backend-announced runtime-context block to the user message it was
373
+ * attached to. No-op if that message is not a user turn or already ends with the
374
+ * identical block. Exported for tests.
375
+ */
376
+ function persistRuntimeContext(messages, idx, text) {
377
+ const m = messages[idx];
378
+ if (!m || m.role !== 'user' || !Array.isArray(m.content))
379
+ return false;
380
+ const lastBlock = m.content[m.content.length - 1];
381
+ if (lastBlock && lastBlock.type === 'text' && lastBlock.text === text)
382
+ return false;
383
+ messages[idx] = { ...m, content: [...m.content, { type: 'text', text }] };
384
+ return true;
385
+ }
386
+ function transcriptOf(messages) {
387
+ const parts = [];
388
+ for (const m of messages) {
389
+ for (const b of m.content) {
390
+ if ((0, types_1.isRuntimeContextBlock)(b))
391
+ continue; // editor scaffolding, not conversation
392
+ if (b.type === 'text' && b.text) {
393
+ parts.push(`${m.role.toUpperCase()}: ${(0, safeSlice_1.sliceSafeEnd)(b.text, 2000)}`);
394
+ }
395
+ else if (b.type === 'tool_use') {
396
+ parts.push(`${m.role.toUpperCase()} [tool: ${b.name}]: ${(0, safeSlice_1.sliceSafeEnd)(JSON.stringify(b.input ?? {}), 400)}`);
397
+ }
398
+ else if (b.type === 'tool_result') {
399
+ parts.push(`TOOL RESULT: ${(0, safeSlice_1.sliceSafeEnd)(String(b.content ?? ''), 600)}`);
400
+ }
401
+ }
402
+ }
403
+ const full = parts.join('\n');
404
+ if (full.length <= MAX_TRANSCRIPT_CHARS)
405
+ return full;
406
+ // Middle-out: keep 40% head, 60% tail (recent context is higher-signal for
407
+ // continuation). Slice on line boundaries so we don't cut a line in half.
408
+ const headBudget = Math.floor(MAX_TRANSCRIPT_CHARS * 0.4);
409
+ const tailBudget = MAX_TRANSCRIPT_CHARS - headBudget;
410
+ const head = (0, safeSlice_1.sliceSafeEnd)(full, headBudget);
411
+ const tail = (0, safeSlice_1.sliceSafeStart)(full, full.length - tailBudget);
412
+ const dropped = full.length - head.length - tail.length;
413
+ return `${head}\n\n[… ${dropped} chars of mid-session transcript elided to fit the summariser's context window …]\n\n${tail}`;
414
+ }
415
+ // ─── Structured progress ledger (GAP E) ────────────────────────────────────────
416
+ //
417
+ // The single biggest long-horizon failure mode (industry-wide "context rot") is
418
+ // that each auto-compaction summarises a transcript that ALREADY contains a prior
419
+ // summary → summary-of-summary → fidelity decays: the agent forgets which files it
420
+ // edited, whether tests passed, what's still open. Prose summarisation is inherently
421
+ // lossy and gets worse every round.
422
+ //
423
+ // Defence: maintain a DETERMINISTIC, append-only ledger of high-signal facts derived
424
+ // directly from tool calls — files created/edited (with count), commands verified
425
+ // (test/build/lint) and their pass/fail, and explicit open TODOs. This is built from
426
+ // structured tool data (NOT model output), so it is LOSSLESS and never degrades. We
427
+ // inject it VERBATIM into every compaction preamble, so no matter how many times the
428
+ // prose summary is re-summarised, the concrete "what changed / what's verified /
429
+ // what's left" facts survive intact across an arbitrarily long run.
430
+ const LEDGER_MAX_FILES = 60; // cap the file list so the preamble can't balloon
431
+ const LEDGER_MAX_NOTES = 20; // cap verification/among notes
432
+ function createLedger() {
433
+ return { filesTouched: new Map(), filesTouchedTotal: 0, verifications: [], testIntegrity: [], testIntegrityTotal: 0, epoch: 0 };
434
+ }
435
+ /** Record one tool call's effect on the ledger (deterministic, no model call). */
436
+ function ledgerRecord(ledger, toolName, input, ok, output, exitCode) {
437
+ if (exports.WRITE_TOOL_NAMES.has(toolName)) {
438
+ if (!ok)
439
+ return; // a FAILED write changed nothing — not a durable fact
440
+ // A successful source write advances the mutation epoch: any verification
441
+ // run after this point has different inputs than runs before it.
442
+ ledger.epoch += 1;
443
+ const p = typeof input?.path === 'string' ? input.path : undefined;
444
+ if (p) {
445
+ const prev = ledger.filesTouched.get(p);
446
+ if (!prev)
447
+ ledger.filesTouchedTotal++;
448
+ // DELETE before SET, so a re-touched path moves to the BACK of the insertion
449
+ // order. `Map.set` on an existing key keeps its ORIGINAL slot, which quietly
450
+ // broke the eviction policy below: a file edited hundreds of times over a long
451
+ // session kept the position of its FIRST edit, so it aged out like a file nobody
452
+ // had looked at since — and on the next edit it was re-inserted as "new", double-
453
+ // counting filesTouchedTotal (which is documented as DISTINCT paths). Making the
454
+ // Map a true LRU-by-touch is what lets the `key !== p` guard below mean anything.
455
+ ledger.filesTouched.delete(p);
456
+ ledger.filesTouched.set(p, { tool: toolName, edits: (prev?.edits ?? 0) + 1 });
457
+ // Bound the Map itself, not just its rendering. LEDGER_MAX_FILES caps how many
458
+ // paths the preamble PRINTS (see ledgerSummary's slice), but the Map was only ever
459
+ // written to — so a multi-hour run touching thousands of files grew it without
460
+ // limit, and it is deliberately retained across every compaction. Evict the
461
+ // least-recently-touched entries once we hold well beyond what can ever be
462
+ // displayed. Hysteresis (evict down to 2× only once we exceed 4×) keeps this an
463
+ // occasional bulk sweep instead of a delete on every single write.
464
+ if (ledger.filesTouched.size > LEDGER_MAX_FILES * 4) {
465
+ for (const key of ledger.filesTouched.keys()) {
466
+ if (ledger.filesTouched.size <= LEDGER_MAX_FILES * 2)
467
+ break;
468
+ if (key !== p)
469
+ ledger.filesTouched.delete(key);
470
+ }
471
+ }
472
+ }
473
+ // Reward-hacking guard: if this write WEAKENED a test file, record it so the
474
+ // signal survives compaction and can be surfaced before the agent finishes.
475
+ const reasons = [];
476
+ // write_file overwrites carry a marker computed by the executor (which had the
477
+ // prior on-disk content) — it detects REMOVED assertions/cases, not just
478
+ // additive skips/tautologies. Prefer it when present.
479
+ const markerReasons = toolName === 'write_file' ? (0, testIntegrity_1.decodeTestIntegrityMarker)(output) : [];
480
+ if (markerReasons.length) {
481
+ reasons.push(...markerReasons);
482
+ }
483
+ else {
484
+ const ti = (0, testIntegrity_1.analyzeWriteToolForTestIntegrity)(toolName, input);
485
+ if (ti?.suspicious)
486
+ reasons.push(...ti.findings.map((f) => f.reason));
487
+ }
488
+ if (reasons.length && p) {
489
+ for (const reason of reasons) {
490
+ ledger.testIntegrity.push({ path: p, reason });
491
+ ledger.testIntegrityTotal++;
492
+ }
493
+ if (ledger.testIntegrity.length > LEDGER_MAX_NOTES * 2) {
494
+ ledger.testIntegrity.splice(0, ledger.testIntegrity.length - LEDGER_MAX_NOTES);
495
+ }
496
+ }
497
+ }
498
+ else if (toolName === 'bash') {
499
+ const cmd = String(input?.command ?? '').trim();
500
+ if (cmd && exports.VERIFY_CMD_RE.test(cmd)) {
501
+ // Record BOTH outcomes: a FAILED test/build is the single most important
502
+ // fact to carry across a compaction (it tells the agent work is NOT done).
503
+ //
504
+ // CRITICAL: `ok` is `result.error === undefined`, which is TRUE even when a
505
+ // test suite exits non-zero (the executor doesn't set `error` for a plain
506
+ // command failure — only for timeout/abort/spawn-fail). So `ok` alone would
507
+ // record a FAILING `npm test` as PASSED. The executor now reports the real
508
+ // process exit code via `exitCode`; a non-zero exit means the verification
509
+ // FAILED regardless of `ok`. Fall back to `ok` only when no exitCode is
510
+ // available (older tools / non-bash paths).
511
+ const passed = exitCode !== undefined ? exitCode === 0 : ok;
512
+ ledger.verifications.push({ cmd: cmd.slice(0, 120), ok: passed, epoch: ledger.epoch });
513
+ if (ledger.verifications.length > LEDGER_MAX_NOTES * 2) {
514
+ ledger.verifications.splice(0, ledger.verifications.length - LEDGER_MAX_NOTES);
515
+ }
516
+ }
517
+ }
518
+ }
519
+ /** Render the ledger as a compact, verbatim block for the compaction preamble. */
520
+ function ledgerSummary(ledger) {
521
+ const lines = [];
522
+ if (ledger.filesTouched.size) {
523
+ const files = [...ledger.filesTouched.entries()];
524
+ // The TAIL, not the head: the Map is ordered least-recently-touched first, so
525
+ // slicing from the front showed the OLDEST files and reliably omitted the ones the
526
+ // agent was working on right now — the opposite of what this preamble is for.
527
+ const shown = files.slice(-LEDGER_MAX_FILES);
528
+ lines.push(`FILES CHANGED THIS SESSION (${ledger.filesTouchedTotal || ledger.filesTouched.size}):`);
529
+ for (const [p, meta] of shown) {
530
+ lines.push(` • ${p} (${meta.tool}${meta.edits > 1 ? ` ×${meta.edits}` : ''})`);
531
+ }
532
+ if (files.length > shown.length)
533
+ lines.push(` • … and ${files.length - shown.length} more`);
534
+ }
535
+ if (ledger.verifications.length) {
536
+ const recent = ledger.verifications.slice(-LEDGER_MAX_NOTES);
537
+ lines.push(`VERIFICATION RUNS (most recent ${recent.length}):`);
538
+ for (const v of recent)
539
+ lines.push(` • [${v.ok ? 'PASS' : 'FAIL'}] ${v.cmd}`);
540
+ }
541
+ if (ledger.testIntegrity.length) {
542
+ const recent = ledger.testIntegrity.slice(-LEDGER_MAX_NOTES);
543
+ lines.push(`⚠ TEST-INTEGRITY ALERTS (test files were weakened — must justify or revert):`);
544
+ for (const t of recent)
545
+ lines.push(` • ${t.path}: ${t.reason}`);
546
+ }
547
+ const flaky = (0, flaky_1.detectFlaky)(ledger.verifications);
548
+ if (flaky.length) {
549
+ lines.push(`⚠ FLAKY TESTS (same command flipped PASS↔FAIL with no edit between — a green run proves nothing):`);
550
+ for (const f of flaky.slice(0, LEDGER_MAX_NOTES)) {
551
+ lines.push(` • ${f.cmd} (${f.passes} pass / ${f.fails} fail at identical code)`);
552
+ }
553
+ }
554
+ return lines.join('\n');
555
+ }
556
+ // How many of the most-recent messages keep their tool_result content verbatim.
557
+ // Older tool_result bodies are the bulk of a large body and are the safest thing
558
+ // to shed first (the model has already acted on them), so we replace their content
559
+ // with a short stub while KEEPING the block (so tool_use/tool_result pairing and
560
+ // turn structure stay intact — unlike summarisation, which drops whole turns).
561
+ exports.PRUNE_KEEP_RECENT = 8;
562
+ const PRUNE_STUB_KEEP_CHARS = 400; // keep a short head of each pruned result for context
563
+ // Marker sentinel appended to a pruned tool_result's content. We detect
564
+ // "already pruned" by this suffix rather than by an out-of-schema field on the
565
+ // block, because the block object is serialised verbatim onto the request body
566
+ // and forwarded to Anthropic — any extra property (e.g. a `_pruned` flag) would
567
+ // be rejected as an unknown field on a content block (400). Encoding the state
568
+ // inside the (string) content keeps the wire payload schema-clean AND idempotent.
569
+ const PRUNE_MARKER = '\n\n[… ';
570
+ const PRUNE_MARKER_TAIL = ' pruned to conserve context. Re-run the tool if you need the full result.]';
571
+ /**
572
+ * Lossy-but-structure-preserving prune: shrink OLD, large tool_result blocks in
573
+ * place, keeping the last PRUNE_KEEP_RECENT messages untouched. This is tried
574
+ * BEFORE summarisation because it:
575
+ * • keeps every turn and every tool_use/tool_result pair (API stays valid),
576
+ * • never makes an extra model call (summarisation does — cost + latency),
577
+ * • degrades gracefully on repeat (summarise-of-summarise loses the most on
578
+ * long runs; pruning just trims already-consumed output further).
579
+ *
580
+ * IMPORTANT: pruned state is encoded in the content string (PRUNE_MARKER_TAIL
581
+ * suffix), NOT as an extra property on the block — a stray field on a content
582
+ * block is rejected by the Anthropic API as an unknown key (400). This keeps the
583
+ * serialised body schema-clean while remaining idempotent across repeat calls.
584
+ *
585
+ * Returns the number of bytes reclaimed (0 if nothing was prunable).
586
+ *
587
+ * `minReclaimBytes` (default 0): if the TOTAL prunable amount is below this, the
588
+ * function makes NO changes and returns 0. This is a cache-safety gate — pruning
589
+ * even one old block changes the request prefix and invalidates the message-level
590
+ * prompt cache, so a tiny prune would bust the cache (re-write at 1.25×) for
591
+ * almost no size win. Measuring first, then applying only if worthwhile, keeps
592
+ * the "don't bust cache for a trivial gain" contract truly atomic (the old code
593
+ * mutated first and let the caller decide, which had already invalidated the
594
+ * cache by the time the caller declined).
595
+ */
596
+ function pruneOldToolResults(messages, minReclaimBytes = 0) {
597
+ const cutoff = messages.length - exports.PRUNE_KEEP_RECENT;
598
+ if (cutoff <= 1)
599
+ return 0;
600
+ // Collect prunable blocks + measure the total reclaim WITHOUT mutating yet.
601
+ const targets = [];
602
+ let total = 0;
603
+ for (let i = 0; i < cutoff; i++) {
604
+ const m = messages[i];
605
+ if (!Array.isArray(m.content))
606
+ continue;
607
+ for (const b of m.content) {
608
+ if (b.type !== 'tool_result')
609
+ continue;
610
+ const text = typeof b.content === 'string' ? b.content : JSON.stringify(b.content ?? '');
611
+ if (text.endsWith(PRUNE_MARKER_TAIL))
612
+ continue; // already pruned (idempotent)
613
+ if (text.length <= PRUNE_STUB_KEEP_CHARS + 80)
614
+ continue; // already small
615
+ // MUST use the surrogate-safe slice: a raw `text.slice(0, N)` landing
616
+ // between the high/low half of an emoji or CJK-extension glyph leaves a
617
+ // lone surrogate in the stub. That string is still valid JS but breaks
618
+ // when JSON.stringify'd onto the wire — Anthropic rejects the WHOLE
619
+ // request with a deterministic 400 "no low surrogate in string" that
620
+ // repeats identically on every retry (the corrupted payload never
621
+ // changes). This is exactly the class of bug util/safeSlice.ts exists
622
+ // to prevent; this call site just never got migrated to it.
623
+ const head = (0, safeSlice_1.sliceSafeEnd)(text, PRUNE_STUB_KEEP_CHARS);
624
+ const omitted = text.length - head.length;
625
+ targets.push({ block: b, head, omitted });
626
+ total += omitted;
627
+ }
628
+ }
629
+ // Cache-safety gate: not worth busting the prompt cache for a trivial reclaim.
630
+ if (total < minReclaimBytes)
631
+ return 0;
632
+ // Worthwhile — apply the stubs.
633
+ let reclaimed = 0;
634
+ for (const { block, head, omitted } of targets) {
635
+ block.content = `${head}${PRUNE_MARKER}${omitted} chars of earlier tool output${PRUNE_MARKER_TAIL}`;
636
+ reclaimed += omitted;
637
+ }
638
+ return reclaimed;
639
+ }
640
+ /**
641
+ * Compact `messages` in place: summarise everything before a safe cut point and
642
+ * replace it with a summary preamble. Returns true if compaction happened.
643
+ */
644
+ /** Extract the first user turn's plain text — the ORIGINAL task/goal. */
645
+ function originalTaskText(messages) {
646
+ const first = messages.find((m) => m.role === 'user');
647
+ if (!first || !Array.isArray(first.content))
648
+ return '';
649
+ return first.content
650
+ .filter((b) => b.type === 'text' && b.text && !(0, types_1.isRuntimeContextBlock)(b))
651
+ .map((b) => b.text)
652
+ .join('\n')
653
+ .trim();
654
+ }
655
+ /**
656
+ * The summariser's own model call failed — as opposed to running fine but not
657
+ * shrinking anything.
658
+ *
659
+ * These two outcomes used to be one `return false`, and collapsing them was a
660
+ * real bug: the in-loop circuit breaker disables auto-compaction permanently
661
+ * after COMPACT_MAX_FAILURES, on the sound theory that a compaction which cannot
662
+ * reclaim bytes will never start. But a summariser that THREW says nothing about
663
+ * whether compaction would help — only that the network/provider was unavailable
664
+ * for a moment. Feeding those into the same counter meant three transient blips
665
+ * (a 529 burst, a brief outage, a rate-limit spike) permanently switched off the
666
+ * one mechanism keeping the context under control, and the run then died at the
667
+ * context wall minutes later with its own recovery already disabled.
668
+ */
669
+ class CompactionUnavailableError extends Error {
670
+ constructor() {
671
+ super('Compaction summariser was unavailable');
672
+ this.name = 'CompactionUnavailableError';
673
+ }
674
+ }
675
+ exports.CompactionUnavailableError = CompactionUnavailableError;
676
+ function makeCachedSummarizer(messages, requestOptions, abortSignal) {
677
+ return async (instruction) => {
678
+ const last = messages[messages.length - 1];
679
+ if (!last || last.role !== 'user')
680
+ return '';
681
+ const lastContent = typeof last.content === 'string'
682
+ ? [{ type: 'text', text: last.content }]
683
+ : [...last.content];
684
+ const probe = [
685
+ ...messages.slice(0, -1),
686
+ { ...last, content: [...lastContent, { type: 'text', text: instruction }] },
687
+ ];
688
+ const reply = await (0, client_1.streamChat)(probe, { ...requestOptions(), abortSignal, allowRestartAfterRender: true }, () => { });
689
+ return reply.content
690
+ .filter((b) => b.type === 'text')
691
+ .map((b) => b.text ?? '')
692
+ .join('')
693
+ .trim();
694
+ };
695
+ }
696
+ const CACHED_SUMMARY_INSTRUCTION = `[Context compaction — this is an automated request from the agent runtime, not the user.]\n` +
697
+ `Do NOT call any tools and do NOT continue the task. Reply with text only: a concise bullet-point ` +
698
+ `summary of this whole session so far that you will need to continue the work — the user's goal, ` +
699
+ `key decisions, files changed (and how), commands run and their outcome, unresolved problems, and ` +
700
+ `user preferences. Max 400 words.`;
701
+ async function autoCompactMessages(messages, options, ledger, summarizeCached) {
702
+ const cut = findSafeCutIndex(messages, messages.length - exports.COMPACT_KEEP_MIN);
703
+ if (cut < 2)
704
+ return false; // nothing meaningful to fold
705
+ // PreCompact: only now that compaction will really happen (the early return above would
706
+ // have made the hook fire for a no-op). Main agent only; an observer, it cannot veto.
707
+ if ((options._depth ?? 0) === 0) {
708
+ await (0, hooks_1.runLifecycleHooks)((0, hooks_1.loadHooks)(options.workDir).PreCompact, 'PreCompact', options.workDir, { trigger: 'auto', session_id: options.sessionId ?? '', messages_to_summarize: cut }, 'auto', (0, hooks_1.hookRunOptsFor)(options)).catch(() => undefined);
709
+ }
710
+ // Sizes measured here, not in firePostCompact: by then `messages` has already
711
+ // been spliced and the "before" number no longer exists.
712
+ const beforeCompact = messages.length;
713
+ const afterCompact = cut > 0 ? messages.length - cut + 1 : messages.length;
714
+ const toSummarize = messages.slice(0, cut);
715
+ const kept = messages.slice(cut);
716
+ const firePostCompact = async () => {
717
+ if ((options._depth ?? 0) !== 0)
718
+ return;
719
+ // Host callback: the loop rewrote the context, so the caller's own trail
720
+ // (event log, ledger, telemetry) must show it.
721
+ try {
722
+ options.onCompact?.({ trigger: 'auto', before: beforeCompact, after: afterCompact });
723
+ }
724
+ catch { /* a host callback must never break the run */ }
725
+ await (0, hooks_1.runLifecycleHooks)((0, hooks_1.loadHooks)(options.workDir).PostCompact, 'PostCompact', options.workDir, { trigger: 'auto', session_id: options.sessionId ?? '', messages_summarized: cut }, 'auto', (0, hooks_1.hookRunOptsFor)(options)).catch(() => undefined);
726
+ };
727
+ // Pin the ORIGINAL task verbatim. findSafeCutIndex can (and on a long single
728
+ // run usually does) cut PAST the first user turn, folding the user's actual
729
+ // goal into the lossy summary — after a few compactions the agent drifts off
730
+ // what it was asked to do. We re-inject the first user turn's text verbatim
731
+ // into the replacement preamble so the objective survives every compaction.
732
+ // (We cannot keep it as a separate user message: the API requires alternating
733
+ // roles and kept[0] is already an assistant turn — two user turns would 400.)
734
+ const originalTask = originalTaskText(toSummarize);
735
+ const summaryPrompt = `Summarize this coding-session transcript into concise bullet points the assistant needs to continue the work: ` +
736
+ `key decisions, files changed (and how), commands run, unresolved problems, and user preferences. Max 400 words.\n\n` +
737
+ transcriptOf(toSummarize);
738
+ let summary = '';
739
+ // Cache-friendly path first; any failure or empty reply falls back to the standalone
740
+ // transcript summariser below, which always works but pays for the history uncached.
741
+ if (summarizeCached) {
742
+ try {
743
+ summary = await summarizeCached(CACHED_SUMMARY_INSTRUCTION);
744
+ }
745
+ catch (err) {
746
+ if (options.abortSignal?.aborted || err.name === 'AbortError')
747
+ throw err;
748
+ summary = '';
749
+ }
750
+ }
751
+ if (!summary)
752
+ try {
753
+ const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
754
+ // Run on the SAME model as the actual conversation. A previous version
755
+ // forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
756
+ // "bullet-point this transcript" task doesn't need the user's tier — but
757
+ // that silently billed Anthropic (and made a real network call to a
758
+ // provider the user may not have configured/paid for) even when the
759
+ // whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
760
+ // is already paying for is used for compaction too, so there is never a
761
+ // surprise charge on a different provider. Falls back to the same
762
+ // default as the main loop (see `runAgentLoop`) only when no model was
763
+ // set at all.
764
+ //
765
+ // Cheapest model of the SAME vendor (Claude Code runs this kind of work on
766
+ // Haiku). Safe to downgrade HERE because this fallback sends a fresh
767
+ // transcript — it shares no cached prefix with the conversation, unlike
768
+ // summarizeCached above, which must stay on the conversation's own model.
769
+ // Never crosses providers; falls back to the session model if the
770
+ // catalogue is not loaded.
771
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
772
+ mode: 'ask', // summariser must not call tools; ask-mode discourages action
773
+ env: options.env,
774
+ clientType: options.clientType,
775
+ abortSignal: options.abortSignal,
776
+ // The summariser renders NOTHING (onEvent below is a no-op) and its result is
777
+ // read only from the returned message, so a restart has nothing to roll back —
778
+ // always safe. Worth enabling: a blip here used to abandon compaction entirely,
779
+ // which then let the very next turn hit the context wall it was meant to prevent.
780
+ allowRestartAfterRender: true,
781
+ }, () => { });
782
+ summary = reply.content
783
+ .filter((b) => b.type === 'text')
784
+ .map((b) => b.text ?? '')
785
+ .join('')
786
+ .trim();
787
+ }
788
+ catch {
789
+ // Summarisation FAILED — the model call itself threw (network blip, 529,
790
+ // provider quota). Distinguished from "ran fine but didn't help" by the
791
+ // caller, because the two must not feed the same circuit breaker: three
792
+ // transient network errors would otherwise permanently disable compaction
793
+ // for the rest of the run, leaving the context to grow until the turn dies
794
+ // with no recovery left. Leave history as is; the turn may still fit.
795
+ throw new CompactionUnavailableError();
796
+ }
797
+ if (!summary)
798
+ return false;
799
+ // Replace the summarized head with a single user summary message. The cut is
800
+ // at a turn boundary (kept[0] is an assistant message — see findSafeCutIndex),
801
+ // so `user(summary) → assistant(kept[0])` is a valid, well-ordered sequence
802
+ // and no orphaned tool_result is left behind. We intentionally do NOT insert
803
+ // an assistant-ack here: that would put two assistant messages back-to-back
804
+ // (kept[0] is already an assistant), which the API rejects.
805
+ const taskBlock = originalTask
806
+ ? `ORIGINAL TASK (verbatim — keep working toward this, do not lose sight of it):\n${originalTask}\n\n`
807
+ : '';
808
+ // GAP E — the deterministic ledger (files changed + verification pass/fail) is
809
+ // injected VERBATIM, so these concrete facts never decay through repeated
810
+ // summary-of-summary compactions the way the prose summary does.
811
+ const ledgerText = ledger ? ledgerSummary(ledger) : '';
812
+ const ledgerBlock = ledgerText
813
+ ? `PROGRESS LEDGER (authoritative, machine-tracked — trust this over the prose summary for what changed/verified):\n${ledgerText}\n\n`
814
+ : '';
815
+ messages.splice(0, cut, { role: 'user', content: [{ type: 'text', text: `[Auto-compacted ${toSummarize.length} earlier messages]\n\n${taskBlock}${ledgerBlock}Summary of the earlier conversation so far:\n${summary}\n\nContinue the work from here.` }] });
816
+ // `kept` follows automatically since splice only replaced the head.
817
+ void kept;
818
+ await firePostCompact();
819
+ return true;
820
+ }
821
+ // ─── Periodic memory compaction ────────────────────────────────────────────
822
+ // A memory file (project or global — see agent/memory.ts) can grow large over
823
+ // many sessions since memory_write only ever appends. writeMemory() already
824
+ // applies an immediate, synchronous byte-cap eviction (oldest entries dropped)
825
+ // as a hard backstop, but that's a blunt instrument — this periodically
826
+ // consolidates the file with a real LLM summarization pass instead, so old
827
+ // facts are condensed into fewer, denser bullets rather than silently lost.
828
+ // Checked opportunistically right after a successful memory_write (see the
829
+ // call site below) rather than on every tool call — cheap to check (a single
830
+ // file stat), and memory_write is the only thing that can push a file over
831
+ // the trigger threshold in the first place.
832
+ async function maybeCompactMemory(scope, options) {
833
+ try {
834
+ await (0, memory_1.compactMemoryIfNeeded)(scope, options.workDir, async (prompt) => {
835
+ const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: prompt }] }], {
836
+ // Same reasoning as autoCompactMessages' summariser: run on the same
837
+ // model as the actual conversation rather than forcing a fixed tier
838
+ // (which silently billed Anthropic regardless of the user's provider).
839
+ // Cheapest SAME-vendor model: a fresh prompt with no shared cache prefix,
840
+ // and merging a few memory bullets does not need the session's top tier.
841
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
842
+ mode: 'ask',
843
+ env: options.env,
844
+ clientType: options.clientType,
845
+ abortSignal: options.abortSignal,
846
+ // Same as the transcript summariser: no rendered output, result read only from
847
+ // the returned message, so restarting on a blip is always safe.
848
+ allowRestartAfterRender: true,
849
+ }, () => { });
850
+ return reply.content
851
+ .filter((b) => b.type === 'text')
852
+ .map((b) => b.text ?? '')
853
+ .join('');
854
+ });
855
+ }
856
+ catch {
857
+ // Best-effort — a failed/aborted compaction just means the file stays as-is
858
+ // until the next memory_write call tries again; writeMemory's synchronous
859
+ // byte cap already bounds worst-case growth in the meantime.
860
+ }
861
+ }
862
+ // ─── Resume-time proactive compaction ─────────────────────────────────────────
863
+ //
864
+ // The in-loop auto-compact above only reacts to `lastPromptTokens`, which is
865
+ // populated from the PREVIOUS turn's usage event. On a freshly-resumed session
866
+ // (opening an old chat from history and sending the first new message) there is
867
+ // no previous turn in this process — `lastPromptTokens` starts at 0 — so the
868
+ // token-pressure trigger never fires for turn 0, and the byte-pressure trigger
869
+ // only catches truly huge sessions (MAX_BODY_BYTES is sized to stay under the
870
+ // backend's 25 MB body limit, not to bound cost — 8 MB of tool-heavy JSON is
871
+ // already on the order of the 1M-token context window itself). The result: a
872
+ // resumed session comfortably under both guards, but still hundreds of
873
+ // thousands of tokens, gets sent to the model at FULL PRICE on the very first
874
+ // message after resume, silently, every time.
875
+ //
876
+ // This function closes that gap: call it once, right after loading a stored
877
+ // session and BEFORE the user's next message is sent, so the expensive
878
+ // resend is compacted proactively instead of being missed by both in-loop
879
+ // guards. It reuses the exact same threshold/mechanics as the in-loop guard
880
+ // (cheap prune first, then summarising compaction) so behaviour stays
881
+ // consistent whether compaction happens at resume-time or mid-run.
882
+ const RESUME_CHARS_PER_TOKEN = 4; // rough, conservative estimate for JSON/code-heavy transcripts
883
+ /** Rough token estimate for a resumed transcript — no API round-trip needed. */
884
+ function estimateTokensRough(messages) {
885
+ return Math.ceil(estimateBodyBytes(messages) / RESUME_CHARS_PER_TOKEN);
886
+ }
887
+ /**
888
+ * Proactively compact `messages` in place if resuming this session would blow
889
+ * past the auto-compact threshold on the very first turn. Returns true if any
890
+ * compaction happened (so the caller can surface a one-line notice to the
891
+ * user). Safe to call on any message array, including empty/small ones (no-op).
892
+ *
893
+ * `model` picks the right context window (mirrors runAgentLoop's own lookup);
894
+ * `onNotice` is optional — pass it to show the same "auto-compacted" message
895
+ * the in-loop path shows, so the behaviour is visually consistent.
896
+ */
897
+ async function compactMessagesForResume(messages, opts) {
898
+ if (messages.length <= exports.COMPACT_KEEP_MIN + 2)
899
+ return false;
900
+ const settings = (0, rules_1.loadSettings)(opts.workDir);
901
+ if (!resolveAutoCompact(undefined, settings.raw))
902
+ return false;
903
+ // Live catalogue first (see contextWindowFor's doc comment above for why),
904
+ // same fallback chain this call site always used otherwise.
905
+ const contextWindow = (0, modelCatalogue_1.liveContextWindowFor)(opts.model, exports.MODEL_CONTEXT_TOKENS[opts.model ?? 'turbo'] ?? 1000000);
906
+ let bodyBytes = estimateBodyBytes(messages);
907
+ let tokenGuess = estimateTokensRough(messages);
908
+ // Prune fires at the EARLY threshold (mirrors the in-loop guard); summarisation
909
+ // only at the late one. On resume this matters most: a stored session is resent
910
+ // whole on the first turn, so shedding old tool_result bulk up front is exactly
911
+ // what stops that first message being billed at full size.
912
+ const limits = compactionLimits(contextWindow, settings.raw);
913
+ const overPruneThreshold = () => tokenGuess > limits.prune || bodyBytes > exports.MAX_BODY_BYTES;
914
+ const overCompactThreshold = () => tokenGuess > limits.compact || bodyBytes > exports.MAX_BODY_BYTES;
915
+ if (!overPruneThreshold())
916
+ return false;
917
+ let compacted = false;
918
+ // Cheap pass first — shrinks old tool_result blocks with no model call. The
919
+ // reclaim floor is enforced atomically inside pruneOldToolResults (measures
920
+ // first, mutates only if worthwhile), so a declined prune leaves the cache intact.
921
+ if (messages.length > exports.PRUNE_KEEP_RECENT + 2) {
922
+ const reclaimed = pruneOldToolResults(messages, exports.PRUNE_MIN_RECLAIM_BYTES);
923
+ if (reclaimed > 0) {
924
+ bodyBytes = estimateBodyBytes(messages);
925
+ tokenGuess = estimateTokensRough(messages);
926
+ compacted = true;
927
+ opts.onNotice?.(`\n\u267b\ufe0f Trimmed ~${(reclaimed / (1024 * 1024)).toFixed(1)}MB of older tool output before resuming this chat.\n`);
928
+ }
929
+ }
930
+ // If still over the LATE (summarise) threshold, fall through to summarising
931
+ // compaction — same mechanism the in-loop guard uses, so this can safely loop
932
+ // (a single summarisation pass may still leave a very long session over it).
933
+ // A session between the prune and summarise thresholds is left as-is after the
934
+ // cheap prune: no model call needed, prefix already shrunk.
935
+ let guard = 0;
936
+ while (overCompactThreshold() && messages.length > exports.COMPACT_KEEP_MIN + 2 && guard < 5) {
937
+ guard += 1;
938
+ // No ledger at resume time — the ledger is per-run, in-memory, and would
939
+ // have been created fresh anyway since this is a new process/run. The
940
+ // ORIGINAL TASK verbatim pin (inside autoCompactMessages) still applies.
941
+ // A summariser failure is not fatal HERE. This runs before the session is
942
+ // handed back to the user, so the worst case is resuming with a longer (more
943
+ // expensive) prefix — strictly better than refusing to resume at all. The
944
+ // in-loop caller treats the same signal differently, because there it must
945
+ // decide whether to arm a circuit breaker.
946
+ let did;
947
+ try {
948
+ did = await autoCompactMessages(messages, {
949
+ workDir: opts.workDir,
950
+ model: opts.model,
951
+ clientType: opts.clientType,
952
+ env: opts.env,
953
+ onText: () => { },
954
+ onToolUse: () => { },
955
+ onToolResult: () => { },
956
+ onUsage: () => { },
957
+ requestPermission: async () => false,
958
+ });
959
+ }
960
+ catch (err) {
961
+ if (err instanceof CompactionUnavailableError)
962
+ break;
963
+ throw err;
964
+ }
965
+ if (!did)
966
+ break;
967
+ compacted = true;
968
+ bodyBytes = estimateBodyBytes(messages);
969
+ tokenGuess = estimateTokensRough(messages);
970
+ }
971
+ if (compacted) {
972
+ opts.onNotice?.(`\n\u267b\ufe0f Auto-compacted this chat's earlier history before resuming, to avoid resending it at full cost.\n`);
973
+ }
974
+ return compacted;
975
+ }
976
+ //# sourceMappingURL=compaction.js.map