@nexrall/code-core 1.4.65 → 1.4.67
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/agent/agentTypes.d.ts +8 -4
- package/dist/agent/agentTypes.d.ts.map +1 -1
- package/dist/agent/agentTypes.js +20 -147
- package/dist/agent/askOnce.d.ts +12 -0
- package/dist/agent/askOnce.d.ts.map +1 -0
- package/dist/agent/askOnce.js +41 -0
- package/dist/agent/compaction.d.ts +244 -0
- package/dist/agent/compaction.d.ts.map +1 -0
- package/dist/agent/compaction.js +976 -0
- package/dist/agent/fileLocks.d.ts +34 -0
- package/dist/agent/fileLocks.d.ts.map +1 -0
- package/dist/agent/fileLocks.js +114 -0
- package/dist/agent/hooks.d.ts +324 -0
- package/dist/agent/hooks.d.ts.map +1 -0
- package/dist/agent/hooks.js +1228 -0
- package/dist/agent/iterationPolicy.d.ts +121 -0
- package/dist/agent/iterationPolicy.d.ts.map +1 -0
- package/dist/agent/iterationPolicy.js +297 -0
- package/dist/agent/lifecycleHost.d.ts +55 -0
- package/dist/agent/lifecycleHost.d.ts.map +1 -0
- package/dist/agent/lifecycleHost.js +294 -0
- package/dist/agent/loop.d.ts +11 -491
- package/dist/agent/loop.d.ts.map +1 -1
- package/dist/agent/loop.js +418 -3032
- package/dist/agent/memory.d.ts +11 -0
- package/dist/agent/memory.d.ts.map +1 -1
- package/dist/agent/memory.js +23 -3
- package/dist/agent/planMode.d.ts +14 -0
- package/dist/agent/planMode.d.ts.map +1 -1
- package/dist/agent/planMode.js +146 -17
- package/dist/agent/securityLint.js +2 -2
- package/dist/agent/sharedTasks.d.ts +7 -0
- package/dist/agent/sharedTasks.d.ts.map +1 -1
- package/dist/agent/sharedTasks.js +16 -0
- package/dist/agent/subAgentBudget.d.ts +65 -0
- package/dist/agent/subAgentBudget.d.ts.map +1 -0
- package/dist/agent/subAgentBudget.js +269 -0
- package/dist/agent/subTask.d.ts +6 -0
- package/dist/agent/subTask.d.ts.map +1 -0
- package/dist/agent/subTask.js +713 -0
- package/dist/agent/subTaskSupport.d.ts +156 -0
- package/dist/agent/subTaskSupport.d.ts.map +1 -0
- package/dist/agent/subTaskSupport.js +409 -0
- package/dist/agent/toolDescriptions.d.ts +3 -0
- package/dist/agent/toolDescriptions.d.ts.map +1 -0
- package/dist/agent/toolDescriptions.js +116 -0
- package/dist/api/client.d.ts +1 -1
- package/dist/api/client.d.ts.map +1 -1
- package/dist/api/client.js +17 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/mcp/client.d.ts +104 -0
- package/dist/mcp/client.d.ts.map +1 -1
- package/dist/mcp/client.js +136 -2
- package/dist/mcp/httpClient.d.ts +17 -1
- package/dist/mcp/httpClient.d.ts.map +1 -1
- package/dist/mcp/httpClient.js +120 -19
- package/dist/mcp/manager.d.ts +77 -2
- package/dist/mcp/manager.d.ts.map +1 -1
- package/dist/mcp/manager.js +275 -9
- package/dist/mcp/sseClient.d.ts +7 -1
- package/dist/mcp/sseClient.d.ts.map +1 -1
- package/dist/mcp/sseClient.js +45 -1
- package/dist/mcp/stats.d.ts +41 -0
- package/dist/mcp/stats.d.ts.map +1 -0
- package/dist/mcp/stats.js +108 -0
- package/dist/permissions/destructive.d.ts +11 -7
- package/dist/permissions/destructive.d.ts.map +1 -1
- package/dist/permissions/destructive.js +59 -8
- package/dist/permissions/destructiveTokens.d.ts +34 -0
- package/dist/permissions/destructiveTokens.d.ts.map +1 -0
- package/dist/permissions/destructiveTokens.js +475 -0
- package/dist/permissions/modePolicy.d.ts +12 -8
- package/dist/permissions/modePolicy.d.ts.map +1 -1
- package/dist/permissions/modePolicy.js +19 -12
- package/dist/permissions/rules.d.ts +4 -1
- package/dist/permissions/rules.d.ts.map +1 -1
- package/dist/permissions/rules.js +29 -0
- package/dist/types.d.ts +47 -3
- package/dist/types.d.ts.map +1 -1
- package/package.json +8 -7
|
@@ -0,0 +1,976 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.CompactionUnavailableError = exports.PRUNE_KEEP_RECENT = exports.VERIFY_CMD_RE = exports.WRITE_TOOL_NAMES = exports.MAX_BODY_BYTES = exports.COMPACT_MAX_FAILURES = exports.COMPACT_MIN_RECLAIM_BYTES = exports.COMPACT_KEEP_MIN = exports._lastApiCallEndedAt = exports.PRUNE_MIN_RECLAIM_BYTES = exports.MODEL_CONTEXT_TOKENS = void 0;
|
|
4
|
+
exports.contextWindowFor = contextWindowFor;
|
|
5
|
+
exports.compactionLimits = compactionLimits;
|
|
6
|
+
exports.compactionThresholds = compactionThresholds;
|
|
7
|
+
exports.pruneReclaimFloor = pruneReclaimFloor;
|
|
8
|
+
exports.estimateBodyBytes = estimateBodyBytes;
|
|
9
|
+
exports.resolveAutoCompact = resolveAutoCompact;
|
|
10
|
+
exports.resolveVerificationNudge = resolveVerificationNudge;
|
|
11
|
+
exports.allowsTestOnlyWrite = allowsTestOnlyWrite;
|
|
12
|
+
exports.findSafeCutIndex = findSafeCutIndex;
|
|
13
|
+
exports.persistRuntimeContext = persistRuntimeContext;
|
|
14
|
+
exports.transcriptOf = transcriptOf;
|
|
15
|
+
exports.createLedger = createLedger;
|
|
16
|
+
exports.ledgerRecord = ledgerRecord;
|
|
17
|
+
exports.ledgerSummary = ledgerSummary;
|
|
18
|
+
exports.pruneOldToolResults = pruneOldToolResults;
|
|
19
|
+
exports.makeCachedSummarizer = makeCachedSummarizer;
|
|
20
|
+
exports.autoCompactMessages = autoCompactMessages;
|
|
21
|
+
exports.maybeCompactMemory = maybeCompactMemory;
|
|
22
|
+
exports.estimateTokensRough = estimateTokensRough;
|
|
23
|
+
exports.compactMessagesForResume = compactMessagesForResume;
|
|
24
|
+
const types_1 = require("../types");
|
|
25
|
+
const testIntegrity_1 = require("./testIntegrity");
|
|
26
|
+
const flaky_1 = require("./flaky");
|
|
27
|
+
const client_1 = require("../api/client");
|
|
28
|
+
const rules_1 = require("../permissions/rules");
|
|
29
|
+
const memory_1 = require("./memory");
|
|
30
|
+
const modelCatalogue_1 = require("./modelCatalogue");
|
|
31
|
+
const safeSlice_1 = require("../util/safeSlice");
|
|
32
|
+
const hooks_1 = require("./hooks");
|
|
33
|
+
// ─── Auto-compact ─────────────────────────────────────────────────────────────
|
|
34
|
+
//
|
|
35
|
+
// When the conversation's prompt size approaches the model's context window,
|
|
36
|
+
// summarise the older portion automatically (Claude-Code style) instead of
|
|
37
|
+
// letting the request fail or forcing the user to run /compact by hand.
|
|
38
|
+
//
|
|
39
|
+
// Compaction only happens at a turn boundary (top of the loop, before the next
|
|
40
|
+
// streamChat) and only cuts at a "safe" user message — one with no tool_result
|
|
41
|
+
// blocks — so tool_use/tool_result pairing is never broken.
|
|
42
|
+
// Context window per model, keyed by BOTH the tier aliases and the real model
|
|
43
|
+
// ids the picker now sends.
|
|
44
|
+
//
|
|
45
|
+
// This table drives auto-compaction, so a wrong number is expensive in one
|
|
46
|
+
// direction and merely wasteful in the other: too small compacts early and
|
|
47
|
+
// summarises lossily; too LARGE means compaction never fires before the real
|
|
48
|
+
// ceiling and the turn dies on a provider 400 — mid-conversation, on exactly
|
|
49
|
+
// the long sessions compaction exists to protect.
|
|
50
|
+
//
|
|
51
|
+
// It previously held only turbo/pro/ultra, all at 1M, which was correct while
|
|
52
|
+
// every tier was a 1M-context Claude. With several providers selectable by
|
|
53
|
+
// name that assumption breaks hard: gpt-4o-mini is 128K, i.e. 8x smaller than
|
|
54
|
+
// the value this table would have guessed for it.
|
|
55
|
+
//
|
|
56
|
+
// The Anthropic figures mirror the backend registry
|
|
57
|
+
// (services/providers/modelRegistry.js). The OpenAI figures were MEASURED
|
|
58
|
+
// against the live API rather than read from docs — note that gpt-5.4 is
|
|
59
|
+
// 922_000, not a round 1M, and gpt-4.1 is 1_047_576.
|
|
60
|
+
exports.MODEL_CONTEXT_TOKENS = {
|
|
61
|
+
// Tier aliases — still sent by older clients, saved settings and sub-agent
|
|
62
|
+
// frontmatter, so they must keep resolving.
|
|
63
|
+
turbo: 1000000,
|
|
64
|
+
pro: 1000000,
|
|
65
|
+
ultra: 1000000,
|
|
66
|
+
fast: 200000,
|
|
67
|
+
// Anthropic, by real model id.
|
|
68
|
+
'claude-sonnet-5-5': 1000000,
|
|
69
|
+
'claude-opus-5-5': 1000000,
|
|
70
|
+
'claude-fable-5-1': 1000000,
|
|
71
|
+
'claude-haiku-4-5-20251001': 200000,
|
|
72
|
+
// OpenAI, by real model id (measured).
|
|
73
|
+
'gpt-5.4': 922000,
|
|
74
|
+
'gpt-5.4-mini': 272000,
|
|
75
|
+
'gpt-4.1': 1047576,
|
|
76
|
+
'gpt-4o-mini': 128000,
|
|
77
|
+
// GPT-5.6 family: MEASURED 2026-09-03 via a 400 (same technique as
|
|
78
|
+
// gpt-5.4's 922_000 above) — and it is the EXACT SAME 922,000-token
|
|
79
|
+
// ceiling on all three sizes. This CONTRADICTS the publicly documented
|
|
80
|
+
// figure (openai.com/index/gpt-5-6 + OpenRouter's model card both
|
|
81
|
+
// advertise 1,050,000) — see backend services/providers/modelRegistry.js's
|
|
82
|
+
// own comment on these rows for the measured 400 body. Guessing 1.05M here
|
|
83
|
+
// would fire auto-compaction ~12% past the real wall.
|
|
84
|
+
'gpt-5.6-sol': 922000,
|
|
85
|
+
'gpt-5.6-terra': 922000,
|
|
86
|
+
'gpt-5.6-luna': 922000,
|
|
87
|
+
// DeepSeek, by real model id (documented — see backend/services/providers/
|
|
88
|
+
// modelRegistry.js's own TODO(unverified-by-400): DeepSeek accepts an
|
|
89
|
+
// oversized max_completion_tokens without rejecting it, so there was no 400
|
|
90
|
+
// to measure the ceiling from the way the OpenAI rows above were).
|
|
91
|
+
'deepseek-flash': 1048576,
|
|
92
|
+
// Qwen (DashScope), by real model id (documented max input, same caveat).
|
|
93
|
+
'qwen3.7-max': 991800,
|
|
94
|
+
// Z.ai (GLM), by real model id. contextWindow is documented (Z.ai/
|
|
95
|
+
// Cloudflare Workers AI model cards, both list 1,048,576) rather than
|
|
96
|
+
// measured — a ~400K-token request was ACCEPTED (200), not rejected, so
|
|
97
|
+
// there was no 400 to read a real ceiling out of. maxOutputTokens IS
|
|
98
|
+
// measured: `max_tokens: 999999` was rejected with the ceiling in the
|
|
99
|
+
// error body (backend services/providers/modelRegistry.js's glm-5.3 row
|
|
100
|
+
// has the full verification notes).
|
|
101
|
+
'glm-5.3': 1048576,
|
|
102
|
+
};
|
|
103
|
+
/**
|
|
104
|
+
* Context window (tokens) for a model alias or real model id.
|
|
105
|
+
*
|
|
106
|
+
* Checks the LIVE catalogue (GET /api/code/models, modelCatalogue.ts) first —
|
|
107
|
+
* populated once per process by whichever client fetched it (CLI at session
|
|
108
|
+
* start, VS Code via _postModelCatalogue, desktop via its main-process
|
|
109
|
+
* fetch) — falling back to this hard-coded table when no live data exists yet
|
|
110
|
+
* (offline, older backend, or the catalogue simply hasn't been fetched by
|
|
111
|
+
* this call site). This is what lets a model added to the backend registry
|
|
112
|
+
* (services/providers/modelRegistry.js) get the CORRECT context window here
|
|
113
|
+
* even before this table is updated by hand for a new nexrall-code release —
|
|
114
|
+
* exactly the class of bug GPT-5.6's 922K-vs-1.05M mismatch was (see
|
|
115
|
+
* modelRegistry.js's own comment on that row).
|
|
116
|
+
*
|
|
117
|
+
* The static-table fallback is deliberately the SMALLEST window in the table
|
|
118
|
+
* rather than the largest. An unknown model is most likely a newly added one
|
|
119
|
+
* this client build predates, and guessing high is the failure that cannot be
|
|
120
|
+
* recovered from: the turn hits a provider 400 with no chance to compact.
|
|
121
|
+
* Guessing low only costs an earlier, lossy compaction — annoying, not broken.
|
|
122
|
+
*/
|
|
123
|
+
function contextWindowFor(model) {
|
|
124
|
+
const fallback = exports.MODEL_CONTEXT_TOKENS[model ?? 'turbo'] ?? 128000;
|
|
125
|
+
return (0, modelCatalogue_1.liveContextWindowFor)(model, fallback);
|
|
126
|
+
}
|
|
127
|
+
// ── Compaction thresholds (cost control) ─────────────────────────────────────
|
|
128
|
+
// Two independent triggers, deliberately at DIFFERENT levels:
|
|
129
|
+
//
|
|
130
|
+
// • PRUNE threshold (cheap, lossy-but-structure-preserving, NO model call):
|
|
131
|
+
// fires EARLY. Every turn a large history is resent, cache-read alone
|
|
132
|
+
// (0.10× input) is still billed on the whole prefix — on a 700K-token
|
|
133
|
+
// session that is real money accruing per turn long before the 1M wall.
|
|
134
|
+
// Anthropic's own server-side compaction defaults its trigger to 150K
|
|
135
|
+
// input tokens (docs: compact_20260112 default trigger 150000). We mirror
|
|
136
|
+
// that intent: start shedding already-consumed tool_result bulk at ~120K
|
|
137
|
+
// tokens (see compactionLimits) so the per-turn cache-read bill stops growing,
|
|
138
|
+
// WITHOUT paying for a summariser model call and WITHOUT dropping any turn
|
|
139
|
+
// (pruneOldToolResults keeps every tool_use/tool_result pair intact).
|
|
140
|
+
//
|
|
141
|
+
// • SUMMARISE threshold (a model call, lossy: drops whole turns): Claude Code's
|
|
142
|
+
// ~167K line (see compactionLimits). Later than prune, because summarise-of-summarise is the main cause of an
|
|
143
|
+
// agent "forgetting" earlier work. Only when cheap pruning can't keep the
|
|
144
|
+
// prompt under this line do we fall through to summarisation.
|
|
145
|
+
//
|
|
146
|
+
// Both are overridable via env for power users / tests.
|
|
147
|
+
function envFraction(name, fallback) {
|
|
148
|
+
const v = Number(process.env[name]);
|
|
149
|
+
return Number.isFinite(v) && v > 0 && v < 1 ? v : fallback;
|
|
150
|
+
}
|
|
151
|
+
// Where auto-compaction fires, in TOKENS — Claude Code's rule, not a fraction of a 1M window.
|
|
152
|
+
//
|
|
153
|
+
// Claude Code summarises at `window − min(maxOutput, 20K) − 13K buffer`, on a 200K window
|
|
154
|
+
// (~167K tokens). We used to wait for 80% of a 1M window (~800K): every request of a long
|
|
155
|
+
// run re-read that whole prefix from cache, so a session cost ~5× more per request than
|
|
156
|
+
// the same work in Claude Code long before anything was shed. The window used for this is
|
|
157
|
+
// capped (default 200K, like Claude Code) — the model's real 1M window still bounds what
|
|
158
|
+
// the backend will accept; this only decides when WE tidy up.
|
|
159
|
+
//
|
|
160
|
+
// NEXRALL_COMPACT_WINDOW / settings.json "autoCompactWindow": the cap (tokens). Set it
|
|
161
|
+
// to e.g. 1000000 to use the whole window before compacting.
|
|
162
|
+
// NEXRALL_PRUNE_THRESHOLD / NEXRALL_COMPACT_THRESHOLD: fractions of that capped window.
|
|
163
|
+
const DEFAULT_COMPACT_WINDOW = 200000;
|
|
164
|
+
const COMPACT_OUTPUT_RESERVE = 20000; // Claude Code: min(model max output, 20K)
|
|
165
|
+
const COMPACT_BUFFER_TOKENS = 13000; // Claude Code's autocompact buffer
|
|
166
|
+
const DEFAULT_PRUNE_FRACTION = 0.6; // cheap lossless prune ~120K, before the ~167K summarise
|
|
167
|
+
function compactWindowCap(settingsRaw) {
|
|
168
|
+
const fromEnv = Number(process.env.NEXRALL_COMPACT_WINDOW);
|
|
169
|
+
if (Number.isFinite(fromEnv) && fromEnv >= 50000)
|
|
170
|
+
return fromEnv;
|
|
171
|
+
const fromSettings = Number(settingsRaw?.autoCompactWindow);
|
|
172
|
+
if (Number.isFinite(fromSettings) && fromSettings >= 50000)
|
|
173
|
+
return fromSettings;
|
|
174
|
+
return DEFAULT_COMPACT_WINDOW;
|
|
175
|
+
}
|
|
176
|
+
/** Token counts at which auto-prune and auto-compact (summarise) fire for a model window. */
|
|
177
|
+
function compactionLimits(contextWindow, settingsRaw) {
|
|
178
|
+
const eff = Math.min(contextWindow, compactWindowCap(settingsRaw));
|
|
179
|
+
const compactFrac = envFraction('NEXRALL_COMPACT_THRESHOLD', 0);
|
|
180
|
+
const pruneFrac = envFraction('NEXRALL_PRUNE_THRESHOLD', 0);
|
|
181
|
+
const compact = compactFrac
|
|
182
|
+
? eff * compactFrac
|
|
183
|
+
: Math.max(eff * 0.5, eff - Math.min(COMPACT_OUTPUT_RESERVE, eff * 0.1) - COMPACT_BUFFER_TOKENS);
|
|
184
|
+
const prune = Math.min(pruneFrac ? eff * pruneFrac : eff * DEFAULT_PRUNE_FRACTION, compact * 0.9);
|
|
185
|
+
return { prune: Math.floor(prune), compact: Math.floor(compact) };
|
|
186
|
+
}
|
|
187
|
+
/** Auto-prune / auto-compact thresholds as fractions of the context window — for UI display (e.g. `/context`). */
|
|
188
|
+
function compactionThresholds(contextWindow = 1000000, settingsRaw) {
|
|
189
|
+
const { prune, compact } = compactionLimits(contextWindow, settingsRaw);
|
|
190
|
+
return { prune: prune / contextWindow, compact: compact / contextWindow };
|
|
191
|
+
}
|
|
192
|
+
// Only bother pruning if it reclaims a meaningful amount — a tiny prune busts
|
|
193
|
+
// the message-level prompt cache (the pruned prefix changes) for little gain,
|
|
194
|
+
// so we require at least this many bytes reclaimed before accepting a prune.
|
|
195
|
+
// Sized for the ~120K-token prune line: at 256 KB a prune could rarely reclaim enough to
|
|
196
|
+
// qualify, so sessions skipped straight to the lossy summariser.
|
|
197
|
+
exports.PRUNE_MIN_RECLAIM_BYTES = 96 * 1024; // 96 KB
|
|
198
|
+
// Cache-aware prune floor. The 96 KB floor exists ONLY to avoid busting a WARM prompt
|
|
199
|
+
// cache for a small gain. Once the session has been idle past the provider cache TTL
|
|
200
|
+
// (Anthropic/OpenAI: 5 min), the whole prefix is re-written on the next request anyway,
|
|
201
|
+
// so a prune at that moment costs nothing extra — accept a much smaller reclaim then.
|
|
202
|
+
// Keyed by sessionId (same scheme as _subAgentBudgets) because runAgentLoop runs once
|
|
203
|
+
// per user turn and the idle gap that matters is BETWEEN turns.
|
|
204
|
+
const PRUNE_MIN_RECLAIM_BYTES_COLD = 16 * 1024; // 16 KB
|
|
205
|
+
const CACHE_COLD_AFTER_MS = 5 * 60000;
|
|
206
|
+
exports._lastApiCallEndedAt = new Map();
|
|
207
|
+
function pruneReclaimFloor(lastCallEndedAt, now = Date.now()) {
|
|
208
|
+
return lastCallEndedAt !== undefined && now - lastCallEndedAt >= CACHE_COLD_AFTER_MS
|
|
209
|
+
? PRUNE_MIN_RECLAIM_BYTES_COLD
|
|
210
|
+
: exports.PRUNE_MIN_RECLAIM_BYTES;
|
|
211
|
+
}
|
|
212
|
+
exports.COMPACT_KEEP_MIN = 6; // always keep at least the last N messages verbatim
|
|
213
|
+
/**
|
|
214
|
+
* Bytes a compaction must reclaim to count as productive.
|
|
215
|
+
*
|
|
216
|
+
* Deliberately much smaller than PRUNE_MIN_RECLAIM_BYTES: a prune declines when the
|
|
217
|
+
* gain isn't worth busting the prompt cache, whereas by the time we are summarising
|
|
218
|
+
* we are already committed to rewriting the prefix — the only question is whether the
|
|
219
|
+
* summariser is making ANY headway. 32 KB is small enough that a genuinely useful
|
|
220
|
+
* compaction always clears it, large enough that shuffling a few bytes doesn't.
|
|
221
|
+
*/
|
|
222
|
+
exports.COMPACT_MIN_RECLAIM_BYTES = 32 * 1024; // 32 KB
|
|
223
|
+
/**
|
|
224
|
+
* Consecutive non-productive compaction attempts before auto-compaction is switched
|
|
225
|
+
* off for the rest of the run.
|
|
226
|
+
*
|
|
227
|
+
* 3 rather than 1 because the failure is often transient — a summariser stream that
|
|
228
|
+
* blipped will usually succeed on the next turn, and giving up instantly would lose
|
|
229
|
+
* the safety net for a whole long session over one network hiccup. 3 also bounds the
|
|
230
|
+
* wasted spend: at most three summariser calls, not hundreds.
|
|
231
|
+
*/
|
|
232
|
+
exports.COMPACT_MAX_FAILURES = 3;
|
|
233
|
+
// Byte-level safety net, independent of the token estimate.
|
|
234
|
+
//
|
|
235
|
+
// Tool-heavy sessions on large codebases accumulate many tool_result blocks
|
|
236
|
+
// (read_file / bash / search output). The token count can still look "under
|
|
237
|
+
// budget" while the SERIALISED body has grown to tens of MB — the char↔token
|
|
238
|
+
// ratio for JSON/code/logs is highly variable, so a token threshold alone does
|
|
239
|
+
// NOT bound the request body size. The backend rejects bodies over its limit
|
|
240
|
+
// (413), which the token-based compactor never anticipates because:
|
|
241
|
+
// • it reacts to lastPromptTokens from the PREVIOUS turn's usage event, so on
|
|
242
|
+
// a freshly-resumed (already-large) session it is 0 and never fires, and
|
|
243
|
+
// • 80% × 1M tokens of tool_result can be 25–45 MB — far past any body limit.
|
|
244
|
+
// This guard measures the ACTUAL body bytes before each send and forces a
|
|
245
|
+
// compaction whenever it crosses the threshold, regardless of the token count.
|
|
246
|
+
// Kept comfortably under the server's 25 MB /api/code limit.
|
|
247
|
+
exports.MAX_BODY_BYTES = 8 * 1024 * 1024; // 8 MB
|
|
248
|
+
/** Approximate serialised request-body size (bytes) for the messages array. */
|
|
249
|
+
function estimateBodyBytes(messages) {
|
|
250
|
+
try {
|
|
251
|
+
return Buffer.byteLength(JSON.stringify(messages), 'utf-8');
|
|
252
|
+
}
|
|
253
|
+
catch {
|
|
254
|
+
return 0; // circular/unserialisable — don't block on the estimate
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
function resolveAutoCompact(fromOptions, rawSettings) {
|
|
258
|
+
if (typeof fromOptions === 'boolean')
|
|
259
|
+
return fromOptions;
|
|
260
|
+
const env = (process.env.NEXRALL_AUTO_COMPACT ?? '').toLowerCase();
|
|
261
|
+
if (env === '0' || env === 'false' || env === 'off')
|
|
262
|
+
return false;
|
|
263
|
+
if (env === '1' || env === 'true' || env === 'on')
|
|
264
|
+
return true;
|
|
265
|
+
const s = rawSettings?.autoCompact;
|
|
266
|
+
if (typeof s === 'boolean')
|
|
267
|
+
return s;
|
|
268
|
+
return true;
|
|
269
|
+
}
|
|
270
|
+
/** Opt-out for the one-shot verification nudge (GAP D). Defaults to on. */
|
|
271
|
+
function resolveVerificationNudge(rawSettings) {
|
|
272
|
+
const env = (process.env.NEXRALL_VERIFY_NUDGE ?? '').toLowerCase();
|
|
273
|
+
if (env === '0' || env === 'false' || env === 'off')
|
|
274
|
+
return false;
|
|
275
|
+
if (env === '1' || env === 'true' || env === 'on')
|
|
276
|
+
return true;
|
|
277
|
+
const s = rawSettings?.verifyNudge;
|
|
278
|
+
if (typeof s === 'boolean')
|
|
279
|
+
return s;
|
|
280
|
+
return true;
|
|
281
|
+
}
|
|
282
|
+
/** Tools that mutate the filesystem — used by the verification nudge (GAP D). */
|
|
283
|
+
exports.WRITE_TOOL_NAMES = new Set(['write_file', 'edit_file', 'multi_edit', 'delete_file', 'move_file', 'copy_file', 'notebook_edit']);
|
|
284
|
+
/**
|
|
285
|
+
* May an agent restricted to `testFilesOnly` perform this tool call?
|
|
286
|
+
*
|
|
287
|
+
* A tool allowlist is all-or-nothing per tool: granting `edit_file` grants it for
|
|
288
|
+
* every path in the repo. A user-defined test-writer agent needs write access to produce
|
|
289
|
+
* tests, but must NOT be able to "fix" production source so a failing test goes
|
|
290
|
+
* green — the single most common way a test-writing agent destroys the signal it
|
|
291
|
+
* was asked to create. Its prompt says so; this makes it a refusal rather than a
|
|
292
|
+
* request.
|
|
293
|
+
*
|
|
294
|
+
* Pure + exported so the rules are testable directly, without running a real
|
|
295
|
+
* sub-agent.
|
|
296
|
+
*
|
|
297
|
+
* KNOWN LIMIT, stated rather than hidden: this gates the file TOOLS, not `bash`.
|
|
298
|
+
* A determined model could still write source via `bash: echo ... > src/x.ts`.
|
|
299
|
+
* Closing that means parsing shell redirection, which is not reliably doable — so
|
|
300
|
+
* this is a strong guardrail against the realistic failure mode, not a sandbox.
|
|
301
|
+
* Real isolation is the sandbox config (tools/sandbox.ts), a separate mechanism.
|
|
302
|
+
*/
|
|
303
|
+
function allowsTestOnlyWrite(tool, input) {
|
|
304
|
+
// Non-write tools are unaffected: reading, searching and running tests are all
|
|
305
|
+
// essential to writing a test.
|
|
306
|
+
//
|
|
307
|
+
// WRITE_TOOL_NAMES deliberately excludes `create_directory`: isTestFile matches
|
|
308
|
+
// FILE paths, so a legitimate `create_directory('test/helpers')` would be
|
|
309
|
+
// refused and the agent could not scaffold the tree it needs — while an empty
|
|
310
|
+
// directory cannot damage production code, and files placed in it are still
|
|
311
|
+
// checked individually.
|
|
312
|
+
if (!exports.WRITE_TOOL_NAMES.has(tool))
|
|
313
|
+
return true;
|
|
314
|
+
// EVERY path the call could affect must be a test file, not just `path`:
|
|
315
|
+
// move_file takes {source, dest} and copy_file {source, destination}, so
|
|
316
|
+
// checking `path` alone would let `move_file src/index.ts -> /tmp/x` through and
|
|
317
|
+
// remove production code by relocating it.
|
|
318
|
+
//
|
|
319
|
+
// `source` is skipped for notebook_edit specifically, where it is the CELL
|
|
320
|
+
// CONTENT rather than a path — treating a blob of code as a path would refuse
|
|
321
|
+
// every legitimate notebook edit.
|
|
322
|
+
const pathKeys = tool === 'notebook_edit'
|
|
323
|
+
? ['path']
|
|
324
|
+
: ['path', 'source', 'dest', 'destination'];
|
|
325
|
+
const candidates = pathKeys
|
|
326
|
+
.map((k) => input?.[k])
|
|
327
|
+
.filter((v) => typeof v === 'string' && v.length > 0);
|
|
328
|
+
// An unrecognised write shape (no path-like argument at all) is refused rather
|
|
329
|
+
// than allowed through, so a future tool cannot silently become a hole here.
|
|
330
|
+
if (candidates.length === 0)
|
|
331
|
+
return false;
|
|
332
|
+
return candidates.every((p) => (0, testIntegrity_1.isTestFile)(p));
|
|
333
|
+
}
|
|
334
|
+
/** Heuristic: does a bash command look like it's running tests/build/lint/typecheck? (GAP D) */
|
|
335
|
+
exports.VERIFY_CMD_RE = /\b(npm|yarn|pnpm)\s+(run\s+)?(test|build|lint|typecheck|tsc)\b|\bpytest\b|\bgo\s+(test|vet|build)\b|\btsc\b|\beslint\b|\bcargo\s+(test|build|check)\b/i;
|
|
336
|
+
/**
|
|
337
|
+
* Find the latest index ≤ maxIdx where history can be cut safely.
|
|
338
|
+
*
|
|
339
|
+
* A safe cut point is a **turn boundary**: an assistant message (which always
|
|
340
|
+
* begins a fresh turn after a user message). Cutting there guarantees that
|
|
341
|
+
* `messages[cut..]` starts with an assistant whose `tool_use` blocks are all
|
|
342
|
+
* answered by `tool_result`s that remain in the kept slice — so we never orphan
|
|
343
|
+
* a tool_result (which the API rejects). We deliberately allow cutting across
|
|
344
|
+
* tool_result-bearing user messages: the OLD implementation only cut at a
|
|
345
|
+
* *non*-tool_result user message, which never exists inside a single long
|
|
346
|
+
* agentic run (every user turn is a tool_result), so compaction was a no-op
|
|
347
|
+
* exactly when a long task needs it most.
|
|
348
|
+
*/
|
|
349
|
+
function findSafeCutIndex(messages, maxIdx) {
|
|
350
|
+
for (let i = Math.min(maxIdx, messages.length - 1); i >= 2; i--) {
|
|
351
|
+
if (messages[i].role === 'assistant')
|
|
352
|
+
return i;
|
|
353
|
+
}
|
|
354
|
+
return -1;
|
|
355
|
+
}
|
|
356
|
+
// Hard ceiling on the transcript we hand to the summariser. Per-block truncation
|
|
357
|
+
// alone does NOT bound the total: a very long run has thousands of blocks, so the
|
|
358
|
+
// concatenated transcript can itself exceed the summariser call's context window →
|
|
359
|
+
// the summarise request 400s → autoCompactMessages returns false → NO compaction
|
|
360
|
+
// happens exactly when the session is largest (the context-wall failure mode).
|
|
361
|
+
// ~600K chars ≈ 150K tokens, well under a 1M window even with prompt overhead.
|
|
362
|
+
const MAX_TRANSCRIPT_CHARS = 600000;
|
|
363
|
+
/**
|
|
364
|
+
* Render messages to a plain-text transcript for the summariser (tool noise
|
|
365
|
+
* truncated per-block AND the whole transcript hard-capped). When the transcript
|
|
366
|
+
* would exceed MAX_TRANSCRIPT_CHARS we keep the HEAD (original task + early
|
|
367
|
+
* decisions) and the TAIL (most-recent, highest-signal context) and drop the
|
|
368
|
+
* middle — a middle-out elision that preserves both "what we set out to do" and
|
|
369
|
+
* "where we are now", which is what the continuation summary needs most.
|
|
370
|
+
*/
|
|
371
|
+
/**
|
|
372
|
+
* Appends the backend-announced runtime-context block to the user message it was
|
|
373
|
+
* attached to. No-op if that message is not a user turn or already ends with the
|
|
374
|
+
* identical block. Exported for tests.
|
|
375
|
+
*/
|
|
376
|
+
function persistRuntimeContext(messages, idx, text) {
|
|
377
|
+
const m = messages[idx];
|
|
378
|
+
if (!m || m.role !== 'user' || !Array.isArray(m.content))
|
|
379
|
+
return false;
|
|
380
|
+
const lastBlock = m.content[m.content.length - 1];
|
|
381
|
+
if (lastBlock && lastBlock.type === 'text' && lastBlock.text === text)
|
|
382
|
+
return false;
|
|
383
|
+
messages[idx] = { ...m, content: [...m.content, { type: 'text', text }] };
|
|
384
|
+
return true;
|
|
385
|
+
}
|
|
386
|
+
function transcriptOf(messages) {
|
|
387
|
+
const parts = [];
|
|
388
|
+
for (const m of messages) {
|
|
389
|
+
for (const b of m.content) {
|
|
390
|
+
if ((0, types_1.isRuntimeContextBlock)(b))
|
|
391
|
+
continue; // editor scaffolding, not conversation
|
|
392
|
+
if (b.type === 'text' && b.text) {
|
|
393
|
+
parts.push(`${m.role.toUpperCase()}: ${(0, safeSlice_1.sliceSafeEnd)(b.text, 2000)}`);
|
|
394
|
+
}
|
|
395
|
+
else if (b.type === 'tool_use') {
|
|
396
|
+
parts.push(`${m.role.toUpperCase()} [tool: ${b.name}]: ${(0, safeSlice_1.sliceSafeEnd)(JSON.stringify(b.input ?? {}), 400)}`);
|
|
397
|
+
}
|
|
398
|
+
else if (b.type === 'tool_result') {
|
|
399
|
+
parts.push(`TOOL RESULT: ${(0, safeSlice_1.sliceSafeEnd)(String(b.content ?? ''), 600)}`);
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
const full = parts.join('\n');
|
|
404
|
+
if (full.length <= MAX_TRANSCRIPT_CHARS)
|
|
405
|
+
return full;
|
|
406
|
+
// Middle-out: keep 40% head, 60% tail (recent context is higher-signal for
|
|
407
|
+
// continuation). Slice on line boundaries so we don't cut a line in half.
|
|
408
|
+
const headBudget = Math.floor(MAX_TRANSCRIPT_CHARS * 0.4);
|
|
409
|
+
const tailBudget = MAX_TRANSCRIPT_CHARS - headBudget;
|
|
410
|
+
const head = (0, safeSlice_1.sliceSafeEnd)(full, headBudget);
|
|
411
|
+
const tail = (0, safeSlice_1.sliceSafeStart)(full, full.length - tailBudget);
|
|
412
|
+
const dropped = full.length - head.length - tail.length;
|
|
413
|
+
return `${head}\n\n[… ${dropped} chars of mid-session transcript elided to fit the summariser's context window …]\n\n${tail}`;
|
|
414
|
+
}
|
|
415
|
+
// ─── Structured progress ledger (GAP E) ────────────────────────────────────────
|
|
416
|
+
//
|
|
417
|
+
// The single biggest long-horizon failure mode (industry-wide "context rot") is
|
|
418
|
+
// that each auto-compaction summarises a transcript that ALREADY contains a prior
|
|
419
|
+
// summary → summary-of-summary → fidelity decays: the agent forgets which files it
|
|
420
|
+
// edited, whether tests passed, what's still open. Prose summarisation is inherently
|
|
421
|
+
// lossy and gets worse every round.
|
|
422
|
+
//
|
|
423
|
+
// Defence: maintain a DETERMINISTIC, append-only ledger of high-signal facts derived
|
|
424
|
+
// directly from tool calls — files created/edited (with count), commands verified
|
|
425
|
+
// (test/build/lint) and their pass/fail, and explicit open TODOs. This is built from
|
|
426
|
+
// structured tool data (NOT model output), so it is LOSSLESS and never degrades. We
|
|
427
|
+
// inject it VERBATIM into every compaction preamble, so no matter how many times the
|
|
428
|
+
// prose summary is re-summarised, the concrete "what changed / what's verified /
|
|
429
|
+
// what's left" facts survive intact across an arbitrarily long run.
|
|
430
|
+
const LEDGER_MAX_FILES = 60; // cap the file list so the preamble can't balloon
|
|
431
|
+
const LEDGER_MAX_NOTES = 20; // cap verification/among notes
|
|
432
|
+
function createLedger() {
|
|
433
|
+
return { filesTouched: new Map(), filesTouchedTotal: 0, verifications: [], testIntegrity: [], testIntegrityTotal: 0, epoch: 0 };
|
|
434
|
+
}
|
|
435
|
+
/** Record one tool call's effect on the ledger (deterministic, no model call). */
|
|
436
|
+
function ledgerRecord(ledger, toolName, input, ok, output, exitCode) {
|
|
437
|
+
if (exports.WRITE_TOOL_NAMES.has(toolName)) {
|
|
438
|
+
if (!ok)
|
|
439
|
+
return; // a FAILED write changed nothing — not a durable fact
|
|
440
|
+
// A successful source write advances the mutation epoch: any verification
|
|
441
|
+
// run after this point has different inputs than runs before it.
|
|
442
|
+
ledger.epoch += 1;
|
|
443
|
+
const p = typeof input?.path === 'string' ? input.path : undefined;
|
|
444
|
+
if (p) {
|
|
445
|
+
const prev = ledger.filesTouched.get(p);
|
|
446
|
+
if (!prev)
|
|
447
|
+
ledger.filesTouchedTotal++;
|
|
448
|
+
// DELETE before SET, so a re-touched path moves to the BACK of the insertion
|
|
449
|
+
// order. `Map.set` on an existing key keeps its ORIGINAL slot, which quietly
|
|
450
|
+
// broke the eviction policy below: a file edited hundreds of times over a long
|
|
451
|
+
// session kept the position of its FIRST edit, so it aged out like a file nobody
|
|
452
|
+
// had looked at since — and on the next edit it was re-inserted as "new", double-
|
|
453
|
+
// counting filesTouchedTotal (which is documented as DISTINCT paths). Making the
|
|
454
|
+
// Map a true LRU-by-touch is what lets the `key !== p` guard below mean anything.
|
|
455
|
+
ledger.filesTouched.delete(p);
|
|
456
|
+
ledger.filesTouched.set(p, { tool: toolName, edits: (prev?.edits ?? 0) + 1 });
|
|
457
|
+
// Bound the Map itself, not just its rendering. LEDGER_MAX_FILES caps how many
|
|
458
|
+
// paths the preamble PRINTS (see ledgerSummary's slice), but the Map was only ever
|
|
459
|
+
// written to — so a multi-hour run touching thousands of files grew it without
|
|
460
|
+
// limit, and it is deliberately retained across every compaction. Evict the
|
|
461
|
+
// least-recently-touched entries once we hold well beyond what can ever be
|
|
462
|
+
// displayed. Hysteresis (evict down to 2× only once we exceed 4×) keeps this an
|
|
463
|
+
// occasional bulk sweep instead of a delete on every single write.
|
|
464
|
+
if (ledger.filesTouched.size > LEDGER_MAX_FILES * 4) {
|
|
465
|
+
for (const key of ledger.filesTouched.keys()) {
|
|
466
|
+
if (ledger.filesTouched.size <= LEDGER_MAX_FILES * 2)
|
|
467
|
+
break;
|
|
468
|
+
if (key !== p)
|
|
469
|
+
ledger.filesTouched.delete(key);
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
// Reward-hacking guard: if this write WEAKENED a test file, record it so the
|
|
474
|
+
// signal survives compaction and can be surfaced before the agent finishes.
|
|
475
|
+
const reasons = [];
|
|
476
|
+
// write_file overwrites carry a marker computed by the executor (which had the
|
|
477
|
+
// prior on-disk content) — it detects REMOVED assertions/cases, not just
|
|
478
|
+
// additive skips/tautologies. Prefer it when present.
|
|
479
|
+
const markerReasons = toolName === 'write_file' ? (0, testIntegrity_1.decodeTestIntegrityMarker)(output) : [];
|
|
480
|
+
if (markerReasons.length) {
|
|
481
|
+
reasons.push(...markerReasons);
|
|
482
|
+
}
|
|
483
|
+
else {
|
|
484
|
+
const ti = (0, testIntegrity_1.analyzeWriteToolForTestIntegrity)(toolName, input);
|
|
485
|
+
if (ti?.suspicious)
|
|
486
|
+
reasons.push(...ti.findings.map((f) => f.reason));
|
|
487
|
+
}
|
|
488
|
+
if (reasons.length && p) {
|
|
489
|
+
for (const reason of reasons) {
|
|
490
|
+
ledger.testIntegrity.push({ path: p, reason });
|
|
491
|
+
ledger.testIntegrityTotal++;
|
|
492
|
+
}
|
|
493
|
+
if (ledger.testIntegrity.length > LEDGER_MAX_NOTES * 2) {
|
|
494
|
+
ledger.testIntegrity.splice(0, ledger.testIntegrity.length - LEDGER_MAX_NOTES);
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
else if (toolName === 'bash') {
|
|
499
|
+
const cmd = String(input?.command ?? '').trim();
|
|
500
|
+
if (cmd && exports.VERIFY_CMD_RE.test(cmd)) {
|
|
501
|
+
// Record BOTH outcomes: a FAILED test/build is the single most important
|
|
502
|
+
// fact to carry across a compaction (it tells the agent work is NOT done).
|
|
503
|
+
//
|
|
504
|
+
// CRITICAL: `ok` is `result.error === undefined`, which is TRUE even when a
|
|
505
|
+
// test suite exits non-zero (the executor doesn't set `error` for a plain
|
|
506
|
+
// command failure — only for timeout/abort/spawn-fail). So `ok` alone would
|
|
507
|
+
// record a FAILING `npm test` as PASSED. The executor now reports the real
|
|
508
|
+
// process exit code via `exitCode`; a non-zero exit means the verification
|
|
509
|
+
// FAILED regardless of `ok`. Fall back to `ok` only when no exitCode is
|
|
510
|
+
// available (older tools / non-bash paths).
|
|
511
|
+
const passed = exitCode !== undefined ? exitCode === 0 : ok;
|
|
512
|
+
ledger.verifications.push({ cmd: cmd.slice(0, 120), ok: passed, epoch: ledger.epoch });
|
|
513
|
+
if (ledger.verifications.length > LEDGER_MAX_NOTES * 2) {
|
|
514
|
+
ledger.verifications.splice(0, ledger.verifications.length - LEDGER_MAX_NOTES);
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
}
|
|
519
|
+
/** Render the ledger as a compact, verbatim block for the compaction preamble. */
|
|
520
|
+
function ledgerSummary(ledger) {
|
|
521
|
+
const lines = [];
|
|
522
|
+
if (ledger.filesTouched.size) {
|
|
523
|
+
const files = [...ledger.filesTouched.entries()];
|
|
524
|
+
// The TAIL, not the head: the Map is ordered least-recently-touched first, so
|
|
525
|
+
// slicing from the front showed the OLDEST files and reliably omitted the ones the
|
|
526
|
+
// agent was working on right now — the opposite of what this preamble is for.
|
|
527
|
+
const shown = files.slice(-LEDGER_MAX_FILES);
|
|
528
|
+
lines.push(`FILES CHANGED THIS SESSION (${ledger.filesTouchedTotal || ledger.filesTouched.size}):`);
|
|
529
|
+
for (const [p, meta] of shown) {
|
|
530
|
+
lines.push(` • ${p} (${meta.tool}${meta.edits > 1 ? ` ×${meta.edits}` : ''})`);
|
|
531
|
+
}
|
|
532
|
+
if (files.length > shown.length)
|
|
533
|
+
lines.push(` • … and ${files.length - shown.length} more`);
|
|
534
|
+
}
|
|
535
|
+
if (ledger.verifications.length) {
|
|
536
|
+
const recent = ledger.verifications.slice(-LEDGER_MAX_NOTES);
|
|
537
|
+
lines.push(`VERIFICATION RUNS (most recent ${recent.length}):`);
|
|
538
|
+
for (const v of recent)
|
|
539
|
+
lines.push(` • [${v.ok ? 'PASS' : 'FAIL'}] ${v.cmd}`);
|
|
540
|
+
}
|
|
541
|
+
if (ledger.testIntegrity.length) {
|
|
542
|
+
const recent = ledger.testIntegrity.slice(-LEDGER_MAX_NOTES);
|
|
543
|
+
lines.push(`⚠ TEST-INTEGRITY ALERTS (test files were weakened — must justify or revert):`);
|
|
544
|
+
for (const t of recent)
|
|
545
|
+
lines.push(` • ${t.path}: ${t.reason}`);
|
|
546
|
+
}
|
|
547
|
+
const flaky = (0, flaky_1.detectFlaky)(ledger.verifications);
|
|
548
|
+
if (flaky.length) {
|
|
549
|
+
lines.push(`⚠ FLAKY TESTS (same command flipped PASS↔FAIL with no edit between — a green run proves nothing):`);
|
|
550
|
+
for (const f of flaky.slice(0, LEDGER_MAX_NOTES)) {
|
|
551
|
+
lines.push(` • ${f.cmd} (${f.passes} pass / ${f.fails} fail at identical code)`);
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
return lines.join('\n');
|
|
555
|
+
}
|
|
556
|
+
// How many of the most-recent messages keep their tool_result content verbatim.
|
|
557
|
+
// Older tool_result bodies are the bulk of a large body and are the safest thing
|
|
558
|
+
// to shed first (the model has already acted on them), so we replace their content
|
|
559
|
+
// with a short stub while KEEPING the block (so tool_use/tool_result pairing and
|
|
560
|
+
// turn structure stay intact — unlike summarisation, which drops whole turns).
|
|
561
|
+
exports.PRUNE_KEEP_RECENT = 8;
|
|
562
|
+
const PRUNE_STUB_KEEP_CHARS = 400; // keep a short head of each pruned result for context
|
|
563
|
+
// Marker sentinel appended to a pruned tool_result's content. We detect
|
|
564
|
+
// "already pruned" by this suffix rather than by an out-of-schema field on the
|
|
565
|
+
// block, because the block object is serialised verbatim onto the request body
|
|
566
|
+
// and forwarded to Anthropic — any extra property (e.g. a `_pruned` flag) would
|
|
567
|
+
// be rejected as an unknown field on a content block (400). Encoding the state
|
|
568
|
+
// inside the (string) content keeps the wire payload schema-clean AND idempotent.
|
|
569
|
+
const PRUNE_MARKER = '\n\n[… ';
|
|
570
|
+
const PRUNE_MARKER_TAIL = ' pruned to conserve context. Re-run the tool if you need the full result.]';
|
|
571
|
+
/**
|
|
572
|
+
* Lossy-but-structure-preserving prune: shrink OLD, large tool_result blocks in
|
|
573
|
+
* place, keeping the last PRUNE_KEEP_RECENT messages untouched. This is tried
|
|
574
|
+
* BEFORE summarisation because it:
|
|
575
|
+
* • keeps every turn and every tool_use/tool_result pair (API stays valid),
|
|
576
|
+
* • never makes an extra model call (summarisation does — cost + latency),
|
|
577
|
+
* • degrades gracefully on repeat (summarise-of-summarise loses the most on
|
|
578
|
+
* long runs; pruning just trims already-consumed output further).
|
|
579
|
+
*
|
|
580
|
+
* IMPORTANT: pruned state is encoded in the content string (PRUNE_MARKER_TAIL
|
|
581
|
+
* suffix), NOT as an extra property on the block — a stray field on a content
|
|
582
|
+
* block is rejected by the Anthropic API as an unknown key (400). This keeps the
|
|
583
|
+
* serialised body schema-clean while remaining idempotent across repeat calls.
|
|
584
|
+
*
|
|
585
|
+
* Returns the number of bytes reclaimed (0 if nothing was prunable).
|
|
586
|
+
*
|
|
587
|
+
* `minReclaimBytes` (default 0): if the TOTAL prunable amount is below this, the
|
|
588
|
+
* function makes NO changes and returns 0. This is a cache-safety gate — pruning
|
|
589
|
+
* even one old block changes the request prefix and invalidates the message-level
|
|
590
|
+
* prompt cache, so a tiny prune would bust the cache (re-write at 1.25×) for
|
|
591
|
+
* almost no size win. Measuring first, then applying only if worthwhile, keeps
|
|
592
|
+
* the "don't bust cache for a trivial gain" contract truly atomic (the old code
|
|
593
|
+
* mutated first and let the caller decide, which had already invalidated the
|
|
594
|
+
* cache by the time the caller declined).
|
|
595
|
+
*/
|
|
596
|
+
function pruneOldToolResults(messages, minReclaimBytes = 0) {
|
|
597
|
+
const cutoff = messages.length - exports.PRUNE_KEEP_RECENT;
|
|
598
|
+
if (cutoff <= 1)
|
|
599
|
+
return 0;
|
|
600
|
+
// Collect prunable blocks + measure the total reclaim WITHOUT mutating yet.
|
|
601
|
+
const targets = [];
|
|
602
|
+
let total = 0;
|
|
603
|
+
for (let i = 0; i < cutoff; i++) {
|
|
604
|
+
const m = messages[i];
|
|
605
|
+
if (!Array.isArray(m.content))
|
|
606
|
+
continue;
|
|
607
|
+
for (const b of m.content) {
|
|
608
|
+
if (b.type !== 'tool_result')
|
|
609
|
+
continue;
|
|
610
|
+
const text = typeof b.content === 'string' ? b.content : JSON.stringify(b.content ?? '');
|
|
611
|
+
if (text.endsWith(PRUNE_MARKER_TAIL))
|
|
612
|
+
continue; // already pruned (idempotent)
|
|
613
|
+
if (text.length <= PRUNE_STUB_KEEP_CHARS + 80)
|
|
614
|
+
continue; // already small
|
|
615
|
+
// MUST use the surrogate-safe slice: a raw `text.slice(0, N)` landing
|
|
616
|
+
// between the high/low half of an emoji or CJK-extension glyph leaves a
|
|
617
|
+
// lone surrogate in the stub. That string is still valid JS but breaks
|
|
618
|
+
// when JSON.stringify'd onto the wire — Anthropic rejects the WHOLE
|
|
619
|
+
// request with a deterministic 400 "no low surrogate in string" that
|
|
620
|
+
// repeats identically on every retry (the corrupted payload never
|
|
621
|
+
// changes). This is exactly the class of bug util/safeSlice.ts exists
|
|
622
|
+
// to prevent; this call site just never got migrated to it.
|
|
623
|
+
const head = (0, safeSlice_1.sliceSafeEnd)(text, PRUNE_STUB_KEEP_CHARS);
|
|
624
|
+
const omitted = text.length - head.length;
|
|
625
|
+
targets.push({ block: b, head, omitted });
|
|
626
|
+
total += omitted;
|
|
627
|
+
}
|
|
628
|
+
}
|
|
629
|
+
// Cache-safety gate: not worth busting the prompt cache for a trivial reclaim.
|
|
630
|
+
if (total < minReclaimBytes)
|
|
631
|
+
return 0;
|
|
632
|
+
// Worthwhile — apply the stubs.
|
|
633
|
+
let reclaimed = 0;
|
|
634
|
+
for (const { block, head, omitted } of targets) {
|
|
635
|
+
block.content = `${head}${PRUNE_MARKER}${omitted} chars of earlier tool output${PRUNE_MARKER_TAIL}`;
|
|
636
|
+
reclaimed += omitted;
|
|
637
|
+
}
|
|
638
|
+
return reclaimed;
|
|
639
|
+
}
|
|
640
|
+
/**
|
|
641
|
+
* Compact `messages` in place: summarise everything before a safe cut point and
|
|
642
|
+
* replace it with a summary preamble. Returns true if compaction happened.
|
|
643
|
+
*/
|
|
644
|
+
/** Extract the first user turn's plain text — the ORIGINAL task/goal. */
|
|
645
|
+
function originalTaskText(messages) {
|
|
646
|
+
const first = messages.find((m) => m.role === 'user');
|
|
647
|
+
if (!first || !Array.isArray(first.content))
|
|
648
|
+
return '';
|
|
649
|
+
return first.content
|
|
650
|
+
.filter((b) => b.type === 'text' && b.text && !(0, types_1.isRuntimeContextBlock)(b))
|
|
651
|
+
.map((b) => b.text)
|
|
652
|
+
.join('\n')
|
|
653
|
+
.trim();
|
|
654
|
+
}
|
|
655
|
+
/**
|
|
656
|
+
* The summariser's own model call failed — as opposed to running fine but not
|
|
657
|
+
* shrinking anything.
|
|
658
|
+
*
|
|
659
|
+
* These two outcomes used to be one `return false`, and collapsing them was a
|
|
660
|
+
* real bug: the in-loop circuit breaker disables auto-compaction permanently
|
|
661
|
+
* after COMPACT_MAX_FAILURES, on the sound theory that a compaction which cannot
|
|
662
|
+
* reclaim bytes will never start. But a summariser that THREW says nothing about
|
|
663
|
+
* whether compaction would help — only that the network/provider was unavailable
|
|
664
|
+
* for a moment. Feeding those into the same counter meant three transient blips
|
|
665
|
+
* (a 529 burst, a brief outage, a rate-limit spike) permanently switched off the
|
|
666
|
+
* one mechanism keeping the context under control, and the run then died at the
|
|
667
|
+
* context wall minutes later with its own recovery already disabled.
|
|
668
|
+
*/
|
|
669
|
+
class CompactionUnavailableError extends Error {
|
|
670
|
+
constructor() {
|
|
671
|
+
super('Compaction summariser was unavailable');
|
|
672
|
+
this.name = 'CompactionUnavailableError';
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
exports.CompactionUnavailableError = CompactionUnavailableError;
|
|
676
|
+
function makeCachedSummarizer(messages, requestOptions, abortSignal) {
|
|
677
|
+
return async (instruction) => {
|
|
678
|
+
const last = messages[messages.length - 1];
|
|
679
|
+
if (!last || last.role !== 'user')
|
|
680
|
+
return '';
|
|
681
|
+
const lastContent = typeof last.content === 'string'
|
|
682
|
+
? [{ type: 'text', text: last.content }]
|
|
683
|
+
: [...last.content];
|
|
684
|
+
const probe = [
|
|
685
|
+
...messages.slice(0, -1),
|
|
686
|
+
{ ...last, content: [...lastContent, { type: 'text', text: instruction }] },
|
|
687
|
+
];
|
|
688
|
+
const reply = await (0, client_1.streamChat)(probe, { ...requestOptions(), abortSignal, allowRestartAfterRender: true }, () => { });
|
|
689
|
+
return reply.content
|
|
690
|
+
.filter((b) => b.type === 'text')
|
|
691
|
+
.map((b) => b.text ?? '')
|
|
692
|
+
.join('')
|
|
693
|
+
.trim();
|
|
694
|
+
};
|
|
695
|
+
}
|
|
696
|
+
const CACHED_SUMMARY_INSTRUCTION = `[Context compaction — this is an automated request from the agent runtime, not the user.]\n` +
|
|
697
|
+
`Do NOT call any tools and do NOT continue the task. Reply with text only: a concise bullet-point ` +
|
|
698
|
+
`summary of this whole session so far that you will need to continue the work — the user's goal, ` +
|
|
699
|
+
`key decisions, files changed (and how), commands run and their outcome, unresolved problems, and ` +
|
|
700
|
+
`user preferences. Max 400 words.`;
|
|
701
|
+
async function autoCompactMessages(messages, options, ledger, summarizeCached) {
|
|
702
|
+
const cut = findSafeCutIndex(messages, messages.length - exports.COMPACT_KEEP_MIN);
|
|
703
|
+
if (cut < 2)
|
|
704
|
+
return false; // nothing meaningful to fold
|
|
705
|
+
// PreCompact: only now that compaction will really happen (the early return above would
|
|
706
|
+
// have made the hook fire for a no-op). Main agent only; an observer, it cannot veto.
|
|
707
|
+
if ((options._depth ?? 0) === 0) {
|
|
708
|
+
await (0, hooks_1.runLifecycleHooks)((0, hooks_1.loadHooks)(options.workDir).PreCompact, 'PreCompact', options.workDir, { trigger: 'auto', session_id: options.sessionId ?? '', messages_to_summarize: cut }, 'auto', (0, hooks_1.hookRunOptsFor)(options)).catch(() => undefined);
|
|
709
|
+
}
|
|
710
|
+
// Sizes measured here, not in firePostCompact: by then `messages` has already
|
|
711
|
+
// been spliced and the "before" number no longer exists.
|
|
712
|
+
const beforeCompact = messages.length;
|
|
713
|
+
const afterCompact = cut > 0 ? messages.length - cut + 1 : messages.length;
|
|
714
|
+
const toSummarize = messages.slice(0, cut);
|
|
715
|
+
const kept = messages.slice(cut);
|
|
716
|
+
const firePostCompact = async () => {
|
|
717
|
+
if ((options._depth ?? 0) !== 0)
|
|
718
|
+
return;
|
|
719
|
+
// Host callback: the loop rewrote the context, so the caller's own trail
|
|
720
|
+
// (event log, ledger, telemetry) must show it.
|
|
721
|
+
try {
|
|
722
|
+
options.onCompact?.({ trigger: 'auto', before: beforeCompact, after: afterCompact });
|
|
723
|
+
}
|
|
724
|
+
catch { /* a host callback must never break the run */ }
|
|
725
|
+
await (0, hooks_1.runLifecycleHooks)((0, hooks_1.loadHooks)(options.workDir).PostCompact, 'PostCompact', options.workDir, { trigger: 'auto', session_id: options.sessionId ?? '', messages_summarized: cut }, 'auto', (0, hooks_1.hookRunOptsFor)(options)).catch(() => undefined);
|
|
726
|
+
};
|
|
727
|
+
// Pin the ORIGINAL task verbatim. findSafeCutIndex can (and on a long single
|
|
728
|
+
// run usually does) cut PAST the first user turn, folding the user's actual
|
|
729
|
+
// goal into the lossy summary — after a few compactions the agent drifts off
|
|
730
|
+
// what it was asked to do. We re-inject the first user turn's text verbatim
|
|
731
|
+
// into the replacement preamble so the objective survives every compaction.
|
|
732
|
+
// (We cannot keep it as a separate user message: the API requires alternating
|
|
733
|
+
// roles and kept[0] is already an assistant turn — two user turns would 400.)
|
|
734
|
+
const originalTask = originalTaskText(toSummarize);
|
|
735
|
+
const summaryPrompt = `Summarize this coding-session transcript into concise bullet points the assistant needs to continue the work: ` +
|
|
736
|
+
`key decisions, files changed (and how), commands run, unresolved problems, and user preferences. Max 400 words.\n\n` +
|
|
737
|
+
transcriptOf(toSummarize);
|
|
738
|
+
let summary = '';
|
|
739
|
+
// Cache-friendly path first; any failure or empty reply falls back to the standalone
|
|
740
|
+
// transcript summariser below, which always works but pays for the history uncached.
|
|
741
|
+
if (summarizeCached) {
|
|
742
|
+
try {
|
|
743
|
+
summary = await summarizeCached(CACHED_SUMMARY_INSTRUCTION);
|
|
744
|
+
}
|
|
745
|
+
catch (err) {
|
|
746
|
+
if (options.abortSignal?.aborted || err.name === 'AbortError')
|
|
747
|
+
throw err;
|
|
748
|
+
summary = '';
|
|
749
|
+
}
|
|
750
|
+
}
|
|
751
|
+
if (!summary)
|
|
752
|
+
try {
|
|
753
|
+
const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
|
|
754
|
+
// Run on the SAME model as the actual conversation. A previous version
|
|
755
|
+
// forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
|
|
756
|
+
// "bullet-point this transcript" task doesn't need the user's tier — but
|
|
757
|
+
// that silently billed Anthropic (and made a real network call to a
|
|
758
|
+
// provider the user may not have configured/paid for) even when the
|
|
759
|
+
// whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
|
|
760
|
+
// is already paying for is used for compaction too, so there is never a
|
|
761
|
+
// surprise charge on a different provider. Falls back to the same
|
|
762
|
+
// default as the main loop (see `runAgentLoop`) only when no model was
|
|
763
|
+
// set at all.
|
|
764
|
+
//
|
|
765
|
+
// Cheapest model of the SAME vendor (Claude Code runs this kind of work on
|
|
766
|
+
// Haiku). Safe to downgrade HERE because this fallback sends a fresh
|
|
767
|
+
// transcript — it shares no cached prefix with the conversation, unlike
|
|
768
|
+
// summarizeCached above, which must stay on the conversation's own model.
|
|
769
|
+
// Never crosses providers; falls back to the session model if the
|
|
770
|
+
// catalogue is not loaded.
|
|
771
|
+
model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
|
|
772
|
+
mode: 'ask', // summariser must not call tools; ask-mode discourages action
|
|
773
|
+
env: options.env,
|
|
774
|
+
clientType: options.clientType,
|
|
775
|
+
abortSignal: options.abortSignal,
|
|
776
|
+
// The summariser renders NOTHING (onEvent below is a no-op) and its result is
|
|
777
|
+
// read only from the returned message, so a restart has nothing to roll back —
|
|
778
|
+
// always safe. Worth enabling: a blip here used to abandon compaction entirely,
|
|
779
|
+
// which then let the very next turn hit the context wall it was meant to prevent.
|
|
780
|
+
allowRestartAfterRender: true,
|
|
781
|
+
}, () => { });
|
|
782
|
+
summary = reply.content
|
|
783
|
+
.filter((b) => b.type === 'text')
|
|
784
|
+
.map((b) => b.text ?? '')
|
|
785
|
+
.join('')
|
|
786
|
+
.trim();
|
|
787
|
+
}
|
|
788
|
+
catch {
|
|
789
|
+
// Summarisation FAILED — the model call itself threw (network blip, 529,
|
|
790
|
+
// provider quota). Distinguished from "ran fine but didn't help" by the
|
|
791
|
+
// caller, because the two must not feed the same circuit breaker: three
|
|
792
|
+
// transient network errors would otherwise permanently disable compaction
|
|
793
|
+
// for the rest of the run, leaving the context to grow until the turn dies
|
|
794
|
+
// with no recovery left. Leave history as is; the turn may still fit.
|
|
795
|
+
throw new CompactionUnavailableError();
|
|
796
|
+
}
|
|
797
|
+
if (!summary)
|
|
798
|
+
return false;
|
|
799
|
+
// Replace the summarized head with a single user summary message. The cut is
|
|
800
|
+
// at a turn boundary (kept[0] is an assistant message — see findSafeCutIndex),
|
|
801
|
+
// so `user(summary) → assistant(kept[0])` is a valid, well-ordered sequence
|
|
802
|
+
// and no orphaned tool_result is left behind. We intentionally do NOT insert
|
|
803
|
+
// an assistant-ack here: that would put two assistant messages back-to-back
|
|
804
|
+
// (kept[0] is already an assistant), which the API rejects.
|
|
805
|
+
const taskBlock = originalTask
|
|
806
|
+
? `ORIGINAL TASK (verbatim — keep working toward this, do not lose sight of it):\n${originalTask}\n\n`
|
|
807
|
+
: '';
|
|
808
|
+
// GAP E — the deterministic ledger (files changed + verification pass/fail) is
|
|
809
|
+
// injected VERBATIM, so these concrete facts never decay through repeated
|
|
810
|
+
// summary-of-summary compactions the way the prose summary does.
|
|
811
|
+
const ledgerText = ledger ? ledgerSummary(ledger) : '';
|
|
812
|
+
const ledgerBlock = ledgerText
|
|
813
|
+
? `PROGRESS LEDGER (authoritative, machine-tracked — trust this over the prose summary for what changed/verified):\n${ledgerText}\n\n`
|
|
814
|
+
: '';
|
|
815
|
+
messages.splice(0, cut, { role: 'user', content: [{ type: 'text', text: `[Auto-compacted ${toSummarize.length} earlier messages]\n\n${taskBlock}${ledgerBlock}Summary of the earlier conversation so far:\n${summary}\n\nContinue the work from here.` }] });
|
|
816
|
+
// `kept` follows automatically since splice only replaced the head.
|
|
817
|
+
void kept;
|
|
818
|
+
await firePostCompact();
|
|
819
|
+
return true;
|
|
820
|
+
}
|
|
821
|
+
// ─── Periodic memory compaction ────────────────────────────────────────────
|
|
822
|
+
// A memory file (project or global — see agent/memory.ts) can grow large over
|
|
823
|
+
// many sessions since memory_write only ever appends. writeMemory() already
|
|
824
|
+
// applies an immediate, synchronous byte-cap eviction (oldest entries dropped)
|
|
825
|
+
// as a hard backstop, but that's a blunt instrument — this periodically
|
|
826
|
+
// consolidates the file with a real LLM summarization pass instead, so old
|
|
827
|
+
// facts are condensed into fewer, denser bullets rather than silently lost.
|
|
828
|
+
// Checked opportunistically right after a successful memory_write (see the
|
|
829
|
+
// call site below) rather than on every tool call — cheap to check (a single
|
|
830
|
+
// file stat), and memory_write is the only thing that can push a file over
|
|
831
|
+
// the trigger threshold in the first place.
|
|
832
|
+
async function maybeCompactMemory(scope, options) {
|
|
833
|
+
try {
|
|
834
|
+
await (0, memory_1.compactMemoryIfNeeded)(scope, options.workDir, async (prompt) => {
|
|
835
|
+
const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: prompt }] }], {
|
|
836
|
+
// Same reasoning as autoCompactMessages' summariser: run on the same
|
|
837
|
+
// model as the actual conversation rather than forcing a fixed tier
|
|
838
|
+
// (which silently billed Anthropic regardless of the user's provider).
|
|
839
|
+
// Cheapest SAME-vendor model: a fresh prompt with no shared cache prefix,
|
|
840
|
+
// and merging a few memory bullets does not need the session's top tier.
|
|
841
|
+
model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
|
|
842
|
+
mode: 'ask',
|
|
843
|
+
env: options.env,
|
|
844
|
+
clientType: options.clientType,
|
|
845
|
+
abortSignal: options.abortSignal,
|
|
846
|
+
// Same as the transcript summariser: no rendered output, result read only from
|
|
847
|
+
// the returned message, so restarting on a blip is always safe.
|
|
848
|
+
allowRestartAfterRender: true,
|
|
849
|
+
}, () => { });
|
|
850
|
+
return reply.content
|
|
851
|
+
.filter((b) => b.type === 'text')
|
|
852
|
+
.map((b) => b.text ?? '')
|
|
853
|
+
.join('');
|
|
854
|
+
});
|
|
855
|
+
}
|
|
856
|
+
catch {
|
|
857
|
+
// Best-effort — a failed/aborted compaction just means the file stays as-is
|
|
858
|
+
// until the next memory_write call tries again; writeMemory's synchronous
|
|
859
|
+
// byte cap already bounds worst-case growth in the meantime.
|
|
860
|
+
}
|
|
861
|
+
}
|
|
862
|
+
// ─── Resume-time proactive compaction ─────────────────────────────────────────
|
|
863
|
+
//
|
|
864
|
+
// The in-loop auto-compact above only reacts to `lastPromptTokens`, which is
|
|
865
|
+
// populated from the PREVIOUS turn's usage event. On a freshly-resumed session
|
|
866
|
+
// (opening an old chat from history and sending the first new message) there is
|
|
867
|
+
// no previous turn in this process — `lastPromptTokens` starts at 0 — so the
|
|
868
|
+
// token-pressure trigger never fires for turn 0, and the byte-pressure trigger
|
|
869
|
+
// only catches truly huge sessions (MAX_BODY_BYTES is sized to stay under the
|
|
870
|
+
// backend's 25 MB body limit, not to bound cost — 8 MB of tool-heavy JSON is
|
|
871
|
+
// already on the order of the 1M-token context window itself). The result: a
|
|
872
|
+
// resumed session comfortably under both guards, but still hundreds of
|
|
873
|
+
// thousands of tokens, gets sent to the model at FULL PRICE on the very first
|
|
874
|
+
// message after resume, silently, every time.
|
|
875
|
+
//
|
|
876
|
+
// This function closes that gap: call it once, right after loading a stored
|
|
877
|
+
// session and BEFORE the user's next message is sent, so the expensive
|
|
878
|
+
// resend is compacted proactively instead of being missed by both in-loop
|
|
879
|
+
// guards. It reuses the exact same threshold/mechanics as the in-loop guard
|
|
880
|
+
// (cheap prune first, then summarising compaction) so behaviour stays
|
|
881
|
+
// consistent whether compaction happens at resume-time or mid-run.
|
|
882
|
+
const RESUME_CHARS_PER_TOKEN = 4; // rough, conservative estimate for JSON/code-heavy transcripts
|
|
883
|
+
/** Rough token estimate for a resumed transcript — no API round-trip needed. */
|
|
884
|
+
function estimateTokensRough(messages) {
|
|
885
|
+
return Math.ceil(estimateBodyBytes(messages) / RESUME_CHARS_PER_TOKEN);
|
|
886
|
+
}
|
|
887
|
+
/**
|
|
888
|
+
* Proactively compact `messages` in place if resuming this session would blow
|
|
889
|
+
* past the auto-compact threshold on the very first turn. Returns true if any
|
|
890
|
+
* compaction happened (so the caller can surface a one-line notice to the
|
|
891
|
+
* user). Safe to call on any message array, including empty/small ones (no-op).
|
|
892
|
+
*
|
|
893
|
+
* `model` picks the right context window (mirrors runAgentLoop's own lookup);
|
|
894
|
+
* `onNotice` is optional — pass it to show the same "auto-compacted" message
|
|
895
|
+
* the in-loop path shows, so the behaviour is visually consistent.
|
|
896
|
+
*/
|
|
897
|
+
async function compactMessagesForResume(messages, opts) {
|
|
898
|
+
if (messages.length <= exports.COMPACT_KEEP_MIN + 2)
|
|
899
|
+
return false;
|
|
900
|
+
const settings = (0, rules_1.loadSettings)(opts.workDir);
|
|
901
|
+
if (!resolveAutoCompact(undefined, settings.raw))
|
|
902
|
+
return false;
|
|
903
|
+
// Live catalogue first (see contextWindowFor's doc comment above for why),
|
|
904
|
+
// same fallback chain this call site always used otherwise.
|
|
905
|
+
const contextWindow = (0, modelCatalogue_1.liveContextWindowFor)(opts.model, exports.MODEL_CONTEXT_TOKENS[opts.model ?? 'turbo'] ?? 1000000);
|
|
906
|
+
let bodyBytes = estimateBodyBytes(messages);
|
|
907
|
+
let tokenGuess = estimateTokensRough(messages);
|
|
908
|
+
// Prune fires at the EARLY threshold (mirrors the in-loop guard); summarisation
|
|
909
|
+
// only at the late one. On resume this matters most: a stored session is resent
|
|
910
|
+
// whole on the first turn, so shedding old tool_result bulk up front is exactly
|
|
911
|
+
// what stops that first message being billed at full size.
|
|
912
|
+
const limits = compactionLimits(contextWindow, settings.raw);
|
|
913
|
+
const overPruneThreshold = () => tokenGuess > limits.prune || bodyBytes > exports.MAX_BODY_BYTES;
|
|
914
|
+
const overCompactThreshold = () => tokenGuess > limits.compact || bodyBytes > exports.MAX_BODY_BYTES;
|
|
915
|
+
if (!overPruneThreshold())
|
|
916
|
+
return false;
|
|
917
|
+
let compacted = false;
|
|
918
|
+
// Cheap pass first — shrinks old tool_result blocks with no model call. The
|
|
919
|
+
// reclaim floor is enforced atomically inside pruneOldToolResults (measures
|
|
920
|
+
// first, mutates only if worthwhile), so a declined prune leaves the cache intact.
|
|
921
|
+
if (messages.length > exports.PRUNE_KEEP_RECENT + 2) {
|
|
922
|
+
const reclaimed = pruneOldToolResults(messages, exports.PRUNE_MIN_RECLAIM_BYTES);
|
|
923
|
+
if (reclaimed > 0) {
|
|
924
|
+
bodyBytes = estimateBodyBytes(messages);
|
|
925
|
+
tokenGuess = estimateTokensRough(messages);
|
|
926
|
+
compacted = true;
|
|
927
|
+
opts.onNotice?.(`\n\u267b\ufe0f Trimmed ~${(reclaimed / (1024 * 1024)).toFixed(1)}MB of older tool output before resuming this chat.\n`);
|
|
928
|
+
}
|
|
929
|
+
}
|
|
930
|
+
// If still over the LATE (summarise) threshold, fall through to summarising
|
|
931
|
+
// compaction — same mechanism the in-loop guard uses, so this can safely loop
|
|
932
|
+
// (a single summarisation pass may still leave a very long session over it).
|
|
933
|
+
// A session between the prune and summarise thresholds is left as-is after the
|
|
934
|
+
// cheap prune: no model call needed, prefix already shrunk.
|
|
935
|
+
let guard = 0;
|
|
936
|
+
while (overCompactThreshold() && messages.length > exports.COMPACT_KEEP_MIN + 2 && guard < 5) {
|
|
937
|
+
guard += 1;
|
|
938
|
+
// No ledger at resume time — the ledger is per-run, in-memory, and would
|
|
939
|
+
// have been created fresh anyway since this is a new process/run. The
|
|
940
|
+
// ORIGINAL TASK verbatim pin (inside autoCompactMessages) still applies.
|
|
941
|
+
// A summariser failure is not fatal HERE. This runs before the session is
|
|
942
|
+
// handed back to the user, so the worst case is resuming with a longer (more
|
|
943
|
+
// expensive) prefix — strictly better than refusing to resume at all. The
|
|
944
|
+
// in-loop caller treats the same signal differently, because there it must
|
|
945
|
+
// decide whether to arm a circuit breaker.
|
|
946
|
+
let did;
|
|
947
|
+
try {
|
|
948
|
+
did = await autoCompactMessages(messages, {
|
|
949
|
+
workDir: opts.workDir,
|
|
950
|
+
model: opts.model,
|
|
951
|
+
clientType: opts.clientType,
|
|
952
|
+
env: opts.env,
|
|
953
|
+
onText: () => { },
|
|
954
|
+
onToolUse: () => { },
|
|
955
|
+
onToolResult: () => { },
|
|
956
|
+
onUsage: () => { },
|
|
957
|
+
requestPermission: async () => false,
|
|
958
|
+
});
|
|
959
|
+
}
|
|
960
|
+
catch (err) {
|
|
961
|
+
if (err instanceof CompactionUnavailableError)
|
|
962
|
+
break;
|
|
963
|
+
throw err;
|
|
964
|
+
}
|
|
965
|
+
if (!did)
|
|
966
|
+
break;
|
|
967
|
+
compacted = true;
|
|
968
|
+
bodyBytes = estimateBodyBytes(messages);
|
|
969
|
+
tokenGuess = estimateTokensRough(messages);
|
|
970
|
+
}
|
|
971
|
+
if (compacted) {
|
|
972
|
+
opts.onNotice?.(`\n\u267b\ufe0f Auto-compacted this chat's earlier history before resuming, to avoid resending it at full cost.\n`);
|
|
973
|
+
}
|
|
974
|
+
return compacted;
|
|
975
|
+
}
|
|
976
|
+
//# sourceMappingURL=compaction.js.map
|