@myagentroam/agent 0.9.64 → 0.9.66
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/main.js +13 -9
- package/dist/model/openai-responses.js +26 -5
- package/dist/prompts/output.d.ts +1 -0
- package/dist/prompts/output.js +1 -0
- package/dist/runtime/context-gc-runtime.js +47 -10
- package/dist/runtime/context-gc.d.ts +8 -1
- package/dist/runtime/context-gc.js +170 -16
- package/dist/runtime/context-projection.js +3 -1
- package/dist/sdk/agent.js +111 -39
- package/dist/sdk/types.d.ts +1 -0
- package/dist/session/context-gc-checkpoint.d.ts +9 -1
- package/dist/session/context-gc-checkpoint.js +130 -12
- package/dist/session/context-gc-replacements.d.ts +6 -3
- package/dist/session/context-gc-replacements.js +75 -15
- package/dist/session/context-gc-subagent.d.ts +12 -0
- package/dist/session/context-gc-subagent.js +79 -7
- package/dist/session/jsonl-store.d.ts +15 -0
- package/dist/session/jsonl-store.js +15 -0
- package/dist/subagent/scheduler.d.ts +14 -1
- package/dist/subagent/scheduler.js +46 -3
- package/dist/subagent/session-controller.d.ts +4 -1
- package/dist/subagent/session-controller.js +118 -22
- package/dist/tools/agent-wait.js +12 -8
- package/dist/tools/execution-limits.d.ts +1 -0
- package/dist/tools/execution-limits.js +1 -0
- package/dist/tools/view-image.js +1 -1
- package/package.json +1 -1
package/dist/cli/main.js
CHANGED
|
@@ -226,8 +226,7 @@ async function main() {
|
|
|
226
226
|
stdout.write(HELP);
|
|
227
227
|
return;
|
|
228
228
|
}
|
|
229
|
-
|
|
230
|
-
if (command === 'models') {
|
|
229
|
+
if (arguments_[0] === 'models') {
|
|
231
230
|
await modelsCommand(home, arguments_, codexAuthFile);
|
|
232
231
|
return;
|
|
233
232
|
}
|
|
@@ -240,6 +239,7 @@ async function main() {
|
|
|
240
239
|
: parseReasoningEffort(reasoningEffortOption, '--reasoning-effort');
|
|
241
240
|
const access = (takeOption(arguments_, '--access') ?? 'on-request');
|
|
242
241
|
const format = takeOption(arguments_, '--format') ?? 'text';
|
|
242
|
+
const command = arguments_[0];
|
|
243
243
|
if (access !== 'on-request' && access !== 'full-access')
|
|
244
244
|
throw new MarAgentError('MAR_AGENT_CLI_ARGUMENT_INVALID', 'Access must be on-request or full-access.');
|
|
245
245
|
const selectedModel = catalog.models.find((model) => model.id === modelId && model.enabled);
|
|
@@ -249,14 +249,18 @@ async function main() {
|
|
|
249
249
|
throw new MarAgentError('MAR_AGENT_CLI_ARGUMENT_INVALID', `Reasoning effort "${reasoningEffort}" is not supported by model "${selectedModel.id}".`);
|
|
250
250
|
const interactive = command !== 'run';
|
|
251
251
|
let prompt = command === 'run' ? arguments_.slice(1).join(' ') : '';
|
|
252
|
-
|
|
252
|
+
// `run` does not need stdin until an approval or question occurs. Create
|
|
253
|
+
// readline lazily so a closed or pseudo-TTY stream in an external harness
|
|
254
|
+
// cannot fail the execution before its first model request.
|
|
255
|
+
let terminal;
|
|
256
|
+
const requireTerminal = () => (terminal ??= createInterface({ input: stdin, output: stdout }));
|
|
253
257
|
const host = new LocalAgentHost({
|
|
254
258
|
homeDirectory: home,
|
|
255
259
|
workspace,
|
|
256
260
|
approve: async (call) => {
|
|
257
261
|
if (!stdin.isTTY)
|
|
258
262
|
return 'denied';
|
|
259
|
-
const answer = await
|
|
263
|
+
const answer = await requireTerminal().question(`Approve ${call.name}? [y/N] `);
|
|
260
264
|
return /^y(es)?$/i.test(answer.trim()) ? 'approved' : 'denied';
|
|
261
265
|
},
|
|
262
266
|
question: async (request) => {
|
|
@@ -266,7 +270,7 @@ async function main() {
|
|
|
266
270
|
for (const question of request.questions) {
|
|
267
271
|
const value = question;
|
|
268
272
|
const id = typeof value.id === 'string' ? value.id : String(Object.keys(answers).length);
|
|
269
|
-
answers[id] = await
|
|
273
|
+
answers[id] = await requireTerminal().question(`${typeof value.question === 'string' ? value.question : JSON.stringify(question)} `);
|
|
270
274
|
}
|
|
271
275
|
return answers;
|
|
272
276
|
}
|
|
@@ -298,7 +302,7 @@ async function main() {
|
|
|
298
302
|
const listed = (await agent.listSessions());
|
|
299
303
|
for (const item of listed.sessions ?? [])
|
|
300
304
|
stdout.write(`${item.id}\t${item.title}\t${item.lastActivityAt}\n`);
|
|
301
|
-
resumeId = (await
|
|
305
|
+
resumeId = (await requireTerminal().question('Session ID: ')).trim();
|
|
302
306
|
if (!resumeId)
|
|
303
307
|
throw new MarAgentError('MAR_AGENT_CLI_ARGUMENT_INVALID', 'Session id is required.');
|
|
304
308
|
}
|
|
@@ -314,7 +318,7 @@ async function main() {
|
|
|
314
318
|
if (activeHandle)
|
|
315
319
|
activeHandle.cancel();
|
|
316
320
|
else
|
|
317
|
-
terminal
|
|
321
|
+
terminal?.close();
|
|
318
322
|
process.exitCode = 130;
|
|
319
323
|
if (interrupts > 1)
|
|
320
324
|
activeHandle?.cancel();
|
|
@@ -323,7 +327,7 @@ async function main() {
|
|
|
323
327
|
try {
|
|
324
328
|
do {
|
|
325
329
|
if (interactive)
|
|
326
|
-
prompt = await
|
|
330
|
+
prompt = await requireTerminal().question('> ');
|
|
327
331
|
if (!prompt.trim())
|
|
328
332
|
break;
|
|
329
333
|
const invocation = parseExecutionInvocation(prompt);
|
|
@@ -360,7 +364,7 @@ async function main() {
|
|
|
360
364
|
}
|
|
361
365
|
}
|
|
362
366
|
finally {
|
|
363
|
-
terminal
|
|
367
|
+
terminal?.close();
|
|
364
368
|
await agent.dispose();
|
|
365
369
|
}
|
|
366
370
|
}
|
|
@@ -4,7 +4,7 @@ import { describeMarAgentError, MarAgentError } from '../error.js';
|
|
|
4
4
|
import { resolveModelDefaultReasoningEffort } from './configuration.js';
|
|
5
5
|
import WebSocket from 'ws';
|
|
6
6
|
import { isRetryableUpstreamOverloadCode, number, parseSse, record, text } from './sse.js';
|
|
7
|
-
import { connectSocks5Target, fetchModelResponse, isRetryableModelTransportError, modelRetryMaxAttempts, isRetryableModelRequestError, modelRetryDelay, waitForModelRetry } from './http.js';
|
|
7
|
+
import { connectSocks5Target, fetchModelResponse, isRetryableModelHttpStatus, isRetryableModelTransportError, modelRetryMaxAttempts, isRetryableModelRequestError, modelRetryDelay, waitForModelRetry } from './http.js';
|
|
8
8
|
import { acquireModelCredential, appendModelCredentialHeaders } from './runtime-credential.js';
|
|
9
9
|
const WEBSOCKET_IDLE_TIMEOUT_MS = 300_000;
|
|
10
10
|
export const OPENAI_RESPONSES_EARLY_EOF_REASON = 'OPENAI_RESPONSES_EARLY_EOF';
|
|
@@ -742,7 +742,11 @@ class OpenAiResponsesTurnSession {
|
|
|
742
742
|
if (socket?.readyState !== WebSocket.OPEN) {
|
|
743
743
|
invalidateResponsesWebSocket(this.state, true);
|
|
744
744
|
throw new MarAgentError('MAR_AGENT_MODEL_CONNECTION_FAILED', 'OpenAI Responses WebSocket connection is unavailable.', {
|
|
745
|
-
details: {
|
|
745
|
+
details: {
|
|
746
|
+
transport: 'WEBSOCKET',
|
|
747
|
+
continuation: input.continuation,
|
|
748
|
+
reason: 'MODEL_TRANSPORT_FAILURE'
|
|
749
|
+
},
|
|
746
750
|
retryable: true
|
|
747
751
|
});
|
|
748
752
|
}
|
|
@@ -1003,15 +1007,32 @@ async function waitForWebSocketOpen(socket, signal) {
|
|
|
1003
1007
|
reject(signal.reason);
|
|
1004
1008
|
});
|
|
1005
1009
|
const open = () => settle(resolve);
|
|
1006
|
-
const error = (cause) => settle(() => reject(new MarAgentError('MAR_AGENT_MODEL_CONNECTION_FAILED', 'OpenAI Responses WebSocket connection failed.', {
|
|
1007
|
-
|
|
1010
|
+
const error = (cause) => settle(() => reject(new MarAgentError('MAR_AGENT_MODEL_CONNECTION_FAILED', 'OpenAI Responses WebSocket connection failed.', {
|
|
1011
|
+
details: { reason: 'MODEL_TRANSPORT_FAILURE', transport: 'WEBSOCKET' },
|
|
1012
|
+
retryable: true,
|
|
1013
|
+
cause
|
|
1014
|
+
})));
|
|
1015
|
+
const close = () => settle(() => reject(new MarAgentError('MAR_AGENT_MODEL_CONNECTION_FAILED', 'OpenAI Responses WebSocket closed during connection.', {
|
|
1016
|
+
details: { reason: 'MODEL_TRANSPORT_FAILURE', transport: 'WEBSOCKET' },
|
|
1017
|
+
retryable: true
|
|
1018
|
+
})));
|
|
1008
1019
|
const unexpectedResponse = (_request, response) => settle(() => {
|
|
1009
1020
|
const status = response.statusCode ?? 0;
|
|
1010
1021
|
response.resume();
|
|
1011
1022
|
terminateWebSocket(socket);
|
|
1023
|
+
const retryable = status !== 401 && isRetryableModelHttpStatus(status);
|
|
1024
|
+
const retryAfter = response.headers['retry-after'];
|
|
1025
|
+
const retryAfterValue = Array.isArray(retryAfter) ? retryAfter[0] : retryAfter;
|
|
1012
1026
|
reject(new MarAgentError(status === 401 ? 'MAR_AGENT_MODEL_AUTH_FAILED' : 'MAR_AGENT_MODEL_CONNECTION_FAILED', status === 401
|
|
1013
1027
|
? 'OpenAI Responses WebSocket authentication failed.'
|
|
1014
|
-
: 'OpenAI Responses WebSocket handshake failed.', {
|
|
1028
|
+
: 'OpenAI Responses WebSocket handshake failed.', {
|
|
1029
|
+
details: {
|
|
1030
|
+
status,
|
|
1031
|
+
transport: 'WEBSOCKET',
|
|
1032
|
+
...(retryAfterValue === undefined ? {} : { retryAfter: retryAfterValue })
|
|
1033
|
+
},
|
|
1034
|
+
...(retryable ? { retryable: true } : {})
|
|
1035
|
+
}));
|
|
1015
1036
|
});
|
|
1016
1037
|
signal.addEventListener('abort', abort, { once: true });
|
|
1017
1038
|
socket.once('open', open);
|
package/dist/prompts/output.d.ts
CHANGED
|
@@ -1 +1,2 @@
|
|
|
1
1
|
export declare const outputStylePrompt = "# Output and communication style\nBe direct, natural, and concise. Match detail and structure to the user's needs and the task's complexity. Lead with the result; include actual verification, material risks, and unfinished work when relevant. Avoid repeating progress, dumping logs, or adding empty sections. For a requested review, lead with actionable findings ordered by severity and file evidence; if none were found, say so and note validation gaps. Preserve exact paths, identifiers, error codes, citations, and useful line locations when they are evidence. Never expose hidden reasoning.";
|
|
2
|
+
export declare const modelOutputContinuationMessage = "Continue the previous response from exactly where it stopped. Do not repeat completed content.";
|
package/dist/prompts/output.js
CHANGED
|
@@ -1 +1,2 @@
|
|
|
1
1
|
export const outputStylePrompt = "# Output and communication style\nBe direct, natural, and concise. Match detail and structure to the user's needs and the task's complexity. Lead with the result; include actual verification, material risks, and unfinished work when relevant. Avoid repeating progress, dumping logs, or adding empty sections. For a requested review, lead with actionable findings ordered by severity and file evidence; if none were found, say so and note validation gaps. Preserve exact paths, identifiers, error codes, citations, and useful line locations when they are evidence. Never expose hidden reasoning.";
|
|
2
|
+
export const modelOutputContinuationMessage = 'Continue the previous response from exactly where it stopped. Do not repeat completed content.';
|
|
@@ -1,18 +1,20 @@
|
|
|
1
1
|
import { isDeepStrictEqual } from 'node:util';
|
|
2
2
|
import { ContextGcCheckpointCache, contextGcMinimumUsageGrowth, contextGcReplacementKey, contextGcSourceProjectionHash } from '../session/context-gc-checkpoint.js';
|
|
3
|
-
import { contextGcEligibleUsage, inspectContextGc } from './context-gc.js';
|
|
3
|
+
import { CONTEXT_GC_POLICY, contextGcEligibleUsage, inspectContextGc } from './context-gc.js';
|
|
4
4
|
import { buildContextProjection } from './context-projection.js';
|
|
5
5
|
/** Only scheduling metadata survives a call; records and validation caches do not. */
|
|
6
6
|
export class ContextGcRuntime {
|
|
7
7
|
#scope = '';
|
|
8
8
|
#lastUsage;
|
|
9
9
|
#checkpoint;
|
|
10
|
+
#rolloutsSinceCheckpoint = 0;
|
|
10
11
|
#failures = 0;
|
|
11
12
|
#remaining = 0;
|
|
12
13
|
reset() {
|
|
13
14
|
this.#scope = '';
|
|
14
15
|
this.#lastUsage = undefined;
|
|
15
16
|
this.#checkpoint = undefined;
|
|
17
|
+
this.#rolloutsSinceCheckpoint = 0;
|
|
16
18
|
this.#failures = 0;
|
|
17
19
|
this.#remaining = 0;
|
|
18
20
|
}
|
|
@@ -39,26 +41,39 @@ export class ContextGcRuntime {
|
|
|
39
41
|
this.#remaining = 0;
|
|
40
42
|
}
|
|
41
43
|
this.#lastUsage = usage;
|
|
44
|
+
if (this.#checkpoint)
|
|
45
|
+
this.#rolloutsSinceCheckpoint++;
|
|
42
46
|
const maximumInterval = usage / input.contextWindowTokens >= 0.8 ? 2 : 8;
|
|
43
47
|
this.#remaining = Math.min(this.#remaining, maximumInterval - 1);
|
|
44
48
|
if (this.#remaining > 0) {
|
|
45
49
|
this.#remaining--;
|
|
46
50
|
return false;
|
|
47
51
|
}
|
|
48
|
-
if (this.#checkpoint?.modelId === input.modelId &&
|
|
49
|
-
this.#checkpoint.contextWindowTokens === input.contextWindowTokens &&
|
|
50
|
-
|
|
52
|
+
if (this.#checkpoint?.trigger.modelId === input.modelId &&
|
|
53
|
+
this.#checkpoint.trigger.contextWindowTokens === input.contextWindowTokens &&
|
|
54
|
+
!this.#checkpoint.protectedCandidatesPending &&
|
|
55
|
+
this.#rolloutsSinceCheckpoint < CONTEXT_GC_POLICY.largeExecProtectionRollouts &&
|
|
56
|
+
usage - this.#checkpoint.trigger.contextInputTokens <
|
|
51
57
|
contextGcMinimumUsageGrowth(input.contextWindowTokens))
|
|
52
58
|
return false;
|
|
53
59
|
const attempt = await attemptContextGc(input);
|
|
54
60
|
if (attempt.result) {
|
|
55
|
-
this.#checkpoint = {
|
|
61
|
+
this.#checkpoint = {
|
|
62
|
+
trigger: { ...attempt.result.projection.checkpoint.trigger },
|
|
63
|
+
protectedCandidatesPending: attempt.result.projection.checkpoint.metrics.protectedCandidateTokens > 0
|
|
64
|
+
};
|
|
65
|
+
this.#rolloutsSinceCheckpoint = 0;
|
|
56
66
|
this.#failures = 0;
|
|
57
67
|
this.#remaining = 1;
|
|
58
68
|
}
|
|
59
69
|
else {
|
|
60
70
|
if (attempt.checkpointTrigger)
|
|
61
|
-
this.#checkpoint = {
|
|
71
|
+
this.#checkpoint = {
|
|
72
|
+
trigger: { ...attempt.checkpointTrigger },
|
|
73
|
+
protectedCandidatesPending: attempt.protectedCandidatesPending === true
|
|
74
|
+
};
|
|
75
|
+
if (attempt.checkpointTrigger)
|
|
76
|
+
this.#rolloutsSinceCheckpoint = 0;
|
|
62
77
|
this.#failures = Math.min(3, this.#failures + 1);
|
|
63
78
|
this.#remaining =
|
|
64
79
|
Math.min(2 ** this.#failures, maximumInterval, attempt.retryAfterRollouts ?? Infinity) - 1;
|
|
@@ -85,9 +100,22 @@ async function attemptContextGc(input) {
|
|
|
85
100
|
const checkpointCache = new ContextGcCheckpointCache();
|
|
86
101
|
const previous = checkpointCache.read(records, undefined, input.workspace);
|
|
87
102
|
if (previous) {
|
|
88
|
-
deferred = {
|
|
89
|
-
|
|
90
|
-
|
|
103
|
+
deferred = {
|
|
104
|
+
...deferred,
|
|
105
|
+
checkpointTrigger: previous.payload.trigger,
|
|
106
|
+
protectedCandidatesPending: previous.payload.metrics.protectedCandidateTokens > 0
|
|
107
|
+
};
|
|
108
|
+
const rolloutGrowth = records.filter((record) => record.sequence > previous.record.sequence &&
|
|
109
|
+
record.type === 'model.attempt' &&
|
|
110
|
+
isRecord(record.payload) &&
|
|
111
|
+
record.payload.purpose === 'EXECUTION' &&
|
|
112
|
+
record.payload.state === 'SUCCEEDED').length;
|
|
113
|
+
const minimumRollouts = previous.payload.metrics.protectedCandidateTokens > 0
|
|
114
|
+
? 2
|
|
115
|
+
: CONTEXT_GC_POLICY.largeExecProtectionRollouts;
|
|
116
|
+
if (rolloutGrowth < minimumRollouts &&
|
|
117
|
+
contextUsage - previous.payload.trigger.contextInputTokens <
|
|
118
|
+
contextGcMinimumUsageGrowth(input.contextWindowTokens))
|
|
91
119
|
return deferred;
|
|
92
120
|
}
|
|
93
121
|
const projection = buildContextProjection(records, input.workspace, checkpointCache);
|
|
@@ -107,7 +135,13 @@ async function attemptContextGc(input) {
|
|
|
107
135
|
activeProcessIds: input.activeProcessIds ?? (await input.getActiveProcessIds?.()) ?? new Set()
|
|
108
136
|
});
|
|
109
137
|
if (inspection.retryAfterRollouts !== undefined)
|
|
110
|
-
deferred = {
|
|
138
|
+
deferred = {
|
|
139
|
+
...deferred,
|
|
140
|
+
protectedCandidatesPending: true,
|
|
141
|
+
retryAfterRollouts: inspection.retryAfterRollouts
|
|
142
|
+
};
|
|
143
|
+
else if (!inspection.plan)
|
|
144
|
+
deferred = { ...deferred, protectedCandidatesPending: false };
|
|
111
145
|
const plan = inspection.plan;
|
|
112
146
|
if (!plan)
|
|
113
147
|
return deferred;
|
|
@@ -161,3 +195,6 @@ async function matchingCheckpointWasPersisted(store, sessionId, draft, workspace
|
|
|
161
195
|
});
|
|
162
196
|
return matches ? { records: state.contextRecords, projection } : false;
|
|
163
197
|
}
|
|
198
|
+
function isRecord(value) {
|
|
199
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
200
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { ClientToolDefinition, ModelMessage, ModelUsage } from '../model/contracts.js';
|
|
2
|
-
import { ContextGcCheckpointCache, type ContextGcCheckpointDraft, type ContextGcEconomicsPolicy, type ContextGcMetrics, type ContextGcPlatform } from '../session/context-gc-checkpoint.js';
|
|
2
|
+
import { ContextGcCheckpointCache, type ContextGcCheckpointDraft, type ContextGcEconomicsPolicy, type ContextGcMetrics, type ContextGcPlatform, type ContextGcReplacement } from '../session/context-gc-checkpoint.js';
|
|
3
3
|
import type { SessionRecord } from '../session/jsonl-store.js';
|
|
4
4
|
export declare const CONTEXT_GC_POLICY: ContextGcEconomicsPolicy;
|
|
5
5
|
interface ContextGcEconomicsInput {
|
|
@@ -28,6 +28,11 @@ export interface ContextGcPlan {
|
|
|
28
28
|
readonly draft: ContextGcCheckpointDraft;
|
|
29
29
|
readonly metrics: ContextGcMetrics;
|
|
30
30
|
}
|
|
31
|
+
interface ContextGcCacheEconomicsProfile {
|
|
32
|
+
readonly reclaimedTokens: number;
|
|
33
|
+
readonly reclaimedCachedTokens: number;
|
|
34
|
+
readonly cacheRebuildPenaltyCostUnits: number;
|
|
35
|
+
}
|
|
31
36
|
export interface ContextGcInspection {
|
|
32
37
|
readonly plan?: ContextGcPlan;
|
|
33
38
|
readonly retryAfterRollouts?: number;
|
|
@@ -54,6 +59,8 @@ export declare function contextGcMinimumReclaimedTokens(contextUsageTokens: numb
|
|
|
54
59
|
export declare function contextGcEligibleUsage(usage: ModelUsage, contextWindowTokens: number): number | undefined;
|
|
55
60
|
export declare function planContextGc(input: ContextGcPlannerInput): ContextGcPlan | undefined;
|
|
56
61
|
export declare function inspectContextGc(input: ContextGcPlannerInput): ContextGcInspection;
|
|
62
|
+
export declare function contextGcMarginalCacheInvestmentPaysBack(current: ContextGcCacheEconomicsProfile, combined: ContextGcCacheEconomicsProfile): boolean;
|
|
63
|
+
export declare function contextGcEarliestReplacementSequence(replacements: readonly ContextGcReplacement[]): number;
|
|
57
64
|
export declare function contextGcImminentCompactAvoidanceCredit(input: {
|
|
58
65
|
readonly contextUsageTokens: number;
|
|
59
66
|
readonly contextWindowTokens: number;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { ContextGcCheckpointCache, contextGcBaseFromRecords, contextGcLogicalPathKey, contextGcMinimumUsageGrowth, contextGcRecordSourceHash, contextGcTargetKey, normalizeContextGcReplacements } from '../session/context-gc-checkpoint.js';
|
|
2
|
-
import { contextGcApplyPatchInputReplacement, contextGcExecInputReplacement, contextGcExecOutputReplacement, contextGcPatchResultFromCompletedPayload, contextGcStaleReadReplacement, normalizeContextGcExecResult } from '../session/context-gc-replacements.js';
|
|
3
|
-
import { contextGcSubagentNeedsRetainedPriorFinal, contextGcSubagentOutputReplacement, contextGcSubagentPriorFinalIsRetained, contextGcSubagentSnapshotIsSuperseded, contextGcSubagentSnapshotTurnId, normalizeContextGcSubagentSnapshot } from '../session/context-gc-subagent.js';
|
|
2
|
+
import { contextGcApplyPatchInputReplacement, contextGcExecInputReplacement, contextGcExecOutputReplacement, contextGcPatchResultFromCompletedPayload, contextGcStaleReadReplacement, contextGcWebOutputReplacement, normalizeContextGcExecResult } from '../session/context-gc-replacements.js';
|
|
3
|
+
import { contextGcSubagentNeedsRetainedPriorFinal, contextGcSubagentOutputReplacement, contextGcSubagentPriorFinalIsRetained, contextGcSubagentSnapshotIsPreciselyCovered, contextGcSubagentSnapshotIsSuperseded, contextGcSubagentSnapshotTurnId, normalizeContextGcSubagentSnapshot } from '../session/context-gc-subagent.js';
|
|
4
4
|
import { parseReadContextOutput } from '../tools/read-context.js';
|
|
5
5
|
import { CONTEXT_GC_TOKEN_ESTIMATOR, contextGcCacheImpact, contextGcMessageTokens, contextGcPrefixTokens, contextGcReclaimedTokens } from './context-gc-tokens.js';
|
|
6
6
|
import { applyContextGcReplacementsToMessages, providerToolCallId, providerToolName } from './context-projection.js';
|
|
@@ -27,7 +27,8 @@ const MINIMUM_CONTEXT_WINDOW_REDUCTION_RATIO = 0.05;
|
|
|
27
27
|
const COMPACT_CREDIT_ACTIVATION_RATIO = 0.8;
|
|
28
28
|
const COMPACT_THRESHOLD_RATIO = 0.9;
|
|
29
29
|
const IMMINENT_COMPACT_CREDIT_DISCOUNT = 0.8;
|
|
30
|
-
const
|
|
30
|
+
const LARGE_TERMINAL_OUTPUT_TOKENS = 4_096;
|
|
31
|
+
const HIGH_WATER_LARGE_TERMINAL_OUTPUT_TOKENS = 2_048;
|
|
31
32
|
export function evaluateContextGcEconomics(input) {
|
|
32
33
|
const reclaimedTokens = Math.max(0, input.reclaimedTokens);
|
|
33
34
|
const cacheReadTokens = Math.max(0, input.cacheReadTokens);
|
|
@@ -134,8 +135,12 @@ export function inspectContextGc(input) {
|
|
|
134
135
|
const previous = input.checkpointCache.read(input.records, undefined, input.workspace);
|
|
135
136
|
if (previous) {
|
|
136
137
|
const minimumUsageGrowth = contextGcMinimumUsageGrowth(input.contextWindowTokens);
|
|
137
|
-
|
|
138
|
-
|
|
138
|
+
const protectedCandidatesPending = previous.payload.metrics.protectedCandidateTokens > 0;
|
|
139
|
+
const rolloutGrowth = normalRolloutOrdinal - previous.payload.trigger.normalRolloutOrdinal;
|
|
140
|
+
if (rolloutGrowth < 2 ||
|
|
141
|
+
(!protectedCandidatesPending &&
|
|
142
|
+
rolloutGrowth < CONTEXT_GC_POLICY.largeExecProtectionRollouts &&
|
|
143
|
+
contextUsageTokens - previous.payload.trigger.contextInputTokens < minimumUsageGrowth))
|
|
139
144
|
return {};
|
|
140
145
|
}
|
|
141
146
|
if (input.messages.length !== input.messageSources.length)
|
|
@@ -161,29 +166,114 @@ export function inspectContextGc(input) {
|
|
|
161
166
|
protectedCandidateTokens: deterministicProtectedTokens
|
|
162
167
|
});
|
|
163
168
|
if (stageAPlan?.economics.passes) {
|
|
164
|
-
const
|
|
165
|
-
|
|
169
|
+
const exec = collectExecCandidates(input, groups, existingTargets, usageRatio);
|
|
170
|
+
const web = collectWebCandidates(input, groups, existingTargets, usageRatio);
|
|
171
|
+
const protectedCandidateTokens = deterministicProtectedTokens + exec.protectedTokens + web.protectedTokens;
|
|
172
|
+
const stageAPlanWithProtectedCandidates = evaluateCandidateSet(input, existing, deterministicCandidates, {
|
|
173
|
+
stage: 'STAGE_A',
|
|
174
|
+
contextUsageTokens,
|
|
175
|
+
usageRatio,
|
|
176
|
+
usageRecord,
|
|
177
|
+
normalRolloutOrdinal,
|
|
178
|
+
protectedCandidateTokens
|
|
179
|
+
}) ?? stageAPlan;
|
|
180
|
+
const marginal = mergeMarginalLargeOutputCandidates(input, existing, deterministicCandidates, [...exec.eligible, ...web.eligible], stageAPlanWithProtectedCandidates, {
|
|
181
|
+
contextUsageTokens,
|
|
182
|
+
usageRatio,
|
|
183
|
+
usageRecord,
|
|
184
|
+
normalRolloutOrdinal,
|
|
185
|
+
protectedCandidateTokens
|
|
186
|
+
});
|
|
187
|
+
const plan = finalizePlan(input, marginal ?? stageAPlanWithProtectedCandidates);
|
|
188
|
+
const retryAfterRollouts = Math.min(exec.retryAfterRollouts, web.retryAfterRollouts);
|
|
189
|
+
return plan
|
|
190
|
+
? {
|
|
191
|
+
plan,
|
|
192
|
+
...(protectedCandidateTokens > 0 && Number.isFinite(retryAfterRollouts)
|
|
193
|
+
? { retryAfterRollouts }
|
|
194
|
+
: {})
|
|
195
|
+
}
|
|
196
|
+
: {};
|
|
166
197
|
}
|
|
167
|
-
const exec = collectExecCandidates(input, groups, existingTargets);
|
|
168
|
-
const
|
|
198
|
+
const exec = collectExecCandidates(input, groups, existingTargets, usageRatio);
|
|
199
|
+
const web = collectWebCandidates(input, groups, existingTargets, usageRatio);
|
|
200
|
+
const combined = evaluateCandidateSet(input, existing, [...deterministicCandidates, ...exec.eligible, ...web.eligible], {
|
|
169
201
|
stage: 'STAGE_A_AND_EXEC',
|
|
170
202
|
contextUsageTokens,
|
|
171
203
|
usageRatio,
|
|
172
204
|
usageRecord,
|
|
173
205
|
normalRolloutOrdinal,
|
|
174
|
-
protectedCandidateTokens: deterministicProtectedTokens + exec.protectedTokens
|
|
206
|
+
protectedCandidateTokens: deterministicProtectedTokens + exec.protectedTokens + web.protectedTokens
|
|
175
207
|
});
|
|
176
208
|
if (combined?.economics.passes) {
|
|
177
209
|
const plan = finalizePlan(input, combined);
|
|
178
210
|
return plan ? { plan } : {};
|
|
179
211
|
}
|
|
180
|
-
const retryAfterRollouts = Math.min(stageA.retryAfterRollouts, subagent.retryAfterRollouts, exec.retryAfterRollouts);
|
|
181
|
-
return deterministicProtectedTokens + exec.protectedTokens >=
|
|
212
|
+
const retryAfterRollouts = Math.min(stageA.retryAfterRollouts, subagent.retryAfterRollouts, exec.retryAfterRollouts, web.retryAfterRollouts);
|
|
213
|
+
return deterministicProtectedTokens + exec.protectedTokens + web.protectedTokens >=
|
|
182
214
|
contextGcMinimumReclaimedTokens(contextUsageTokens, input.contextWindowTokens) &&
|
|
183
215
|
Number.isFinite(retryAfterRollouts)
|
|
184
216
|
? { retryAfterRollouts }
|
|
185
217
|
: {};
|
|
186
218
|
}
|
|
219
|
+
function mergeMarginalLargeOutputCandidates(input, existing, deterministicCandidates, execCandidates, stageAPlan, state) {
|
|
220
|
+
let selected = [...deterministicCandidates];
|
|
221
|
+
let current = stageAPlan;
|
|
222
|
+
const stageAEarliestSequence = contextGcEarliestReplacementSequence(deterministicCandidates.flatMap((candidate) => candidate.replacements));
|
|
223
|
+
for (const candidate of [...execCandidates].sort((left, right) => right.evidenceSequence - left.evidenceSequence)) {
|
|
224
|
+
const evaluated = evaluateCandidateSet(input, existing, [...selected, candidate], {
|
|
225
|
+
stage: 'STAGE_A_AND_EXEC',
|
|
226
|
+
...state
|
|
227
|
+
});
|
|
228
|
+
if (!evaluated?.economics.passes || evaluated.reclaimedTokens <= current.reclaimedTokens)
|
|
229
|
+
continue;
|
|
230
|
+
const doesNotMoveInvalidationEarlier = contextGcEarliestReplacementSequence(candidate.replacements) >= stageAEarliestSequence;
|
|
231
|
+
if (!doesNotMoveInvalidationEarlier &&
|
|
232
|
+
!contextGcMarginalCacheInvestmentPaysBack({
|
|
233
|
+
reclaimedTokens: current.reclaimedTokens,
|
|
234
|
+
reclaimedCachedTokens: current.reclaimedCachedTokens,
|
|
235
|
+
cacheRebuildPenaltyCostUnits: current.economics.cacheRebuildPenaltyCostUnits
|
|
236
|
+
}, {
|
|
237
|
+
reclaimedTokens: evaluated.reclaimedTokens,
|
|
238
|
+
reclaimedCachedTokens: evaluated.reclaimedCachedTokens,
|
|
239
|
+
cacheRebuildPenaltyCostUnits: evaluated.economics.cacheRebuildPenaltyCostUnits
|
|
240
|
+
}))
|
|
241
|
+
continue;
|
|
242
|
+
selected = [...selected, candidate];
|
|
243
|
+
current = evaluated;
|
|
244
|
+
}
|
|
245
|
+
return current === stageAPlan ? undefined : current;
|
|
246
|
+
}
|
|
247
|
+
export function contextGcMarginalCacheInvestmentPaysBack(current, combined) {
|
|
248
|
+
const marginalPenalty = combined.cacheRebuildPenaltyCostUnits - current.cacheRebuildPenaltyCostUnits;
|
|
249
|
+
if (marginalPenalty <= 0)
|
|
250
|
+
return true;
|
|
251
|
+
const firstRolloutSavings = (profile) => {
|
|
252
|
+
const reclaimedCachedTokens = clamp(profile.reclaimedCachedTokens, 0, Math.max(0, profile.reclaimedTokens));
|
|
253
|
+
return (reclaimedCachedTokens * CONTEXT_GC_POLICY.cacheReadCostRate +
|
|
254
|
+
Math.max(0, profile.reclaimedTokens - reclaimedCachedTokens) *
|
|
255
|
+
CONTEXT_GC_POLICY.normalInputCostRate);
|
|
256
|
+
};
|
|
257
|
+
const marginalFirstRolloutSavings = firstRolloutSavings(combined) - firstRolloutSavings(current);
|
|
258
|
+
const marginalSteadyRolloutSavings = Math.max(0, combined.reclaimedTokens - current.reclaimedTokens) *
|
|
259
|
+
CONTEXT_GC_POLICY.cacheReadCostRate;
|
|
260
|
+
for (let rollouts = 1; rollouts <= CONTEXT_GC_POLICY.immediatePaybackRollouts; rollouts++) {
|
|
261
|
+
const net = -marginalPenalty +
|
|
262
|
+
marginalFirstRolloutSavings +
|
|
263
|
+
Math.max(0, rollouts - 1) * marginalSteadyRolloutSavings;
|
|
264
|
+
if (net > 0 && net / marginalPenalty >= CONTEXT_GC_POLICY.minimumCacheInvestmentRoi)
|
|
265
|
+
return true;
|
|
266
|
+
}
|
|
267
|
+
return false;
|
|
268
|
+
}
|
|
269
|
+
export function contextGcEarliestReplacementSequence(replacements) {
|
|
270
|
+
return Math.min(...replacements.flatMap((replacement) => [
|
|
271
|
+
replacement.target.sequence,
|
|
272
|
+
...(replacement.kind === 'TOOL_CALL_INPUT'
|
|
273
|
+
? replacement.providerAliasTargets.map((target) => target.sequence)
|
|
274
|
+
: [])
|
|
275
|
+
]));
|
|
276
|
+
}
|
|
187
277
|
function finalizePlan(input, evaluated) {
|
|
188
278
|
if (!evaluated)
|
|
189
279
|
return undefined;
|
|
@@ -302,6 +392,7 @@ function evaluateCandidateSet(input, existing, candidates, state) {
|
|
|
302
392
|
return {
|
|
303
393
|
...state,
|
|
304
394
|
economics,
|
|
395
|
+
reclaimedCachedTokens: cache.reclaimedCachedTokens,
|
|
305
396
|
replacements,
|
|
306
397
|
beforeTokens,
|
|
307
398
|
afterTokens,
|
|
@@ -444,7 +535,9 @@ function collectSubagentCandidates(input, groups, existingTargets) {
|
|
|
444
535
|
})
|
|
445
536
|
.sort((left, right) => left.group.result.record.sequence - right.group.result.record.sequence);
|
|
446
537
|
for (const [index, source] of invocations.entries()) {
|
|
447
|
-
if (source.group.toolName !== 'agent_wait' &&
|
|
538
|
+
if (source.group.toolName !== 'agent_wait' &&
|
|
539
|
+
source.group.toolName !== 'agent_message' &&
|
|
540
|
+
source.group.toolName !== 'agent_cancel')
|
|
448
541
|
continue;
|
|
449
542
|
const sourceTarget = targetFor(source.group.result.record, source.group.result.message);
|
|
450
543
|
if (!sourceTarget || existingTargets.has(contextGcTargetKey(sourceTarget)))
|
|
@@ -461,7 +554,8 @@ function collectSubagentCandidates(input, groups, existingTargets) {
|
|
|
461
554
|
.slice(index + 1)
|
|
462
555
|
.find((candidate) => candidate.snapshot.agentId === source.snapshot.agentId &&
|
|
463
556
|
candidate.turnId === source.turnId &&
|
|
464
|
-
contextGcSubagentSnapshotIsSuperseded(source.snapshot, candidate.snapshot, retainedPriorFinal?.snapshot)
|
|
557
|
+
(contextGcSubagentSnapshotIsSuperseded(source.snapshot, candidate.snapshot, retainedPriorFinal?.snapshot) ||
|
|
558
|
+
contextGcSubagentSnapshotIsPreciselyCovered(source.snapshot, candidate.snapshot)));
|
|
465
559
|
if (!terminal)
|
|
466
560
|
continue;
|
|
467
561
|
const sourceContent = source.group.result.message.content;
|
|
@@ -508,7 +602,7 @@ function collectSubagentCandidates(input, groups, existingTargets) {
|
|
|
508
602
|
}
|
|
509
603
|
return { eligible, protectedTokens, retryAfterRollouts };
|
|
510
604
|
}
|
|
511
|
-
function collectExecCandidates(input, groups, existingTargets) {
|
|
605
|
+
function collectExecCandidates(input, groups, existingTargets, usageRatio) {
|
|
512
606
|
const eligible = [];
|
|
513
607
|
let protectedTokens = 0;
|
|
514
608
|
let retryAfterRollouts = Infinity;
|
|
@@ -568,7 +662,10 @@ function collectExecCandidates(input, groups, existingTargets) {
|
|
|
568
662
|
invocation.group.result.message,
|
|
569
663
|
...invocation.group.aliases.map((alias) => alias.message)
|
|
570
664
|
]));
|
|
571
|
-
|
|
665
|
+
const minimumVisibleTokens = usageRatio >= COMPACT_CREDIT_ACTIVATION_RATIO
|
|
666
|
+
? HIGH_WATER_LARGE_TERMINAL_OUTPUT_TOKENS
|
|
667
|
+
: LARGE_TERMINAL_OUTPUT_TOKENS;
|
|
668
|
+
if (visibleTokens < minimumVisibleTokens)
|
|
572
669
|
continue;
|
|
573
670
|
const evidenceSequence = final.group.result.record.sequence;
|
|
574
671
|
const observedRollouts = successfulRolloutsAfter(input.records, evidenceSequence, input.modelId, input.contextWindowTokens);
|
|
@@ -617,12 +714,69 @@ function collectExecCandidates(input, groups, existingTargets) {
|
|
|
617
714
|
}
|
|
618
715
|
return { eligible, protectedTokens, retryAfterRollouts };
|
|
619
716
|
}
|
|
717
|
+
function collectWebCandidates(input, groups, existingTargets, usageRatio) {
|
|
718
|
+
const eligible = [];
|
|
719
|
+
let protectedTokens = 0;
|
|
720
|
+
let retryAfterRollouts = Infinity;
|
|
721
|
+
const minimumVisibleTokens = usageRatio >= COMPACT_CREDIT_ACTIVATION_RATIO
|
|
722
|
+
? HIGH_WATER_LARGE_TERMINAL_OUTPUT_TOKENS
|
|
723
|
+
: LARGE_TERMINAL_OUTPUT_TOKENS;
|
|
724
|
+
for (const group of [...groups.values()].sort((left, right) => (left.result?.record.sequence ?? Number.MAX_SAFE_INTEGER) -
|
|
725
|
+
(right.result?.record.sequence ?? Number.MAX_SAFE_INTEGER))) {
|
|
726
|
+
if (!isContextGcWebToolName(group.toolName) ||
|
|
727
|
+
!group.call ||
|
|
728
|
+
!group.result ||
|
|
729
|
+
group.failed ||
|
|
730
|
+
!group.completed ||
|
|
731
|
+
group.call.message.role !== 'tool_call' ||
|
|
732
|
+
group.result.message.role !== 'tool' ||
|
|
733
|
+
typeof group.result.message.content !== 'string' ||
|
|
734
|
+
!isRecord(group.completed.payload) ||
|
|
735
|
+
group.completed.payload.type !== 'tool.completed' ||
|
|
736
|
+
group.completed.payload.callId !== group.callId ||
|
|
737
|
+
group.completed.payload.toolName !== group.toolName ||
|
|
738
|
+
typeof group.completed.payload.truncated !== 'boolean')
|
|
739
|
+
continue;
|
|
740
|
+
const target = targetFor(group.result.record, group.result.message);
|
|
741
|
+
if (!target || existingTargets.has(contextGcTargetKey(target)))
|
|
742
|
+
continue;
|
|
743
|
+
const visibleTokens = contextGcMessageTokens(input, [group.result.message]);
|
|
744
|
+
if (visibleTokens < minimumVisibleTokens)
|
|
745
|
+
continue;
|
|
746
|
+
const evidenceSequence = group.result.record.sequence;
|
|
747
|
+
const observedRollouts = successfulRolloutsAfter(input.records, evidenceSequence, input.modelId, input.contextWindowTokens);
|
|
748
|
+
const candidate = {
|
|
749
|
+
replacements: [
|
|
750
|
+
{
|
|
751
|
+
kind: 'WEB_OUTPUT',
|
|
752
|
+
toolName: group.toolName,
|
|
753
|
+
callId: group.callId,
|
|
754
|
+
target,
|
|
755
|
+
replacementContent: contextGcWebOutputReplacement(group.toolName, group.result.message.content, group.completed.payload.truncated),
|
|
756
|
+
evidenceSequence
|
|
757
|
+
}
|
|
758
|
+
],
|
|
759
|
+
observedRollouts,
|
|
760
|
+
evidenceSequence
|
|
761
|
+
};
|
|
762
|
+
if (observedRollouts >= CONTEXT_GC_POLICY.largeExecProtectionRollouts)
|
|
763
|
+
eligible.push(candidate);
|
|
764
|
+
else {
|
|
765
|
+
protectedTokens += candidateTokens(input, candidate);
|
|
766
|
+
retryAfterRollouts = Math.min(retryAfterRollouts, CONTEXT_GC_POLICY.largeExecProtectionRollouts - observedRollouts);
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
return { eligible, protectedTokens, retryAfterRollouts };
|
|
770
|
+
}
|
|
620
771
|
function isContextGcSubagentToolName(value) {
|
|
621
772
|
return (value === 'agent_start' ||
|
|
622
773
|
value === 'agent_wait' ||
|
|
623
774
|
value === 'agent_message' ||
|
|
624
775
|
value === 'agent_cancel');
|
|
625
776
|
}
|
|
777
|
+
function isContextGcWebToolName(value) {
|
|
778
|
+
return value === 'web_search' || value === 'web_fetch';
|
|
779
|
+
}
|
|
626
780
|
function buildToolGroups(records, messages, sources) {
|
|
627
781
|
const groups = new Map();
|
|
628
782
|
const group = (callId, toolName = '') => {
|
|
@@ -60,7 +60,9 @@ export function applyContextGcReplacementsToMessages(messages, messageSources, r
|
|
|
60
60
|
};
|
|
61
61
|
else if (replacement.kind === 'TOOL_CALL_INPUT' && next.role === 'tool_call')
|
|
62
62
|
next = { ...next, arguments: structuredClone(replacement.replacementArguments) };
|
|
63
|
-
else if ((replacement.kind === 'EXEC_OUTPUT' ||
|
|
63
|
+
else if ((replacement.kind === 'EXEC_OUTPUT' ||
|
|
64
|
+
replacement.kind === 'SUBAGENT_OUTPUT' ||
|
|
65
|
+
replacement.kind === 'WEB_OUTPUT') &&
|
|
64
66
|
next.role === 'tool')
|
|
65
67
|
next = { ...next, content: replacement.replacementContent };
|
|
66
68
|
}
|