@shanepadgett/tau-agent 0.20.1 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/context-pruning/README.md +18 -114
- package/extensions/context-pruning/index.ts +74 -74
- package/extensions/context-pruning/projection.ts +40 -96
- package/extensions/context-pruning/prune.ts +138 -283
- package/extensions/context-pruning/render.ts +99 -35
- package/extensions/context-pruning/settings.ts +14 -15
- package/extensions/explore/read-cache.ts +31 -7
- package/extensions/handoff/README.md +7 -0
- package/extensions/handoff/index.ts +139 -0
- package/extensions/handoff/model.ts +71 -0
- package/extensions/subagent/index.ts +3 -1
- package/extensions/tau-help/help.md +5 -1
- package/package.json +2 -2
- package/schemas/tau.schema.json +15 -12
- package/shared/context-pruning-state.ts +69 -273
- package/extensions/context-pruning/file-evidence.ts +0 -265
|
@@ -1,135 +1,39 @@
|
|
|
1
1
|
# Context Pruning
|
|
2
2
|
|
|
3
|
-
Context Pruning
|
|
3
|
+
Context Pruning gives the agent direct control over its future model context. It does not delete or rewrite the saved conversation.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
`context_prune` creates a hard checkpoint. Everything before that checkpoint leaves future model input unless the agent explicitly carries it forward. Messages after the checkpoint remain unchanged.
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
Before calling the tool, the agent writes visible prose containing the durable conclusions, user constraints, conditional relevance, and next action that must survive. That prose is part of the checkpoint turn.
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
## Selections
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
- A `context_prune` call decides what old evidence the agent wants to remove or retain.
|
|
13
|
-
- Tau then checks whether that exact removal is safe and whether it saves enough tokens.
|
|
11
|
+
The tool accepts three required lists:
|
|
14
12
|
|
|
15
|
-
|
|
13
|
+
- `keepToolCalls` retains exact earlier tool exchanges by tool-call ID. Parallel calls remain independently selectable: retaining one call does not retain its siblings.
|
|
14
|
+
- `keepFiles` reads each selected file from disk and carries its complete current contents forward as a fresh autoread snapshot. It does not require an earlier complete read.
|
|
15
|
+
- `deferFiles` carries forward a short advisory note explaining why a file is irrelevant now and when to reconsider it.
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
Duplicate selections are collapsed. A file selected in both `keepFiles` and `deferFiles` is kept. Missing, unreadable, or otherwise unsnapshotable files produce warnings in the successful tool result; they do not block the checkpoint or other file snapshots.
|
|
18
18
|
|
|
19
|
-
|
|
19
|
+
Context Pruning does not impose a minimum token saving and does not reject a checkpoint because old context has unusual bookkeeping. Pi supplies the current provider context. Tau removes unretained tool calls and their matching results with the same ID, drops other pre-checkpoint messages, and leaves the checkpoint turn and later messages intact.
|
|
20
20
|
|
|
21
|
-
|
|
22
|
-
- `keepToolCalls`: exact earlier tool calls and results that must remain available.
|
|
23
|
-
- `deferFiles`: files that are currently irrelevant but may matter under a stated condition.
|
|
21
|
+
## Automatic and manual requests
|
|
24
22
|
|
|
25
|
-
|
|
23
|
+
Tau checks context growth after tool-using turns and can send progressively stronger private instructions from `nudgeInstructions`. Growth is measured from the first tool-using turn after the latest checkpoint. Branch navigation and compaction reconstruct that baseline from the active branch.
|
|
26
24
|
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
Tau prepares the result without changing the active context. It estimates the token count before and after the proposed removal, including any replacement file contents and deferred-file note. It applies the prune only when the estimated saving is at least `minimumReclaimTokens`.
|
|
30
|
-
|
|
31
|
-
After a successful call, Tau removes the selected evidence from future model input. The saved conversation remains unchanged. The tool result records exactly which tool exchanges and autoreads were removed, retained, or refreshed.
|
|
32
|
-
|
|
33
|
-
## Why a prune is skipped
|
|
34
|
-
|
|
35
|
-
A skipped prune removes nothing and publishes no deferred-file note or replacement file contents. Tau skips when any of these checks fails:
|
|
36
|
-
|
|
37
|
-
1. Context Pruning is disabled.
|
|
38
|
-
2. Tau does not have a usable copy of the exact messages most recently sent to the model.
|
|
39
|
-
3. The `context_prune` call is missing from the active branch.
|
|
40
|
-
4. Another tool call appears beside `context_prune` in the same assistant message.
|
|
41
|
-
5. The current model input contains duplicate, mismatched, or orphaned tool records that Tau cannot safely remove. An unmatched call from an explicitly aborted assistant response is discarded and does not block pruning.
|
|
42
|
-
6. A retained tool-call ID is duplicated, missing, or is not a complete exchange in the current model input.
|
|
43
|
-
7. Two selected file paths resolve to the same file, including aliases and symbolic links across `keepFiles` and `deferFiles`.
|
|
44
|
-
8. A retained file does not have an earlier complete-file read, has malformed read evidence, cannot currently be read as UTF-8, or exceeds the 1 MiB complete-file limit.
|
|
45
|
-
9. The estimated saving is below `minimumReclaimTokens`.
|
|
46
|
-
|
|
47
|
-
Cancellation, a session lifecycle change during preparation, and failures while publishing an already prepared prune are reported as tool failures rather than skipped prunes.
|
|
48
|
-
|
|
49
|
-
### “The latest provider-context projection is unavailable”
|
|
50
|
-
|
|
51
|
-
This error uses an internal term. In plain language, Tau did not have a usable copy of the exact message list the model had just received.
|
|
52
|
-
|
|
53
|
-
Tau normally saves that list while Pi prepares a model turn. Tau clears it on session start, branch changes, compaction, and shutdown. It also refuses to save a list with invalid tool-call/result pairing. If the list is missing or belongs to an earlier session state, `context_prune` stops before checking the agent’s selections or estimating savings.
|
|
54
|
-
|
|
55
|
-
Pi can save an aborted assistant response that requested a tool but never received a result. Tau discards that abandoned call when preparing later model input. Other unmatched calls and results still fail validation because Tau cannot prove whether their evidence is incomplete.
|
|
56
|
-
|
|
57
|
-
For example, a row that says `context_prune 29 selections` followed by this error means:
|
|
58
|
-
|
|
59
|
-
- the agent supplied 29 items across the three selection lists;
|
|
60
|
-
- Tau did not inspect or apply those selections;
|
|
61
|
-
- no old tool calls were removed;
|
|
62
|
-
- the context percentage did not cause the rejection;
|
|
63
|
-
- the current implementation does not reconstruct the missing list during that call.
|
|
64
|
-
|
|
65
|
-
The response tells the agent not to retry immediately because repeated calls in the same state would usually fail for the same reason. A later model turn should normally give Tau a new list. Repeated occurrences indicate a feature defect or a session history that Tau cannot safely process.
|
|
66
|
-
|
|
67
|
-
## Retaining files
|
|
68
|
-
|
|
69
|
-
`keepFiles` preserves complete current file knowledge, not necessarily the original read row.
|
|
70
|
-
|
|
71
|
-
Each selected file must have been read completely earlier in the current model input. A partial read is insufficient. Tau then reads the file from disk again and chooses the cheaper valid representation:
|
|
72
|
-
|
|
73
|
-
- Keep an existing complete snapshot when it still matches the file.
|
|
74
|
-
- Keep a baseline plus later read diffs when that chain reconstructs the current file and costs no more than a fresh snapshot.
|
|
75
|
-
- Add one fresh complete snapshot when the earlier evidence is stale, broken, or more expensive.
|
|
76
|
-
|
|
77
|
-
The entire prune is skipped if any selected file fails. Every retained file must currently exist, be valid UTF-8, and fit within the 1 MiB complete-file snapshot limit. Read the complete file before selecting it for retention.
|
|
78
|
-
|
|
79
|
-
Paths are resolved from the session working directory. Paths outside it are stored as absolute paths. Existing symbolic links and other aliases are resolved before duplicate checks.
|
|
80
|
-
|
|
81
|
-
The required `relevance` text explains the agent’s selection. It does not loosen any validation rule.
|
|
82
|
-
|
|
83
|
-
## Retaining tool exchanges
|
|
84
|
-
|
|
85
|
-
`keepToolCalls` retains both sides of an earlier exchange: the assistant’s tool call and the matching tool result. The ID must occur exactly once, the tool names must match, and the result must follow the call.
|
|
86
|
-
|
|
87
|
-
Retention is exact. Selecting one call does not retain nearby calls from the same assistant message. Tau can remove one parallel tool exchange while keeping another, then removes the empty assistant message if no content remains.
|
|
88
|
-
|
|
89
|
-
The required `relevance` text explains why the exchange matters. Tau does not use that text to infer additional exchanges to retain.
|
|
90
|
-
|
|
91
|
-
## Deferring files
|
|
92
|
-
|
|
93
|
-
`deferFiles` does not read or preserve file contents. It adds a hidden advisory note telling the model why each file was deferred and when to reconsider it. The latest successful anchor replaces the previous deferred-file list.
|
|
94
|
-
|
|
95
|
-
A deferred path may be missing. It still participates in canonical path and duplicate checks. Its `reason` and `relevantWhen` text are passed to the model as written.
|
|
96
|
-
|
|
97
|
-
## Automatic hints
|
|
98
|
-
|
|
99
|
-
Tau checks context usage only after a turn that produced at least one tool result. It emits at most one marker for the strongest newly crossed boundary.
|
|
100
|
-
|
|
101
|
-
With the defaults:
|
|
102
|
-
|
|
103
|
-
- `nudgeEveryPercent: 20` allows markers after enough context growth to cross 20% intervals.
|
|
104
|
-
- `pressurePercent: 50` makes a marker a pressure hint only when raw usage is greater than 50%.
|
|
105
|
-
|
|
106
|
-
Below the pressure threshold, the hidden instruction says no prune is required unless broad exploration has converged or substantial evidence is already irrelevant. Above it, the instruction asks the agent to finish the current coherent step and prune when the projected saving is substantial. A hint never calls the tool or bypasses its checks.
|
|
107
|
-
|
|
108
|
-
After a successful prune, the first later tool-using turn records a new usage baseline and emits no automatic marker. Tau waits for context usage to grow by another `nudgeEveryPercent` from that baseline. The baseline and crossed boundaries are stored with the active branch so compaction and branch navigation do not immediately repeat hints.
|
|
109
|
-
|
|
110
|
-
The visible marker is compact. The full instruction stays hidden, and the agent is told not to discuss internal context management.
|
|
111
|
-
|
|
112
|
-
## Manual requests
|
|
113
|
-
|
|
114
|
-
Run `/prune` with no arguments to ask the agent to create an anchor and continue unfinished work. The command starts an agent turn immediately. It does not force a successful prune or bypass file, tool-pairing, current-message, lifecycle, or minimum-savings checks.
|
|
115
|
-
|
|
116
|
-
Extra arguments produce `Usage: /prune`. When Context Pruning is disabled, the command reports that state and does not start a prune turn.
|
|
117
|
-
|
|
118
|
-
Asking the agent to prune in ordinary chat has the same execution boundaries once it calls `context_prune`.
|
|
25
|
+
Run `/prune` with no arguments to ask the agent to create a checkpoint and continue its task immediately.
|
|
119
26
|
|
|
120
27
|
## Branches, compaction, and display
|
|
121
28
|
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
Context Pruning does not cancel or replace Pi’s manual, threshold, or overflow compaction. Compaction clears Tau’s saved model-input list; Tau must receive another model-input event before a later prune can use it.
|
|
29
|
+
Checkpoints belong to the active branch. Switching branches rebuilds the latest checkpoint, deferred-file state, automatic-hint baseline, and warning-colored pruned rows from that branch.
|
|
125
30
|
|
|
126
|
-
|
|
31
|
+
Normal Pi compaction remains independent. When a compaction no longer includes an old checkpoint turn, that checkpoint has nothing left to filter.
|
|
127
32
|
|
|
128
33
|
## Settings
|
|
129
34
|
|
|
130
35
|
Settings live under `extensions.contextPruning` in Tau settings.
|
|
131
36
|
|
|
132
|
-
- `enabled`: enables the tool, `/prune`,
|
|
133
|
-
- `nudgeEveryPercent`:
|
|
134
|
-
- `
|
|
135
|
-
- `minimumReclaimTokens`: positive integer minimum estimated saving required to apply a prune. Defaults to `8000`.
|
|
37
|
+
- `enabled`: enables the tool, `/prune`, projection, markers, and branch replay. Defaults to `true`.
|
|
38
|
+
- `nudgeEveryPercent`: context-growth interval between automatic hints, from `1` through `100`. Defaults to `20`.
|
|
39
|
+
- `nudgeInstructions`: ordered list of one through five nonempty instructions. Later reminders repeat the final instruction. Defaults to three escalating instructions.
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
import {
|
|
2
2
|
defineTool,
|
|
3
|
-
type ContextEvent,
|
|
4
3
|
type ExtensionAPI,
|
|
5
4
|
type ExtensionContext,
|
|
6
5
|
type SessionEntry,
|
|
@@ -8,7 +7,7 @@ import {
|
|
|
8
7
|
import {
|
|
9
8
|
replayContextPruningState,
|
|
10
9
|
setContextPruningEnabled,
|
|
11
|
-
type
|
|
10
|
+
type ContextPruneDetailsV2,
|
|
12
11
|
} from "../../shared/context-pruning-state.ts";
|
|
13
12
|
import { emitTauEvent, onTauEvent } from "../../shared/events.ts";
|
|
14
13
|
import { loadTauExtensionSettings } from "../../shared/settings/load.ts";
|
|
@@ -16,23 +15,16 @@ import { createToolRowStateStore } from "../../shared/tool-row-state.ts";
|
|
|
16
15
|
import { contextPruneParameters, executeContextPrune } from "./prune.ts";
|
|
17
16
|
import { projectContext } from "./projection.ts";
|
|
18
17
|
import {
|
|
19
|
-
|
|
18
|
+
parseContextPruningNudgeDetailsV2,
|
|
20
19
|
renderContextPruneCall,
|
|
21
20
|
renderContextPruneResult,
|
|
22
21
|
renderContextPruningNudge,
|
|
23
|
-
type
|
|
22
|
+
type ContextPruningNudgeDetailsV2,
|
|
24
23
|
} from "./render.ts";
|
|
25
24
|
import contextPruningSettings from "./settings.ts";
|
|
26
25
|
|
|
27
|
-
interface ProjectionCache {
|
|
28
|
-
generation: number;
|
|
29
|
-
input: ContextEvent["messages"];
|
|
30
|
-
stateKey: string;
|
|
31
|
-
messages: ContextEvent["messages"];
|
|
32
|
-
}
|
|
33
|
-
|
|
34
26
|
const TOOL_DESCRIPTION =
|
|
35
|
-
"Create a context
|
|
27
|
+
"Create a hard context checkpoint after broad exploration converges, when a context-pruning nudge directs it, or when stale evidence has accumulated. Everything before the checkpoint is removed from future model context unless selected for retention. Immediately before calling, state durable conclusions, conditional relevance, and the next action in visible prose.";
|
|
36
28
|
const NUDGE_MESSAGE_TYPE = "tau.context-pruning.nudge";
|
|
37
29
|
const NUDGE_BASELINE_ENTRY_TYPE = "tau.context-pruning.nudge-baseline";
|
|
38
30
|
|
|
@@ -40,15 +32,15 @@ interface NudgeState {
|
|
|
40
32
|
anchorToolCallId: string | undefined;
|
|
41
33
|
growthBaselinePercent: number | undefined;
|
|
42
34
|
highestBoundary: number;
|
|
35
|
+
highestTier: number;
|
|
36
|
+
terminalTierReached: boolean;
|
|
43
37
|
}
|
|
44
38
|
|
|
45
39
|
export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
46
40
|
let enabled = false;
|
|
47
41
|
let lifecycleGeneration = 0;
|
|
48
|
-
let projectionCache: ProjectionCache | undefined;
|
|
49
|
-
let minimumReclaimTokens = contextPruningSettings.defaults.minimumReclaimTokens;
|
|
50
42
|
let nudgeEveryPercent = contextPruningSettings.defaults.nudgeEveryPercent;
|
|
51
|
-
let
|
|
43
|
+
let nudgeInstructions = contextPruningSettings.defaults.nudgeInstructions;
|
|
52
44
|
let toolRegistered = false;
|
|
53
45
|
let commandRegistered = false;
|
|
54
46
|
let visualRows = new Set<string>();
|
|
@@ -56,9 +48,11 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
56
48
|
anchorToolCallId: undefined,
|
|
57
49
|
growthBaselinePercent: 0,
|
|
58
50
|
highestBoundary: 0,
|
|
51
|
+
highestTier: 0,
|
|
52
|
+
terminalTierReached: false,
|
|
59
53
|
};
|
|
60
54
|
const rowState = createToolRowStateStore(pi, "context-pruning.tool-row-state");
|
|
61
|
-
pi.registerMessageRenderer<
|
|
55
|
+
pi.registerMessageRenderer<ContextPruningNudgeDetailsV2>(NUDGE_MESSAGE_TYPE, (message, _options, theme) =>
|
|
62
56
|
renderContextPruningNudge(message.details, theme),
|
|
63
57
|
);
|
|
64
58
|
|
|
@@ -76,7 +70,6 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
76
70
|
|
|
77
71
|
const clearEphemeralState = () => {
|
|
78
72
|
lifecycleGeneration += 1;
|
|
79
|
-
projectionCache = undefined;
|
|
80
73
|
};
|
|
81
74
|
const setContextPruneToolActive = (active: boolean) => {
|
|
82
75
|
if (!toolRegistered) return;
|
|
@@ -105,37 +98,32 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
105
98
|
const settings = await loadTauExtensionSettings(ctx, contextPruningSettings);
|
|
106
99
|
if (generation !== lifecycleGeneration) return;
|
|
107
100
|
enabled = settings.enabled;
|
|
108
|
-
minimumReclaimTokens = settings.minimumReclaimTokens;
|
|
109
101
|
nudgeEveryPercent = settings.nudgeEveryPercent;
|
|
110
|
-
|
|
102
|
+
nudgeInstructions = settings.nudgeInstructions;
|
|
111
103
|
setContextPruningEnabled(enabled);
|
|
112
104
|
if (enabled && !toolRegistered) {
|
|
113
105
|
pi.registerTool(
|
|
114
|
-
defineTool<typeof contextPruneParameters,
|
|
106
|
+
defineTool<typeof contextPruneParameters, ContextPruneDetailsV2>({
|
|
115
107
|
name: "context_prune",
|
|
116
108
|
label: "context_prune",
|
|
117
109
|
description: TOOL_DESCRIPTION,
|
|
118
110
|
promptSnippet:
|
|
119
111
|
"Prune substantial stale tool evidence after stating durable conclusions and the next action",
|
|
120
112
|
promptGuidelines: [
|
|
121
|
-
"Use context_prune after broad exploration converges,
|
|
122
|
-
"
|
|
123
|
-
"
|
|
113
|
+
"Use context_prune after broad exploration converges, when a context-pruning nudge directs it, or when substantial irrelevant evidence has accumulated.",
|
|
114
|
+
"A final-tier context-pruning nudge means preserve durable conclusions and prune before further tool work.",
|
|
115
|
+
"Everything before context_prune leaves future model context unless selected in keepFiles or keepToolCalls, so preserve durable conclusions, user constraints, conditional relevance, and the next action in visible prose immediately before calling it.",
|
|
124
116
|
],
|
|
125
117
|
parameters: contextPruneParameters,
|
|
126
118
|
executionMode: "sequential",
|
|
127
119
|
async execute(toolCallId, params, signal, _onUpdate, executionContext) {
|
|
128
|
-
const cache = projectionCache;
|
|
129
120
|
return executeContextPrune({
|
|
130
|
-
pi,
|
|
131
121
|
toolCallId,
|
|
132
122
|
params,
|
|
133
123
|
signal,
|
|
134
124
|
ctx: executionContext,
|
|
135
|
-
|
|
125
|
+
generation: lifecycleGeneration,
|
|
136
126
|
currentGeneration: () => lifecycleGeneration,
|
|
137
|
-
currentEnabled: () => enabled,
|
|
138
|
-
minimumReclaimTokens,
|
|
139
127
|
});
|
|
140
128
|
},
|
|
141
129
|
renderCall(args, theme, context) {
|
|
@@ -170,17 +158,20 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
170
158
|
commandContext.sessionManager.getBranch(),
|
|
171
159
|
true,
|
|
172
160
|
).latestAnchorToolCallId;
|
|
173
|
-
pi.sendMessage<
|
|
161
|
+
pi.sendMessage<ContextPruningNudgeDetailsV2>(
|
|
174
162
|
{
|
|
175
163
|
customType: NUDGE_MESSAGE_TYPE,
|
|
176
164
|
content: manualPruneSteeringMessage(),
|
|
177
165
|
display: true,
|
|
178
166
|
details: {
|
|
179
|
-
v:
|
|
167
|
+
v: 2,
|
|
180
168
|
kind: "manual",
|
|
181
169
|
percent: null,
|
|
182
170
|
boundary: null,
|
|
183
|
-
|
|
171
|
+
reminder: null,
|
|
172
|
+
tier: null,
|
|
173
|
+
tierCount: null,
|
|
174
|
+
tierFloor: null,
|
|
184
175
|
anchorToolCallId: anchorToolCallId ?? null,
|
|
185
176
|
growthBaselinePercent: null,
|
|
186
177
|
},
|
|
@@ -207,7 +198,13 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
207
198
|
enabled = false;
|
|
208
199
|
setContextPruneToolActive(false);
|
|
209
200
|
visualRows.clear();
|
|
210
|
-
nudgeState = {
|
|
201
|
+
nudgeState = {
|
|
202
|
+
anchorToolCallId: undefined,
|
|
203
|
+
growthBaselinePercent: 0,
|
|
204
|
+
highestBoundary: 0,
|
|
205
|
+
highestTier: 0,
|
|
206
|
+
terminalTierReached: false,
|
|
207
|
+
};
|
|
211
208
|
setContextPruningEnabled(false);
|
|
212
209
|
pushVisualSnapshot();
|
|
213
210
|
});
|
|
@@ -233,61 +230,53 @@ export default function contextPruningExtension(pi: ExtensionAPI): void {
|
|
|
233
230
|
return undefined;
|
|
234
231
|
}
|
|
235
232
|
const baseline = nudgeState.growthBaselinePercent ?? 0;
|
|
236
|
-
|
|
237
|
-
|
|
233
|
+
const reminder = Math.floor((percent - baseline) / nudgeEveryPercent);
|
|
234
|
+
if (reminder < 1) return undefined;
|
|
235
|
+
const boundary = baseline + reminder * nudgeEveryPercent;
|
|
238
236
|
if (boundary <= nudgeState.highestBoundary) return undefined;
|
|
239
|
-
const
|
|
240
|
-
const
|
|
241
|
-
|
|
237
|
+
const tierCount = nudgeInstructions.length;
|
|
238
|
+
const tierFloor = nudgeState.terminalTierReached
|
|
239
|
+
? tierCount
|
|
240
|
+
: Math.min(nudgeState.highestTier, tierCount);
|
|
241
|
+
const tier = Math.max(Math.min(reminder, tierCount), tierFloor);
|
|
242
|
+
const instruction = nudgeInstructions[tier - 1] ?? nudgeInstructions[0];
|
|
243
|
+
const details: ContextPruningNudgeDetailsV2 = {
|
|
244
|
+
v: 2,
|
|
242
245
|
kind: "automatic",
|
|
243
246
|
percent,
|
|
244
247
|
boundary,
|
|
245
|
-
|
|
248
|
+
reminder,
|
|
249
|
+
tier,
|
|
250
|
+
tierCount,
|
|
251
|
+
tierFloor,
|
|
246
252
|
anchorToolCallId: activeAnchor ?? null,
|
|
247
253
|
growthBaselinePercent: baseline,
|
|
248
254
|
};
|
|
249
|
-
pi.sendMessage<
|
|
255
|
+
pi.sendMessage<ContextPruningNudgeDetailsV2>(
|
|
250
256
|
{
|
|
251
257
|
customType: NUDGE_MESSAGE_TYPE,
|
|
252
|
-
content: automaticPruneSteeringMessage(
|
|
258
|
+
content: automaticPruneSteeringMessage(instruction, tier === tierCount),
|
|
253
259
|
display: true,
|
|
254
260
|
details,
|
|
255
261
|
},
|
|
256
262
|
{ deliverAs: "steer" },
|
|
257
263
|
);
|
|
258
264
|
nudgeState.highestBoundary = boundary;
|
|
265
|
+
nudgeState.highestTier = Math.max(nudgeState.highestTier, tier);
|
|
266
|
+
nudgeState.terminalTierReached ||= tier === tierCount;
|
|
259
267
|
return undefined;
|
|
260
268
|
});
|
|
261
269
|
|
|
262
270
|
pi.on("context", (event, ctx) => {
|
|
263
271
|
if (!enabled) return undefined;
|
|
264
272
|
const state = replayContextPruningState(ctx.sessionManager.getBranch(), true);
|
|
265
|
-
const
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
if (
|
|
271
|
-
projectionCache?.generation === lifecycleGeneration &&
|
|
272
|
-
projectionCache.input === event.messages &&
|
|
273
|
-
projectionCache.stateKey === stateKey
|
|
274
|
-
) {
|
|
275
|
-
return { messages: projectionCache.messages };
|
|
276
|
-
}
|
|
277
|
-
|
|
278
|
-
try {
|
|
279
|
-
const messages = projectContext(event.messages, state);
|
|
280
|
-
projectionCache = { generation: lifecycleGeneration, input: event.messages, stateKey, messages };
|
|
281
|
-
const nextRows = new Set([...state.prunedToolCallIds, ...state.prunedAutoreadRowIds]);
|
|
282
|
-
if (!setsEqual(visualRows, nextRows)) {
|
|
283
|
-
visualRows = nextRows;
|
|
284
|
-
pushVisualSnapshot();
|
|
285
|
-
}
|
|
286
|
-
return { messages };
|
|
287
|
-
} catch {
|
|
288
|
-
projectionCache = undefined;
|
|
289
|
-
return undefined;
|
|
273
|
+
const messages = projectContext(event.messages, state);
|
|
274
|
+
const nextRows = new Set([...state.prunedToolCallIds, ...state.prunedAutoreadRowIds]);
|
|
275
|
+
if (!setsEqual(visualRows, nextRows)) {
|
|
276
|
+
visualRows = nextRows;
|
|
277
|
+
pushVisualSnapshot();
|
|
290
278
|
}
|
|
279
|
+
return { messages };
|
|
291
280
|
});
|
|
292
281
|
}
|
|
293
282
|
|
|
@@ -300,6 +289,8 @@ function setsEqual(left: ReadonlySet<string>, right: ReadonlySet<string>): boole
|
|
|
300
289
|
function reconstructNudgeState(branch: readonly SessionEntry[], anchorToolCallId: string | undefined): NudgeState {
|
|
301
290
|
let growthBaselinePercent = anchorToolCallId === undefined ? 0 : undefined;
|
|
302
291
|
let highestBoundary = 0;
|
|
292
|
+
let highestTier = 0;
|
|
293
|
+
let terminalTierReached = false;
|
|
303
294
|
let anchorResultIndex = -1;
|
|
304
295
|
if (anchorToolCallId !== undefined) {
|
|
305
296
|
anchorResultIndex = branch.findIndex(
|
|
@@ -326,19 +317,27 @@ function reconstructNudgeState(branch: readonly SessionEntry[], anchorToolCallId
|
|
|
326
317
|
continue;
|
|
327
318
|
}
|
|
328
319
|
if (entry.type !== "custom_message" || entry.customType !== NUDGE_MESSAGE_TYPE) continue;
|
|
329
|
-
const details =
|
|
320
|
+
const details = parseContextPruningNudgeDetailsV2(entry.details);
|
|
330
321
|
if (
|
|
331
322
|
!details ||
|
|
332
323
|
details.kind !== "automatic" ||
|
|
324
|
+
index <= anchorResultIndex ||
|
|
333
325
|
details.anchorToolCallId !== (anchorToolCallId ?? null) ||
|
|
334
326
|
details.boundary === null ||
|
|
335
|
-
details.growthBaselinePercent === null
|
|
327
|
+
details.growthBaselinePercent === null ||
|
|
328
|
+
(growthBaselinePercent !== undefined && details.growthBaselinePercent !== growthBaselinePercent)
|
|
336
329
|
)
|
|
337
330
|
continue;
|
|
331
|
+
const expectedTierFloor = terminalTierReached
|
|
332
|
+
? details.tierCount
|
|
333
|
+
: Math.min(highestTier, details.tierCount);
|
|
334
|
+
if (details.boundary <= highestBoundary || details.tierFloor !== expectedTierFloor) continue;
|
|
338
335
|
highestBoundary = Math.max(highestBoundary, details.boundary);
|
|
336
|
+
highestTier = Math.max(highestTier, details.tier);
|
|
337
|
+
terminalTierReached ||= details.tier === details.tierCount;
|
|
339
338
|
growthBaselinePercent = details.growthBaselinePercent;
|
|
340
339
|
}
|
|
341
|
-
return { anchorToolCallId, growthBaselinePercent, highestBoundary };
|
|
340
|
+
return { anchorToolCallId, growthBaselinePercent, highestBoundary, highestTier, terminalTierReached };
|
|
342
341
|
}
|
|
343
342
|
|
|
344
343
|
function parseNudgeBaseline(value: unknown): { v: 1; anchorToolCallId: string; baselinePercent: number } | undefined {
|
|
@@ -365,15 +364,16 @@ function parseNudgeBaseline(value: unknown): { v: 1; anchorToolCallId: string; b
|
|
|
365
364
|
};
|
|
366
365
|
}
|
|
367
366
|
|
|
368
|
-
function automaticPruneSteeringMessage(
|
|
367
|
+
function automaticPruneSteeringMessage(instruction: string, finalTier: boolean): string {
|
|
369
368
|
const silent =
|
|
370
369
|
"Internal context-management instruction. Follow it silently. Do not mention or acknowledge context percentages, prune messages, or internal context management.";
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
370
|
+
const protocol =
|
|
371
|
+
"When pruning, first preserve durable conclusions, user constraints, conditional relevance, and the next action in visible prose, then call context_prune.";
|
|
372
|
+
return finalTier
|
|
373
|
+
? `${silent} ${instruction} This is the final reminder tier. Create a context anchor before further tool work. ${protocol}`
|
|
374
|
+
: `${silent} ${instruction} ${protocol}`;
|
|
375
375
|
}
|
|
376
376
|
|
|
377
377
|
function manualPruneSteeringMessage(): string {
|
|
378
|
-
return "Internal context-management instruction. Follow it silently without mentioning this request. Create a context
|
|
378
|
+
return "Internal context-management instruction. Follow it silently without mentioning this request. Create a hard context checkpoint with context_prune, then continue unfinished work. First preserve durable conclusions, user constraints, conditional relevance, and the next action in visible prose.";
|
|
379
379
|
}
|
|
@@ -3,116 +3,60 @@ import type { ActiveContextPruningState } from "../../shared/context-pruning-sta
|
|
|
3
3
|
|
|
4
4
|
type ContextMessage = ContextEvent["messages"][number];
|
|
5
5
|
|
|
6
|
-
const NUDGE_MESSAGE_TYPE = "tau.context-pruning.nudge";
|
|
7
|
-
const DEFERRED_MESSAGE_TYPE = "tau.context-pruning.deferred";
|
|
8
|
-
|
|
9
6
|
export function projectContext(
|
|
10
7
|
messages: readonly ContextMessage[],
|
|
11
8
|
state: ActiveContextPruningState,
|
|
12
9
|
): ContextMessage[] {
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
const prunedToolCallIds = state.latestAnchorToolCallId === undefined ? new Set<string>() : state.prunedToolCallIds;
|
|
17
|
-
const prunedAutoreadRowIds =
|
|
18
|
-
state.latestAnchorToolCallId === undefined ? new Set<string>() : state.prunedAutoreadRowIds;
|
|
19
|
-
const projected: ContextMessage[] = [];
|
|
20
|
-
|
|
21
|
-
for (let index = 0; index < messages.length; index += 1) {
|
|
10
|
+
if (state.latestAnchorToolCallId === undefined) return [...messages];
|
|
11
|
+
let anchorIndex = -1;
|
|
12
|
+
for (let index = messages.length - 1; index >= 0; index -= 1) {
|
|
22
13
|
const message = messages[index];
|
|
23
|
-
if (message.role === "toolResult" && prunedToolCallIds.has(message.toolCallId)) continue;
|
|
24
14
|
if (
|
|
25
|
-
message
|
|
26
|
-
message.
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
)
|
|
36
|
-
continue;
|
|
37
|
-
if (message.role !== "assistant") {
|
|
38
|
-
projected.push(message);
|
|
39
|
-
continue;
|
|
15
|
+
message?.role === "assistant" &&
|
|
16
|
+
message.content.some(
|
|
17
|
+
(block) =>
|
|
18
|
+
block.type === "toolCall" &&
|
|
19
|
+
block.id === state.latestAnchorToolCallId &&
|
|
20
|
+
block.name === "context_prune",
|
|
21
|
+
)
|
|
22
|
+
) {
|
|
23
|
+
anchorIndex = index;
|
|
24
|
+
break;
|
|
40
25
|
}
|
|
41
|
-
|
|
42
|
-
const removeThinking = anchorBoundary !== undefined && index <= anchorBoundary;
|
|
43
|
-
let changed = false;
|
|
44
|
-
const content = message.content.filter((block) => {
|
|
45
|
-
const remove =
|
|
46
|
-
(removeThinking && block.type === "thinking") ||
|
|
47
|
-
(block.type === "toolCall" && (prunedToolCallIds.has(block.id) || abandonedToolCallIds.has(block.id)));
|
|
48
|
-
if (remove) changed = true;
|
|
49
|
-
return !remove;
|
|
50
|
-
});
|
|
51
|
-
if (content.length === 0) continue;
|
|
52
|
-
projected.push(changed ? { ...message, content } : message);
|
|
53
26
|
}
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
function isPrunedAutoread(value: unknown, prunedRowIds: ReadonlySet<string>): boolean {
|
|
60
|
-
if (typeof value !== "object" || value === null || Array.isArray(value)) return false;
|
|
61
|
-
const rowId = (value as Record<string, unknown>).rowId;
|
|
62
|
-
return typeof rowId === "string" && prunedRowIds.has(rowId);
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
function visibleAnchorBoundary(
|
|
66
|
-
messages: readonly ContextMessage[],
|
|
67
|
-
anchorToolCallId: string | undefined,
|
|
68
|
-
pairs: ReadonlyMap<string, { callIndex: number; resultIndex: number }>,
|
|
69
|
-
): number | undefined {
|
|
70
|
-
if (anchorToolCallId === undefined) return undefined;
|
|
71
|
-
const pair = pairs.get(anchorToolCallId);
|
|
72
|
-
if (!pair) return undefined;
|
|
73
|
-
const call = messages[pair.callIndex];
|
|
74
|
-
const result = messages[pair.resultIndex];
|
|
75
|
-
if (call?.role !== "assistant" || result?.role !== "toolResult" || result.toolName !== "context_prune")
|
|
76
|
-
return undefined;
|
|
77
|
-
const block = call.content.find((item) => item.type === "toolCall" && item.id === anchorToolCallId);
|
|
78
|
-
return block?.type === "toolCall" && block.name === "context_prune" ? pair.resultIndex : undefined;
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
function indexToolPairs(
|
|
82
|
-
messages: readonly ContextMessage[],
|
|
83
|
-
abandonedToolCallIds?: Set<string>,
|
|
84
|
-
): Map<string, { callIndex: number; resultIndex: number }> {
|
|
85
|
-
const calls = new Map<string, { index: number; aborted: boolean }>();
|
|
86
|
-
const results = new Map<string, number>();
|
|
87
|
-
for (let index = 0; index < messages.length; index += 1) {
|
|
27
|
+
if (anchorIndex < 0) return [...messages];
|
|
28
|
+
const retainedCallNames = new Map<string, string>();
|
|
29
|
+
const retainedResultNames = new Map<string, string>();
|
|
30
|
+
for (let index = 0; index < anchorIndex; index += 1) {
|
|
88
31
|
const message = messages[index];
|
|
89
|
-
if (message
|
|
32
|
+
if (message?.role === "assistant") {
|
|
90
33
|
for (const block of message.content) {
|
|
91
|
-
if (block.type
|
|
92
|
-
|
|
93
|
-
|
|
34
|
+
if (block.type === "toolCall" && state.retainedToolCallIds.has(block.id)) {
|
|
35
|
+
retainedCallNames.set(block.id, block.name);
|
|
36
|
+
}
|
|
94
37
|
}
|
|
95
|
-
} else if (message
|
|
96
|
-
|
|
97
|
-
throw new Error(`Duplicate tool result in projected context: ${message.toolCallId}`);
|
|
98
|
-
results.set(message.toolCallId, index);
|
|
38
|
+
} else if (message?.role === "toolResult" && state.retainedToolCallIds.has(message.toolCallId)) {
|
|
39
|
+
retainedResultNames.set(message.toolCallId, message.toolName);
|
|
99
40
|
}
|
|
100
41
|
}
|
|
42
|
+
const retainableToolCallIds = new Set(
|
|
43
|
+
[...retainedCallNames].flatMap(([id, name]) => (retainedResultNames.get(id) === name ? [id] : [])),
|
|
44
|
+
);
|
|
101
45
|
|
|
102
|
-
const
|
|
103
|
-
for (
|
|
104
|
-
const
|
|
105
|
-
if (
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
}
|
|
110
|
-
throw new Error(`Orphaned tool call in projected context: ${id}`);
|
|
46
|
+
const projected: ContextMessage[] = [];
|
|
47
|
+
for (let index = 0; index < anchorIndex; index += 1) {
|
|
48
|
+
const message = messages[index];
|
|
49
|
+
if (!message) continue;
|
|
50
|
+
if (message.role === "toolResult") {
|
|
51
|
+
if (retainableToolCallIds.has(message.toolCallId)) projected.push(message);
|
|
52
|
+
continue;
|
|
111
53
|
}
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
54
|
+
if (message.role !== "assistant") continue;
|
|
55
|
+
const content = message.content.filter(
|
|
56
|
+
(block) => block.type === "toolCall" && retainableToolCallIds.has(block.id),
|
|
57
|
+
);
|
|
58
|
+
if (content.length > 0) projected.push({ ...message, content });
|
|
116
59
|
}
|
|
117
|
-
|
|
60
|
+
projected.push(...messages.slice(anchorIndex));
|
|
61
|
+
return projected;
|
|
118
62
|
}
|