@shanepadgett/tau-agent 0.28.1 → 0.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/docs/context.md +14 -9
  2. package/docs/subagents.md +7 -5
  3. package/extensions/cache-diagnostics/index.ts +4 -5
  4. package/extensions/context/README.md +8 -5
  5. package/extensions/context/definitions.ts +28 -17
  6. package/extensions/context/evidence.ts +4 -3
  7. package/extensions/context/index.ts +57 -35
  8. package/extensions/context/panel.ts +10 -9
  9. package/extensions/context/projection.ts +141 -0
  10. package/extensions/context/state.ts +30 -0
  11. package/extensions/explore/README.md +1 -1
  12. package/extensions/explore/ast/read/hook.ts +5 -35
  13. package/extensions/explore/ast/read/policy.ts +0 -32
  14. package/extensions/explore/index.ts +2 -0
  15. package/extensions/explore/outline-injection.ts +151 -0
  16. package/extensions/explore/settings.ts +0 -8
  17. package/extensions/ideas/browser.ts +27 -16
  18. package/extensions/review/README.md +11 -0
  19. package/extensions/review/index.ts +135 -0
  20. package/extensions/review/model.ts +144 -0
  21. package/extensions/review/panel.ts +128 -0
  22. package/extensions/review/session.ts +106 -0
  23. package/extensions/runtime-context/README.md +1 -1
  24. package/extensions/runtime-context/index.ts +3 -66
  25. package/extensions/script-runner/README.md +7 -0
  26. package/extensions/script-runner/index.ts +275 -0
  27. package/extensions/stash/browser.ts +30 -18
  28. package/extensions/subagent/README.md +17 -4
  29. package/extensions/subagent/agents/context-sync.md +2 -2
  30. package/extensions/subagent/agents/scout.md +49 -52
  31. package/extensions/subagent/agents.ts +0 -1
  32. package/extensions/subagent/cmux-dashboard.ts +7 -3
  33. package/extensions/subagent/index.ts +105 -4
  34. package/extensions/subagent/panel.ts +124 -0
  35. package/extensions/subagent/render.ts +2 -1
  36. package/extensions/subagent/run.ts +9 -9
  37. package/extensions/subagent/runtime.ts +5 -5
  38. package/extensions/subagent/session-resource.ts +19 -127
  39. package/extensions/subagent/settings.ts +18 -0
  40. package/extensions/tau-help/help.md +12 -8
  41. package/extensions/working-memory/README.md +9 -0
  42. package/extensions/working-memory/checkpoint.ts +175 -0
  43. package/extensions/working-memory/index.ts +338 -0
  44. package/extensions/working-memory/memory.ts +266 -0
  45. package/extensions/working-memory/render.ts +178 -0
  46. package/extensions/working-memory/settings.ts +38 -0
  47. package/extensions/working-memory/state.ts +152 -0
  48. package/package.json +2 -2
  49. package/schemas/tau.schema.json +35 -55
  50. package/shared/context-messages.ts +19 -0
  51. package/shared/events.ts +5 -0
  52. package/shared/full-file-knowledge.ts +0 -1
  53. package/shared/injected-context.ts +2 -2
  54. package/shared/isolated-session.ts +151 -0
  55. package/shared/outline-injection.ts +56 -0
  56. package/extensions/context-pruning/README.md +0 -39
  57. package/extensions/context-pruning/index.ts +0 -382
  58. package/extensions/context-pruning/projection.ts +0 -60
  59. package/extensions/context-pruning/prune.ts +0 -199
  60. package/extensions/context-pruning/render.ts +0 -251
  61. package/extensions/context-pruning/settings.ts +0 -39
  62. package/extensions/subagent/agents/review.md +0 -68
  63. package/extensions/turn-budget/README.md +0 -14
  64. package/extensions/turn-budget/index.ts +0 -116
  65. package/extensions/turn-budget/settings.ts +0 -35
  66. package/shared/context-pruning-state.ts +0 -152
@@ -1,5 +1,6 @@
1
1
  import type { Usage } from "@earendil-works/pi-ai";
2
2
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
3
+ import { createIsolatedSessionResource, type IsolatedSessionResource } from "../../shared/isolated-session.ts";
3
4
  import type { AgentDefinition } from "./agents.ts";
4
5
  import { buildColdResumePrompt, retainSubagentTurn } from "./resume.ts";
5
6
  import {
@@ -15,7 +16,6 @@ import {
15
16
  type SubagentPhase,
16
17
  type SubagentThread,
17
18
  } from "./run.ts";
18
- import { createSubagentSessionResource, type SubagentSessionResource } from "./session-resource.ts";
19
19
 
20
20
  const MAX_RETAINED_THREADS = 16;
21
21
  const GLOBAL_CONCURRENCY = 4;
@@ -216,7 +216,7 @@ export class SubagentRuntime {
216
216
  definition?: AgentDefinition;
217
217
  ctx: ExtensionContext;
218
218
  parentModel: string;
219
- parentThinking: string;
219
+ parentThinking: NonNullable<ExtensionContext["thinkingLevel"]>;
220
220
  signal?: AbortSignal;
221
221
  onUpdate?: (details: SubagentDetails) => void | Promise<void>;
222
222
  resolveFreshDefinition: () => Promise<
@@ -286,7 +286,7 @@ export class SubagentRuntime {
286
286
  threadKey?: string;
287
287
  ctx: ExtensionContext;
288
288
  parentModel: string;
289
- parentThinking: string;
289
+ parentThinking: NonNullable<ExtensionContext["thinkingLevel"]>;
290
290
  definition?: AgentDefinition;
291
291
  signal?: AbortSignal;
292
292
  onUpdate?: (details: SubagentDetails) => void | Promise<void>;
@@ -316,7 +316,7 @@ export class SubagentRuntime {
316
316
  let releaseGlobal: (() => void) | undefined;
317
317
  let reservedThread: TrackedThread | undefined;
318
318
  let provisionalThread: TrackedThread | undefined;
319
- let provisionalResource: SubagentSessionResource | undefined;
319
+ let provisionalResource: IsolatedSessionResource | undefined;
320
320
  let reservationToken: symbol | undefined;
321
321
  let admitAdvanced = false;
322
322
  let phase: SubagentPhase = "queue";
@@ -513,7 +513,7 @@ export class SubagentRuntime {
513
513
  phase = "startup";
514
514
  fanOut({ ...active.snapshot, status: "starting", phase: "startup", agent, threadId: thread.id }, true);
515
515
  const oldResource = thread.resource;
516
- provisionalResource = await createSubagentSessionResource(thread.sessionInputs, combined);
516
+ provisionalResource = await createIsolatedSessionResource(thread.sessionInputs, combined);
517
517
  if (!this.isLive(generation, combined) || thread.disposed || this.threads.get(thread.id) !== thread) {
518
518
  await provisionalResource.dispose();
519
519
  provisionalResource = undefined;
@@ -1,12 +1,5 @@
1
- import {
2
- createAgentSession,
3
- DefaultResourceLoader,
4
- getAgentDir,
5
- ModelRuntime,
6
- SessionManager,
7
- type AgentSession,
8
- type ExtensionContext,
9
- } from "@earendil-works/pi-coding-agent";
1
+ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
2
+ import { resolveIsolatedSessionModel, type IsolatedSessionInputs } from "../../shared/isolated-session.ts";
10
3
  import type { AgentDefinition, ThinkingLevel } from "./agents.ts";
11
4
 
12
5
  const CHILD_UI_BLOCKED_METHODS = new Set([
@@ -18,25 +11,10 @@ const CHILD_UI_BLOCKED_METHODS = new Set([
18
11
  "setWorkingIndicator",
19
12
  ]);
20
13
 
21
- type SelectedModel = NonNullable<ExtensionContext["model"]>;
22
- type SelectedProvider = NonNullable<ReturnType<ExtensionContext["modelRegistry"]["getProvider"]>>;
23
-
24
- export interface SubagentSessionInputs {
14
+ export interface SubagentSessionInputs extends IsolatedSessionInputs {
25
15
  definition: AgentDefinition;
26
- extensionPaths: readonly string[];
27
- cwd: string;
28
- model: SelectedModel;
29
16
  modelName: string;
30
- provider: SelectedProvider;
31
- runtimeApiKey: string | undefined;
32
17
  thinkingLevel: ThinkingLevel;
33
- bindTarget: { mode: "print" } | { mode: "tui"; uiContext: ExtensionContext["ui"] };
34
- }
35
-
36
- export interface SubagentSessionResource {
37
- readonly inputs: SubagentSessionInputs;
38
- readonly session: AgentSession;
39
- dispose(): Promise<void>;
40
18
  }
41
19
 
42
20
  function childUiContext(ui: ExtensionContext["ui"]): ExtensionContext["ui"] {
@@ -53,117 +31,31 @@ export async function resolveSubagentSessionInputs(options: {
53
31
  definition: AgentDefinition;
54
32
  extensionPaths: readonly string[];
55
33
  ctx: ExtensionContext;
56
- parentThinkingLevel: string;
34
+ parentThinkingLevel: NonNullable<ExtensionContext["thinkingLevel"]>;
57
35
  signal: AbortSignal;
58
36
  onWarning?: (warning: string) => void;
59
37
  }): Promise<SubagentSessionInputs> {
60
38
  const { definition, extensionPaths, ctx, signal, onWarning } = options;
61
- let model = ctx.model;
62
- let thinkingLevel = options.parentThinkingLevel as ThinkingLevel;
63
- let selectedAuth: Awaited<ReturnType<ExtensionContext["modelRegistry"]["getApiKeyAndHeaders"]>> | undefined;
64
- if (definition.model) {
65
- const separator = definition.model.indexOf("/");
66
- const configured = ctx.modelRegistry.find(
67
- definition.model.slice(0, separator),
68
- definition.model.slice(separator + 1),
69
- );
70
- if (!configured) onWarning?.(`model ${definition.model} is unavailable; using parent model`);
71
- else {
72
- const auth = await ctx.modelRegistry.getApiKeyAndHeaders(configured);
73
- if (!auth.ok) onWarning?.(`model ${definition.model} is unavailable: ${auth.error}; using parent model`);
74
- else {
75
- model = configured;
76
- selectedAuth = auth;
77
- }
78
- }
79
- }
80
- if (definition.thinking) {
81
- const mapped = model?.thinkingLevelMap?.[definition.thinking];
82
- const unsupported =
83
- !model?.reasoning ||
84
- mapped === null ||
85
- ((definition.thinking === "xhigh" || definition.thinking === "max") && mapped === undefined);
86
- if (unsupported)
87
- onWarning?.(`thinking ${definition.thinking} is unavailable for the selected model; using parent thinking`);
88
- else thinkingLevel = definition.thinking;
89
- }
90
- if (!model) throw new Error(`Agent ${definition.name} startup failed: parent has no model`);
91
- const auth = selectedAuth ?? (await ctx.modelRegistry.getApiKeyAndHeaders(model));
92
- if (!auth.ok) throw new Error(`Agent ${definition.name} startup failed: ${auth.error}`);
93
- const provider = ctx.modelRegistry.getProvider(model.provider);
94
- if (!provider) throw new Error(`Agent ${definition.name} startup failed: provider ${model.provider} is unavailable`);
95
- if (signal.aborted) throw new Error(`Agent ${definition.name} startup aborted`);
39
+ const selected = await resolveIsolatedSessionModel({
40
+ label: `Agent ${definition.name}`,
41
+ preferredModel: definition.model,
42
+ preferredThinkingLevel: definition.thinking,
43
+ usePreferredThinkingAfterModelFallback: true,
44
+ ctx,
45
+ parentThinkingLevel: options.parentThinkingLevel,
46
+ signal,
47
+ ...(onWarning === undefined ? {} : { onWarning }),
48
+ });
96
49
  return {
50
+ ...selected,
51
+ label: `Agent ${definition.name}`,
97
52
  definition,
98
53
  extensionPaths: [...extensionPaths],
99
54
  cwd: ctx.cwd,
100
- model,
101
- modelName: `${model.provider}/${model.id}`,
102
- provider,
103
- runtimeApiKey:
104
- auth.apiKey && provider.auth.apiKey && !ctx.modelRegistry.isUsingOAuth(model) ? auth.apiKey : undefined,
105
- thinkingLevel,
55
+ modelName: `${selected.model.provider}/${selected.model.id}`,
56
+ tools: definition.tools,
57
+ customTools: [],
106
58
  bindTarget:
107
59
  ctx.mode === "tui" && ctx.hasUI ? { mode: "tui", uiContext: childUiContext(ctx.ui) } : { mode: "print" },
108
60
  };
109
61
  }
110
-
111
- export async function createSubagentSessionResource(
112
- inputs: SubagentSessionInputs,
113
- signal: AbortSignal,
114
- ): Promise<SubagentSessionResource> {
115
- let session: AgentSession | undefined;
116
- try {
117
- if (signal.aborted) throw new Error(`Agent ${inputs.definition.name} startup aborted`);
118
- const modelRuntime = await ModelRuntime.create();
119
- modelRuntime.registerNativeProvider(inputs.provider);
120
- if (inputs.runtimeApiKey !== undefined)
121
- await modelRuntime.setRuntimeApiKey(inputs.model.provider, inputs.runtimeApiKey, { allowNetwork: false });
122
- if (signal.aborted) throw new Error(`Agent ${inputs.definition.name} startup aborted`);
123
- const resourceLoader = new DefaultResourceLoader({
124
- cwd: inputs.cwd,
125
- agentDir: getAgentDir(),
126
- noExtensions: true,
127
- additionalExtensionPaths: [...inputs.extensionPaths],
128
- });
129
- await resourceLoader.reload();
130
- if (signal.aborted) throw new Error(`Agent ${inputs.definition.name} startup aborted`);
131
- const created = await createAgentSession({
132
- cwd: inputs.cwd,
133
- model: inputs.model,
134
- modelRuntime,
135
- thinkingLevel: inputs.thinkingLevel,
136
- tools: inputs.definition.tools,
137
- excludeTools: ["subagent"],
138
- resourceLoader,
139
- sessionManager: SessionManager.inMemory(inputs.cwd),
140
- });
141
- session = created.session;
142
- if (signal.aborted) throw new Error(`Agent ${inputs.definition.name} startup aborted`);
143
- await session.bindExtensions(inputs.bindTarget);
144
- if (signal.aborted) throw new Error(`Agent ${inputs.definition.name} startup aborted`);
145
- const active = session.getActiveToolNames().sort();
146
- const expected = [...inputs.definition.tools].sort();
147
- if (active.join("\0") !== expected.join("\0") || active.includes("subagent")) {
148
- const missing = expected.filter((tool) => !active.includes(tool));
149
- throw new Error(
150
- `Agent ${inputs.definition.name} startup failed: unavailable tools: ${missing.join(", ") || "active tool mismatch"}`,
151
- );
152
- }
153
- let disposed = false;
154
- return {
155
- inputs,
156
- session,
157
- async dispose() {
158
- if (disposed) return;
159
- disposed = true;
160
- if (session?.isStreaming) await session.abort().catch(() => undefined);
161
- session?.dispose();
162
- },
163
- };
164
- } catch (error) {
165
- if (session?.isStreaming) await session.abort().catch(() => undefined);
166
- session?.dispose();
167
- throw error;
168
- }
169
- }
@@ -0,0 +1,18 @@
1
+ import { Type } from "typebox";
2
+ import { defineTauExtensionSettings } from "../../shared/settings/define.ts";
3
+
4
+ export default defineTauExtensionSettings({
5
+ key: "subagent",
6
+ defaults: { disabled: [] as string[] },
7
+ schema: Type.Object(
8
+ {
9
+ disabled: Type.Optional(
10
+ Type.Array(Type.String({ minLength: 1 }), {
11
+ default: [],
12
+ description: "Agent names unavailable for delegation.",
13
+ }),
14
+ ),
15
+ },
16
+ { additionalProperties: false },
17
+ ),
18
+ });
@@ -32,11 +32,11 @@ Adds `/commit` for semantic commit grouping, review, and committing selected rep
32
32
 
33
33
  ## context
34
34
 
35
- Adds `/context` to select reusable repository work scopes from `.pi/contexts`, and `/context-sync` or `/context-sync <nudge>` for human-driven catalog sync (optional nudge). Escape cancels a running manual sync. When `sync.automation` is on, the coding agent can also run the `context-sync` subagent after meaningful uncommitted work. Sync catalogs durable code and long-lived documentation; recurring scratch, planning, interview, and rough-idea paths belong in `validation.ignoreGlobs`. `sync.enabled` is the master switch for command, automation, and validation auto-run. Entry `files` are autoread; entry `anchors` are unloaded navigation paths. Context validation is off by default; when on (and sync enabled), Tau auto-runs context-sync on failure. Folder names are tabs, TOML files are concepts, and TOML sections are selectable entries.
35
+ Adds `/context` to set branch-local reusable repository work scopes from `.pi/contexts`, and `/context-sync` or `/context-sync <nudge>` for human-driven catalog sync. Active entries produce one ephemeral per-call projection instead of transcript messages. Entry `read` paths supply exact contents, `outline` paths use Explore, and `references` stay unloaded. Clear all selections and confirm to remove active context. Escape cancels a running manual sync. When `sync.automation` is on, coding agent can also run `context-sync` after meaningful uncommitted work. Sync catalogs durable code and long-lived documentation; recurring scratch, planning, interview, and rough-idea paths belong in `validation.ignoreGlobs`. `sync.enabled` is master switch for command, automation, and validation auto-run. Context validation is off by default; when on (and sync enabled), Tau auto-runs context-sync on failure. Folder names are tabs, TOML files are concepts, and TOML sections are selectable entries.
36
36
 
37
- ## context-pruning
37
+ ## working-memory
38
38
 
39
- Gives the agent `context_prune` for creating a hard context checkpoint after broad exploration converges. Everything before the checkpoint leaves future model input unless the agent retains an exact tool exchange or carries a file forward as a fresh snapshot. File snapshot failures are reported without blocking the checkpoint. Deferred files remain as a small conditional note. Context markers use an ordered instruction ladder that escalates from informational guidance to pruning before further tool work. Run `/prune` with no arguments to request a checkpoint manually. Session history remains unchanged, and checkpoints, markers, and pruned-row state follow the active branch.
39
+ Gives agent `working_memory` for selective hard checkpoints. Model-only references identify useful conversation evidence and complete tool exchanges. Requested source files return as structural outlines, while deferred files remain cheap conditional reminders. Everything else before checkpoint leaves future model input without changing saved session. Advisory reminders begin at 40k active-context tokens. Run `/prune` to request reassessment manually.
40
40
 
41
41
  ## explore
42
42
 
@@ -74,6 +74,10 @@ Adds `/publish` to create a tagged release, trigger trusted npm publishing in Gi
74
74
 
75
75
  Adds `/qna` for when the agent has asked you several questions in chat and you want a friendly UI for answering them on your own terms. It is only active when you manually run the command.
76
76
 
77
+ ## review
78
+
79
+ Adds `/review` for explicit isolated review of current Git changes. Choose `simplify`, `architecture`, or `correctness`, or run a mode directly. Results stay outside agent context until you send them from result view, and can be exported under `.pi/tau/reviews/`. `/review show` reopens latest result on current session branch.
80
+
77
81
  ## reference
78
82
 
79
83
  Adds `/reference` to manage separate repositories kept outside the current project for inspiration or comparison. Add one with `/reference new <git-url>`, update it, switch its referenced branch, or open it in an editor. Select references and explain why they matter; Tau then puts their paths and that reason into the editor for the agent. References stay outside the project so the agent does not wander into unrelated code unless you explicitly point it there.
@@ -86,6 +90,10 @@ Shows a compact display-only marker after each run with wall time and model cost
86
90
 
87
91
  Supplies the agent with the current local date and an initial root directory snapshot as hidden session context.
88
92
 
93
+ ## script-runner
94
+
95
+ Gives the agent a first-class `script_runner` tool to execute Python and TypeScript instead of bash. On failure it returns a `scriptId`; the agent retries with targeted `{oldText,newText}` edits against the script it already wrote rather than resending the whole script. Languages are detected from the environment (Python via `python3`/`python`; TypeScript via Node `--experimental-strip-types`, Node 22.6+). The tool registers only available languages and is hidden from the prompt if neither is present.
96
+
89
97
  ## silent-command-runner
90
98
 
91
99
  Runs configured commands while keeping their output out of agent context when that is useful.
@@ -100,7 +108,7 @@ Adds `Alt+S` to stash the current prompt draft and `/pop` to browse stashed draf
100
108
 
101
109
  ## subagent
102
110
 
103
- Gives Tau a subagent delegation tool for isolated, focused work. The built-in `review` agent performs adversarial, read-only code reviews; `scout` finds local files, symbols, data flow, constraints, and unknowns without changing anything; `web-research` handles external research. Known files can be autoread as line-numbered snapshots into a fresh or retained child turn. Tau can continue a retained child thread when follow-up work depends on its prior reads and reasoning. You can also create your own subagents in the supported subagent directories. Ask Tau how to do it and have it consult the extension’s own documentation; the built-in agents show the pattern. Each subagent can register its own model, tools, and pool of display names. Reused pool names get numeric suffixes. In interactive cmux sessions, Tau opens one temporary Markdown dashboard for live subagent progress; it does not change how children run and closes shortly after the active cohort finishes.
111
+ Gives Tau a subagent delegation tool for isolated, focused work. Run `/agents` to enable or disable individual agents for the current session, or set `extensions.subagent.disabled` in Tau settings for a persistent choice. `scout` is substantial multi-hop local code lookup that would chew parent context; facts only, not small digs; `web-research` handles external research. Known files can be autoread as line-numbered snapshots into a fresh or retained child turn. Tau can continue a retained child thread when follow-up work depends on its prior reads and reasoning. You can also create your own subagents in supported subagent directories. Ask Tau how to do it and have it consult extension documentation; built-in agents show pattern. Each subagent can register its own model, tools, and pool of display names. Reused pool names get numeric suffixes. In interactive cmux sessions, Tau opens one temporary Markdown dashboard for live subagent progress; it does not change how children run and closes shortly after active cohort finishes.
104
112
 
105
113
  ## tau-help
106
114
 
@@ -114,10 +122,6 @@ Adds `/tau`, `/tau init [--global|--project]`, and `/tau doctor` for Tau setup a
114
122
 
115
123
  Progressively exposes specialist tools through `load_tools`. Tau normally loads the fixed `web`, `image`, and `appshot` groups itself when needed; supported providers can preserve more prompt-cache reuse.
116
124
 
117
- ## turn-budget
118
-
119
- Tracks and limits agent turns to keep work bounded.
120
-
121
125
  ## web
122
126
 
123
127
  Gives the agent compact `websearch`, `webfetch`, and `codesearch` tools for web and implementation research.
@@ -0,0 +1,9 @@
1
+ # Working Memory
2
+
3
+ Working Memory gives agent selective checkpoints without changing saved conversation.
4
+
5
+ `working_memory` keeps referenced conversation evidence and complete tool exchanges, carries requested source files as structural outlines, and records deferred files as conditional reminders. Everything else before checkpoint leaves future model context.
6
+
7
+ Automatic reminders begin at 40,000 active-context tokens and remain advisory. Run `/prune` to request reassessment manually.
8
+
9
+ Settings live under `extensions.workingMemory`.
@@ -0,0 +1,175 @@
1
+ import { resolve } from "node:path";
2
+ import { type ExtensionAPI, type ExtensionContext, type SessionEntry } from "@earendil-works/pi-coding-agent";
3
+ import { type Static, Type } from "typebox";
4
+ import { truncateBoundedHead } from "../../shared/bounded-text-result.ts";
5
+ import { requestOutlineInjections, type PreparedOutlineInjection } from "../../shared/outline-injection.ts";
6
+ import { buildMemoryCatalog } from "./memory.ts";
7
+ import { WORKING_MEMORY_TOOL, type DeferredFile, type WorkingMemoryCheckpointDetailsV1 } from "./state.ts";
8
+
9
+ const PATH = Type.String({ minLength: 1, maxLength: 500, pattern: "\\S" });
10
+
11
+ export const workingMemoryParameters = Type.Object(
12
+ {
13
+ continuation: Type.String({ minLength: 1, maxLength: 8_000, pattern: "\\S" }),
14
+ keep: Type.Array(Type.String({ minLength: 3, maxLength: 100 }), { maxItems: 100 }),
15
+ outlineFiles: Type.Array(PATH, { maxItems: 12 }),
16
+ deferFiles: Type.Array(
17
+ Type.Object(
18
+ {
19
+ path: PATH,
20
+ reason: Type.String({ minLength: 1, maxLength: 300, pattern: "\\S" }),
21
+ relevantWhen: Type.String({ minLength: 1, maxLength: 300, pattern: "\\S" }),
22
+ },
23
+ { additionalProperties: false },
24
+ ),
25
+ { maxItems: 8 },
26
+ ),
27
+ },
28
+ { additionalProperties: false },
29
+ );
30
+
31
+ export type WorkingMemoryInput = Static<typeof workingMemoryParameters>;
32
+
33
+ interface ExecuteWorkingMemoryOptions {
34
+ pi: Pick<ExtensionAPI, "events">;
35
+ toolCallId: string;
36
+ params: WorkingMemoryInput;
37
+ signal: AbortSignal | undefined;
38
+ ctx: ExtensionContext;
39
+ generation: number;
40
+ currentGeneration(): number;
41
+ }
42
+
43
+ export interface WorkingMemoryExecution {
44
+ result: {
45
+ content: Array<{ type: "text"; text: string }>;
46
+ details: WorkingMemoryCheckpointDetailsV1;
47
+ };
48
+ outlines: PreparedOutlineInjection[];
49
+ }
50
+
51
+ export async function executeWorkingMemory(options: ExecuteWorkingMemoryOptions): Promise<WorkingMemoryExecution> {
52
+ assertCurrent(options);
53
+ const branch = options.ctx.sessionManager.getBranch();
54
+ const catalog = buildMemoryCatalog(branch);
55
+ const requestedRefs = [...new Set(options.params.keep)];
56
+ const retained = requestedRefs
57
+ .flatMap((ref) => {
58
+ const unit = catalog.get(ref);
59
+ return unit ? [unit] : [];
60
+ })
61
+ .sort((left, right) => left.order - right.order || left.suborder - right.suborder);
62
+ const retainedRefs = retained.map((unit) => unit.ref);
63
+ const warnings = requestedRefs
64
+ .filter((ref) => !catalog.has(ref))
65
+ .map((ref) => `${ref}: memory reference is unavailable and was not retained`);
66
+
67
+ const outlinePaths = dedupePaths(options.params.outlineFiles, options.ctx.cwd);
68
+ const outlineResponse = await requestOutlineInjections(options.pi, {
69
+ cwd: options.ctx.cwd,
70
+ batchId: options.toolCallId,
71
+ paths: outlinePaths,
72
+ signal: options.signal,
73
+ isLifecycleCurrent: () => options.generation === options.currentGeneration(),
74
+ });
75
+ warnings.push(...outlineResponse.warnings.map((warning) => boundedWarning(warning)));
76
+ assertCurrent(options);
77
+
78
+ const outlinedKeys = new Set(
79
+ outlineResponse.messages.map((message) => resolve(options.ctx.cwd, message.details.path)),
80
+ );
81
+ const deferredFiles: DeferredFile[] = [];
82
+ const deferredKeys = new Set<string>();
83
+ for (const file of options.params.deferFiles) {
84
+ const path = normalizePath(file.path);
85
+ const key = resolve(options.ctx.cwd, path);
86
+ if (outlinedKeys.has(key) || deferredKeys.has(key)) continue;
87
+ deferredKeys.add(key);
88
+ deferredFiles.push({ path, reason: file.reason.trim(), relevantWhen: file.relevantWhen.trim() });
89
+ }
90
+
91
+ const anchorIndex = findAnchorEntry(branch, options.toolCallId);
92
+ const retainedSet = new Set(retainedRefs);
93
+ const preAnchorUnits = [...catalog.values()].filter((unit) => unit.order < anchorIndex);
94
+ const prunedRowIds = [...new Set(preAnchorUnits.flatMap((unit) => (retainedSet.has(unit.ref) ? [] : unit.rowIds)))];
95
+ const details: WorkingMemoryCheckpointDetailsV1 = {
96
+ v: 1,
97
+ anchorToolCallId: options.toolCallId,
98
+ retainedRefs,
99
+ retainedLabels: retained.map((unit) => ({ ref: unit.ref, label: unit.label, preview: unit.preview })),
100
+ prunedRowIds,
101
+ outlinedFiles: outlineResponse.messages.map((message) => ({
102
+ path: message.details.path,
103
+ rowId: message.details.rowId,
104
+ })),
105
+ deferredFiles,
106
+ removedUnits: Math.max(0, preAnchorUnits.length - retained.length),
107
+ warnings,
108
+ };
109
+ const markdown = formatResult(options.params.continuation.trim(), deferredFiles, warnings);
110
+ return {
111
+ result: { content: [{ type: "text", text: truncateBoundedHead(markdown).content }], details },
112
+ outlines: outlineResponse.messages,
113
+ };
114
+ }
115
+
116
+ function formatResult(continuation: string, deferred: readonly DeferredFile[], warnings: readonly string[]): string {
117
+ const sections = [`## Continue\n\n${continuation}`];
118
+ if (deferred.length > 0) {
119
+ sections.push(
120
+ `## Deferred files\n\n${deferred
121
+ .map((file) => `- \`${escapeCode(file.path)}\` — ${file.reason} Reconsider when: ${file.relevantWhen}.`)
122
+ .join("\n")}`,
123
+ );
124
+ }
125
+ if (warnings.length > 0) sections.push(`## Warnings\n\n${warnings.map((warning) => `- ${warning}`).join("\n")}`);
126
+ return sections.join("\n\n");
127
+ }
128
+
129
+ function dedupePaths(paths: readonly string[], cwd: string): string[] {
130
+ const keys = new Set<string>();
131
+ const result: string[] = [];
132
+ for (const raw of paths) {
133
+ const path = normalizePath(raw);
134
+ const key = resolve(cwd, path);
135
+ if (keys.has(key)) continue;
136
+ keys.add(key);
137
+ result.push(path);
138
+ }
139
+ return result;
140
+ }
141
+
142
+ function findAnchorEntry(branch: readonly SessionEntry[], toolCallId: string): number {
143
+ for (let index = branch.length - 1; index >= 0; index -= 1) {
144
+ const entry = branch[index];
145
+ if (
146
+ entry?.type === "message" &&
147
+ entry.message.role === "assistant" &&
148
+ entry.message.content.some(
149
+ (block) => block.type === "toolCall" && block.id === toolCallId && block.name === WORKING_MEMORY_TOOL,
150
+ )
151
+ ) {
152
+ return index;
153
+ }
154
+ }
155
+ return -1;
156
+ }
157
+
158
+ function normalizePath(path: string): string {
159
+ return path.trim().replace(/^@/, "");
160
+ }
161
+
162
+ function escapeCode(path: string): string {
163
+ return path.replaceAll("`", "\\`");
164
+ }
165
+
166
+ function boundedWarning(warning: string): string {
167
+ return warning.length <= 500 ? warning : `${warning.slice(0, 499)}…`;
168
+ }
169
+
170
+ function assertCurrent(options: ExecuteWorkingMemoryOptions): void {
171
+ options.signal?.throwIfAborted();
172
+ if (options.generation !== options.currentGeneration()) {
173
+ throw new Error("Working-memory checkpoint crossed a session lifecycle boundary");
174
+ }
175
+ }