@shanepadgett/tau-agent 0.29.0 → 0.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/docs/context.md +14 -9
  2. package/docs/subagents.md +7 -5
  3. package/extensions/cache-diagnostics/index.ts +3 -4
  4. package/extensions/context/README.md +8 -5
  5. package/extensions/context/definitions.ts +28 -17
  6. package/extensions/context/evidence.ts +4 -3
  7. package/extensions/context/index.ts +57 -35
  8. package/extensions/context/panel.ts +10 -9
  9. package/extensions/context/projection.ts +141 -0
  10. package/extensions/context/state.ts +30 -0
  11. package/extensions/explore/ast/read/hook.ts +3 -1
  12. package/extensions/ideas/browser.ts +27 -16
  13. package/extensions/review/README.md +11 -0
  14. package/extensions/review/index.ts +135 -0
  15. package/extensions/review/model.ts +144 -0
  16. package/extensions/review/panel.ts +128 -0
  17. package/extensions/review/session.ts +106 -0
  18. package/extensions/script-runner/README.md +7 -0
  19. package/extensions/script-runner/index.ts +275 -0
  20. package/extensions/stash/browser.ts +30 -18
  21. package/extensions/subagent/README.md +17 -4
  22. package/extensions/subagent/agents/context-sync.md +2 -2
  23. package/extensions/subagent/agents/scout.md +48 -61
  24. package/extensions/subagent/agents.ts +0 -1
  25. package/extensions/subagent/cmux-dashboard.ts +7 -3
  26. package/extensions/subagent/index.ts +105 -4
  27. package/extensions/subagent/panel.ts +124 -0
  28. package/extensions/subagent/render.ts +2 -1
  29. package/extensions/subagent/run.ts +9 -9
  30. package/extensions/subagent/runtime.ts +5 -5
  31. package/extensions/subagent/session-resource.ts +19 -127
  32. package/extensions/subagent/settings.ts +18 -0
  33. package/extensions/tau-help/help.md +11 -7
  34. package/extensions/working-memory/README.md +1 -1
  35. package/extensions/working-memory/checkpoint.ts +3 -4
  36. package/extensions/working-memory/index.ts +25 -7
  37. package/extensions/working-memory/memory.ts +53 -82
  38. package/package.json +2 -2
  39. package/schemas/tau.schema.json +9 -23
  40. package/shared/context-messages.ts +19 -0
  41. package/shared/injected-context.ts +2 -2
  42. package/shared/isolated-session.ts +151 -0
  43. package/extensions/subagent/agents/review.md +0 -75
  44. package/extensions/turn-budget/README.md +0 -14
  45. package/extensions/turn-budget/index.ts +0 -116
  46. package/extensions/turn-budget/settings.ts +0 -35
@@ -0,0 +1,275 @@
1
+ import { execFileSync } from "node:child_process";
2
+ import { randomBytes } from "node:crypto";
3
+ import { mkdtemp, rm, writeFile } from "node:fs/promises";
4
+ import { tmpdir } from "node:os";
5
+ import { join } from "node:path";
6
+ import { StringEnum } from "@earendil-works/pi-ai";
7
+ import {
8
+ DEFAULT_MAX_BYTES,
9
+ DEFAULT_MAX_LINES,
10
+ defineTool,
11
+ type ExecResult,
12
+ type ExtensionAPI,
13
+ type Theme,
14
+ truncateTail,
15
+ } from "@earendil-works/pi-coding-agent";
16
+ import { Text } from "@earendil-works/pi-tui";
17
+ import { Type } from "typebox";
18
+
19
+ type Language = "python" | "typescript";
20
+
21
+ interface Runtimes {
22
+ python: string | undefined;
23
+ typescript: string | undefined;
24
+ }
25
+
26
+ interface StoredScript {
27
+ language: Language;
28
+ source: string;
29
+ }
30
+
31
+ const TIMEOUT_MS = 120_000;
32
+ const MAX_STORED = 8;
33
+
34
+ function detectRuntimes(): Runtimes {
35
+ let python: string | undefined;
36
+ for (const cmd of ["python3", "python"] as const) {
37
+ try {
38
+ execFileSync(cmd, ["--version"], { stdio: ["ignore", "pipe", "ignore"] });
39
+ python = cmd;
40
+ break;
41
+ } catch {
42
+ // runtime not installed
43
+ }
44
+ }
45
+ const [major, minor] = process.versions.node.split(".").map(Number);
46
+ let typescript: string | undefined;
47
+ if (major > 22 || (major === 22 && minor >= 6)) {
48
+ typescript = process.execPath;
49
+ }
50
+ return { python, typescript };
51
+ }
52
+
53
+ function capitalize(lang: Language): string {
54
+ return lang === "python" ? "Python" : "TypeScript";
55
+ }
56
+
57
+ function newScriptId(): string {
58
+ return randomBytes(5).toString("base64url").slice(0, 7);
59
+ }
60
+
61
+ function scrubPath(text: string, file: string, dir: string): string {
62
+ if (!text) return text;
63
+ return text.replaceAll(file, "<script>").replaceAll(dir, "<tmpdir>");
64
+ }
65
+
66
+ function applyEdits(source: string, edits: ReadonlyArray<{ oldText: string; newText: string }>): string {
67
+ let next = source;
68
+ for (const edit of edits) {
69
+ const idx = next.indexOf(edit.oldText);
70
+ if (idx === -1) {
71
+ throw new Error(
72
+ "An edits oldText was not found in the script. Copy the exact text from the script you wrote.",
73
+ );
74
+ }
75
+ next = next.slice(0, idx) + edit.newText + next.slice(idx + edit.oldText.length);
76
+ }
77
+ return next;
78
+ }
79
+
80
+ function renderEditsPreview(edits: ReadonlyArray<{ oldText: string; newText: string }>, theme: Theme): string {
81
+ return edits
82
+ .map((edit) => {
83
+ const oldLines = edit.oldText
84
+ .split("\n")
85
+ .map((line) => theme.fg("error", `- ${line}`))
86
+ .join("\n");
87
+ const newLines = edit.newText
88
+ .split("\n")
89
+ .map((line) => theme.fg("success", `+ ${line}`))
90
+ .join("\n");
91
+ return `${oldLines}\n${newLines}`;
92
+ })
93
+ .join("\n");
94
+ }
95
+
96
+ export default function scriptRunnerExtension(pi: ExtensionAPI): void {
97
+ const runtimes = detectRuntimes();
98
+ const detected = (["python", "typescript"] as const).filter(
99
+ (lang): lang is Language => (lang === "python" ? runtimes.python : runtimes.typescript) !== undefined,
100
+ );
101
+ if (detected.length === 0) return;
102
+
103
+ const langPhrase = detected.map(capitalize).join(" or ");
104
+
105
+ const scripts = new Map<string, StoredScript>();
106
+ let tempDir: string | undefined;
107
+
108
+ function remember(scriptId: string, script: StoredScript): void {
109
+ scripts.set(scriptId, script);
110
+ while (scripts.size > MAX_STORED) {
111
+ const oldest = scripts.keys().next().value;
112
+ if (oldest === undefined) break;
113
+ scripts.delete(oldest);
114
+ }
115
+ }
116
+
117
+ function resolveCommand(language: Language): string {
118
+ if (language === "python") {
119
+ const cmd = runtimes.python;
120
+ if (!cmd) throw new Error("Python is not available on this machine.");
121
+ return cmd;
122
+ }
123
+ const cmd = runtimes.typescript;
124
+ if (!cmd) throw new Error("TypeScript is not available (requires Node >= 22.6).");
125
+ return cmd;
126
+ }
127
+
128
+ async function ensureTempDir(): Promise<string> {
129
+ if (tempDir) return tempDir;
130
+ tempDir = await mkdtemp(join(tmpdir(), "tau-script-runner-"));
131
+ return tempDir;
132
+ }
133
+
134
+ async function runScript(
135
+ language: Language,
136
+ command: string,
137
+ source: string,
138
+ cwd: string,
139
+ signal: AbortSignal | undefined,
140
+ ): Promise<ExecResult> {
141
+ const dir = await ensureTempDir();
142
+ const file = join(dir, language === "python" ? "_run.py" : "_run.ts");
143
+ await writeFile(file, source, "utf8");
144
+ const args = language === "python" ? [file] : ["--experimental-strip-types", file];
145
+ const result = await pi.exec(command, args, { cwd, signal, timeout: TIMEOUT_MS });
146
+ return {
147
+ ...result,
148
+ stdout: scrubPath(result.stdout, file, dir),
149
+ stderr: scrubPath(result.stderr, file, dir),
150
+ };
151
+ }
152
+
153
+ const paramsSchema = Type.Object(
154
+ {
155
+ language: StringEnum(detected, { description: "Execution language available on this machine." }),
156
+ script: Type.Optional(
157
+ Type.String({ description: "Full script source for a new run. Omit when retrying with edits." }),
158
+ ),
159
+ scriptId: Type.Optional(
160
+ Type.String({ description: "scriptId returned by a failed run. Required when retrying with edits." }),
161
+ ),
162
+ edits: Type.Optional(
163
+ Type.Array(
164
+ Type.Object({
165
+ oldText: Type.String({ description: "Exact text currently in the script." }),
166
+ newText: Type.String({ description: "Replacement text." }),
167
+ }),
168
+ {
169
+ description:
170
+ "Targeted oldText/newText patches applied to the stored script before rerun. Prefer this over resending the whole script.",
171
+ },
172
+ ),
173
+ ),
174
+ },
175
+ { additionalProperties: false },
176
+ );
177
+
178
+ const tool = defineTool<typeof paramsSchema, undefined>({
179
+ name: "script_runner",
180
+ label: "Script Runner",
181
+ description: `Run a ${langPhrase} script in the project working directory and return its output. On a non-zero exit, returns a scriptId; retry by sending targeted {oldText,newText} edits against the script you already wrote (same language + scriptId) instead of resending the whole script. Available on this machine: ${langPhrase}.`,
182
+ promptSnippet: `Run ${langPhrase} scripts; fix failures with targeted edits instead of rewriting the whole script.`,
183
+ promptGuidelines: [
184
+ `Prefer script_runner over bash for ${langPhrase} when computation, data handling, or bulk file work is cleaner than chaining built-in tools.`,
185
+ `When script_runner fails, do not resend the full script. Call script_runner again with the same language, the returned scriptId, and an edits array of {oldText,newText} patches against the script you just wrote. Only resend a full script if the approach itself was wrong.`,
186
+ `script_runner never exposes the script file path, and you already know the script you sent. Never try to read it back.`,
187
+ ],
188
+ parameters: paramsSchema,
189
+ async execute(_toolCallId, params, signal, onUpdate, ctx) {
190
+ const language = params.language;
191
+ if (signal?.aborted) {
192
+ return { content: [{ type: "text", text: "Cancelled." }], details: undefined };
193
+ }
194
+
195
+ const command = resolveCommand(language);
196
+ const edits = params.edits;
197
+ let scriptId: string | undefined = params.scriptId;
198
+ let source: string;
199
+
200
+ if (edits && edits.length > 0) {
201
+ if (!scriptId) throw new Error("edits require the scriptId returned by the failed run.");
202
+ const stored = scripts.get(scriptId);
203
+ if (!stored) {
204
+ throw new Error(
205
+ `No stored script for scriptId ${scriptId}. It may have been evicted; resend the full script.`,
206
+ );
207
+ }
208
+ if (stored.language !== language) {
209
+ throw new Error(`Language mismatch: scriptId ${scriptId} is ${stored.language}, not ${language}.`);
210
+ }
211
+ source = applyEdits(stored.source, edits);
212
+ } else {
213
+ if (typeof params.script !== "string" || params.script.length === 0) {
214
+ throw new Error("Provide a script for a new run, or edits + scriptId to retry.");
215
+ }
216
+ source = params.script;
217
+ if (!scriptId) scriptId = newScriptId();
218
+ }
219
+
220
+ remember(scriptId, { language, source });
221
+ await onUpdate?.({ content: [{ type: "text", text: `Running ${language}...` }], details: undefined });
222
+
223
+ const result = await runScript(language, command, source, ctx.cwd, signal);
224
+ if (result.code === 0 && !result.killed) {
225
+ scripts.delete(scriptId);
226
+ const trunc = truncateTail(result.stdout.trim(), {
227
+ maxLines: DEFAULT_MAX_LINES,
228
+ maxBytes: DEFAULT_MAX_BYTES,
229
+ });
230
+ const out = trunc.content.trim();
231
+ const note = trunc.truncated
232
+ ? `\n\n[output truncated: kept tail ${trunc.outputLines} / ${trunc.totalLines} lines]`
233
+ : "";
234
+ return {
235
+ content: [{ type: "text", text: out ? `${out}${note}` : `(no output)${note}` }],
236
+ details: undefined,
237
+ };
238
+ }
239
+
240
+ const diag = result.stderr.trim() || result.stdout.trim();
241
+ const trunc = truncateTail(diag, { maxLines: DEFAULT_MAX_LINES, maxBytes: DEFAULT_MAX_BYTES });
242
+ const note = trunc.truncated
243
+ ? `\n\n[output truncated: kept tail ${trunc.outputLines} / ${trunc.totalLines} lines]`
244
+ : "";
245
+ const detail = trunc.content ? `${trunc.content}${note}\n\n` : "";
246
+ throw new Error(
247
+ `${detail}scriptId: ${scriptId}\nRetry with edits: [{oldText,newText}] using the same language and scriptId; do not resend the full script.`,
248
+ );
249
+ },
250
+ renderCall(args, theme, context) {
251
+ const text = (context.lastComponent as Text | undefined) ?? new Text("", 0, 0);
252
+ const header = `${theme.fg("toolTitle", theme.bold("script_runner"))} ${theme.fg("muted", args.language)}`;
253
+ const edits = args.edits;
254
+ const body =
255
+ edits && edits.length > 0
256
+ ? renderEditsPreview(edits, theme)
257
+ : theme.fg("accent", theme.bold(args.script ?? ""));
258
+ text.setText(body ? `${header}\n${body}` : header);
259
+ return text;
260
+ },
261
+ });
262
+
263
+ pi.registerTool(tool);
264
+
265
+ pi.on("session_shutdown", () => {
266
+ scripts.clear();
267
+ const dir = tempDir;
268
+ tempDir = undefined;
269
+ if (dir) {
270
+ void rm(dir, { recursive: true, force: true }).catch(() => {
271
+ // best-effort cleanup
272
+ });
273
+ }
274
+ });
275
+ }
@@ -6,13 +6,14 @@ import {
6
6
  type TextRecordSelectPanelConfig,
7
7
  type TextRecordSelectResult,
8
8
  } from "@shanepadgett/tau-tui";
9
+ import { errorText } from "../../shared/text.ts";
9
10
  import { loadStashes, removeStash, type Stash, stashFilePath } from "./store.ts";
10
11
 
11
- const CONFIG: Omit<TextRecordSelectPanelConfig, "path"> = {
12
+ const CONFIG: Omit<TextRecordSelectPanelConfig<Stash>, "path" | "destructiveAction"> = {
12
13
  title: "Stash",
13
14
  emptyMessage: "No stashed prompts. Use the stash shortcut while typing to stash.",
14
15
  primaryLabel: "pop",
15
- actions: [{ id: "discard", key: Key.ctrl("d"), hint: rawHint("ctrl+d", "discard") }],
16
+ actions: [],
16
17
  expandActiveItem: false,
17
18
  };
18
19
 
@@ -24,20 +25,9 @@ export async function browseStash(ctx: ExtensionCommandContext): Promise<Stash |
24
25
 
25
26
  const path = await stashFilePath(ctx.cwd);
26
27
 
27
- while (true) {
28
- const stashes = await loadStashes(ctx.cwd);
29
- const result = await show(ctx, stashes, path);
30
-
31
- if (result.kind === "cancel") return undefined;
32
- if (result.kind === "primary") return result.item;
33
-
34
- // discard: drop the stashed prompt without restoring it.
35
- const ok = await ctx.ui.confirm("Discard stashed prompt?", result.item.text);
36
- if (ok) {
37
- await removeStash(ctx.cwd, result.item.id);
38
- ctx.ui.notify("Stash discarded.", "info");
39
- }
40
- }
28
+ const stashes = await loadStashes(ctx.cwd);
29
+ const result = await show(ctx, stashes, path);
30
+ return result.kind === "primary" ? result.item : undefined;
41
31
  }
42
32
 
43
33
  async function show(
@@ -45,7 +35,29 @@ async function show(
45
35
  stashes: readonly Stash[],
46
36
  path: string,
47
37
  ): Promise<TextRecordSelectResult<Stash>> {
48
- return ctx.ui.custom<TextRecordSelectResult<Stash>>((_tui, theme, _keybindings, done) =>
49
- createTextRecordSelectPanel(theme, stashes, { ...CONFIG, path }, done),
38
+ return ctx.ui.custom<TextRecordSelectResult<Stash>>((tui, theme, _keybindings, done) =>
39
+ createTextRecordSelectPanel(
40
+ tui,
41
+ theme,
42
+ stashes,
43
+ {
44
+ ...CONFIG,
45
+ path,
46
+ destructiveAction: {
47
+ id: "discard",
48
+ key: Key.ctrl("d"),
49
+ hint: rawHint("ctrl+d", "discard"),
50
+ confirmLabel: () => "Discard stashed prompt?",
51
+ runningLabel: "Discarding stashed prompt…",
52
+ onConfirm: async (item) => {
53
+ const next = await removeStash(ctx.cwd, item.id);
54
+ ctx.ui.notify("Stash discarded.", "info");
55
+ return next;
56
+ },
57
+ onError: (error) => ctx.ui.notify(`Stash discard failed: ${errorText(error)}`, "error"),
58
+ },
59
+ },
60
+ done,
61
+ ),
50
62
  );
51
63
  }
@@ -8,17 +8,30 @@ Each fresh child also gets a display name from its agent definition. The name st
8
8
 
9
9
  Tau includes these built-in agents:
10
10
 
11
- - `review` performs adversarial, read-only code review for correctness, runtime risks, duplication, and over- or under-engineering.
12
- - `scout` finds local files, symbols, data flow, constraints, and unknowns without changing anything.
11
+ - `scout` does substantial multi-hop local code lookup that would chew parent context; paths, declarations, imports, references, call edges; facts only. Skip small digs.
13
12
  - `web-research` researches web and code sources with `websearch`, `codesearch`, and `webfetch`.
14
13
  - `context-sync` maps meaningful uncommitted work into `.pi/contexts`. Agent-driven use is `extensions.context.sync.automation` (requires `sync.enabled`). `/context-sync` is the manual/nudge path when sync is enabled. Validation can auto-run it when validation and sync are enabled.
15
14
 
16
15
  Ask Tau to delegate a task, or let it call `subagent` with an agent name and task. Children use the parent's current working directory and inherit its model and thinking level unless their definition overrides either value. They do not receive the parent conversation. Tau loads only the extensions that own a child's declared tools, so unrelated extension hooks do not run in child sessions. When a child must inspect another repository, put its exact absolute path in the delegated task.
17
16
 
18
- When the relevant files are already known, pass them with the call so Tau can autoread them into that child turn:
17
+ Run `/agents` to enable or disable agents for the current session. Press Space to stage each toggle, then Enter to apply the changes. Session choices follow the current session branch and do not change Tau settings. Agents disabled in Tau settings appear as `disabled by Tau settings` and cannot be enabled from this command. Disable agents persistently with `extensions.subagent.disabled`:
18
+
19
+ ```json
20
+ {
21
+ "extensions": {
22
+ "subagent": {
23
+ "disabled": ["web-research"]
24
+ }
25
+ }
26
+ }
27
+ ```
28
+
29
+ Disabled agents are hidden from the parent prompt and cannot start or continue a child thread.
30
+
31
+ When relevant files are already known, pass them with the call so Tau can autoread them into that child turn:
19
32
 
20
33
  ```text
21
- subagent({ agent: "review", task: "Review the runtime change", files: ["src/runtime.ts", "test/runtime.test.ts"] })
34
+ subagent({ agent: "scout", task: "Trace the runtime change", files: ["src/runtime.ts", "test/runtime.test.ts"] })
22
35
  ```
23
36
 
24
37
  Paths may be relative to the parent's current working directory or absolute. Tau reads current snapshots when the turn starts and includes line numbers so the child can cite them without another read. Missing files appear as failed autoread context; they do not stop the child. Keep the list focused because the complete snapshots use the child's context window. Files can also be supplied on a retained-thread follow-up.
@@ -23,7 +23,7 @@ thinking: high
23
23
 
24
24
  You maintain the living repository context map under `.pi/contexts`.
25
25
 
26
- Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries. Entry `files` are eager autoread paths. Entry `anchors` are lazy navigation paths. Preserve an existing path's loading class when it already appears anywhere in the catalog. New paths default to eager `files`.
26
+ Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries. Every entry has three explicit loading modes: `read` for exact complete contents, `outline` for structural Explore outlines, and `references` for unloaded navigation paths. Preserve an existing path's loading mode when it already appears anywhere in the catalog. New paths default to `references`. Promote recurring source entry points to `outline`. Use `read` only when exact wording is routinely required, such as repository instructions or a small authoritative specification.
27
27
 
28
28
  ## Tools
29
29
 
@@ -65,7 +65,7 @@ Before placing any path, answer out loud in order:
65
65
  2. **Concept** — Inside that domain, which subsystem TOML? Reuse, new, split, or merge?
66
66
  3. **Entry** — Which work scope? Update, new, split, delete, or move between concepts/domains?
67
67
  4. **Bloat** — Did this touch make an entry/concept a junk drawer? Split now if yes.
68
- 5. **Membership** — Assign files/anchors only under the winners. Every eligible changed non-deleted file must belong somewhere. Remove every stale catalog path.
68
+ 5. **Membership** — Assign read/outline/references only under the winners. Every entry must contain all three arrays, even when an array is empty. Every eligible changed non-deleted file must belong somewhere. Remove every stale catalog path.
69
69
 
70
70
  Path stuffing into the nearest feature bucket without climbing the ladder is failure.
71
71
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: scout
3
- description: Tiered, AST-first local discovery of files, declarations, data flow, constraints, and unknowns without changes
3
+ description: "Substantial multi-hop local code lookup that would chew parent context; paths, declarations, imports, references, call edges; facts only. Skip small digs"
4
4
  tools:
5
5
  - read
6
6
  - bash
@@ -14,8 +14,6 @@ tools:
14
14
  - callees
15
15
  - references
16
16
  - implementations
17
- - impact
18
- - context
19
17
  - working_memory
20
18
  names:
21
19
  - Pathfinder
@@ -27,42 +25,43 @@ model: openai-codex/gpt-5.6-luna
27
25
  thinking: high
28
26
  ---
29
27
 
30
- Stay inside task. Answer only what was asked. No mutations, side quests, background sweeps, or unasked advice.
28
+ You are a read-only repository retrieval worker. Locate requested source evidence and return exact cited facts.
31
29
 
32
- Delegating prompt controls output. Otherwise use smallest matching shape below.
30
+ Do not diagnose bugs, explain causes, infer runtime behavior, evaluate correctness, assess consequences, recommend changes, choose between alternatives, or make design decisions. The parent agent owns all interpretation and judgment.
31
+
32
+ If a task mixes lookup with judgment, perform only its concrete lookup portion and list the unanswered judgment under `Parent question`. If no concrete lookup exists, return `Parent question:` followed by the request. Do not attempt to answer it.
33
+
34
+ Stay inside task. No mutations, side quests, background sweeps, or unasked advice.
35
+
36
+ ## Allowed work
37
+
38
+ - Find files, declarations, literals, configuration values, registrations, and tests.
39
+ - List imports, references, callers, callees, implementations, and other direct syntactic relationships.
40
+ - Retrieve exact signatures or declaration bodies requested by parent.
41
+ - Confirm whether an exact source pattern exists within a stated scope.
42
+ - Report ambiguity or missing evidence without resolving it through inference.
33
43
 
34
44
  ## Evidence ladder
35
45
 
36
- Use cheapest source that proves each claim. Skip steps when task supplies exact path or declaration. Escalate only when current evidence cannot answer.
46
+ Use cheapest source that proves each returned fact. Skip steps when task supplies exact path or declaration. Escalate only when current evidence cannot complete requested lookup.
37
47
 
38
48
  1. **Supplied context** — Treat current line-numbered task files as authoritative this turn.
39
49
  2. **Paths and literals** — Use read-only `bash` (`ls`, `find`, `rg`/`grep`) for narrow path discovery, exact text, registrations, and unsupported formats. Use ranged `read` for formatting or source without structural support.
40
- 3. **Structure** — Default to `outline` for known files/packages and unfamiliar supported subtrees. Use `discover` when reuse intent is known but path or exact name is not. Use `ast_search` for source shapes.
41
- 4. **Exact declarations** — Use `show` with path + name (+ line when needed). Prefer `signature`; add docs, body, imports, or context lines only when question requires them.
42
- 5. **Focused relationships** — After resolving a declaration, use `callers`, `callees`, `references`, or `implementations` for one direct relationship question. Use `deps` and `reverse_deps` for file imports, not declaration calls.
43
- 6. **Composition** — Use `impact` for full one-hop declaration plus transitive file blast radius. Use `context` for one budgeted declaration pack when nearby bodies and relationships answer faster than separate calls.
44
-
45
- Structural results prove bounded syntax, not runtime dispatch. Preserve exact, inferred, and ambiguous labels. Do not turn ambiguous sites into claimed impact.
46
-
47
- ## Exploration discipline
50
+ 3. **Structure** — Default to `outline` for known files/packages and unfamiliar supported subtrees. Use `discover` when requested declaration path or exact name is unknown. Use `ast_search` for source shapes.
51
+ 4. **Exact declarations** — Use `show` with path + name (+ line when needed). Prefer `signature`; add docs, body, imports, or context lines only when explicitly required.
52
+ 5. **Direct relationships** — After resolving a declaration, use `callers`, `callees`, `references`, or `implementations` for one direct relationship lookup. Use `deps` and `reverse_deps` for file imports, not declaration calls.
48
53
 
49
- - Narrow each call around one unanswered claim. Prefer structural summaries and signatures over full source, and batch only independent questions whose results stay small.
50
- - Let each result reduce the search space. Do not fan out across every plausible path, repeat evidence through another tool, or use tools merely to increase coverage.
51
- - Keep a short mental set of proven facts, live unknowns, and candidate paths. Drop rejected branches as soon as evidence rules them out.
52
- - During long or branching work, use `working_memory` when stale evidence would burden the next phase: after ruling out branches, after finishing a distinct phase, before switching to a materially different search, or when reminded to reassess.
53
- - At a checkpoint, keep decisive or expensive evidence, carry active file structure as outlines when bodies are no longer needed, and defer known paths only when a clear condition would make them relevant. Continuation should preserve task, proven constraints, live unknowns, and next search step.
54
- - Do not checkpoint a small search or prune coherent evidence still needed for the current line of reasoning.
54
+ Structural results prove bounded syntax, not runtime dispatch. Preserve exact, inferred, and ambiguous labels emitted by tools. Never convert an ambiguous result into a fact.
55
55
 
56
- ## Search procedure
56
+ ## Search discipline
57
57
 
58
- 1. Extract target, question, scope, and requested output shape.
59
- 2. List required claims and select cheapest evidence for each.
60
- 3. Start from supplied paths and names. Search outward only for required relationships.
61
- 4. For reuse, run `discover`, then inspect selected candidates with `show`.
62
- 5. For unknown source shape, run `ast_search`, then inspect only selected enclosing declarations.
63
- 6. For behavior or data flow, orient target, follow focused relationships, then retrieve only declarations needed to explain flow.
64
- 7. For impact, use `impact`; use focused relationship tools only when one section needs closer evidence.
65
- 8. Stop when requested claims are supported. Put material gaps under `Unknowns`.
58
+ - Extract concrete target, lookup type, scope, and requested output shape.
59
+ - Narrow each call around one missing fact. Prefer structural summaries and signatures over full source.
60
+ - Start from supplied paths and names. Search outward only as needed to locate requested evidence.
61
+ - Batch only independent lookups whose results will stay small.
62
+ - Do not fan out across plausible explanations or collect evidence for a theory.
63
+ - Stop when requested evidence has been found or bounded search cannot find it.
64
+ - During a long inventory, use `working_memory` only when stale evidence would burden the remaining lookup. Do not checkpoint a small search.
66
65
 
67
66
  Absolute paths may point to read-only reference repositories outside cwd.
68
67
 
@@ -72,49 +71,37 @@ Use relevant sections only. Omit empty sections.
72
71
 
73
72
  ### Locate
74
73
 
75
- `path:start-end` — declaration — match reason
74
+ `path:start-end` — declaration or match — exact reason it matches
76
75
 
77
- ### Explain behavior
78
-
79
- - `Entry:` `path:start-end` — declaration
80
- - `Flow:` ordered steps; one cited fact each
81
- - `Result:` observed outcome
82
-
83
- ### Trace data
84
-
85
- - `Source:` cited origin
86
- - `Transforms:` ordered, cited transformations
87
- - `Consumers:` cited uses
88
-
89
- ### Find references or impact
76
+ ### Inventory
90
77
 
91
- - `Direct references:` cited relationships with certainty
92
- - `Editable scopes:` declarations requiring inspection or change
93
- - `Behavior affected:` evidence-backed consequences
94
- - `Unknowns:` remaining uncertainty
78
+ `path:start-end` — declaration or match — source-defined role
95
79
 
96
- ### Verify a claim
80
+ State searched scope when completeness matters.
97
81
 
98
- - `Verdict:` `yes`, `no`, `partially`, or `unknown`
99
- - `Evidence:` cited facts
100
- - `Qualification:` only when needed
82
+ ### Direct relationships
101
83
 
102
- ### Compare
84
+ - `Relationship:` caller, callee, import, reference, or implementation
85
+ - `Source:` cited declaration
86
+ - `Target:` cited declaration
87
+ - `Certainty:` exact or ambiguous
103
88
 
104
- - `Shared:` cited similarities
105
- - `Differences:` cited by aspect
106
- - `Relevant consequence:` requested consequences only
89
+ ### Exact pattern check
107
90
 
108
- ### Inventory
91
+ - `Found:` yes or no within searched scope
92
+ - `Scope:` paths or subtree searched
93
+ - `Matches:` exact citations when found
109
94
 
110
- `path:start-end` — declaration — role
95
+ ### Unresolved
111
96
 
112
- When completeness matters, state searched scope. If uncertain, say why.
97
+ - `Missing evidence:` requested lookup that could not be found
98
+ - `Ambiguity:` competing exact matches the tools could not disambiguate
99
+ - `Parent question:` diagnosis, explanation, evaluation, consequence, recommendation, or decision left to parent
113
100
 
114
101
  ## Reporting rules
115
102
 
116
- - Every material code claim needs exact path, line range, and declaration when one exists.
103
+ - Every returned code fact needs exact path and line range. Include declaration name when one exists.
117
104
  - Cite ranges returned by tools. Never estimate line numbers.
118
- - Separate fact from inference. Label inference.
119
105
  - Quote smallest useful fragment.
120
- - No preamble, search log, generic repository summary, repeated evidence, or unasked next steps.
106
+ - Describe only what source directly contains or what a structural tool directly reports.
107
+ - No preamble, search log, repository summary, causal explanation, conclusions, or next-step advice.
@@ -138,7 +138,6 @@ async function loadScope(
138
138
  if (!required) return new Map();
139
139
  const reason = error instanceof Error ? error.message : "directory unavailable";
140
140
  return new Map([
141
- ["review", [{ path: directory, name: "review", reason: `packaged agents unavailable: ${reason}` }]],
142
141
  [
143
142
  "web-research",
144
143
  [{ path: directory, name: "web-research", reason: `packaged agents unavailable: ${reason}` }],
@@ -124,13 +124,17 @@ export function formatDashboardMarkdown(snapshots: readonly SubagentInvocationSn
124
124
  lines.push("_No subagents._", "");
125
125
  return lines.join("\n");
126
126
  }
127
- lines.push("| Agent | State | Last tool | Calls | Time |", "| --- | --- | --- | ---: | ---: |");
127
+ lines.push(
128
+ "| Agent | State | Last tool | Calls | Cost | Ctx | Time |",
129
+ "| --- | --- | --- | ---: | ---: | ---: | ---: |",
130
+ );
128
131
  for (const details of ordered) {
129
132
  const latest = details.actions.at(-1);
130
133
  const currentTool = details.currentActivity?.match(/^\S+/)?.[0];
131
134
  const lastTool = currentTool ?? latest?.tool ?? "";
135
+ const ctx = typeof details.contextPercent === "number" ? `${details.contextPercent.toFixed(1)}%` : "—";
132
136
  lines.push(
133
- `| ${tableCell(`${details.displayName} (${details.agent})`, 48)} | ${dashboardState(details.status)} | ${tableCell(lastTool, 32) || "—"} | ${details.toolCalls} | ${elapsed(details.durationMs)} |`,
137
+ `| ${tableCell(`${details.displayName} (${details.agent})`, 48)} | ${dashboardState(details.status)} | ${tableCell(lastTool, 32) || "—"} | ${details.toolCalls} | $${details.usage.cost.toFixed(4)} | ${ctx} | ${elapsed(details.durationMs)} |`,
134
138
  );
135
139
  }
136
140
  lines.push("", "## Inputs", "");
@@ -138,7 +142,7 @@ export function formatDashboardMarkdown(snapshots: readonly SubagentInvocationSn
138
142
  lines.push(
139
143
  `### ${tableCell(details.displayName, 80)}`,
140
144
  "",
141
- `${tableCell(details.agent, 80)} · ${tableCell(details.invocationId, 80)}`,
145
+ tableCell(details.agent, 80),
142
146
  "",
143
147
  quote(details.task) || "> _(empty)_",
144
148
  );