@shanepadgett/tau-agent 0.29.0 → 0.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/context.md +14 -9
- package/docs/subagents.md +7 -5
- package/extensions/cache-diagnostics/index.ts +3 -4
- package/extensions/context/README.md +8 -5
- package/extensions/context/definitions.ts +28 -17
- package/extensions/context/evidence.ts +4 -3
- package/extensions/context/index.ts +57 -35
- package/extensions/context/panel.ts +10 -9
- package/extensions/context/projection.ts +141 -0
- package/extensions/context/state.ts +30 -0
- package/extensions/explore/ast/read/hook.ts +3 -1
- package/extensions/ideas/browser.ts +27 -16
- package/extensions/review/README.md +11 -0
- package/extensions/review/index.ts +135 -0
- package/extensions/review/model.ts +144 -0
- package/extensions/review/panel.ts +128 -0
- package/extensions/review/session.ts +106 -0
- package/extensions/script-runner/README.md +7 -0
- package/extensions/script-runner/index.ts +275 -0
- package/extensions/stash/browser.ts +30 -18
- package/extensions/subagent/README.md +17 -4
- package/extensions/subagent/agents/context-sync.md +2 -2
- package/extensions/subagent/agents/scout.md +48 -61
- package/extensions/subagent/agents.ts +0 -1
- package/extensions/subagent/cmux-dashboard.ts +7 -3
- package/extensions/subagent/index.ts +105 -4
- package/extensions/subagent/panel.ts +124 -0
- package/extensions/subagent/render.ts +2 -1
- package/extensions/subagent/run.ts +9 -9
- package/extensions/subagent/runtime.ts +5 -5
- package/extensions/subagent/session-resource.ts +19 -127
- package/extensions/subagent/settings.ts +18 -0
- package/extensions/tau-help/help.md +11 -7
- package/extensions/working-memory/README.md +1 -1
- package/extensions/working-memory/checkpoint.ts +3 -4
- package/extensions/working-memory/index.ts +25 -7
- package/extensions/working-memory/memory.ts +53 -82
- package/package.json +2 -2
- package/schemas/tau.schema.json +9 -23
- package/shared/context-messages.ts +19 -0
- package/shared/injected-context.ts +2 -2
- package/shared/isolated-session.ts +151 -0
- package/extensions/subagent/agents/review.md +0 -75
- package/extensions/turn-budget/README.md +0 -14
- package/extensions/turn-budget/index.ts +0 -116
- package/extensions/turn-budget/settings.ts +0 -35
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
import { execFileSync } from "node:child_process";
|
|
2
|
+
import { randomBytes } from "node:crypto";
|
|
3
|
+
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import { join } from "node:path";
|
|
6
|
+
import { StringEnum } from "@earendil-works/pi-ai";
|
|
7
|
+
import {
|
|
8
|
+
DEFAULT_MAX_BYTES,
|
|
9
|
+
DEFAULT_MAX_LINES,
|
|
10
|
+
defineTool,
|
|
11
|
+
type ExecResult,
|
|
12
|
+
type ExtensionAPI,
|
|
13
|
+
type Theme,
|
|
14
|
+
truncateTail,
|
|
15
|
+
} from "@earendil-works/pi-coding-agent";
|
|
16
|
+
import { Text } from "@earendil-works/pi-tui";
|
|
17
|
+
import { Type } from "typebox";
|
|
18
|
+
|
|
19
|
+
type Language = "python" | "typescript";
|
|
20
|
+
|
|
21
|
+
interface Runtimes {
|
|
22
|
+
python: string | undefined;
|
|
23
|
+
typescript: string | undefined;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
interface StoredScript {
|
|
27
|
+
language: Language;
|
|
28
|
+
source: string;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const TIMEOUT_MS = 120_000;
|
|
32
|
+
const MAX_STORED = 8;
|
|
33
|
+
|
|
34
|
+
function detectRuntimes(): Runtimes {
|
|
35
|
+
let python: string | undefined;
|
|
36
|
+
for (const cmd of ["python3", "python"] as const) {
|
|
37
|
+
try {
|
|
38
|
+
execFileSync(cmd, ["--version"], { stdio: ["ignore", "pipe", "ignore"] });
|
|
39
|
+
python = cmd;
|
|
40
|
+
break;
|
|
41
|
+
} catch {
|
|
42
|
+
// runtime not installed
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
const [major, minor] = process.versions.node.split(".").map(Number);
|
|
46
|
+
let typescript: string | undefined;
|
|
47
|
+
if (major > 22 || (major === 22 && minor >= 6)) {
|
|
48
|
+
typescript = process.execPath;
|
|
49
|
+
}
|
|
50
|
+
return { python, typescript };
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function capitalize(lang: Language): string {
|
|
54
|
+
return lang === "python" ? "Python" : "TypeScript";
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function newScriptId(): string {
|
|
58
|
+
return randomBytes(5).toString("base64url").slice(0, 7);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function scrubPath(text: string, file: string, dir: string): string {
|
|
62
|
+
if (!text) return text;
|
|
63
|
+
return text.replaceAll(file, "<script>").replaceAll(dir, "<tmpdir>");
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function applyEdits(source: string, edits: ReadonlyArray<{ oldText: string; newText: string }>): string {
|
|
67
|
+
let next = source;
|
|
68
|
+
for (const edit of edits) {
|
|
69
|
+
const idx = next.indexOf(edit.oldText);
|
|
70
|
+
if (idx === -1) {
|
|
71
|
+
throw new Error(
|
|
72
|
+
"An edits oldText was not found in the script. Copy the exact text from the script you wrote.",
|
|
73
|
+
);
|
|
74
|
+
}
|
|
75
|
+
next = next.slice(0, idx) + edit.newText + next.slice(idx + edit.oldText.length);
|
|
76
|
+
}
|
|
77
|
+
return next;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function renderEditsPreview(edits: ReadonlyArray<{ oldText: string; newText: string }>, theme: Theme): string {
|
|
81
|
+
return edits
|
|
82
|
+
.map((edit) => {
|
|
83
|
+
const oldLines = edit.oldText
|
|
84
|
+
.split("\n")
|
|
85
|
+
.map((line) => theme.fg("error", `- ${line}`))
|
|
86
|
+
.join("\n");
|
|
87
|
+
const newLines = edit.newText
|
|
88
|
+
.split("\n")
|
|
89
|
+
.map((line) => theme.fg("success", `+ ${line}`))
|
|
90
|
+
.join("\n");
|
|
91
|
+
return `${oldLines}\n${newLines}`;
|
|
92
|
+
})
|
|
93
|
+
.join("\n");
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export default function scriptRunnerExtension(pi: ExtensionAPI): void {
|
|
97
|
+
const runtimes = detectRuntimes();
|
|
98
|
+
const detected = (["python", "typescript"] as const).filter(
|
|
99
|
+
(lang): lang is Language => (lang === "python" ? runtimes.python : runtimes.typescript) !== undefined,
|
|
100
|
+
);
|
|
101
|
+
if (detected.length === 0) return;
|
|
102
|
+
|
|
103
|
+
const langPhrase = detected.map(capitalize).join(" or ");
|
|
104
|
+
|
|
105
|
+
const scripts = new Map<string, StoredScript>();
|
|
106
|
+
let tempDir: string | undefined;
|
|
107
|
+
|
|
108
|
+
function remember(scriptId: string, script: StoredScript): void {
|
|
109
|
+
scripts.set(scriptId, script);
|
|
110
|
+
while (scripts.size > MAX_STORED) {
|
|
111
|
+
const oldest = scripts.keys().next().value;
|
|
112
|
+
if (oldest === undefined) break;
|
|
113
|
+
scripts.delete(oldest);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function resolveCommand(language: Language): string {
|
|
118
|
+
if (language === "python") {
|
|
119
|
+
const cmd = runtimes.python;
|
|
120
|
+
if (!cmd) throw new Error("Python is not available on this machine.");
|
|
121
|
+
return cmd;
|
|
122
|
+
}
|
|
123
|
+
const cmd = runtimes.typescript;
|
|
124
|
+
if (!cmd) throw new Error("TypeScript is not available (requires Node >= 22.6).");
|
|
125
|
+
return cmd;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
async function ensureTempDir(): Promise<string> {
|
|
129
|
+
if (tempDir) return tempDir;
|
|
130
|
+
tempDir = await mkdtemp(join(tmpdir(), "tau-script-runner-"));
|
|
131
|
+
return tempDir;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
async function runScript(
|
|
135
|
+
language: Language,
|
|
136
|
+
command: string,
|
|
137
|
+
source: string,
|
|
138
|
+
cwd: string,
|
|
139
|
+
signal: AbortSignal | undefined,
|
|
140
|
+
): Promise<ExecResult> {
|
|
141
|
+
const dir = await ensureTempDir();
|
|
142
|
+
const file = join(dir, language === "python" ? "_run.py" : "_run.ts");
|
|
143
|
+
await writeFile(file, source, "utf8");
|
|
144
|
+
const args = language === "python" ? [file] : ["--experimental-strip-types", file];
|
|
145
|
+
const result = await pi.exec(command, args, { cwd, signal, timeout: TIMEOUT_MS });
|
|
146
|
+
return {
|
|
147
|
+
...result,
|
|
148
|
+
stdout: scrubPath(result.stdout, file, dir),
|
|
149
|
+
stderr: scrubPath(result.stderr, file, dir),
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const paramsSchema = Type.Object(
|
|
154
|
+
{
|
|
155
|
+
language: StringEnum(detected, { description: "Execution language available on this machine." }),
|
|
156
|
+
script: Type.Optional(
|
|
157
|
+
Type.String({ description: "Full script source for a new run. Omit when retrying with edits." }),
|
|
158
|
+
),
|
|
159
|
+
scriptId: Type.Optional(
|
|
160
|
+
Type.String({ description: "scriptId returned by a failed run. Required when retrying with edits." }),
|
|
161
|
+
),
|
|
162
|
+
edits: Type.Optional(
|
|
163
|
+
Type.Array(
|
|
164
|
+
Type.Object({
|
|
165
|
+
oldText: Type.String({ description: "Exact text currently in the script." }),
|
|
166
|
+
newText: Type.String({ description: "Replacement text." }),
|
|
167
|
+
}),
|
|
168
|
+
{
|
|
169
|
+
description:
|
|
170
|
+
"Targeted oldText/newText patches applied to the stored script before rerun. Prefer this over resending the whole script.",
|
|
171
|
+
},
|
|
172
|
+
),
|
|
173
|
+
),
|
|
174
|
+
},
|
|
175
|
+
{ additionalProperties: false },
|
|
176
|
+
);
|
|
177
|
+
|
|
178
|
+
const tool = defineTool<typeof paramsSchema, undefined>({
|
|
179
|
+
name: "script_runner",
|
|
180
|
+
label: "Script Runner",
|
|
181
|
+
description: `Run a ${langPhrase} script in the project working directory and return its output. On a non-zero exit, returns a scriptId; retry by sending targeted {oldText,newText} edits against the script you already wrote (same language + scriptId) instead of resending the whole script. Available on this machine: ${langPhrase}.`,
|
|
182
|
+
promptSnippet: `Run ${langPhrase} scripts; fix failures with targeted edits instead of rewriting the whole script.`,
|
|
183
|
+
promptGuidelines: [
|
|
184
|
+
`Prefer script_runner over bash for ${langPhrase} when computation, data handling, or bulk file work is cleaner than chaining built-in tools.`,
|
|
185
|
+
`When script_runner fails, do not resend the full script. Call script_runner again with the same language, the returned scriptId, and an edits array of {oldText,newText} patches against the script you just wrote. Only resend a full script if the approach itself was wrong.`,
|
|
186
|
+
`script_runner never exposes the script file path, and you already know the script you sent. Never try to read it back.`,
|
|
187
|
+
],
|
|
188
|
+
parameters: paramsSchema,
|
|
189
|
+
async execute(_toolCallId, params, signal, onUpdate, ctx) {
|
|
190
|
+
const language = params.language;
|
|
191
|
+
if (signal?.aborted) {
|
|
192
|
+
return { content: [{ type: "text", text: "Cancelled." }], details: undefined };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
const command = resolveCommand(language);
|
|
196
|
+
const edits = params.edits;
|
|
197
|
+
let scriptId: string | undefined = params.scriptId;
|
|
198
|
+
let source: string;
|
|
199
|
+
|
|
200
|
+
if (edits && edits.length > 0) {
|
|
201
|
+
if (!scriptId) throw new Error("edits require the scriptId returned by the failed run.");
|
|
202
|
+
const stored = scripts.get(scriptId);
|
|
203
|
+
if (!stored) {
|
|
204
|
+
throw new Error(
|
|
205
|
+
`No stored script for scriptId ${scriptId}. It may have been evicted; resend the full script.`,
|
|
206
|
+
);
|
|
207
|
+
}
|
|
208
|
+
if (stored.language !== language) {
|
|
209
|
+
throw new Error(`Language mismatch: scriptId ${scriptId} is ${stored.language}, not ${language}.`);
|
|
210
|
+
}
|
|
211
|
+
source = applyEdits(stored.source, edits);
|
|
212
|
+
} else {
|
|
213
|
+
if (typeof params.script !== "string" || params.script.length === 0) {
|
|
214
|
+
throw new Error("Provide a script for a new run, or edits + scriptId to retry.");
|
|
215
|
+
}
|
|
216
|
+
source = params.script;
|
|
217
|
+
if (!scriptId) scriptId = newScriptId();
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
remember(scriptId, { language, source });
|
|
221
|
+
await onUpdate?.({ content: [{ type: "text", text: `Running ${language}...` }], details: undefined });
|
|
222
|
+
|
|
223
|
+
const result = await runScript(language, command, source, ctx.cwd, signal);
|
|
224
|
+
if (result.code === 0 && !result.killed) {
|
|
225
|
+
scripts.delete(scriptId);
|
|
226
|
+
const trunc = truncateTail(result.stdout.trim(), {
|
|
227
|
+
maxLines: DEFAULT_MAX_LINES,
|
|
228
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
229
|
+
});
|
|
230
|
+
const out = trunc.content.trim();
|
|
231
|
+
const note = trunc.truncated
|
|
232
|
+
? `\n\n[output truncated: kept tail ${trunc.outputLines} / ${trunc.totalLines} lines]`
|
|
233
|
+
: "";
|
|
234
|
+
return {
|
|
235
|
+
content: [{ type: "text", text: out ? `${out}${note}` : `(no output)${note}` }],
|
|
236
|
+
details: undefined,
|
|
237
|
+
};
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
const diag = result.stderr.trim() || result.stdout.trim();
|
|
241
|
+
const trunc = truncateTail(diag, { maxLines: DEFAULT_MAX_LINES, maxBytes: DEFAULT_MAX_BYTES });
|
|
242
|
+
const note = trunc.truncated
|
|
243
|
+
? `\n\n[output truncated: kept tail ${trunc.outputLines} / ${trunc.totalLines} lines]`
|
|
244
|
+
: "";
|
|
245
|
+
const detail = trunc.content ? `${trunc.content}${note}\n\n` : "";
|
|
246
|
+
throw new Error(
|
|
247
|
+
`${detail}scriptId: ${scriptId}\nRetry with edits: [{oldText,newText}] using the same language and scriptId; do not resend the full script.`,
|
|
248
|
+
);
|
|
249
|
+
},
|
|
250
|
+
renderCall(args, theme, context) {
|
|
251
|
+
const text = (context.lastComponent as Text | undefined) ?? new Text("", 0, 0);
|
|
252
|
+
const header = `${theme.fg("toolTitle", theme.bold("script_runner"))} ${theme.fg("muted", args.language)}`;
|
|
253
|
+
const edits = args.edits;
|
|
254
|
+
const body =
|
|
255
|
+
edits && edits.length > 0
|
|
256
|
+
? renderEditsPreview(edits, theme)
|
|
257
|
+
: theme.fg("accent", theme.bold(args.script ?? ""));
|
|
258
|
+
text.setText(body ? `${header}\n${body}` : header);
|
|
259
|
+
return text;
|
|
260
|
+
},
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
pi.registerTool(tool);
|
|
264
|
+
|
|
265
|
+
pi.on("session_shutdown", () => {
|
|
266
|
+
scripts.clear();
|
|
267
|
+
const dir = tempDir;
|
|
268
|
+
tempDir = undefined;
|
|
269
|
+
if (dir) {
|
|
270
|
+
void rm(dir, { recursive: true, force: true }).catch(() => {
|
|
271
|
+
// best-effort cleanup
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
});
|
|
275
|
+
}
|
|
@@ -6,13 +6,14 @@ import {
|
|
|
6
6
|
type TextRecordSelectPanelConfig,
|
|
7
7
|
type TextRecordSelectResult,
|
|
8
8
|
} from "@shanepadgett/tau-tui";
|
|
9
|
+
import { errorText } from "../../shared/text.ts";
|
|
9
10
|
import { loadStashes, removeStash, type Stash, stashFilePath } from "./store.ts";
|
|
10
11
|
|
|
11
|
-
const CONFIG: Omit<TextRecordSelectPanelConfig
|
|
12
|
+
const CONFIG: Omit<TextRecordSelectPanelConfig<Stash>, "path" | "destructiveAction"> = {
|
|
12
13
|
title: "Stash",
|
|
13
14
|
emptyMessage: "No stashed prompts. Use the stash shortcut while typing to stash.",
|
|
14
15
|
primaryLabel: "pop",
|
|
15
|
-
actions: [
|
|
16
|
+
actions: [],
|
|
16
17
|
expandActiveItem: false,
|
|
17
18
|
};
|
|
18
19
|
|
|
@@ -24,20 +25,9 @@ export async function browseStash(ctx: ExtensionCommandContext): Promise<Stash |
|
|
|
24
25
|
|
|
25
26
|
const path = await stashFilePath(ctx.cwd);
|
|
26
27
|
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
if (result.kind === "cancel") return undefined;
|
|
32
|
-
if (result.kind === "primary") return result.item;
|
|
33
|
-
|
|
34
|
-
// discard: drop the stashed prompt without restoring it.
|
|
35
|
-
const ok = await ctx.ui.confirm("Discard stashed prompt?", result.item.text);
|
|
36
|
-
if (ok) {
|
|
37
|
-
await removeStash(ctx.cwd, result.item.id);
|
|
38
|
-
ctx.ui.notify("Stash discarded.", "info");
|
|
39
|
-
}
|
|
40
|
-
}
|
|
28
|
+
const stashes = await loadStashes(ctx.cwd);
|
|
29
|
+
const result = await show(ctx, stashes, path);
|
|
30
|
+
return result.kind === "primary" ? result.item : undefined;
|
|
41
31
|
}
|
|
42
32
|
|
|
43
33
|
async function show(
|
|
@@ -45,7 +35,29 @@ async function show(
|
|
|
45
35
|
stashes: readonly Stash[],
|
|
46
36
|
path: string,
|
|
47
37
|
): Promise<TextRecordSelectResult<Stash>> {
|
|
48
|
-
return ctx.ui.custom<TextRecordSelectResult<Stash>>((
|
|
49
|
-
createTextRecordSelectPanel(
|
|
38
|
+
return ctx.ui.custom<TextRecordSelectResult<Stash>>((tui, theme, _keybindings, done) =>
|
|
39
|
+
createTextRecordSelectPanel(
|
|
40
|
+
tui,
|
|
41
|
+
theme,
|
|
42
|
+
stashes,
|
|
43
|
+
{
|
|
44
|
+
...CONFIG,
|
|
45
|
+
path,
|
|
46
|
+
destructiveAction: {
|
|
47
|
+
id: "discard",
|
|
48
|
+
key: Key.ctrl("d"),
|
|
49
|
+
hint: rawHint("ctrl+d", "discard"),
|
|
50
|
+
confirmLabel: () => "Discard stashed prompt?",
|
|
51
|
+
runningLabel: "Discarding stashed prompt…",
|
|
52
|
+
onConfirm: async (item) => {
|
|
53
|
+
const next = await removeStash(ctx.cwd, item.id);
|
|
54
|
+
ctx.ui.notify("Stash discarded.", "info");
|
|
55
|
+
return next;
|
|
56
|
+
},
|
|
57
|
+
onError: (error) => ctx.ui.notify(`Stash discard failed: ${errorText(error)}`, "error"),
|
|
58
|
+
},
|
|
59
|
+
},
|
|
60
|
+
done,
|
|
61
|
+
),
|
|
50
62
|
);
|
|
51
63
|
}
|
|
@@ -8,17 +8,30 @@ Each fresh child also gets a display name from its agent definition. The name st
|
|
|
8
8
|
|
|
9
9
|
Tau includes these built-in agents:
|
|
10
10
|
|
|
11
|
-
- `
|
|
12
|
-
- `scout` finds local files, symbols, data flow, constraints, and unknowns without changing anything.
|
|
11
|
+
- `scout` does substantial multi-hop local code lookup that would chew parent context; paths, declarations, imports, references, call edges; facts only. Skip small digs.
|
|
13
12
|
- `web-research` researches web and code sources with `websearch`, `codesearch`, and `webfetch`.
|
|
14
13
|
- `context-sync` maps meaningful uncommitted work into `.pi/contexts`. Agent-driven use is `extensions.context.sync.automation` (requires `sync.enabled`). `/context-sync` is the manual/nudge path when sync is enabled. Validation can auto-run it when validation and sync are enabled.
|
|
15
14
|
|
|
16
15
|
Ask Tau to delegate a task, or let it call `subagent` with an agent name and task. Children use the parent's current working directory and inherit its model and thinking level unless their definition overrides either value. They do not receive the parent conversation. Tau loads only the extensions that own a child's declared tools, so unrelated extension hooks do not run in child sessions. When a child must inspect another repository, put its exact absolute path in the delegated task.
|
|
17
16
|
|
|
18
|
-
|
|
17
|
+
Run `/agents` to enable or disable agents for the current session. Press Space to stage each toggle, then Enter to apply the changes. Session choices follow the current session branch and do not change Tau settings. Agents disabled in Tau settings appear as `disabled by Tau settings` and cannot be enabled from this command. Disable agents persistently with `extensions.subagent.disabled`:
|
|
18
|
+
|
|
19
|
+
```json
|
|
20
|
+
{
|
|
21
|
+
"extensions": {
|
|
22
|
+
"subagent": {
|
|
23
|
+
"disabled": ["web-research"]
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Disabled agents are hidden from the parent prompt and cannot start or continue a child thread.
|
|
30
|
+
|
|
31
|
+
When relevant files are already known, pass them with the call so Tau can autoread them into that child turn:
|
|
19
32
|
|
|
20
33
|
```text
|
|
21
|
-
subagent({ agent: "
|
|
34
|
+
subagent({ agent: "scout", task: "Trace the runtime change", files: ["src/runtime.ts", "test/runtime.test.ts"] })
|
|
22
35
|
```
|
|
23
36
|
|
|
24
37
|
Paths may be relative to the parent's current working directory or absolute. Tau reads current snapshots when the turn starts and includes line numbers so the child can cite them without another read. Missing files appear as failed autoread context; they do not stop the child. Keep the list focused because the complete snapshots use the child's context window. Files can also be supplied on a retained-thread follow-up.
|
|
@@ -23,7 +23,7 @@ thinking: high
|
|
|
23
23
|
|
|
24
24
|
You maintain the living repository context map under `.pi/contexts`.
|
|
25
25
|
|
|
26
|
-
Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries.
|
|
26
|
+
Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries. Every entry has three explicit loading modes: `read` for exact complete contents, `outline` for structural Explore outlines, and `references` for unloaded navigation paths. Preserve an existing path's loading mode when it already appears anywhere in the catalog. New paths default to `references`. Promote recurring source entry points to `outline`. Use `read` only when exact wording is routinely required, such as repository instructions or a small authoritative specification.
|
|
27
27
|
|
|
28
28
|
## Tools
|
|
29
29
|
|
|
@@ -65,7 +65,7 @@ Before placing any path, answer out loud in order:
|
|
|
65
65
|
2. **Concept** — Inside that domain, which subsystem TOML? Reuse, new, split, or merge?
|
|
66
66
|
3. **Entry** — Which work scope? Update, new, split, delete, or move between concepts/domains?
|
|
67
67
|
4. **Bloat** — Did this touch make an entry/concept a junk drawer? Split now if yes.
|
|
68
|
-
5. **Membership** — Assign
|
|
68
|
+
5. **Membership** — Assign read/outline/references only under the winners. Every entry must contain all three arrays, even when an array is empty. Every eligible changed non-deleted file must belong somewhere. Remove every stale catalog path.
|
|
69
69
|
|
|
70
70
|
Path stuffing into the nearest feature bucket without climbing the ladder is failure.
|
|
71
71
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: scout
|
|
3
|
-
description:
|
|
3
|
+
description: "Substantial multi-hop local code lookup that would chew parent context; paths, declarations, imports, references, call edges; facts only. Skip small digs"
|
|
4
4
|
tools:
|
|
5
5
|
- read
|
|
6
6
|
- bash
|
|
@@ -14,8 +14,6 @@ tools:
|
|
|
14
14
|
- callees
|
|
15
15
|
- references
|
|
16
16
|
- implementations
|
|
17
|
-
- impact
|
|
18
|
-
- context
|
|
19
17
|
- working_memory
|
|
20
18
|
names:
|
|
21
19
|
- Pathfinder
|
|
@@ -27,42 +25,43 @@ model: openai-codex/gpt-5.6-luna
|
|
|
27
25
|
thinking: high
|
|
28
26
|
---
|
|
29
27
|
|
|
30
|
-
|
|
28
|
+
You are a read-only repository retrieval worker. Locate requested source evidence and return exact cited facts.
|
|
31
29
|
|
|
32
|
-
|
|
30
|
+
Do not diagnose bugs, explain causes, infer runtime behavior, evaluate correctness, assess consequences, recommend changes, choose between alternatives, or make design decisions. The parent agent owns all interpretation and judgment.
|
|
31
|
+
|
|
32
|
+
If a task mixes lookup with judgment, perform only its concrete lookup portion and list the unanswered judgment under `Parent question`. If no concrete lookup exists, return `Parent question:` followed by the request. Do not attempt to answer it.
|
|
33
|
+
|
|
34
|
+
Stay inside task. No mutations, side quests, background sweeps, or unasked advice.
|
|
35
|
+
|
|
36
|
+
## Allowed work
|
|
37
|
+
|
|
38
|
+
- Find files, declarations, literals, configuration values, registrations, and tests.
|
|
39
|
+
- List imports, references, callers, callees, implementations, and other direct syntactic relationships.
|
|
40
|
+
- Retrieve exact signatures or declaration bodies requested by parent.
|
|
41
|
+
- Confirm whether an exact source pattern exists within a stated scope.
|
|
42
|
+
- Report ambiguity or missing evidence without resolving it through inference.
|
|
33
43
|
|
|
34
44
|
## Evidence ladder
|
|
35
45
|
|
|
36
|
-
Use cheapest source that proves each
|
|
46
|
+
Use cheapest source that proves each returned fact. Skip steps when task supplies exact path or declaration. Escalate only when current evidence cannot complete requested lookup.
|
|
37
47
|
|
|
38
48
|
1. **Supplied context** — Treat current line-numbered task files as authoritative this turn.
|
|
39
49
|
2. **Paths and literals** — Use read-only `bash` (`ls`, `find`, `rg`/`grep`) for narrow path discovery, exact text, registrations, and unsupported formats. Use ranged `read` for formatting or source without structural support.
|
|
40
|
-
3. **Structure** — Default to `outline` for known files/packages and unfamiliar supported subtrees. Use `discover` when
|
|
41
|
-
4. **Exact declarations** — Use `show` with path + name (+ line when needed). Prefer `signature`; add docs, body, imports, or context lines only when
|
|
42
|
-
5. **
|
|
43
|
-
6. **Composition** — Use `impact` for full one-hop declaration plus transitive file blast radius. Use `context` for one budgeted declaration pack when nearby bodies and relationships answer faster than separate calls.
|
|
44
|
-
|
|
45
|
-
Structural results prove bounded syntax, not runtime dispatch. Preserve exact, inferred, and ambiguous labels. Do not turn ambiguous sites into claimed impact.
|
|
46
|
-
|
|
47
|
-
## Exploration discipline
|
|
50
|
+
3. **Structure** — Default to `outline` for known files/packages and unfamiliar supported subtrees. Use `discover` when requested declaration path or exact name is unknown. Use `ast_search` for source shapes.
|
|
51
|
+
4. **Exact declarations** — Use `show` with path + name (+ line when needed). Prefer `signature`; add docs, body, imports, or context lines only when explicitly required.
|
|
52
|
+
5. **Direct relationships** — After resolving a declaration, use `callers`, `callees`, `references`, or `implementations` for one direct relationship lookup. Use `deps` and `reverse_deps` for file imports, not declaration calls.
|
|
48
53
|
|
|
49
|
-
|
|
50
|
-
- Let each result reduce the search space. Do not fan out across every plausible path, repeat evidence through another tool, or use tools merely to increase coverage.
|
|
51
|
-
- Keep a short mental set of proven facts, live unknowns, and candidate paths. Drop rejected branches as soon as evidence rules them out.
|
|
52
|
-
- During long or branching work, use `working_memory` when stale evidence would burden the next phase: after ruling out branches, after finishing a distinct phase, before switching to a materially different search, or when reminded to reassess.
|
|
53
|
-
- At a checkpoint, keep decisive or expensive evidence, carry active file structure as outlines when bodies are no longer needed, and defer known paths only when a clear condition would make them relevant. Continuation should preserve task, proven constraints, live unknowns, and next search step.
|
|
54
|
-
- Do not checkpoint a small search or prune coherent evidence still needed for the current line of reasoning.
|
|
54
|
+
Structural results prove bounded syntax, not runtime dispatch. Preserve exact, inferred, and ambiguous labels emitted by tools. Never convert an ambiguous result into a fact.
|
|
55
55
|
|
|
56
|
-
## Search
|
|
56
|
+
## Search discipline
|
|
57
57
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
8. Stop when requested claims are supported. Put material gaps under `Unknowns`.
|
|
58
|
+
- Extract concrete target, lookup type, scope, and requested output shape.
|
|
59
|
+
- Narrow each call around one missing fact. Prefer structural summaries and signatures over full source.
|
|
60
|
+
- Start from supplied paths and names. Search outward only as needed to locate requested evidence.
|
|
61
|
+
- Batch only independent lookups whose results will stay small.
|
|
62
|
+
- Do not fan out across plausible explanations or collect evidence for a theory.
|
|
63
|
+
- Stop when requested evidence has been found or bounded search cannot find it.
|
|
64
|
+
- During a long inventory, use `working_memory` only when stale evidence would burden the remaining lookup. Do not checkpoint a small search.
|
|
66
65
|
|
|
67
66
|
Absolute paths may point to read-only reference repositories outside cwd.
|
|
68
67
|
|
|
@@ -72,49 +71,37 @@ Use relevant sections only. Omit empty sections.
|
|
|
72
71
|
|
|
73
72
|
### Locate
|
|
74
73
|
|
|
75
|
-
`path:start-end` — declaration
|
|
74
|
+
`path:start-end` — declaration or match — exact reason it matches
|
|
76
75
|
|
|
77
|
-
###
|
|
78
|
-
|
|
79
|
-
- `Entry:` `path:start-end` — declaration
|
|
80
|
-
- `Flow:` ordered steps; one cited fact each
|
|
81
|
-
- `Result:` observed outcome
|
|
82
|
-
|
|
83
|
-
### Trace data
|
|
84
|
-
|
|
85
|
-
- `Source:` cited origin
|
|
86
|
-
- `Transforms:` ordered, cited transformations
|
|
87
|
-
- `Consumers:` cited uses
|
|
88
|
-
|
|
89
|
-
### Find references or impact
|
|
76
|
+
### Inventory
|
|
90
77
|
|
|
91
|
-
-
|
|
92
|
-
- `Editable scopes:` declarations requiring inspection or change
|
|
93
|
-
- `Behavior affected:` evidence-backed consequences
|
|
94
|
-
- `Unknowns:` remaining uncertainty
|
|
78
|
+
`path:start-end` — declaration or match — source-defined role
|
|
95
79
|
|
|
96
|
-
|
|
80
|
+
State searched scope when completeness matters.
|
|
97
81
|
|
|
98
|
-
|
|
99
|
-
- `Evidence:` cited facts
|
|
100
|
-
- `Qualification:` only when needed
|
|
82
|
+
### Direct relationships
|
|
101
83
|
|
|
102
|
-
|
|
84
|
+
- `Relationship:` caller, callee, import, reference, or implementation
|
|
85
|
+
- `Source:` cited declaration
|
|
86
|
+
- `Target:` cited declaration
|
|
87
|
+
- `Certainty:` exact or ambiguous
|
|
103
88
|
|
|
104
|
-
|
|
105
|
-
- `Differences:` cited by aspect
|
|
106
|
-
- `Relevant consequence:` requested consequences only
|
|
89
|
+
### Exact pattern check
|
|
107
90
|
|
|
108
|
-
|
|
91
|
+
- `Found:` yes or no within searched scope
|
|
92
|
+
- `Scope:` paths or subtree searched
|
|
93
|
+
- `Matches:` exact citations when found
|
|
109
94
|
|
|
110
|
-
|
|
95
|
+
### Unresolved
|
|
111
96
|
|
|
112
|
-
|
|
97
|
+
- `Missing evidence:` requested lookup that could not be found
|
|
98
|
+
- `Ambiguity:` competing exact matches the tools could not disambiguate
|
|
99
|
+
- `Parent question:` diagnosis, explanation, evaluation, consequence, recommendation, or decision left to parent
|
|
113
100
|
|
|
114
101
|
## Reporting rules
|
|
115
102
|
|
|
116
|
-
- Every
|
|
103
|
+
- Every returned code fact needs exact path and line range. Include declaration name when one exists.
|
|
117
104
|
- Cite ranges returned by tools. Never estimate line numbers.
|
|
118
|
-
- Separate fact from inference. Label inference.
|
|
119
105
|
- Quote smallest useful fragment.
|
|
120
|
-
-
|
|
106
|
+
- Describe only what source directly contains or what a structural tool directly reports.
|
|
107
|
+
- No preamble, search log, repository summary, causal explanation, conclusions, or next-step advice.
|
|
@@ -138,7 +138,6 @@ async function loadScope(
|
|
|
138
138
|
if (!required) return new Map();
|
|
139
139
|
const reason = error instanceof Error ? error.message : "directory unavailable";
|
|
140
140
|
return new Map([
|
|
141
|
-
["review", [{ path: directory, name: "review", reason: `packaged agents unavailable: ${reason}` }]],
|
|
142
141
|
[
|
|
143
142
|
"web-research",
|
|
144
143
|
[{ path: directory, name: "web-research", reason: `packaged agents unavailable: ${reason}` }],
|
|
@@ -124,13 +124,17 @@ export function formatDashboardMarkdown(snapshots: readonly SubagentInvocationSn
|
|
|
124
124
|
lines.push("_No subagents._", "");
|
|
125
125
|
return lines.join("\n");
|
|
126
126
|
}
|
|
127
|
-
lines.push(
|
|
127
|
+
lines.push(
|
|
128
|
+
"| Agent | State | Last tool | Calls | Cost | Ctx | Time |",
|
|
129
|
+
"| --- | --- | --- | ---: | ---: | ---: | ---: |",
|
|
130
|
+
);
|
|
128
131
|
for (const details of ordered) {
|
|
129
132
|
const latest = details.actions.at(-1);
|
|
130
133
|
const currentTool = details.currentActivity?.match(/^\S+/)?.[0];
|
|
131
134
|
const lastTool = currentTool ?? latest?.tool ?? "";
|
|
135
|
+
const ctx = typeof details.contextPercent === "number" ? `${details.contextPercent.toFixed(1)}%` : "—";
|
|
132
136
|
lines.push(
|
|
133
|
-
`| ${tableCell(`${details.displayName} (${details.agent})`, 48)} | ${dashboardState(details.status)} | ${tableCell(lastTool, 32) || "—"} | ${details.toolCalls} | ${elapsed(details.durationMs)} |`,
|
|
137
|
+
`| ${tableCell(`${details.displayName} (${details.agent})`, 48)} | ${dashboardState(details.status)} | ${tableCell(lastTool, 32) || "—"} | ${details.toolCalls} | $${details.usage.cost.toFixed(4)} | ${ctx} | ${elapsed(details.durationMs)} |`,
|
|
134
138
|
);
|
|
135
139
|
}
|
|
136
140
|
lines.push("", "## Inputs", "");
|
|
@@ -138,7 +142,7 @@ export function formatDashboardMarkdown(snapshots: readonly SubagentInvocationSn
|
|
|
138
142
|
lines.push(
|
|
139
143
|
`### ${tableCell(details.displayName, 80)}`,
|
|
140
144
|
"",
|
|
141
|
-
|
|
145
|
+
tableCell(details.agent, 80),
|
|
142
146
|
"",
|
|
143
147
|
quote(details.task) || "> _(empty)_",
|
|
144
148
|
);
|