@shanepadgett/tau-agent 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/docs/subagents.md +25 -6
  2. package/extensions/attention/README.md +1 -0
  3. package/extensions/attention/index.ts +38 -0
  4. package/extensions/context/README.md +13 -3
  5. package/extensions/context/definitions.ts +3 -15
  6. package/extensions/context/evidence.ts +516 -0
  7. package/extensions/context/index.ts +114 -116
  8. package/extensions/context/panel.ts +60 -0
  9. package/extensions/context/settings.ts +30 -1
  10. package/extensions/context/sync.ts +171 -686
  11. package/extensions/context/validation.ts +4 -1
  12. package/extensions/context/write-scope.ts +109 -0
  13. package/extensions/context-pruning/README.md +22 -0
  14. package/extensions/context-pruning/file-evidence.ts +265 -0
  15. package/extensions/context-pruning/index.ts +379 -0
  16. package/extensions/context-pruning/projection.ts +108 -0
  17. package/extensions/context-pruning/prune.ts +346 -0
  18. package/extensions/context-pruning/render.ts +179 -0
  19. package/extensions/context-pruning/settings.ts +41 -0
  20. package/extensions/explore/README.md +3 -1
  21. package/extensions/explore/autoread.ts +85 -22
  22. package/extensions/explore/full-file-knowledge.ts +234 -0
  23. package/extensions/explore/index.ts +2 -3
  24. package/extensions/explore/read-cache.ts +150 -86
  25. package/extensions/explore/read-snapshots.ts +17 -4
  26. package/extensions/explore/read.ts +57 -38
  27. package/extensions/footer/index.ts +62 -60
  28. package/extensions/patch/README.md +1 -1
  29. package/extensions/patch/index.ts +19 -5
  30. package/extensions/run-summary/index.ts +5 -5
  31. package/extensions/silent-command-runner/README.md +1 -1
  32. package/extensions/silent-command-runner/index.ts +45 -28
  33. package/extensions/soul/prompt.ts +3 -1
  34. package/extensions/subagent/README.md +21 -5
  35. package/extensions/subagent/agents/context-sync.md +90 -0
  36. package/extensions/subagent/agents/{generalist.md → dormant/generalist.md} +6 -0
  37. package/extensions/subagent/agents/{scout.md → dormant/scout.md} +6 -0
  38. package/extensions/subagent/agents/review.md +48 -0
  39. package/extensions/subagent/agents/web-research.md +6 -0
  40. package/extensions/subagent/agents.ts +15 -2
  41. package/extensions/subagent/cmux-dashboard.ts +454 -0
  42. package/extensions/subagent/index.ts +181 -234
  43. package/extensions/subagent/render.ts +1 -1
  44. package/extensions/subagent/resume.ts +78 -0
  45. package/extensions/subagent/run.ts +213 -118
  46. package/extensions/subagent/runtime.ts +856 -0
  47. package/extensions/subagent/session-resource.ts +169 -0
  48. package/extensions/tau-help/help.md +7 -3
  49. package/extensions/turn-budget/index.ts +8 -36
  50. package/package.json +2 -2
  51. package/schemas/tau.schema.json +48 -1
  52. package/shared/context-pruning-state.ts +364 -0
  53. package/shared/events.ts +18 -0
  54. package/shared/model-fallback/index.ts +21 -10
  55. package/shared/model-fallback/types.ts +5 -3
  56. package/shared/settings/load.ts +78 -1
  57. package/shared/tool-row-state.ts +21 -1
@@ -4,7 +4,6 @@
4
4
  import { readdir, readFile, stat } from "node:fs/promises";
5
5
  import { homedir } from "node:os";
6
6
  import { join, relative } from "node:path";
7
- import type { AssistantMessage } from "@earendil-works/pi-ai";
8
7
  import type { ExtensionAPI, ExtensionContext, Theme, ThemeColor } from "@earendil-works/pi-coding-agent";
9
8
  import { type Component, truncateToWidth, visibleWidth } from "@earendil-works/pi-tui";
10
9
  import { onTauEvent } from "../../shared/events.ts";
@@ -278,17 +277,20 @@ function isConflict(x: string | undefined, y: string | undefined): boolean {
278
277
  return x === "U" || y === "U" || (x === "A" && y === "A") || (x === "D" && y === "D");
279
278
  }
280
279
 
281
- function sessionCost(ctx: ExtensionContext): UsageSummary {
280
+ export function sessionCost(ctx: Pick<ExtensionContext, "sessionManager">): UsageSummary {
282
281
  let usage = zeroUsage();
283
- for (const entry of ctx.sessionManager.getBranch()) {
284
- if (entry.type !== "message") continue;
285
- if (entry.message.role === "assistant") {
286
- usage = addUsage(usage, normalizeUsage((entry.message as AssistantMessage).usage));
282
+ for (const entry of ctx.sessionManager.getEntries()) {
283
+ if (entry.type === "message" && entry.message.role === "assistant") {
284
+ usage = addUsage(usage, normalizeUsage(entry.message.usage), true);
287
285
  continue;
288
286
  }
289
- if (entry.message.role !== "toolResult" || entry.message.toolName !== "subagent") continue;
290
- const childUsage = subagentUsage(entry.message.details);
291
- if (childUsage) usage = addUsage(usage, childUsage);
287
+ if (entry.type === "message" && entry.message.role === "toolResult" && entry.message.usage) {
288
+ usage = addUsage(usage, normalizeUsage(entry.message.usage), false);
289
+ continue;
290
+ }
291
+ if ((entry.type === "compaction" || entry.type === "branch_summary") && entry.usage) {
292
+ usage = addUsage(usage, normalizeUsage(entry.usage), false);
293
+ }
292
294
  }
293
295
  return usage;
294
296
  }
@@ -339,37 +341,7 @@ async function scanDailyCost(): Promise<number> {
339
341
  try {
340
342
  if ((await stat(path)).mtimeMs < range.startMs) continue;
341
343
  const raw = await readFile(path, "utf8");
342
- for (const line of raw.split("\n")) {
343
- const entry = parseRecord(line);
344
- if (!entry || entry.type !== "message") continue;
345
- const message = asRecord(entry.message);
346
- if (!message) continue;
347
- const timestamp = timestampMs(message.timestamp, entry.timestamp);
348
- if (timestamp < range.startMs || timestamp >= range.endMs) continue;
349
- let usage: UsageSummary;
350
- let identity: unknown[];
351
- if (message.role === "assistant") {
352
- usage = normalizeUsage(message.usage);
353
- identity = [message.provider, message.model];
354
- } else if (message.role === "toolResult" && message.toolName === "subagent") {
355
- const childUsage = subagentUsage(message.details);
356
- if (!childUsage) continue;
357
- usage = childUsage;
358
- identity = ["subagent", message.toolCallId];
359
- } else continue;
360
- const key = [
361
- ...identity,
362
- timestamp,
363
- usage.input,
364
- usage.output,
365
- usage.cacheRead,
366
- usage.cacheWrite,
367
- usage.cost,
368
- ].join(":");
369
- if (seen.has(key)) continue;
370
- seen.add(key);
371
- total += usage.cost;
372
- }
344
+ total += dailyCostFromJsonl(raw, range, seen);
373
345
  } catch {
374
346
  // One bad session file should not break the footer.
375
347
  }
@@ -378,6 +350,52 @@ async function scanDailyCost(): Promise<number> {
378
350
  return total;
379
351
  }
380
352
 
353
+ export function dailyCostFromJsonl(
354
+ raw: string,
355
+ range: { startMs: number; endMs: number },
356
+ seen: Set<string> = new Set<string>(),
357
+ ): number {
358
+ let total = 0;
359
+ for (const line of raw.split("\n")) {
360
+ const entry = parseRecord(line);
361
+ if (!entry) continue;
362
+ let usageValue: unknown;
363
+ let timestamp: number;
364
+ let identity: unknown[];
365
+ if (entry.type === "message") {
366
+ const message = asRecord(entry.message);
367
+ if (!message) continue;
368
+ timestamp = timestampMs(message.timestamp, entry.timestamp);
369
+ if (message.role === "assistant" && asRecord(message.usage)) {
370
+ usageValue = message.usage;
371
+ identity = ["assistant", message.provider, message.model];
372
+ } else if (message.role === "toolResult" && asRecord(message.usage)) {
373
+ usageValue = message.usage;
374
+ identity = ["toolResult", message.toolName, message.toolCallId];
375
+ } else continue;
376
+ } else if ((entry.type === "compaction" || entry.type === "branch_summary") && asRecord(entry.usage)) {
377
+ usageValue = entry.usage;
378
+ timestamp = timestampMs(undefined, entry.timestamp);
379
+ identity = [entry.type, entry.id];
380
+ } else continue;
381
+ if (timestamp < range.startMs || timestamp >= range.endMs) continue;
382
+ const usage = normalizeUsage(usageValue);
383
+ const key = [
384
+ ...identity,
385
+ timestamp,
386
+ usage.input,
387
+ usage.output,
388
+ usage.cacheRead,
389
+ usage.cacheWrite,
390
+ usage.cost,
391
+ ].join(":");
392
+ if (seen.has(key)) continue;
393
+ seen.add(key);
394
+ total += usage.cost;
395
+ }
396
+ return total;
397
+ }
398
+
381
399
  async function sessionFiles(): Promise<string[]> {
382
400
  const root = join(homedir(), ".pi", "agent", "sessions");
383
401
  const files: string[] = [];
@@ -429,31 +447,15 @@ function normalizeUsage(value: unknown): UsageSummary {
429
447
  };
430
448
  }
431
449
 
432
- function subagentUsage(details: unknown): UsageSummary | undefined {
433
- const usage = asRecord(asRecord(details)?.usage);
434
- if (!usage) return undefined;
435
- const input = numberOrZero(usage.input);
436
- const output = numberOrZero(usage.output);
437
- const cacheRead = numberOrZero(usage.cacheRead);
438
- const cacheWrite = numberOrZero(usage.cacheWrite);
439
- const promptTokens = input + cacheRead + cacheWrite;
440
- return {
441
- input,
442
- output,
443
- cacheRead,
444
- cacheWrite,
445
- latestCacheHitRate: promptTokens > 0 ? (cacheRead / promptTokens) * 100 : undefined,
446
- cost: numberOrZero(usage.cost),
447
- };
448
- }
449
-
450
- function addUsage(left: UsageSummary, right: UsageSummary): UsageSummary {
450
+ function addUsage(left: UsageSummary, right: UsageSummary, updateCacheHitRate: boolean): UsageSummary {
451
451
  return {
452
452
  input: left.input + right.input,
453
453
  output: left.output + right.output,
454
454
  cacheRead: left.cacheRead + right.cacheRead,
455
455
  cacheWrite: left.cacheWrite + right.cacheWrite,
456
- latestCacheHitRate: right.latestCacheHitRate ?? left.latestCacheHitRate,
456
+ latestCacheHitRate: updateCacheHitRate
457
+ ? (right.latestCacheHitRate ?? left.latestCacheHitRate)
458
+ : left.latestCacheHitRate,
457
459
  cost: left.cost + right.cost,
458
460
  };
459
461
  }
@@ -1,6 +1,6 @@
1
1
  # patch
2
2
 
3
- Replaces the built-in `edit` and `write` tools with one multi-file patch tool. The agent applies structured patches to create, edit, move, and delete files while this extension is active.
3
+ Replaces the built-in `edit` and `write` tools with one multi-file patch tool. The agent applies structured patches to create, edit, move, and delete files while this extension is active. For xAI and Grok models, Tau disables `patch` and leaves `edit` and `write` active because those models are unreliable with the patch format.
4
4
 
5
5
  ## What it does
6
6
 
@@ -135,6 +135,19 @@ export default function patchExtension(pi: ExtensionAPI): void {
135
135
  const rowState = createToolRowStateStore(pi, "patch.tool-row-state");
136
136
  pi.registerTool(createPatchTool(rowState));
137
137
 
138
+ function configureMutationTools(model: { provider: string; id: string } | undefined): void {
139
+ const active = new Set(pi.getActiveTools());
140
+ const usesGrok = model?.provider.toLowerCase() === "xai" || model?.id.toLowerCase().includes("grok") === true;
141
+ if (usesGrok) {
142
+ active.delete("patch");
143
+ for (const tool of SUPPRESSED_TOOLS) active.add(tool);
144
+ } else {
145
+ active.add("patch");
146
+ for (const tool of SUPPRESSED_TOOLS) active.delete(tool);
147
+ }
148
+ pi.setActiveTools([...active]);
149
+ }
150
+
138
151
  // AgentToolResult has no isError field, so execute returns are always treated as success.
139
152
  // Override via tool_result to flag partial/failed patches as errors for the model and UI.
140
153
  pi.on("tool_result", async (event, ctx) => {
@@ -160,11 +173,12 @@ export default function patchExtension(pi: ExtensionAPI): void {
160
173
  return { isError: true };
161
174
  });
162
175
 
163
- pi.on("session_start", () => {
176
+ pi.on("session_start", (_event, ctx) => {
164
177
  rowState.clear();
165
- const active = new Set(pi.getActiveTools());
166
- active.add("patch");
167
- for (const tool of SUPPRESSED_TOOLS) active.delete(tool);
168
- pi.setActiveTools([...active]);
178
+ configureMutationTools(ctx.model);
179
+ });
180
+
181
+ pi.on("model_select", (event) => {
182
+ configureMutationTools(event.model);
169
183
  });
170
184
  }
@@ -49,7 +49,7 @@ export default function runSummaryExtension(pi: ExtensionAPI): void {
49
49
  continue;
50
50
  }
51
51
  if (message.role !== "toolResult" || message.toolName !== "subagent") continue;
52
- subagentCost += readSubagentCost(message.details);
52
+ subagentCost += readUsageCost(message.usage);
53
53
  }
54
54
  });
55
55
 
@@ -68,11 +68,11 @@ export default function runSummaryExtension(pi: ExtensionAPI): void {
68
68
  });
69
69
  }
70
70
 
71
- function readSubagentCost(value: unknown): number {
71
+ function readUsageCost(value: unknown): number {
72
72
  if (!value || typeof value !== "object") return 0;
73
- const usage = (value as Record<string, unknown>).usage;
74
- if (!usage || typeof usage !== "object") return 0;
75
- return finiteNonNegative((usage as Record<string, unknown>).cost);
73
+ const cost = (value as Record<string, unknown>).cost;
74
+ if (!cost || typeof cost !== "object") return 0;
75
+ return finiteNonNegative((cost as Record<string, unknown>).total);
76
76
  }
77
77
 
78
78
  function readRunSummary(value: unknown): RunSummary | undefined {
@@ -22,4 +22,4 @@ Configure it in Tau settings under `extensions.silentCommandRunner`:
22
22
  }
23
23
  ```
24
24
 
25
- Passes are notifications only. Failures are shown in chat and sent to the agent.
25
+ Passes are notifications only. Failures are shown in chat and sent to the agent. Tau's ready-for-input notification waits for these checks and stays quiet when a failure starts another agent turn.
@@ -2,6 +2,7 @@ import { readdir, stat } from "node:fs/promises";
2
2
  import { resolve } from "node:path";
3
3
  import { type ExecResult, type ExtensionAPI, keyText, type Theme } from "@earendil-works/pi-coding-agent";
4
4
  import { Box, Text } from "@earendil-works/pi-tui";
5
+ import { emitTauEvent } from "../../shared/events.ts";
5
6
  import { matchGlob, posixPath } from "../../shared/glob.ts";
6
7
  import { loadTauExtensionSettings } from "../../shared/settings/load.ts";
7
8
  import { resolveProjectRoot } from "../../shared/settings/paths.ts";
@@ -78,10 +79,11 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
78
79
  let settings: Settings = normalizeSettings(silentCommandRunnerSettings.defaults);
79
80
  let turnStart = Date.now();
80
81
  let turnPaths = new Set<string>();
81
- let run: Promise<void> | undefined;
82
82
  let abortController: AbortController | undefined;
83
- let lastRunAborted = false;
83
+ let sessionActive = false;
84
84
  let chainActive = false;
85
+ let attentionHoldSequence = 0;
86
+ let attentionHoldId: string | undefined;
85
87
 
86
88
  pi.registerMessageRenderer<FailureDetails>(MESSAGE_TYPE, (message, { expanded }, theme) =>
87
89
  renderFailure(asFailureDetails(message.details), expanded, theme),
@@ -89,10 +91,12 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
89
91
 
90
92
  pi.on("session_start", async (_event, ctx) => {
91
93
  settings = normalizeSettings(await loadTauExtensionSettings(ctx, silentCommandRunnerSettings));
94
+ sessionActive = true;
92
95
  turnStart = Date.now();
93
96
  turnPaths = new Set();
94
- lastRunAborted = false;
95
97
  chainActive = false;
98
+ attentionHoldSequence = 0;
99
+ attentionHoldId = undefined;
96
100
  });
97
101
 
98
102
  pi.on("before_agent_start", async (event, ctx) => {
@@ -102,43 +106,44 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
102
106
  });
103
107
 
104
108
  pi.on("agent_start", async (_event, ctx) => {
105
- lastRunAborted = false;
106
- if (chainActive) return;
109
+ const startingChain = !chainActive;
107
110
  chainActive = true;
108
111
  turnStart = Date.now();
109
112
  if (!settings.enabled || settings.commands.length === 0) {
110
113
  turnPaths = new Set();
111
114
  return;
112
115
  }
116
+ if (startingChain) {
117
+ attentionHoldId = `silent-command-runner:${++attentionHoldSequence}`;
118
+ emitTauEvent(pi, "tau:attention.hold.acquire", { id: attentionHoldId });
119
+ }
113
120
  const projectRoot = await resolveProjectRoot(ctx.cwd);
114
121
  turnPaths = new Set(await walkFiles(projectRoot));
115
122
  });
116
123
 
117
- pi.on("agent_end", (event) => {
118
- lastRunAborted = hasAbortedAssistantMessage(event.messages);
124
+ pi.on("agent_end", async (event, ctx) => {
125
+ if (hasAbortedAssistantMessage(event.messages)) return;
126
+ try {
127
+ await runChangedCommands(ctx.cwd, turnStart, ctx.ui.notify);
128
+ } catch (error: unknown) {
129
+ ctx.ui.notify(`silent-command-runner: ${errorMessage(error)}`, "error");
130
+ }
119
131
  });
120
132
 
121
- pi.on("agent_settled", async (_event, ctx) => {
133
+ pi.on("agent_settled", () => {
122
134
  chainActive = false;
123
- if (lastRunAborted) return;
124
- if (run) return;
125
- run = runChangedCommands(ctx.cwd, turnStart, ctx.ui.notify)
126
- .catch((error: unknown) => {
127
- ctx.ui.notify(`silent-command-runner: ${errorMessage(error)}`, "error");
128
- })
129
- .finally(() => {
130
- run = undefined;
131
- });
132
- await run;
135
+ const holdId = attentionHoldId;
136
+ attentionHoldId = undefined;
137
+ if (holdId) emitTauEvent(pi, "tau:attention.hold.release", { id: holdId, disposition: "notify" });
133
138
  });
134
139
 
135
140
  pi.on("session_shutdown", () => {
141
+ sessionActive = false;
136
142
  abortController?.abort();
137
143
  abortController = undefined;
138
- run = undefined;
139
144
  turnPaths = new Set();
140
- lastRunAborted = false;
141
145
  chainActive = false;
146
+ attentionHoldId = undefined;
142
147
  });
143
148
 
144
149
  async function runChangedCommands(
@@ -151,7 +156,7 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
151
156
  const projectRoot = await resolveProjectRoot(cwd);
152
157
  const paths = await walkFiles(projectRoot);
153
158
  const changed = await scanChangedCommands(projectRoot, settings.commands, paths, turnPaths, turnStart);
154
- if (changed.length === 0) return;
159
+ if (!sessionActive || changed.length === 0) return;
155
160
 
156
161
  notify(
157
162
  changed.length === 1
@@ -160,14 +165,26 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
160
165
  "info",
161
166
  );
162
167
 
163
- abortController = new AbortController();
164
168
  const failures: FailedCommandDetails[] = [];
165
- for (const command of changed) {
166
- if (changed.length > 1) notify(`silent-command-runner: running ${command.name}`, "info");
167
- const result = await runCommand(pi, projectRoot, command, abortController.signal, settings.maxOutputBytes);
168
- if (result.code !== 0 || result.killed) failures.push(result);
169
+ const controller = new AbortController();
170
+ abortController = controller;
171
+ try {
172
+ for (const command of changed) {
173
+ if (changed.length > 1) notify(`silent-command-runner: running ${command.name}`, "info");
174
+ let result: FailedCommandDetails;
175
+ try {
176
+ result = await runCommand(pi, projectRoot, command, controller.signal, settings.maxOutputBytes);
177
+ } catch (error: unknown) {
178
+ if (controller.signal.aborted) return;
179
+ throw error;
180
+ }
181
+ if (controller.signal.aborted) return;
182
+ if (result.code !== 0 || result.killed) failures.push(result);
183
+ }
184
+ } finally {
185
+ if (abortController === controller) abortController = undefined;
169
186
  }
170
- abortController = undefined;
187
+ if (!sessionActive) return;
171
188
 
172
189
  const failedNames = new Set(failures.map((failure) => failure.name));
173
190
  const passed = changed.filter((command) => !failedNames.has(command.name));
@@ -181,7 +198,7 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
181
198
  display: true,
182
199
  details: { failed: failures },
183
200
  },
184
- { triggerTurn: true },
201
+ { deliverAs: "followUp" },
185
202
  );
186
203
  }
187
204
  }
@@ -12,7 +12,7 @@ Human interrupts. Human sometimes idiot. Human sometimes has good idea. Rok thin
12
12
 
13
13
  Build only what human specifically asked for. User ask approves that scope only. No bonus features, new option categories, settings, APIs, UI, commands, docs, output, or public behavior unless human explicitly approved. If Rok sees missing public surface that truly helps, ask first in one line. Do not sneak it into diff.
14
14
 
15
- Every read has job. Start from task path or symbol. Grep for broad search, not for rereading known files. Read only files likely to answer current decision or be edited. Do not chase imports, shared helpers, docs, or callers unless current evidence says they matter. Aimless explore wastes context and dulls Rok. If exploration wandered, prune memory and keep only useful facts.
15
+ Every read has job. Start from task path or symbol. Grep for broad search, not for rereading known files. Read only files likely to answer current decision or be edited. Once file is relevant, request whole file by omitting \`offset\`, \`limit\`, and \`lineNumbers\`. One whole-file read beats repeated ranged reads. Supplied eager snapshot counts as initial read. After file changes, request whole file again; read cache returns useful diff or current source instead of blindly repeating old content. Use ranged reads only for targeted inspection of peripheral files or to continue after tool truncation. Do not chase imports, shared helpers, docs, or callers unless current evidence says they matter. Aimless explore wastes context and dulls Rok. If exploration wandered, prune memory and keep only useful facts.
16
16
 
17
17
  Selected snapshots are authoritative unless edited, changed, or missing needed content.
18
18
 
@@ -20,6 +20,8 @@ Never cut validation, data safety, security, accessibility, explicit user ask, h
20
20
 
21
21
  Question asked? Answer and stop. Simple question gets simple answer. If one sentence works, use one sentence. No plan, caveat list, or options unless needed. Change requested? Smallest correct change. Ambiguous? Ask one practical question.
22
22
 
23
+ Idea, plan, or concept talk: short back-and-forth. Shape first. Add detail only as next layer needs it. Do not expand into full plan, option tree, or writeup unless asked.
24
+
23
25
  If instructed not to run checks or not to do something, obey silently. Do not report the omission in meta-speak.
24
26
 
25
27
  Final chat tiny. User saw tools and will inspect code. Do not tour work. Say only non-obvious thing human needs now. If nothing needs saying, one word.`;
@@ -4,15 +4,25 @@ Subagent delegates one focused task to an isolated child Pi session. It keeps th
4
4
 
5
5
  Agent definitions can override the parent model and thinking level. If an override is unavailable, Tau warns once per session and uses the corresponding parent value.
6
6
 
7
- Tau includes three built-in agents:
7
+ Each fresh child also gets a display name from its agent definition. The name stays with a retained thread. Tau cycles through the configured pool and adds `-2`, `-3`, and so on when a pool name is reused, so parallel calls never collide.
8
8
 
9
- - `generalist` handles focused analysis, review, implementation, or mixed work when no narrower agent fits. Delegated tasks should state their scope and desired depth.
10
- - `scout` explores local files and code with `read`, `grep`, `find`, and `ls`.
9
+ Tau includes these built-in agents:
10
+
11
+ - `review` performs adversarial, read-only code review for correctness, runtime risks, duplication, and over- or under-engineering.
11
12
  - `web-research` researches web and code sources with `websearch`, `codesearch`, and `webfetch`.
13
+ - `context-sync` maps meaningful uncommitted work into `.pi/contexts`. Agent-driven use is `extensions.context.sync.automation` (requires `sync.enabled`). `/context-sync` is the manual/nudge path when sync is enabled. Validation can auto-run it when validation and sync are enabled.
12
14
 
13
15
  Ask Tau to delegate a task, or let it call `subagent` with an agent name and task. Children use the parent's current working directory and inherit its model and thinking level unless their definition overrides either value. They do not receive the parent conversation. Tau loads only the extensions that own a child's declared tools, so unrelated extension hooks do not run in child sessions. When a child must inspect another repository, put its exact absolute path in the delegated task.
14
16
 
15
- Fresh calls return a thread ID. Tau can send feedback or follow-up work to that thread, preserving the child's conversation, prior reads, and tool results. It starts a fresh thread for unrelated work or when retained context is stale or oversized. Threads live for the current parent session. Tau retains up to 16 and evicts the least recently used idle thread when needed. Calls to one thread run sequentially.
17
+ When the relevant files are already known, pass them with the call so Tau can autoread them into that child turn:
18
+
19
+ ```text
20
+ subagent({ agent: "review", task: "Review the runtime change", files: ["src/runtime.ts", "test/runtime.test.ts"] })
21
+ ```
22
+
23
+ Paths may be relative to the parent's current working directory or absolute. Tau reads current snapshots when the turn starts and includes line numbers so the child can cite them without another read. Missing files appear as failed autoread context; they do not stop the child. Keep the list focused because the complete snapshots use the child's context window. Files can also be supplied on a retained-thread follow-up.
24
+
25
+ Fresh calls return a thread ID. Follow-ups within five minutes preserve the complete child conversation. After that, Tau replaces the child session and resumes from prior tasks, exact terminal results, and the paths supplied through `files`. Old file contents, tool calls, intermediate responses, and thinking are dropped without a summarization request. The resumed child reads current source before relying on a retained path. Threads live for the current parent session. Tau retains up to 16 and evicts the least recently used idle thread when needed. Calls to one thread run sequentially.
16
26
 
17
27
  ## Agent definitions
18
28
 
@@ -25,6 +35,10 @@ description: Inspect API declarations and usage
25
35
  tools:
26
36
  - read
27
37
  - grep
38
+ names:
39
+ - Ledger
40
+ - Quill
41
+ - Beacon
28
42
  model: openai-codex/gpt-5.6-sol
29
43
  thinking: medium
30
44
  ---
@@ -32,6 +46,8 @@ thinking: medium
32
46
  Stay within the delegated task. Return exact paths and symbols.
33
47
  ```
34
48
 
35
- `name`, `description`, and `tools` are required. Optional `model` uses `provider/model`; optional `thinking` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, or `max`. Tool names must be unique, and `subagent` cannot be delegated. Named tools and configured models must exist in the normally loaded child Pi environment.
49
+ `name`, `description`, and `tools` are required. Optional `names` is a non-empty list of unique display names; without it, Tau uses the agent name. Optional `model` uses `provider/model`; optional `thinking` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, or `max`. Tool names and display names must be unique within their lists, and `subagent` cannot be delegated. Named tools and configured models must exist in the normally loaded child Pi environment.
36
50
 
37
51
  At most four children run at once. Additional calls wait in order. Returned text is limited to 50 KB or 2,000 lines; complete truncated output is saved to a private temporary file outside project repositories.
52
+
53
+ When Tau runs interactively inside cmux, a single temporary Markdown surface shows waiting and running subagent work beside the parent terminal. It does not change child scheduling, concurrency, or results. The surface closes a couple of seconds after the active cohort finishes. Print mode and non-cmux sessions never open it.
@@ -0,0 +1,90 @@
1
+ ---
2
+ name: context-sync
3
+ description: >-
4
+ Map meaningful uncommitted work into `.pi/contexts` (domains/concepts/entries).
5
+ Call after a coherent batch that adds, moves, renames, or changes ownership of code/docs—not after every trivial edit to paths already correctly filed.
6
+ Prefer once per batch or before commit; skip pure refactors that keep the same membership, typos, and already-covered single-file polish.
7
+ Task may include a short human/steer note. Harness may also auto-run this when context validation is enabled.
8
+ tools:
9
+ - read
10
+ - ls
11
+ - find
12
+ - grep
13
+ - bash
14
+ - patch
15
+ - context_sync_evidence
16
+ names:
17
+ - Cartographer
18
+ - Archivist
19
+ - Indexer
20
+ - Surveyor
21
+ - Curator
22
+ model: openai-codex/gpt-5.6-luna
23
+ thinking: high
24
+ ---
25
+
26
+ You maintain the living repository context map under `.pi/contexts`.
27
+
28
+ Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries. Entry `files` are eager autoread paths. Entry `anchors` are lazy navigation paths. Preserve an existing path's loading class when it already appears anywhere in the catalog. New paths default to eager `files`.
29
+
30
+ ## Tools
31
+
32
+ - `context_sync_evidence` — git dirty set, catalog skeleton, deps, diffs, invariants. Prefer this first.
33
+ - `read` / `ls` / `find` / `grep` — default explore tools for boundary judgment when evidence is not enough.
34
+ - `bash` — only when explore tools cannot answer the question.
35
+ - `patch` — create/update/move/delete only under `.pi/contexts/**` (including whole concept TOML via `*** Delete File`).
36
+
37
+ ### Bash limits
38
+
39
+ Prefer `ls`, `find`, `grep`, and `read` first. Use bash only when those cannot cover the need.
40
+
41
+ Allowed bash beyond explore:
42
+
43
+ - Read-only inspection those tools miss — e.g. `git log` / `git blame` / `git show` on existing commits, `file`, `wc`, small read-only pipelines
44
+ - After `patch` deletes the last file in a `.pi/contexts/**` directory, remove that empty directory with `rmdir` (repeat upward only while dirs stay empty under `.pi/contexts`). Prefer `rmdir` over `rm -r`.
45
+
46
+ Forbidden with bash:
47
+
48
+ - Anything `ls` / `find` / `grep` / `read` already do cleanly
49
+ - Creating, editing, moving, or deleting files (catalog files use `patch` only)
50
+ - Non-empty directory deletes; deletes outside `.pi/contexts`
51
+ - `git add` / `commit` / `push` / `checkout` / `restore` / `reset` / `stash` / branch changes
52
+ - Installers, package managers, builds, tests, formatters, codegen, servers
53
+ - Network fetches that change the tree; secrets; credential or config mutation
54
+
55
+ Catalog file mutations go through `patch` under `.pi/contexts` only. The harness restores out-of-scope writes and fails the run.
56
+
57
+ ## Forced ladder
58
+
59
+ Before placing any path, answer out loud in order:
60
+
61
+ 1. **Domain** — Does this changeset reuse an existing domain tab, birth a new one (core, platform, infrastructure, docs, a feature grown into its own domain), or split a bloated domain?
62
+ 2. **Concept** — Inside that domain, which subsystem TOML? Reuse, new, split, or merge?
63
+ 3. **Entry** — Which work scope? Update, new, split, delete, or move between concepts/domains?
64
+ 4. **Bloat** — Did this touch make an entry/concept a junk drawer? Split now if yes.
65
+ 5. **Membership** — Assign files/anchors only under the winners. Every eligible changed non-deleted file must belong somewhere. Remove every stale catalog path.
66
+
67
+ Path stuffing into the nearest feature bucket without climbing the ladder is failure.
68
+
69
+ ## Change classes
70
+
71
+ - **Additive / local edit** — mostly membership or a new entry under a stable concept.
72
+ - **Semantic move / refactor** — meaning moved even if paths stayed covered. Re-evaluate domain/concept. Moves and splits are required verbs, not optional polish.
73
+
74
+ ## Working loop
75
+
76
+ 1. `context_sync_evidence` section `overview`.
77
+ 2. `catalog` for the full skeleton. Do not invent domains you never inspected.
78
+ 3. Pull `dirty`, `file`, `dependencies`, or `previews` as needed. Use `read` / `ls` / `find` / `grep` for neighbor inspection. Use `bash` only when those cannot cover the question.
79
+ 4. Edit catalog TOML with `patch`.
80
+ 5. `context_sync_evidence` section `invariants`. If failed, continue until it holds or you can explain a hard blocker.
81
+ 6. Final reply: short summary of domain/concept/entry decisions and files touched under `.pi/contexts`. If no catalog edit was required, say why.
82
+
83
+ ## Nudge
84
+
85
+ If the task includes a human nudge, treat it as soft steer. It does not override evidence, eligibility, or the ladder. If the nudge conflicts with the changeset, say so and choose the honest map.
86
+
87
+ ## Stop conditions
88
+
89
+ - Invariants hold and the map reflects honest typology for this changeset, or
90
+ - Hard blocker (secrets, conflicts, missing tools/model) — report it clearly without half-applying a broken map.
@@ -7,6 +7,12 @@ tools:
7
7
  - grep
8
8
  - find
9
9
  - ls
10
+ names:
11
+ - Tinker
12
+ - Rivet
13
+ - Patch
14
+ - Wrench
15
+ - Mender
10
16
  model: openai-codex/gpt-5.6-sol
11
17
  thinking: high
12
18
  ---
@@ -6,6 +6,12 @@ tools:
6
6
  - grep
7
7
  - find
8
8
  - ls
9
+ names:
10
+ - Pathfinder
11
+ - Trailblazer
12
+ - Lookout
13
+ - Tracker
14
+ - Ranger
9
15
  model: openai-codex/gpt-5.6-luna
10
16
  thinking: high
11
17
  ---
@@ -0,0 +1,48 @@
1
+ ---
2
+ name: review
3
+ description: Perform an adversarial, read-only review for correctness, runtime risks, duplication, and over- or under-engineering
4
+ tools:
5
+ - read
6
+ - grep
7
+ - find
8
+ - ls
9
+ - bash
10
+ - context_prune
11
+ names:
12
+ - Auditor
13
+ - Inspector
14
+ - Skeptic
15
+ - Examiner
16
+ - Sentinel
17
+ model: openai-codex/gpt-5.6-sol
18
+ thinking: xhigh
19
+ ---
20
+
21
+ Treat the implementation as untrusted. Try to disprove its correctness before accepting it. Review the delegated scope at full depth, then report only findings supported by concrete evidence.
22
+
23
+ ## Review procedure
24
+
25
+ 1. Establish the requested scope. When reviewing uncommitted work, inspect the relevant diff before reading surrounding code.
26
+ 2. Read every changed path in scope. Trace direct callers, consumers, state transitions, error paths, and tests where they can change the verdict.
27
+ 3. Compare behavior with the stated request, repository rules, and existing conventions.
28
+ 4. Look specifically for:
29
+ - incorrect behavior, runtime failures, races, stale state, bad boundaries, and unsafe error handling;
30
+ - over-engineering, needless wrappers, option bags, tiny single-use helpers, duplicated logic, and abstractions that make the code harder to reason about;
31
+ - under-engineering, missing validation, incomplete wiring, weak tests, and assumptions that should be enforced;
32
+ - tests that only mirror the implementation, miss realistic sequences, or fail to protect the requested behavior;
33
+ - dead code, stale documentation, and obsolete resources left behind by the change.
34
+ 5. Use read-only shell commands for evidence when the other tools cannot answer the question. Do not run formatters, generators, installers, or commands that rewrite the repository.
35
+ 6. After broad exploration converges, use `context_prune` before continuing when substantial stale evidence would otherwise remain.
36
+
37
+ Do not modify files. Do not reward review volume. Reject speculative findings and personal style preferences without a concrete maintenance, correctness, or runtime consequence.
38
+
39
+ ## Output
40
+
41
+ List findings first, ordered by severity. For each finding include:
42
+
43
+ - severity and a direct title;
44
+ - exact file and line evidence;
45
+ - the failure mechanism or maintenance cost;
46
+ - the smallest credible fix direction.
47
+
48
+ Then list unresolved questions that materially affect correctness. If there are no findings, say so plainly and state what you inspected. Do not add a summary that repeats the findings.
@@ -5,6 +5,12 @@ tools:
5
5
  - websearch
6
6
  - codesearch
7
7
  - webfetch
8
+ names:
9
+ - Spider
10
+ - Linkhound
11
+ - Crawler
12
+ - Netscout
13
+ - Wayfinder
8
14
  model: openai-codex/gpt-5.6-sol
9
15
  thinking: medium
10
16
  ---