@shanepadgett/tau-agent 0.16.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/subagents.md +25 -6
- package/extensions/attention/README.md +1 -0
- package/extensions/attention/index.ts +38 -0
- package/extensions/context/README.md +13 -3
- package/extensions/context/definitions.ts +3 -15
- package/extensions/context/evidence.ts +516 -0
- package/extensions/context/index.ts +114 -116
- package/extensions/context/panel.ts +60 -0
- package/extensions/context/settings.ts +30 -1
- package/extensions/context/sync.ts +171 -686
- package/extensions/context/validation.ts +4 -1
- package/extensions/context/write-scope.ts +109 -0
- package/extensions/context-pruning/README.md +22 -0
- package/extensions/context-pruning/file-evidence.ts +265 -0
- package/extensions/context-pruning/index.ts +379 -0
- package/extensions/context-pruning/projection.ts +108 -0
- package/extensions/context-pruning/prune.ts +346 -0
- package/extensions/context-pruning/render.ts +179 -0
- package/extensions/context-pruning/settings.ts +41 -0
- package/extensions/explore/README.md +3 -1
- package/extensions/explore/autoread.ts +85 -22
- package/extensions/explore/full-file-knowledge.ts +234 -0
- package/extensions/explore/index.ts +2 -3
- package/extensions/explore/read-cache.ts +150 -86
- package/extensions/explore/read-snapshots.ts +17 -4
- package/extensions/explore/read.ts +57 -38
- package/extensions/footer/index.ts +62 -60
- package/extensions/run-summary/index.ts +5 -5
- package/extensions/silent-command-runner/README.md +1 -1
- package/extensions/silent-command-runner/index.ts +45 -28
- package/extensions/soul/prompt.ts +3 -1
- package/extensions/subagent/README.md +21 -5
- package/extensions/subagent/agents/context-sync.md +90 -0
- package/extensions/subagent/agents/{generalist.md → dormant/generalist.md} +6 -0
- package/extensions/subagent/agents/{scout.md → dormant/scout.md} +6 -0
- package/extensions/subagent/agents/review.md +48 -0
- package/extensions/subagent/agents/web-research.md +6 -0
- package/extensions/subagent/agents.ts +15 -2
- package/extensions/subagent/cmux-dashboard.ts +454 -0
- package/extensions/subagent/index.ts +181 -234
- package/extensions/subagent/render.ts +1 -1
- package/extensions/subagent/resume.ts +78 -0
- package/extensions/subagent/run.ts +213 -118
- package/extensions/subagent/runtime.ts +856 -0
- package/extensions/subagent/session-resource.ts +169 -0
- package/extensions/tau-help/help.md +6 -2
- package/extensions/turn-budget/index.ts +8 -36
- package/package.json +2 -2
- package/schemas/tau.schema.json +48 -1
- package/shared/context-pruning-state.ts +364 -0
- package/shared/events.ts +18 -0
- package/shared/model-fallback/index.ts +21 -10
- package/shared/model-fallback/types.ts +5 -3
- package/shared/settings/load.ts +78 -1
- package/shared/tool-row-state.ts +21 -1
|
@@ -4,7 +4,6 @@
|
|
|
4
4
|
import { readdir, readFile, stat } from "node:fs/promises";
|
|
5
5
|
import { homedir } from "node:os";
|
|
6
6
|
import { join, relative } from "node:path";
|
|
7
|
-
import type { AssistantMessage } from "@earendil-works/pi-ai";
|
|
8
7
|
import type { ExtensionAPI, ExtensionContext, Theme, ThemeColor } from "@earendil-works/pi-coding-agent";
|
|
9
8
|
import { type Component, truncateToWidth, visibleWidth } from "@earendil-works/pi-tui";
|
|
10
9
|
import { onTauEvent } from "../../shared/events.ts";
|
|
@@ -278,17 +277,20 @@ function isConflict(x: string | undefined, y: string | undefined): boolean {
|
|
|
278
277
|
return x === "U" || y === "U" || (x === "A" && y === "A") || (x === "D" && y === "D");
|
|
279
278
|
}
|
|
280
279
|
|
|
281
|
-
function sessionCost(ctx: ExtensionContext): UsageSummary {
|
|
280
|
+
export function sessionCost(ctx: Pick<ExtensionContext, "sessionManager">): UsageSummary {
|
|
282
281
|
let usage = zeroUsage();
|
|
283
|
-
for (const entry of ctx.sessionManager.
|
|
284
|
-
if (entry.type
|
|
285
|
-
|
|
286
|
-
usage = addUsage(usage, normalizeUsage((entry.message as AssistantMessage).usage));
|
|
282
|
+
for (const entry of ctx.sessionManager.getEntries()) {
|
|
283
|
+
if (entry.type === "message" && entry.message.role === "assistant") {
|
|
284
|
+
usage = addUsage(usage, normalizeUsage(entry.message.usage), true);
|
|
287
285
|
continue;
|
|
288
286
|
}
|
|
289
|
-
if (entry.message.role
|
|
290
|
-
|
|
291
|
-
|
|
287
|
+
if (entry.type === "message" && entry.message.role === "toolResult" && entry.message.usage) {
|
|
288
|
+
usage = addUsage(usage, normalizeUsage(entry.message.usage), false);
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
if ((entry.type === "compaction" || entry.type === "branch_summary") && entry.usage) {
|
|
292
|
+
usage = addUsage(usage, normalizeUsage(entry.usage), false);
|
|
293
|
+
}
|
|
292
294
|
}
|
|
293
295
|
return usage;
|
|
294
296
|
}
|
|
@@ -339,37 +341,7 @@ async function scanDailyCost(): Promise<number> {
|
|
|
339
341
|
try {
|
|
340
342
|
if ((await stat(path)).mtimeMs < range.startMs) continue;
|
|
341
343
|
const raw = await readFile(path, "utf8");
|
|
342
|
-
|
|
343
|
-
const entry = parseRecord(line);
|
|
344
|
-
if (!entry || entry.type !== "message") continue;
|
|
345
|
-
const message = asRecord(entry.message);
|
|
346
|
-
if (!message) continue;
|
|
347
|
-
const timestamp = timestampMs(message.timestamp, entry.timestamp);
|
|
348
|
-
if (timestamp < range.startMs || timestamp >= range.endMs) continue;
|
|
349
|
-
let usage: UsageSummary;
|
|
350
|
-
let identity: unknown[];
|
|
351
|
-
if (message.role === "assistant") {
|
|
352
|
-
usage = normalizeUsage(message.usage);
|
|
353
|
-
identity = [message.provider, message.model];
|
|
354
|
-
} else if (message.role === "toolResult" && message.toolName === "subagent") {
|
|
355
|
-
const childUsage = subagentUsage(message.details);
|
|
356
|
-
if (!childUsage) continue;
|
|
357
|
-
usage = childUsage;
|
|
358
|
-
identity = ["subagent", message.toolCallId];
|
|
359
|
-
} else continue;
|
|
360
|
-
const key = [
|
|
361
|
-
...identity,
|
|
362
|
-
timestamp,
|
|
363
|
-
usage.input,
|
|
364
|
-
usage.output,
|
|
365
|
-
usage.cacheRead,
|
|
366
|
-
usage.cacheWrite,
|
|
367
|
-
usage.cost,
|
|
368
|
-
].join(":");
|
|
369
|
-
if (seen.has(key)) continue;
|
|
370
|
-
seen.add(key);
|
|
371
|
-
total += usage.cost;
|
|
372
|
-
}
|
|
344
|
+
total += dailyCostFromJsonl(raw, range, seen);
|
|
373
345
|
} catch {
|
|
374
346
|
// One bad session file should not break the footer.
|
|
375
347
|
}
|
|
@@ -378,6 +350,52 @@ async function scanDailyCost(): Promise<number> {
|
|
|
378
350
|
return total;
|
|
379
351
|
}
|
|
380
352
|
|
|
353
|
+
export function dailyCostFromJsonl(
|
|
354
|
+
raw: string,
|
|
355
|
+
range: { startMs: number; endMs: number },
|
|
356
|
+
seen: Set<string> = new Set<string>(),
|
|
357
|
+
): number {
|
|
358
|
+
let total = 0;
|
|
359
|
+
for (const line of raw.split("\n")) {
|
|
360
|
+
const entry = parseRecord(line);
|
|
361
|
+
if (!entry) continue;
|
|
362
|
+
let usageValue: unknown;
|
|
363
|
+
let timestamp: number;
|
|
364
|
+
let identity: unknown[];
|
|
365
|
+
if (entry.type === "message") {
|
|
366
|
+
const message = asRecord(entry.message);
|
|
367
|
+
if (!message) continue;
|
|
368
|
+
timestamp = timestampMs(message.timestamp, entry.timestamp);
|
|
369
|
+
if (message.role === "assistant" && asRecord(message.usage)) {
|
|
370
|
+
usageValue = message.usage;
|
|
371
|
+
identity = ["assistant", message.provider, message.model];
|
|
372
|
+
} else if (message.role === "toolResult" && asRecord(message.usage)) {
|
|
373
|
+
usageValue = message.usage;
|
|
374
|
+
identity = ["toolResult", message.toolName, message.toolCallId];
|
|
375
|
+
} else continue;
|
|
376
|
+
} else if ((entry.type === "compaction" || entry.type === "branch_summary") && asRecord(entry.usage)) {
|
|
377
|
+
usageValue = entry.usage;
|
|
378
|
+
timestamp = timestampMs(undefined, entry.timestamp);
|
|
379
|
+
identity = [entry.type, entry.id];
|
|
380
|
+
} else continue;
|
|
381
|
+
if (timestamp < range.startMs || timestamp >= range.endMs) continue;
|
|
382
|
+
const usage = normalizeUsage(usageValue);
|
|
383
|
+
const key = [
|
|
384
|
+
...identity,
|
|
385
|
+
timestamp,
|
|
386
|
+
usage.input,
|
|
387
|
+
usage.output,
|
|
388
|
+
usage.cacheRead,
|
|
389
|
+
usage.cacheWrite,
|
|
390
|
+
usage.cost,
|
|
391
|
+
].join(":");
|
|
392
|
+
if (seen.has(key)) continue;
|
|
393
|
+
seen.add(key);
|
|
394
|
+
total += usage.cost;
|
|
395
|
+
}
|
|
396
|
+
return total;
|
|
397
|
+
}
|
|
398
|
+
|
|
381
399
|
async function sessionFiles(): Promise<string[]> {
|
|
382
400
|
const root = join(homedir(), ".pi", "agent", "sessions");
|
|
383
401
|
const files: string[] = [];
|
|
@@ -429,31 +447,15 @@ function normalizeUsage(value: unknown): UsageSummary {
|
|
|
429
447
|
};
|
|
430
448
|
}
|
|
431
449
|
|
|
432
|
-
function
|
|
433
|
-
const usage = asRecord(asRecord(details)?.usage);
|
|
434
|
-
if (!usage) return undefined;
|
|
435
|
-
const input = numberOrZero(usage.input);
|
|
436
|
-
const output = numberOrZero(usage.output);
|
|
437
|
-
const cacheRead = numberOrZero(usage.cacheRead);
|
|
438
|
-
const cacheWrite = numberOrZero(usage.cacheWrite);
|
|
439
|
-
const promptTokens = input + cacheRead + cacheWrite;
|
|
440
|
-
return {
|
|
441
|
-
input,
|
|
442
|
-
output,
|
|
443
|
-
cacheRead,
|
|
444
|
-
cacheWrite,
|
|
445
|
-
latestCacheHitRate: promptTokens > 0 ? (cacheRead / promptTokens) * 100 : undefined,
|
|
446
|
-
cost: numberOrZero(usage.cost),
|
|
447
|
-
};
|
|
448
|
-
}
|
|
449
|
-
|
|
450
|
-
function addUsage(left: UsageSummary, right: UsageSummary): UsageSummary {
|
|
450
|
+
function addUsage(left: UsageSummary, right: UsageSummary, updateCacheHitRate: boolean): UsageSummary {
|
|
451
451
|
return {
|
|
452
452
|
input: left.input + right.input,
|
|
453
453
|
output: left.output + right.output,
|
|
454
454
|
cacheRead: left.cacheRead + right.cacheRead,
|
|
455
455
|
cacheWrite: left.cacheWrite + right.cacheWrite,
|
|
456
|
-
latestCacheHitRate:
|
|
456
|
+
latestCacheHitRate: updateCacheHitRate
|
|
457
|
+
? (right.latestCacheHitRate ?? left.latestCacheHitRate)
|
|
458
|
+
: left.latestCacheHitRate,
|
|
457
459
|
cost: left.cost + right.cost,
|
|
458
460
|
};
|
|
459
461
|
}
|
|
@@ -49,7 +49,7 @@ export default function runSummaryExtension(pi: ExtensionAPI): void {
|
|
|
49
49
|
continue;
|
|
50
50
|
}
|
|
51
51
|
if (message.role !== "toolResult" || message.toolName !== "subagent") continue;
|
|
52
|
-
subagentCost +=
|
|
52
|
+
subagentCost += readUsageCost(message.usage);
|
|
53
53
|
}
|
|
54
54
|
});
|
|
55
55
|
|
|
@@ -68,11 +68,11 @@ export default function runSummaryExtension(pi: ExtensionAPI): void {
|
|
|
68
68
|
});
|
|
69
69
|
}
|
|
70
70
|
|
|
71
|
-
function
|
|
71
|
+
function readUsageCost(value: unknown): number {
|
|
72
72
|
if (!value || typeof value !== "object") return 0;
|
|
73
|
-
const
|
|
74
|
-
if (!
|
|
75
|
-
return finiteNonNegative((
|
|
73
|
+
const cost = (value as Record<string, unknown>).cost;
|
|
74
|
+
if (!cost || typeof cost !== "object") return 0;
|
|
75
|
+
return finiteNonNegative((cost as Record<string, unknown>).total);
|
|
76
76
|
}
|
|
77
77
|
|
|
78
78
|
function readRunSummary(value: unknown): RunSummary | undefined {
|
|
@@ -22,4 +22,4 @@ Configure it in Tau settings under `extensions.silentCommandRunner`:
|
|
|
22
22
|
}
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
-
Passes are notifications only. Failures are shown in chat and sent to the agent.
|
|
25
|
+
Passes are notifications only. Failures are shown in chat and sent to the agent. Tau's ready-for-input notification waits for these checks and stays quiet when a failure starts another agent turn.
|
|
@@ -2,6 +2,7 @@ import { readdir, stat } from "node:fs/promises";
|
|
|
2
2
|
import { resolve } from "node:path";
|
|
3
3
|
import { type ExecResult, type ExtensionAPI, keyText, type Theme } from "@earendil-works/pi-coding-agent";
|
|
4
4
|
import { Box, Text } from "@earendil-works/pi-tui";
|
|
5
|
+
import { emitTauEvent } from "../../shared/events.ts";
|
|
5
6
|
import { matchGlob, posixPath } from "../../shared/glob.ts";
|
|
6
7
|
import { loadTauExtensionSettings } from "../../shared/settings/load.ts";
|
|
7
8
|
import { resolveProjectRoot } from "../../shared/settings/paths.ts";
|
|
@@ -78,10 +79,11 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
78
79
|
let settings: Settings = normalizeSettings(silentCommandRunnerSettings.defaults);
|
|
79
80
|
let turnStart = Date.now();
|
|
80
81
|
let turnPaths = new Set<string>();
|
|
81
|
-
let run: Promise<void> | undefined;
|
|
82
82
|
let abortController: AbortController | undefined;
|
|
83
|
-
let
|
|
83
|
+
let sessionActive = false;
|
|
84
84
|
let chainActive = false;
|
|
85
|
+
let attentionHoldSequence = 0;
|
|
86
|
+
let attentionHoldId: string | undefined;
|
|
85
87
|
|
|
86
88
|
pi.registerMessageRenderer<FailureDetails>(MESSAGE_TYPE, (message, { expanded }, theme) =>
|
|
87
89
|
renderFailure(asFailureDetails(message.details), expanded, theme),
|
|
@@ -89,10 +91,12 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
89
91
|
|
|
90
92
|
pi.on("session_start", async (_event, ctx) => {
|
|
91
93
|
settings = normalizeSettings(await loadTauExtensionSettings(ctx, silentCommandRunnerSettings));
|
|
94
|
+
sessionActive = true;
|
|
92
95
|
turnStart = Date.now();
|
|
93
96
|
turnPaths = new Set();
|
|
94
|
-
lastRunAborted = false;
|
|
95
97
|
chainActive = false;
|
|
98
|
+
attentionHoldSequence = 0;
|
|
99
|
+
attentionHoldId = undefined;
|
|
96
100
|
});
|
|
97
101
|
|
|
98
102
|
pi.on("before_agent_start", async (event, ctx) => {
|
|
@@ -102,43 +106,44 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
102
106
|
});
|
|
103
107
|
|
|
104
108
|
pi.on("agent_start", async (_event, ctx) => {
|
|
105
|
-
|
|
106
|
-
if (chainActive) return;
|
|
109
|
+
const startingChain = !chainActive;
|
|
107
110
|
chainActive = true;
|
|
108
111
|
turnStart = Date.now();
|
|
109
112
|
if (!settings.enabled || settings.commands.length === 0) {
|
|
110
113
|
turnPaths = new Set();
|
|
111
114
|
return;
|
|
112
115
|
}
|
|
116
|
+
if (startingChain) {
|
|
117
|
+
attentionHoldId = `silent-command-runner:${++attentionHoldSequence}`;
|
|
118
|
+
emitTauEvent(pi, "tau:attention.hold.acquire", { id: attentionHoldId });
|
|
119
|
+
}
|
|
113
120
|
const projectRoot = await resolveProjectRoot(ctx.cwd);
|
|
114
121
|
turnPaths = new Set(await walkFiles(projectRoot));
|
|
115
122
|
});
|
|
116
123
|
|
|
117
|
-
pi.on("agent_end", (event) => {
|
|
118
|
-
|
|
124
|
+
pi.on("agent_end", async (event, ctx) => {
|
|
125
|
+
if (hasAbortedAssistantMessage(event.messages)) return;
|
|
126
|
+
try {
|
|
127
|
+
await runChangedCommands(ctx.cwd, turnStart, ctx.ui.notify);
|
|
128
|
+
} catch (error: unknown) {
|
|
129
|
+
ctx.ui.notify(`silent-command-runner: ${errorMessage(error)}`, "error");
|
|
130
|
+
}
|
|
119
131
|
});
|
|
120
132
|
|
|
121
|
-
pi.on("agent_settled",
|
|
133
|
+
pi.on("agent_settled", () => {
|
|
122
134
|
chainActive = false;
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
.catch((error: unknown) => {
|
|
127
|
-
ctx.ui.notify(`silent-command-runner: ${errorMessage(error)}`, "error");
|
|
128
|
-
})
|
|
129
|
-
.finally(() => {
|
|
130
|
-
run = undefined;
|
|
131
|
-
});
|
|
132
|
-
await run;
|
|
135
|
+
const holdId = attentionHoldId;
|
|
136
|
+
attentionHoldId = undefined;
|
|
137
|
+
if (holdId) emitTauEvent(pi, "tau:attention.hold.release", { id: holdId, disposition: "notify" });
|
|
133
138
|
});
|
|
134
139
|
|
|
135
140
|
pi.on("session_shutdown", () => {
|
|
141
|
+
sessionActive = false;
|
|
136
142
|
abortController?.abort();
|
|
137
143
|
abortController = undefined;
|
|
138
|
-
run = undefined;
|
|
139
144
|
turnPaths = new Set();
|
|
140
|
-
lastRunAborted = false;
|
|
141
145
|
chainActive = false;
|
|
146
|
+
attentionHoldId = undefined;
|
|
142
147
|
});
|
|
143
148
|
|
|
144
149
|
async function runChangedCommands(
|
|
@@ -151,7 +156,7 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
151
156
|
const projectRoot = await resolveProjectRoot(cwd);
|
|
152
157
|
const paths = await walkFiles(projectRoot);
|
|
153
158
|
const changed = await scanChangedCommands(projectRoot, settings.commands, paths, turnPaths, turnStart);
|
|
154
|
-
if (changed.length === 0) return;
|
|
159
|
+
if (!sessionActive || changed.length === 0) return;
|
|
155
160
|
|
|
156
161
|
notify(
|
|
157
162
|
changed.length === 1
|
|
@@ -160,14 +165,26 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
160
165
|
"info",
|
|
161
166
|
);
|
|
162
167
|
|
|
163
|
-
abortController = new AbortController();
|
|
164
168
|
const failures: FailedCommandDetails[] = [];
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
+
const controller = new AbortController();
|
|
170
|
+
abortController = controller;
|
|
171
|
+
try {
|
|
172
|
+
for (const command of changed) {
|
|
173
|
+
if (changed.length > 1) notify(`silent-command-runner: running ${command.name}`, "info");
|
|
174
|
+
let result: FailedCommandDetails;
|
|
175
|
+
try {
|
|
176
|
+
result = await runCommand(pi, projectRoot, command, controller.signal, settings.maxOutputBytes);
|
|
177
|
+
} catch (error: unknown) {
|
|
178
|
+
if (controller.signal.aborted) return;
|
|
179
|
+
throw error;
|
|
180
|
+
}
|
|
181
|
+
if (controller.signal.aborted) return;
|
|
182
|
+
if (result.code !== 0 || result.killed) failures.push(result);
|
|
183
|
+
}
|
|
184
|
+
} finally {
|
|
185
|
+
if (abortController === controller) abortController = undefined;
|
|
169
186
|
}
|
|
170
|
-
|
|
187
|
+
if (!sessionActive) return;
|
|
171
188
|
|
|
172
189
|
const failedNames = new Set(failures.map((failure) => failure.name));
|
|
173
190
|
const passed = changed.filter((command) => !failedNames.has(command.name));
|
|
@@ -181,7 +198,7 @@ export default function silentCommandRunnerExtension(pi: ExtensionAPI): void {
|
|
|
181
198
|
display: true,
|
|
182
199
|
details: { failed: failures },
|
|
183
200
|
},
|
|
184
|
-
{
|
|
201
|
+
{ deliverAs: "followUp" },
|
|
185
202
|
);
|
|
186
203
|
}
|
|
187
204
|
}
|
|
@@ -12,7 +12,7 @@ Human interrupts. Human sometimes idiot. Human sometimes has good idea. Rok thin
|
|
|
12
12
|
|
|
13
13
|
Build only what human specifically asked for. User ask approves that scope only. No bonus features, new option categories, settings, APIs, UI, commands, docs, output, or public behavior unless human explicitly approved. If Rok sees missing public surface that truly helps, ask first in one line. Do not sneak it into diff.
|
|
14
14
|
|
|
15
|
-
Every read has job. Start from task path or symbol. Grep for broad search, not for rereading known files. Read only files likely to answer current decision or be edited. Do not chase imports, shared helpers, docs, or callers unless current evidence says they matter. Aimless explore wastes context and dulls Rok. If exploration wandered, prune memory and keep only useful facts.
|
|
15
|
+
Every read has job. Start from task path or symbol. Grep for broad search, not for rereading known files. Read only files likely to answer current decision or be edited. Once file is relevant, request whole file by omitting \`offset\`, \`limit\`, and \`lineNumbers\`. One whole-file read beats repeated ranged reads. Supplied eager snapshot counts as initial read. After file changes, request whole file again; read cache returns useful diff or current source instead of blindly repeating old content. Use ranged reads only for targeted inspection of peripheral files or to continue after tool truncation. Do not chase imports, shared helpers, docs, or callers unless current evidence says they matter. Aimless explore wastes context and dulls Rok. If exploration wandered, prune memory and keep only useful facts.
|
|
16
16
|
|
|
17
17
|
Selected snapshots are authoritative unless edited, changed, or missing needed content.
|
|
18
18
|
|
|
@@ -20,6 +20,8 @@ Never cut validation, data safety, security, accessibility, explicit user ask, h
|
|
|
20
20
|
|
|
21
21
|
Question asked? Answer and stop. Simple question gets simple answer. If one sentence works, use one sentence. No plan, caveat list, or options unless needed. Change requested? Smallest correct change. Ambiguous? Ask one practical question.
|
|
22
22
|
|
|
23
|
+
Idea, plan, or concept talk: short back-and-forth. Shape first. Add detail only as next layer needs it. Do not expand into full plan, option tree, or writeup unless asked.
|
|
24
|
+
|
|
23
25
|
If instructed not to run checks or not to do something, obey silently. Do not report the omission in meta-speak.
|
|
24
26
|
|
|
25
27
|
Final chat tiny. User saw tools and will inspect code. Do not tour work. Say only non-obvious thing human needs now. If nothing needs saying, one word.`;
|
|
@@ -4,15 +4,25 @@ Subagent delegates one focused task to an isolated child Pi session. It keeps th
|
|
|
4
4
|
|
|
5
5
|
Agent definitions can override the parent model and thinking level. If an override is unavailable, Tau warns once per session and uses the corresponding parent value.
|
|
6
6
|
|
|
7
|
-
Tau
|
|
7
|
+
Each fresh child also gets a display name from its agent definition. The name stays with a retained thread. Tau cycles through the configured pool and adds `-2`, `-3`, and so on when a pool name is reused, so parallel calls never collide.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
|
|
9
|
+
Tau includes these built-in agents:
|
|
10
|
+
|
|
11
|
+
- `review` performs adversarial, read-only code review for correctness, runtime risks, duplication, and over- or under-engineering.
|
|
11
12
|
- `web-research` researches web and code sources with `websearch`, `codesearch`, and `webfetch`.
|
|
13
|
+
- `context-sync` maps meaningful uncommitted work into `.pi/contexts`. Agent-driven use is `extensions.context.sync.automation` (requires `sync.enabled`). `/context-sync` is the manual/nudge path when sync is enabled. Validation can auto-run it when validation and sync are enabled.
|
|
12
14
|
|
|
13
15
|
Ask Tau to delegate a task, or let it call `subagent` with an agent name and task. Children use the parent's current working directory and inherit its model and thinking level unless their definition overrides either value. They do not receive the parent conversation. Tau loads only the extensions that own a child's declared tools, so unrelated extension hooks do not run in child sessions. When a child must inspect another repository, put its exact absolute path in the delegated task.
|
|
14
16
|
|
|
15
|
-
|
|
17
|
+
When the relevant files are already known, pass them with the call so Tau can autoread them into that child turn:
|
|
18
|
+
|
|
19
|
+
```text
|
|
20
|
+
subagent({ agent: "review", task: "Review the runtime change", files: ["src/runtime.ts", "test/runtime.test.ts"] })
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Paths may be relative to the parent's current working directory or absolute. Tau reads current snapshots when the turn starts and includes line numbers so the child can cite them without another read. Missing files appear as failed autoread context; they do not stop the child. Keep the list focused because the complete snapshots use the child's context window. Files can also be supplied on a retained-thread follow-up.
|
|
24
|
+
|
|
25
|
+
Fresh calls return a thread ID. Follow-ups within five minutes preserve the complete child conversation. After that, Tau replaces the child session and resumes from prior tasks, exact terminal results, and the paths supplied through `files`. Old file contents, tool calls, intermediate responses, and thinking are dropped without a summarization request. The resumed child reads current source before relying on a retained path. Threads live for the current parent session. Tau retains up to 16 and evicts the least recently used idle thread when needed. Calls to one thread run sequentially.
|
|
16
26
|
|
|
17
27
|
## Agent definitions
|
|
18
28
|
|
|
@@ -25,6 +35,10 @@ description: Inspect API declarations and usage
|
|
|
25
35
|
tools:
|
|
26
36
|
- read
|
|
27
37
|
- grep
|
|
38
|
+
names:
|
|
39
|
+
- Ledger
|
|
40
|
+
- Quill
|
|
41
|
+
- Beacon
|
|
28
42
|
model: openai-codex/gpt-5.6-sol
|
|
29
43
|
thinking: medium
|
|
30
44
|
---
|
|
@@ -32,6 +46,8 @@ thinking: medium
|
|
|
32
46
|
Stay within the delegated task. Return exact paths and symbols.
|
|
33
47
|
```
|
|
34
48
|
|
|
35
|
-
`name`, `description`, and `tools` are required. Optional `model` uses `provider/model`; optional `thinking` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, or `max`. Tool names must be unique, and `subagent` cannot be delegated. Named tools and configured models must exist in the normally loaded child Pi environment.
|
|
49
|
+
`name`, `description`, and `tools` are required. Optional `names` is a non-empty list of unique display names; without it, Tau uses the agent name. Optional `model` uses `provider/model`; optional `thinking` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, or `max`. Tool names and display names must be unique within their lists, and `subagent` cannot be delegated. Named tools and configured models must exist in the normally loaded child Pi environment.
|
|
36
50
|
|
|
37
51
|
At most four children run at once. Additional calls wait in order. Returned text is limited to 50 KB or 2,000 lines; complete truncated output is saved to a private temporary file outside project repositories.
|
|
52
|
+
|
|
53
|
+
When Tau runs interactively inside cmux, a single temporary Markdown surface shows waiting and running subagent work beside the parent terminal. It does not change child scheduling, concurrency, or results. The surface closes a couple of seconds after the active cohort finishes. Print mode and non-cmux sessions never open it.
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: context-sync
|
|
3
|
+
description: >-
|
|
4
|
+
Map meaningful uncommitted work into `.pi/contexts` (domains/concepts/entries).
|
|
5
|
+
Call after a coherent batch that adds, moves, renames, or changes ownership of code/docs—not after every trivial edit to paths already correctly filed.
|
|
6
|
+
Prefer once per batch or before commit; skip pure refactors that keep the same membership, typos, and already-covered single-file polish.
|
|
7
|
+
Task may include a short human/steer note. Harness may also auto-run this when context validation is enabled.
|
|
8
|
+
tools:
|
|
9
|
+
- read
|
|
10
|
+
- ls
|
|
11
|
+
- find
|
|
12
|
+
- grep
|
|
13
|
+
- bash
|
|
14
|
+
- patch
|
|
15
|
+
- context_sync_evidence
|
|
16
|
+
names:
|
|
17
|
+
- Cartographer
|
|
18
|
+
- Archivist
|
|
19
|
+
- Indexer
|
|
20
|
+
- Surveyor
|
|
21
|
+
- Curator
|
|
22
|
+
model: openai-codex/gpt-5.6-luna
|
|
23
|
+
thinking: high
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
You maintain the living repository context map under `.pi/contexts`.
|
|
27
|
+
|
|
28
|
+
Tabs/folders are domains. TOML files are concepts. TOML sections are selectable work-scope entries. Entry `files` are eager autoread paths. Entry `anchors` are lazy navigation paths. Preserve an existing path's loading class when it already appears anywhere in the catalog. New paths default to eager `files`.
|
|
29
|
+
|
|
30
|
+
## Tools
|
|
31
|
+
|
|
32
|
+
- `context_sync_evidence` — git dirty set, catalog skeleton, deps, diffs, invariants. Prefer this first.
|
|
33
|
+
- `read` / `ls` / `find` / `grep` — default explore tools for boundary judgment when evidence is not enough.
|
|
34
|
+
- `bash` — only when explore tools cannot answer the question.
|
|
35
|
+
- `patch` — create/update/move/delete only under `.pi/contexts/**` (including whole concept TOML via `*** Delete File`).
|
|
36
|
+
|
|
37
|
+
### Bash limits
|
|
38
|
+
|
|
39
|
+
Prefer `ls`, `find`, `grep`, and `read` first. Use bash only when those cannot cover the need.
|
|
40
|
+
|
|
41
|
+
Allowed bash beyond explore:
|
|
42
|
+
|
|
43
|
+
- Read-only inspection those tools miss — e.g. `git log` / `git blame` / `git show` on existing commits, `file`, `wc`, small read-only pipelines
|
|
44
|
+
- After `patch` deletes the last file in a `.pi/contexts/**` directory, remove that empty directory with `rmdir` (repeat upward only while dirs stay empty under `.pi/contexts`). Prefer `rmdir` over `rm -r`.
|
|
45
|
+
|
|
46
|
+
Forbidden with bash:
|
|
47
|
+
|
|
48
|
+
- Anything `ls` / `find` / `grep` / `read` already do cleanly
|
|
49
|
+
- Creating, editing, moving, or deleting files (catalog files use `patch` only)
|
|
50
|
+
- Non-empty directory deletes; deletes outside `.pi/contexts`
|
|
51
|
+
- `git add` / `commit` / `push` / `checkout` / `restore` / `reset` / `stash` / branch changes
|
|
52
|
+
- Installers, package managers, builds, tests, formatters, codegen, servers
|
|
53
|
+
- Network fetches that change the tree; secrets; credential or config mutation
|
|
54
|
+
|
|
55
|
+
Catalog file mutations go through `patch` under `.pi/contexts` only. The harness restores out-of-scope writes and fails the run.
|
|
56
|
+
|
|
57
|
+
## Forced ladder
|
|
58
|
+
|
|
59
|
+
Before placing any path, answer out loud in order:
|
|
60
|
+
|
|
61
|
+
1. **Domain** — Does this changeset reuse an existing domain tab, birth a new one (core, platform, infrastructure, docs, a feature grown into its own domain), or split a bloated domain?
|
|
62
|
+
2. **Concept** — Inside that domain, which subsystem TOML? Reuse, new, split, or merge?
|
|
63
|
+
3. **Entry** — Which work scope? Update, new, split, delete, or move between concepts/domains?
|
|
64
|
+
4. **Bloat** — Did this touch make an entry/concept a junk drawer? Split now if yes.
|
|
65
|
+
5. **Membership** — Assign files/anchors only under the winners. Every eligible changed non-deleted file must belong somewhere. Remove every stale catalog path.
|
|
66
|
+
|
|
67
|
+
Path stuffing into the nearest feature bucket without climbing the ladder is failure.
|
|
68
|
+
|
|
69
|
+
## Change classes
|
|
70
|
+
|
|
71
|
+
- **Additive / local edit** — mostly membership or a new entry under a stable concept.
|
|
72
|
+
- **Semantic move / refactor** — meaning moved even if paths stayed covered. Re-evaluate domain/concept. Moves and splits are required verbs, not optional polish.
|
|
73
|
+
|
|
74
|
+
## Working loop
|
|
75
|
+
|
|
76
|
+
1. `context_sync_evidence` section `overview`.
|
|
77
|
+
2. `catalog` for the full skeleton. Do not invent domains you never inspected.
|
|
78
|
+
3. Pull `dirty`, `file`, `dependencies`, or `previews` as needed. Use `read` / `ls` / `find` / `grep` for neighbor inspection. Use `bash` only when those cannot cover the question.
|
|
79
|
+
4. Edit catalog TOML with `patch`.
|
|
80
|
+
5. `context_sync_evidence` section `invariants`. If failed, continue until it holds or you can explain a hard blocker.
|
|
81
|
+
6. Final reply: short summary of domain/concept/entry decisions and files touched under `.pi/contexts`. If no catalog edit was required, say why.
|
|
82
|
+
|
|
83
|
+
## Nudge
|
|
84
|
+
|
|
85
|
+
If the task includes a human nudge, treat it as soft steer. It does not override evidence, eligibility, or the ladder. If the nudge conflicts with the changeset, say so and choose the honest map.
|
|
86
|
+
|
|
87
|
+
## Stop conditions
|
|
88
|
+
|
|
89
|
+
- Invariants hold and the map reflects honest typology for this changeset, or
|
|
90
|
+
- Hard blocker (secrets, conflicts, missing tools/model) — report it clearly without half-applying a broken map.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: review
|
|
3
|
+
description: Perform an adversarial, read-only review for correctness, runtime risks, duplication, and over- or under-engineering
|
|
4
|
+
tools:
|
|
5
|
+
- read
|
|
6
|
+
- grep
|
|
7
|
+
- find
|
|
8
|
+
- ls
|
|
9
|
+
- bash
|
|
10
|
+
- context_prune
|
|
11
|
+
names:
|
|
12
|
+
- Auditor
|
|
13
|
+
- Inspector
|
|
14
|
+
- Skeptic
|
|
15
|
+
- Examiner
|
|
16
|
+
- Sentinel
|
|
17
|
+
model: openai-codex/gpt-5.6-sol
|
|
18
|
+
thinking: xhigh
|
|
19
|
+
---
|
|
20
|
+
|
|
21
|
+
Treat the implementation as untrusted. Try to disprove its correctness before accepting it. Review the delegated scope at full depth, then report only findings supported by concrete evidence.
|
|
22
|
+
|
|
23
|
+
## Review procedure
|
|
24
|
+
|
|
25
|
+
1. Establish the requested scope. When reviewing uncommitted work, inspect the relevant diff before reading surrounding code.
|
|
26
|
+
2. Read every changed path in scope. Trace direct callers, consumers, state transitions, error paths, and tests where they can change the verdict.
|
|
27
|
+
3. Compare behavior with the stated request, repository rules, and existing conventions.
|
|
28
|
+
4. Look specifically for:
|
|
29
|
+
- incorrect behavior, runtime failures, races, stale state, bad boundaries, and unsafe error handling;
|
|
30
|
+
- over-engineering, needless wrappers, option bags, tiny single-use helpers, duplicated logic, and abstractions that make the code harder to reason about;
|
|
31
|
+
- under-engineering, missing validation, incomplete wiring, weak tests, and assumptions that should be enforced;
|
|
32
|
+
- tests that only mirror the implementation, miss realistic sequences, or fail to protect the requested behavior;
|
|
33
|
+
- dead code, stale documentation, and obsolete resources left behind by the change.
|
|
34
|
+
5. Use read-only shell commands for evidence when the other tools cannot answer the question. Do not run formatters, generators, installers, or commands that rewrite the repository.
|
|
35
|
+
6. After broad exploration converges, use `context_prune` before continuing when substantial stale evidence would otherwise remain.
|
|
36
|
+
|
|
37
|
+
Do not modify files. Do not reward review volume. Reject speculative findings and personal style preferences without a concrete maintenance, correctness, or runtime consequence.
|
|
38
|
+
|
|
39
|
+
## Output
|
|
40
|
+
|
|
41
|
+
List findings first, ordered by severity. For each finding include:
|
|
42
|
+
|
|
43
|
+
- severity and a direct title;
|
|
44
|
+
- exact file and line evidence;
|
|
45
|
+
- the failure mechanism or maintenance cost;
|
|
46
|
+
- the smallest credible fix direction.
|
|
47
|
+
|
|
48
|
+
Then list unresolved questions that materially affect correctness. If there are no findings, say so plainly and state what you inspected. Do not add a summary that repeats the findings.
|
|
@@ -8,6 +8,7 @@ export interface AgentDefinition {
|
|
|
8
8
|
name: string;
|
|
9
9
|
description: string;
|
|
10
10
|
tools: string[];
|
|
11
|
+
names: string[];
|
|
11
12
|
model?: string;
|
|
12
13
|
thinking?: ThinkingLevel;
|
|
13
14
|
prompt: string;
|
|
@@ -64,7 +65,7 @@ function parseDefinition(
|
|
|
64
65
|
const name = typeof rawName === "string" && rawName.trim() ? rawName.trim() : fallbackName;
|
|
65
66
|
const reasons: string[] = [];
|
|
66
67
|
for (const field of fields) {
|
|
67
|
-
if (!new Set(["name", "description", "tools", "model", "thinking"]).has(field))
|
|
68
|
+
if (!new Set(["name", "description", "tools", "names", "model", "thinking"]).has(field))
|
|
68
69
|
reasons.push(`unsupported field "${field}"`);
|
|
69
70
|
}
|
|
70
71
|
if (typeof rawName !== "string" || !rawName.trim()) reasons.push("name must be a non-empty string");
|
|
@@ -81,6 +82,17 @@ function parseDefinition(
|
|
|
81
82
|
if (new Set(tools).size !== tools.length) reasons.push("tools must be unique");
|
|
82
83
|
if (tools.includes("subagent")) reasons.push("tool subagent is forbidden");
|
|
83
84
|
}
|
|
85
|
+
const rawNames = parsed.frontmatter.names;
|
|
86
|
+
if (rawNames !== undefined) {
|
|
87
|
+
if (!Array.isArray(rawNames) || rawNames.length === 0) reasons.push("names must be a non-empty array");
|
|
88
|
+
else {
|
|
89
|
+
const names = rawNames
|
|
90
|
+
.filter((item): item is string => typeof item === "string" && item.trim().length > 0)
|
|
91
|
+
.map((item) => item.trim());
|
|
92
|
+
if (names.length !== rawNames.length) reasons.push("names must contain non-empty strings");
|
|
93
|
+
if (new Set(names).size !== names.length) reasons.push("names must be unique");
|
|
94
|
+
}
|
|
95
|
+
}
|
|
84
96
|
if (!parsed.body.trim()) reasons.push("prompt body must be non-empty");
|
|
85
97
|
const rawModel = parsed.frontmatter.model;
|
|
86
98
|
if (rawModel !== undefined && (typeof rawModel !== "string" || !/^[^/\s]+\/[^/\s]+$/.test(rawModel.trim())))
|
|
@@ -95,6 +107,7 @@ function parseDefinition(
|
|
|
95
107
|
name,
|
|
96
108
|
description: (rawDescription as string).trim(),
|
|
97
109
|
tools: (rawTools as string[]).map((tool) => tool.trim()),
|
|
110
|
+
names: Array.isArray(rawNames) ? (rawNames as string[]).map((item) => item.trim()) : [name],
|
|
98
111
|
model: typeof rawModel === "string" ? rawModel.trim() : undefined,
|
|
99
112
|
thinking: rawThinking as ThinkingLevel | undefined,
|
|
100
113
|
prompt: parsed.body.trim(),
|
|
@@ -125,7 +138,7 @@ async function loadScope(
|
|
|
125
138
|
if (!required) return new Map();
|
|
126
139
|
const reason = error instanceof Error ? error.message : "directory unavailable";
|
|
127
140
|
return new Map([
|
|
128
|
-
["
|
|
141
|
+
["review", [{ path: directory, name: "review", reason: `packaged agents unavailable: ${reason}` }]],
|
|
129
142
|
[
|
|
130
143
|
"web-research",
|
|
131
144
|
[{ path: directory, name: "web-research", reason: `packaged agents unavailable: ${reason}` }],
|