@shanepadgett/tau-agent 0.45.1 → 0.46.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/patch/README.md +1 -1
- package/extensions/patch/index.ts +7 -6
- package/extensions/soul/README.md +1 -1
- package/extensions/soul/index.ts +6 -1
- package/extensions/soul/prompt.ts +7 -1
- package/extensions/tau-help/help.md +5 -3
- package/extensions/tool-approval/README.md +11 -3
- package/extensions/tool-approval/evidence.ts +679 -0
- package/extensions/tool-approval/index.ts +377 -120
- package/package.json +2 -2
- package/shared/events.ts +20 -44
- package/shared/model-fallback/index.ts +10 -4
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# patch
|
|
2
2
|
|
|
3
|
-
Replaces the built-in `edit` and `write` tools with one multi-file patch tool. The agent applies structured patches to create, edit, move, and delete files while this extension is active.
|
|
3
|
+
Replaces the built-in `edit` and `write` tools with one multi-file patch tool. The agent applies structured patches to create, edit, move, and delete files while this extension is active. Tau enables `patch` only for OpenAI and OpenAI Codex models; every other provider keeps `edit` and `write` and loses `patch` because those models apply the patch format less reliably.
|
|
4
4
|
|
|
5
5
|
## What it does
|
|
6
6
|
|
|
@@ -7,6 +7,7 @@ import { renderPatchCall, renderPatchResult } from "./render.ts";
|
|
|
7
7
|
import { formatPatchSummary } from "./summary.ts";
|
|
8
8
|
|
|
9
9
|
const SUPPRESSED_TOOLS = new Set(["edit", "write"]);
|
|
10
|
+
const PATCH_PROVIDERS = new Set(["openai", "openai-codex"]);
|
|
10
11
|
|
|
11
12
|
const patchParams = Type.Object({
|
|
12
13
|
input: Type.String({
|
|
@@ -135,15 +136,15 @@ export default function patchExtension(pi: ExtensionAPI): void {
|
|
|
135
136
|
const rowState = createToolRowStateStore(pi, "patch.tool-row-state");
|
|
136
137
|
pi.registerTool(createPatchTool(rowState));
|
|
137
138
|
|
|
138
|
-
function configureMutationTools(model: { provider: string
|
|
139
|
+
function configureMutationTools(model: { provider: string } | undefined): void {
|
|
139
140
|
const active = new Set(pi.getActiveTools());
|
|
140
|
-
const
|
|
141
|
-
if (
|
|
142
|
-
active.delete("patch");
|
|
143
|
-
for (const tool of SUPPRESSED_TOOLS) active.add(tool);
|
|
144
|
-
} else {
|
|
141
|
+
const usesPatch = model !== undefined && PATCH_PROVIDERS.has(model.provider.toLowerCase());
|
|
142
|
+
if (usesPatch) {
|
|
145
143
|
active.add("patch");
|
|
146
144
|
for (const tool of SUPPRESSED_TOOLS) active.delete(tool);
|
|
145
|
+
} else {
|
|
146
|
+
active.delete("patch");
|
|
147
|
+
for (const tool of SUPPRESSED_TOOLS) active.add(tool);
|
|
147
148
|
}
|
|
148
149
|
pi.setActiveTools([...active]);
|
|
149
150
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Soul
|
|
2
2
|
|
|
3
|
-
Soul supplies Tau's system prompt: communication, discussion, planning, execution, and
|
|
3
|
+
Soul supplies Tau's system prompt: communication, discussion, planning, execution, coding, and tool-use guidance. It is always on.
|
|
4
4
|
|
|
5
5
|
Soul adds Pi documentation pointers, tool guidance, and the context that other Tau extensions supply, such as the local date, directory snapshot, and automatic-check instructions. The date and directory snapshot are captured on the first prompt and again after successful compaction, so they stay fixed across turns, reload, resume, and tree navigation.
|
|
6
6
|
|
package/extensions/soul/index.ts
CHANGED
|
@@ -40,7 +40,12 @@ export default function soulExtension(pi: ExtensionAPI): void {
|
|
|
40
40
|
const active = pi.getActiveTools();
|
|
41
41
|
const guidance = [
|
|
42
42
|
...new Set([
|
|
43
|
-
...(active.includes("bash")
|
|
43
|
+
...(active.includes("bash")
|
|
44
|
+
? [
|
|
45
|
+
"Use bash for file operations like ls, rg, find.",
|
|
46
|
+
"Bash already runs in the working directory; do not cd into it.",
|
|
47
|
+
]
|
|
48
|
+
: []),
|
|
44
49
|
...active.flatMap((name) => options.toolGuidelines[name] ?? []),
|
|
45
50
|
...options.promptGuidelines,
|
|
46
51
|
]),
|
|
@@ -36,11 +36,17 @@ Stay within that scope. Ask before making a consequential choice the user has no
|
|
|
36
36
|
Take the normal supported path. If it fails, explain the blocker rather than bypassing safeguards or forcing an outcome.
|
|
37
37
|
Complete the authorized work and check the result. Report what changed, what was checked, and anything unresolved.
|
|
38
38
|
Give brief progress updates when work takes time or the direction changes.
|
|
39
|
-
When gathering independent information, request it together rather than one item per turn.
|
|
40
39
|
For authorized work, make reasonable low-risk assumptions and proceed. Ask when a missing answer affects correctness, scope, or consequences.
|
|
41
40
|
Keep the final answer proportional to the request. Avoid turning a simple answer into a report with repeated summaries.
|
|
42
41
|
</execution>
|
|
43
42
|
|
|
43
|
+
<tool-use>
|
|
44
|
+
Use the tools available to you for the purpose each is designed for.
|
|
45
|
+
Prefer the file tools for ordinary reads, edits, and writes. Use bash or scripts when they express the work more directly and in fewer tokens: generating files, bulk mechanical transforms, or computation whose intermediate output does not need to be shown.
|
|
46
|
+
Batch independent calls into one response: independent reads, searches, and edits go together, and only calls that depend on earlier results are sequenced.
|
|
47
|
+
Keep tool output small: request only the data you need and prefer compact, high-signal commands over ones that flood the context.
|
|
48
|
+
</tool-use>
|
|
49
|
+
|
|
44
50
|
<coding>
|
|
45
51
|
For a prototype, make the requested idea work with minimal setup and polish.
|
|
46
52
|
For product code, follow the project conventions and reuse existing code, standard libraries, and installed dependencies before adding something new.
|
|
@@ -78,7 +78,7 @@ Adds `/manage-sessions` to browse saved sessions and `/sweep` to archive or dele
|
|
|
78
78
|
|
|
79
79
|
## patch
|
|
80
80
|
|
|
81
|
-
Replaces separate edit/write operations with one multi-file `patch` tool. It can create, rewrite, edit, move, and delete files in one structured call. Fewer tool calls means fewer turns, and each avoided turn prevents the full chat context from being sent again. Tau
|
|
81
|
+
Replaces separate edit/write operations with one multi-file `patch` tool. It can create, rewrite, edit, move, and delete files in one structured call. Fewer tool calls means fewer turns, and each avoided turn prevents the full chat context from being sent again. Tau enables `patch` only for OpenAI and OpenAI Codex models and uses `edit` and `write` for every other provider.
|
|
82
82
|
|
|
83
83
|
## qna
|
|
84
84
|
|
|
@@ -114,7 +114,7 @@ Runs configured commands while keeping their output out of agent context when th
|
|
|
114
114
|
|
|
115
115
|
## soul
|
|
116
116
|
|
|
117
|
-
Supplies Tau's communication, discussion, planning, execution, and coding instructions, plus tool guidance and context from other Tau extensions. The date and directory snapshot stay fixed until successful compaction. Other changes, such as an edited `AGENTS.md` after `/reload`, arrive as appended updates without rewriting earlier instructions.
|
|
117
|
+
Supplies Tau's communication, discussion, planning, execution, and coding instructions, plus tool-use rules, tool guidance, and context from other Tau extensions. The date and directory snapshot stay fixed until successful compaction. Other changes, such as an edited `AGENTS.md` after `/reload`, arrive as appended updates without rewriting earlier instructions.
|
|
118
118
|
|
|
119
119
|
## stash
|
|
120
120
|
|
|
@@ -130,7 +130,9 @@ Adds `/tau`, `/tau init [--global|--project]`, and `/tau doctor` for Tau setup a
|
|
|
130
130
|
|
|
131
131
|
## tool-approval
|
|
132
132
|
|
|
133
|
-
Reviews agent `bash` and `script_runner` requests before they run. Common read-only bash commands skip review. Set `extensions.toolApproval.autoApprove` to run every reviewer-approved request without another confirmation. Those auto-approvals show a user-only marker. The reviewer approves routine local development work. Concrete destructive, system, production, privileged, or security-sensitive effects require human approval with
|
|
133
|
+
Reviews agent `bash` and `script_runner` requests before they run. Common read-only bash commands skip review. Visible, understood requests need one review; hidden local execution targets get bounded inspection and one final review, without repository exploration. Unverified execution behavior requires human approval. Set `extensions.toolApproval.autoApprove` to run every reviewer-approved request without another confirmation. Those auto-approvals show a user-only marker. The reviewer approves routine local development work. Concrete destructive, system, production, privileged, or security-sensitive effects require human approval with a plain-language explanation of the effect, affected target, risk, and recovery difficulty. Changes to inspected files invalidate approval. Reviewer failures fall back to human approval and send an attention notification. In the terminal approval panel, press `n` to add a note to Approve or Reject before choosing. Rejection notes tell the agent why the request was blocked; approval notes reach it with the tool result without changing the request. Reject with a note to ask for a revised request.
|
|
134
|
+
|
|
135
|
+
Approval decisions are saved privately in the session JSONL, including allowlist skips, review stages and models, inspected paths, evidence-gap categories, and user decisions. These records stay out of model context and do not copy scripts, arguments, file contents, or raw errors.
|
|
134
136
|
|
|
135
137
|
## tool-loader
|
|
136
138
|
|
|
@@ -2,11 +2,19 @@
|
|
|
2
2
|
|
|
3
3
|
Reviews agent `bash` and `script_runner` requests before they run.
|
|
4
4
|
|
|
5
|
-
Common read-only bash commands skip review and run immediately. Other bash and every `script_runner` request go to a separate reviewer.
|
|
5
|
+
Common read-only bash commands skip review and run immediately. Other bash and every `script_runner` request go to a separate reviewer. The first review can approve, ask you, or request inspection of directly referenced code. Visible, understood requests need only one review. When inspection is needed, Tau reads exact execution targets and supported direct local dependencies, then makes one final review with the same reviewer model policy. Obvious local script execution cannot be automatically approved without inspection, even if the first reviewer misses it.
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
Inspection is limited to four files, 48 KiB of source, and a two-second file-work budget. Tau does not browse directories, search the repository, run code, or fetch remote code during review. It can inspect common literal Python, Node, Deno, and shell script invocations, local imports, subprocess targets, and package task definitions. Project npm execution settings are included without sending registry credentials. Unsupported Yarn executable configuration requires confirmation. Dynamic paths, unsupported execution forms, missing source, and exhausted limits require human approval with an explanation of what could not be verified. Installed tools and libraries retain their normal trust assumption; approval is a risk filter, not a sandbox or a proof that arbitrary code is safe.
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
For a `script_runner` retry with `scriptId` and edits, Tau reconstructs the full resulting script before review and runs that exact source after approval. A retry with missing or invalid stored source is blocked. When the agent requests several tools at once, Tau reviews up to three requests concurrently, with each review focused on one request. Tau rechecks inspected file contents before using a decision and after human confirmation. Changed files invalidate approval; repeated changes or changes during confirmation block the request until the agent submits it again. These checks do not make shell execution atomic with file inspection.
|
|
10
|
+
|
|
11
|
+
Tau uses the reviewer model for the current provider, then the current chat model if that reviewer is unavailable or fails. If a reviewer model is unavailable or fails, Tau notifies and tries the next one. Both review stages use this existing model policy.
|
|
12
|
+
|
|
13
|
+
Each handled request saves a private decision record in the session JSONL under `tau.tool-approval.decision`. Records include the tool-call ID, tool name, approval or block decision, decision source and reason, review stages and reviewer models, inspected paths, evidence-gap categories, and elapsed time. They also cover allowlist skips, disabled approval, failed reviews, cancellations, and user rejection. These records are excluded from model context and follow the session's storage and deletion lifecycle. They do not copy command arguments, scripts, file contents, reviewer explanations, approval notes, or raw errors. Inspected paths are retained, so file names remain visible in the session.
|
|
14
|
+
|
|
15
|
+
With `autoApprove` enabled, reviewer-approved requests run without another confirmation. Tau shows a user-only marker with the reviewer model after those auto-approvals. Common read-only bash that skips review does not get a marker. Routine local development work should be approved, including requests that modify project files or run inspected scripts. The reviewer asks for human approval when it finds a concrete destructive, system, production, privileged, or security-sensitive effect, or cannot verify important execution behavior.
|
|
16
|
+
|
|
17
|
+
When approval is required, Tau shows one plain-language paragraph explaining what you are allowing, who or what is affected, why approval is needed, and what recovery might involve. It summarizes consequences rather than listing script steps or specialized APIs. If the issue is missing evidence rather than a known danger, it says what could not be checked. For `script_runner`, human confirmation also shows the complete script that will run. If several requests need approval, their confirmation windows open one at a time. If the reviewer fails or returns a malformed decision, Tau asks for direct human approval instead of running it automatically. Tau also sends an attention notification when the approval window opens.
|
|
10
18
|
|
|
11
19
|
In the terminal approval panel, move between Approve and Reject, press `j` or `k` to scroll a script, press `n` to add a note to the highlighted choice, then press Enter to choose. Enter saves an edited note before choosing; Escape cancels note editing or blocks the request from the choice list. A rejection note tells the agent why the request was blocked. An approval note reaches the agent with the tool result; it does not change the request being approved. To ask for a different request, reject it with a note. Long notes are truncated. RPC clients use the standard confirmation dialog without notes.
|
|
12
20
|
|