@shanepadgett/tau-agent 0.41.1 → 0.41.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/auto-name/index.ts +1 -1
- package/extensions/commit/commit-plan.ts +2 -1
- package/extensions/handoff/index.ts +1 -1
- package/extensions/soul/overseer.ts +2 -1
- package/extensions/subagent/index.ts +77 -50
- package/extensions/tool-approval/README.md +2 -2
- package/extensions/tool-approval/index.ts +89 -54
- package/package.json +2 -2
- package/shared/model-fallback/index.ts +5 -4
|
@@ -88,7 +88,7 @@ async function runAutoName(
|
|
|
88
88
|
const ui = ctx.ui;
|
|
89
89
|
try {
|
|
90
90
|
const candidates = await resolveCandidates(ctx, AUTO_NAME_MODELS, true);
|
|
91
|
-
const result = await generateToolValidated(
|
|
91
|
+
const { value: result } = await generateToolValidated(
|
|
92
92
|
{ ui, signal: controller.signal },
|
|
93
93
|
candidates,
|
|
94
94
|
`${NAMING_PROMPT}\n\n${prompt}`,
|
|
@@ -57,7 +57,7 @@ export async function generatePlan(
|
|
|
57
57
|
options?: CommitGenerationOptions,
|
|
58
58
|
): Promise<CommitGroup[]> {
|
|
59
59
|
const prompt = buildPlanPrompt(evidence, previousPlan, regenerationNote);
|
|
60
|
-
|
|
60
|
+
const { value } = await generateToolValidated(
|
|
61
61
|
ctx,
|
|
62
62
|
await resolveEffortCandidates(ctx, commitEffort(evidence.files), { includeParentModel: true }),
|
|
63
63
|
prompt,
|
|
@@ -79,6 +79,7 @@ export async function generatePlan(
|
|
|
79
79
|
notifyOnFallback: true,
|
|
80
80
|
},
|
|
81
81
|
);
|
|
82
|
+
return value;
|
|
82
83
|
}
|
|
83
84
|
|
|
84
85
|
export async function regenerateMessage(
|
|
@@ -130,7 +130,7 @@ async function reviewPrimaryDirective(
|
|
|
130
130
|
toolCalls: readonly ToolSignature[],
|
|
131
131
|
): Promise<PrimaryDirectiveReview> {
|
|
132
132
|
const candidates = await resolveEffortCandidates(ctx, "standard", { includeParentModel: false });
|
|
133
|
-
|
|
133
|
+
const { value } = await generateToolValidated(
|
|
134
134
|
ctx,
|
|
135
135
|
candidates,
|
|
136
136
|
buildReviewPrompt(branch, toolCalls),
|
|
@@ -144,6 +144,7 @@ async function reviewPrimaryDirective(
|
|
|
144
144
|
undefined,
|
|
145
145
|
{ maxAttempts: 1 },
|
|
146
146
|
);
|
|
147
|
+
return value;
|
|
147
148
|
}
|
|
148
149
|
|
|
149
150
|
function buildReviewPrompt(branch: readonly SessionEntry[], toolCalls: readonly ToolSignature[]): string {
|
|
@@ -18,32 +18,29 @@ interface SubagentSessionState {
|
|
|
18
18
|
|
|
19
19
|
const SUBAGENT_SESSION_STATE_TYPE = "tau.subagent.disabled";
|
|
20
20
|
|
|
21
|
-
const params = Type.
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
{
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
{ additionalProperties: false },
|
|
45
|
-
),
|
|
46
|
-
]);
|
|
21
|
+
const params = Type.Object(
|
|
22
|
+
{
|
|
23
|
+
agent: Type.Optional(
|
|
24
|
+
Type.String({
|
|
25
|
+
minLength: 1,
|
|
26
|
+
description: "Fresh agent name. Pass agent or thread, never both.",
|
|
27
|
+
}),
|
|
28
|
+
),
|
|
29
|
+
thread: Type.Optional(
|
|
30
|
+
Type.String({
|
|
31
|
+
minLength: 1,
|
|
32
|
+
description: "Retained thread id. Pass thread or agent, never both.",
|
|
33
|
+
}),
|
|
34
|
+
),
|
|
35
|
+
task: Type.String({ minLength: 1 }),
|
|
36
|
+
files: Type.Optional(
|
|
37
|
+
Type.Array(Type.String({ minLength: 1 }), {
|
|
38
|
+
description: "Files to autoread into the child's context before this turn",
|
|
39
|
+
}),
|
|
40
|
+
),
|
|
41
|
+
},
|
|
42
|
+
{ additionalProperties: false },
|
|
43
|
+
);
|
|
47
44
|
|
|
48
45
|
export default function subagentExtension(pi: ExtensionAPI): void {
|
|
49
46
|
const runtime = new SubagentRuntime(pi);
|
|
@@ -215,7 +212,7 @@ Use \`subagent\` only when an available agent's listed purpose matches the deleg
|
|
|
215
212
|
Available agents for this turn:
|
|
216
213
|
${lines.join("\n") || "none"}
|
|
217
214
|
|
|
218
|
-
Start a fresh thread with \`agent\` and \`task\`. Continue an existing thread with \`thread\` and \`task\`. Reuse a thread when feedback or follow-up work depends on its prior reads and reasoning. Start fresh for unrelated work or when its context is stale or oversized.
|
|
215
|
+
Start a fresh thread with \`agent\` and \`task\`. Continue an existing thread with \`thread\` and \`task\`. Pass exactly one of \`agent\` or \`thread\`, never both and never neither. Reuse a thread when feedback or follow-up work depends on its prior reads and reasoning. Start fresh for unrelated work or when its context is stale or oversized.
|
|
219
216
|
|
|
220
217
|
Pass \`files\` when exact relevant files are already known. Tau autoreads current line-numbered snapshots into that child turn before it starts.
|
|
221
218
|
|
|
@@ -296,25 +293,48 @@ Delegate one focused task per call. Children do not inherit parent messages. Inc
|
|
|
296
293
|
async execute(_id, raw, signal, onUpdate, ctx) {
|
|
297
294
|
const task = raw.task.trim();
|
|
298
295
|
const files = [...new Set((raw.files ?? []).map((path) => path.trim()))];
|
|
299
|
-
const
|
|
296
|
+
const agentName = raw.agent?.trim() || undefined;
|
|
297
|
+
const threadKey = raw.thread?.trim() || undefined;
|
|
300
298
|
const parentModel = ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : "unavailable";
|
|
301
299
|
const parentThinking = pi.getThinkingLevel();
|
|
302
|
-
const agent = continuing ? raw.thread.trim() : raw.agent.trim();
|
|
303
|
-
const threadKey = continuing ? raw.thread.trim() : undefined;
|
|
304
300
|
|
|
305
301
|
failureNotify = (message) => {
|
|
306
302
|
ctx.ui.notify(message, "warning");
|
|
307
303
|
};
|
|
308
304
|
dashboard.setInteractive(ctx.mode === "tui" && ctx.hasUI);
|
|
309
305
|
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
return
|
|
306
|
+
const failQueue = (agent: string, error: string, thread?: string) =>
|
|
307
|
+
failedToolResult(agent, task, "queue", parentModel, parentThinking, error, thread);
|
|
308
|
+
|
|
309
|
+
if (!task || files.some((path) => !path)) {
|
|
310
|
+
return failQueue(
|
|
311
|
+
agentName ?? threadKey ?? "",
|
|
312
|
+
threadKey !== undefined
|
|
313
|
+
? "Subagent continuation requires non-empty thread, task, and file paths"
|
|
314
|
+
: "Subagent input requires non-empty agent, task, and file paths",
|
|
315
|
+
threadKey,
|
|
316
|
+
);
|
|
317
|
+
}
|
|
318
|
+
if (agentName !== undefined && threadKey !== undefined) {
|
|
319
|
+
return failQueue(
|
|
320
|
+
agentName,
|
|
321
|
+
"Subagent input requires exactly one of agent or thread, not both",
|
|
322
|
+
threadKey,
|
|
323
|
+
);
|
|
315
324
|
}
|
|
316
325
|
|
|
317
|
-
|
|
326
|
+
const onUpdateDetails = (details: SubagentDetails) =>
|
|
327
|
+
onUpdate?.({
|
|
328
|
+
content: [
|
|
329
|
+
{
|
|
330
|
+
type: "text",
|
|
331
|
+
text: details.currentActivity ?? details.response ?? `${details.agent}: ${details.status}`,
|
|
332
|
+
},
|
|
333
|
+
],
|
|
334
|
+
details,
|
|
335
|
+
});
|
|
336
|
+
|
|
337
|
+
if (threadKey !== undefined) {
|
|
318
338
|
const disabledContinuation = await disabledContinuationResult(
|
|
319
339
|
ctx,
|
|
320
340
|
threadKey,
|
|
@@ -323,29 +343,36 @@ Delegate one focused task per call. Children do not inherit parent messages. Inc
|
|
|
323
343
|
parentThinking,
|
|
324
344
|
);
|
|
325
345
|
if (disabledContinuation) return disabledContinuation;
|
|
346
|
+
return runtime.execute({
|
|
347
|
+
agent: threadKey,
|
|
348
|
+
task,
|
|
349
|
+
files,
|
|
350
|
+
continuing: true,
|
|
351
|
+
threadKey,
|
|
352
|
+
ctx,
|
|
353
|
+
parentModel,
|
|
354
|
+
parentThinking,
|
|
355
|
+
signal,
|
|
356
|
+
onUpdate: onUpdateDetails,
|
|
357
|
+
resolveFreshDefinition: () => resolveFreshSubagentDefinition(ctx, threadKey),
|
|
358
|
+
});
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
if (agentName === undefined) {
|
|
362
|
+
return failQueue("", "Subagent input requires exactly one of agent or thread");
|
|
326
363
|
}
|
|
327
364
|
|
|
328
365
|
return runtime.execute({
|
|
329
|
-
agent,
|
|
366
|
+
agent: agentName,
|
|
330
367
|
task,
|
|
331
368
|
files,
|
|
332
|
-
continuing,
|
|
333
|
-
threadKey,
|
|
369
|
+
continuing: false,
|
|
334
370
|
ctx,
|
|
335
371
|
parentModel,
|
|
336
372
|
parentThinking,
|
|
337
373
|
signal,
|
|
338
|
-
onUpdate:
|
|
339
|
-
|
|
340
|
-
content: [
|
|
341
|
-
{
|
|
342
|
-
type: "text",
|
|
343
|
-
text: details.currentActivity ?? details.response ?? `${details.agent}: ${details.status}`,
|
|
344
|
-
},
|
|
345
|
-
],
|
|
346
|
-
details,
|
|
347
|
-
}),
|
|
348
|
-
resolveFreshDefinition: () => resolveFreshSubagentDefinition(ctx, agent),
|
|
374
|
+
onUpdate: onUpdateDetails,
|
|
375
|
+
resolveFreshDefinition: () => resolveFreshSubagentDefinition(ctx, agentName),
|
|
349
376
|
});
|
|
350
377
|
},
|
|
351
378
|
renderCall(args, theme, context) {
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
Reviews agent `bash` and `script_runner` requests before they run.
|
|
4
4
|
|
|
5
|
-
Common read-only bash commands skip review and run immediately. Other bash and every `script_runner` request go to a
|
|
5
|
+
Common read-only bash commands skip review and run immediately. Other bash and every `script_runner` request go to a separate reviewer. Tau uses the reviewer model for the current provider, then the current chat model if that reviewer is unavailable or fails. If a reviewer model is unavailable or fails, Tau notifies and tries the next one. The reviewer returns a validated decision and one concise paragraph that explains the request.
|
|
6
6
|
|
|
7
|
-
With `autoApprove` enabled, reviewer-approved requests run without another confirmation. Tau shows a user-only marker after those auto-approvals. Common read-only bash that skips review does not get a marker. Routine local development work should be approved, including requests that modify project files or run scripts. The reviewer asks for human approval only when it finds a concrete destructive, system, production, privileged, or security-sensitive effect.
|
|
7
|
+
With `autoApprove` enabled, reviewer-approved requests run without another confirmation. Tau shows a user-only marker with the reviewer model after those auto-approvals. Common read-only bash that skips review does not get a marker. Routine local development work should be approved, including requests that modify project files or run scripts. The reviewer asks for human approval only when it finds a concrete destructive, system, production, privileged, or security-sensitive effect.
|
|
8
8
|
|
|
9
9
|
When approval is required, Tau shows one paragraph that explains the effect and risk without repeating the request. If the reviewer fails or returns a malformed decision, Tau asks for direct human approval instead of running it automatically. Tau also sends an attention notification when the approval window opens.
|
|
10
10
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Tool } from "@earendil-works/pi-ai";
|
|
1
|
+
import type { ThinkingLevel, Tool } from "@earendil-works/pi-ai";
|
|
2
2
|
import {
|
|
3
3
|
isToolCallEventType,
|
|
4
4
|
type ExtensionAPI,
|
|
@@ -6,11 +6,9 @@ import {
|
|
|
6
6
|
type ToolCallEvent,
|
|
7
7
|
} from "@earendil-works/pi-coding-agent";
|
|
8
8
|
import { Marker } from "@shanepadgett/tau-tui";
|
|
9
|
-
import { Type
|
|
10
|
-
import { Value } from "typebox/value";
|
|
9
|
+
import { Type } from "typebox";
|
|
11
10
|
import { emitAgentBlocked } from "../../shared/agent-blocked.ts";
|
|
12
|
-
import {
|
|
13
|
-
import { generateToolValidated } from "../../shared/model-fallback/index.ts";
|
|
11
|
+
import { generateToolValidated, resolveCandidates } from "../../shared/model-fallback/index.ts";
|
|
14
12
|
import { errorText, truncAt } from "../../shared/text.ts";
|
|
15
13
|
import { loadTauExtensionSettings } from "../../shared/settings/load.ts";
|
|
16
14
|
import { isAllowlistedBash } from "./allowlist.ts";
|
|
@@ -19,34 +17,22 @@ import toolApprovalSettings from "./settings.ts";
|
|
|
19
17
|
const STATUS_KEY = "tool-approval";
|
|
20
18
|
const AUTO_APPROVED_TYPE = "tau.tool-approval.auto-approved";
|
|
21
19
|
|
|
22
|
-
const
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
{
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
decision: Type.Literal("requires_user_approval"),
|
|
39
|
-
summary: SUMMARY_SCHEMA,
|
|
40
|
-
reason: Type.String({
|
|
41
|
-
minLength: 1,
|
|
42
|
-
maxLength: 300,
|
|
43
|
-
pattern: "^[^\\r\\n]+$",
|
|
44
|
-
description: "One concise paragraph that states the concrete high-impact risk requiring approval.",
|
|
45
|
-
}),
|
|
46
|
-
},
|
|
47
|
-
{ additionalProperties: false },
|
|
48
|
-
),
|
|
49
|
-
]);
|
|
20
|
+
const REVIEW_SCHEMA = Type.Object(
|
|
21
|
+
{
|
|
22
|
+
decision: Type.Union([Type.Literal("approved"), Type.Literal("requires_user_approval")]),
|
|
23
|
+
summary: Type.String({
|
|
24
|
+
minLength: 1,
|
|
25
|
+
maxLength: 600,
|
|
26
|
+
description: "One concise paragraph that fully explains what the tool request does.",
|
|
27
|
+
}),
|
|
28
|
+
reason: Type.String({
|
|
29
|
+
maxLength: 300,
|
|
30
|
+
description:
|
|
31
|
+
"Empty when approved. One concise paragraph naming the concrete risk when user approval is required.",
|
|
32
|
+
}),
|
|
33
|
+
},
|
|
34
|
+
{ additionalProperties: false },
|
|
35
|
+
);
|
|
50
36
|
|
|
51
37
|
const REVIEW_SYSTEM_PROMPT = [
|
|
52
38
|
"You are a tool-request safety reviewer.",
|
|
@@ -59,8 +45,8 @@ const REVIEW_SYSTEM_PROMPT = [
|
|
|
59
45
|
"Do not require approval merely because the request writes files, invokes code, uses shell composition, could fail, or has ordinary local side effects.",
|
|
60
46
|
"Routine deletion of generated, temporary, or local project files is ordinary local work. Escalate deletion only when it is broad or difficult to recover.",
|
|
61
47
|
"Default to approved. Uncertainty is not a reason to escalate; require user approval only when the request shows a concrete substantial risk listed above.",
|
|
62
|
-
"The summary must be one concise paragraph
|
|
63
|
-
"
|
|
48
|
+
"The summary must be one concise paragraph. Explain the complete effect of the request without lists, headings, or repeated details.",
|
|
49
|
+
"Always set reason. Use an empty string when approved. When user approval is required, give one concise reason naming the concrete risk without repeating the summary.",
|
|
64
50
|
].join("\n");
|
|
65
51
|
|
|
66
52
|
const REVIEW_TOOL = {
|
|
@@ -69,7 +55,18 @@ const REVIEW_TOOL = {
|
|
|
69
55
|
parameters: REVIEW_SCHEMA,
|
|
70
56
|
} satisfies Tool;
|
|
71
57
|
|
|
72
|
-
|
|
58
|
+
const REVIEW_MODELS: ReadonlyArray<{ provider: string; model: string; reasoning: ThinkingLevel }> = [
|
|
59
|
+
{ provider: "openai", model: "gpt-5.6-luna", reasoning: "medium" },
|
|
60
|
+
{ provider: "openai-codex", model: "gpt-5.6-luna", reasoning: "medium" },
|
|
61
|
+
{ provider: "anthropic", model: "claude-sonnet-5", reasoning: "medium" },
|
|
62
|
+
{ provider: "xai", model: "grok-4.5", reasoning: "low" },
|
|
63
|
+
{ provider: "openrouter", model: "deepseek/deepseek-v4.1-flash", reasoning: "low" },
|
|
64
|
+
{ provider: "opencode-go", model: "deepseek-v4.1-flash", reasoning: "low" },
|
|
65
|
+
];
|
|
66
|
+
|
|
67
|
+
type ToolReview =
|
|
68
|
+
| { decision: "approved"; summary: string }
|
|
69
|
+
| { decision: "requires_user_approval"; summary: string; reason: string };
|
|
73
70
|
type ApprovalToolName = "bash" | "script_runner";
|
|
74
71
|
|
|
75
72
|
interface ToolApprovalRequest {
|
|
@@ -79,6 +76,8 @@ interface ToolApprovalRequest {
|
|
|
79
76
|
|
|
80
77
|
interface AutoApprovedMarker {
|
|
81
78
|
toolName: ApprovalToolName;
|
|
79
|
+
provider: string;
|
|
80
|
+
model: string;
|
|
82
81
|
}
|
|
83
82
|
|
|
84
83
|
export default function toolApprovalExtension(pi: ExtensionAPI): void {
|
|
@@ -91,7 +90,7 @@ export default function toolApprovalExtension(pi: ExtensionAPI): void {
|
|
|
91
90
|
theme,
|
|
92
91
|
state: "complete",
|
|
93
92
|
label: "Auto-approved",
|
|
94
|
-
parts: [toolLabel(marker.toolName)],
|
|
93
|
+
parts: [toolLabel(marker.toolName), `${marker.provider}/${marker.model}`],
|
|
95
94
|
});
|
|
96
95
|
});
|
|
97
96
|
|
|
@@ -109,7 +108,7 @@ export default function toolApprovalExtension(pi: ExtensionAPI): void {
|
|
|
109
108
|
return {
|
|
110
109
|
systemPrompt: `${event.systemPrompt}\n\n${[
|
|
111
110
|
"Known-safe read-only bash commands skip review.",
|
|
112
|
-
"Other bash and every script_runner request are reviewed by a separate
|
|
111
|
+
"Other bash and every script_runner request are reviewed by a separate safety classifier before execution.",
|
|
113
112
|
"Treat classifier approval as a gate, not as permission to hide command intent from the user.",
|
|
114
113
|
"Routine local development requests can be approved automatically.",
|
|
115
114
|
"Requests with destructive, system, production, privileged, or security-sensitive effects require human confirmation.",
|
|
@@ -143,7 +142,7 @@ export default function toolApprovalExtension(pi: ExtensionAPI): void {
|
|
|
143
142
|
|
|
144
143
|
ctx.ui.setStatus(STATUS_KEY, `reviewing ${toolLabel(request.toolName)}`);
|
|
145
144
|
try {
|
|
146
|
-
const review = await reviewToolRequest(ctx, request);
|
|
145
|
+
const { review, provider, model } = await reviewToolRequest(ctx, request);
|
|
147
146
|
if (review.decision === "requires_user_approval") {
|
|
148
147
|
return requestToolApproval(
|
|
149
148
|
pi,
|
|
@@ -154,7 +153,11 @@ export default function toolApprovalExtension(pi: ExtensionAPI): void {
|
|
|
154
153
|
);
|
|
155
154
|
}
|
|
156
155
|
if (settings.autoApprove) {
|
|
157
|
-
pi.appendEntry<AutoApprovedMarker>(AUTO_APPROVED_TYPE, {
|
|
156
|
+
pi.appendEntry<AutoApprovedMarker>(AUTO_APPROVED_TYPE, {
|
|
157
|
+
toolName: request.toolName,
|
|
158
|
+
provider,
|
|
159
|
+
model,
|
|
160
|
+
});
|
|
158
161
|
return undefined;
|
|
159
162
|
}
|
|
160
163
|
return requestToolApproval(
|
|
@@ -198,26 +201,41 @@ function toolLabel(toolName: ApprovalToolName): string {
|
|
|
198
201
|
|
|
199
202
|
function autoApprovedMarker(value: unknown): AutoApprovedMarker | undefined {
|
|
200
203
|
if (!value || typeof value !== "object") return undefined;
|
|
201
|
-
const
|
|
202
|
-
if (toolName !== "bash" && toolName !== "script_runner") return undefined;
|
|
203
|
-
|
|
204
|
+
const record = value as AutoApprovedMarker;
|
|
205
|
+
if (record.toolName !== "bash" && record.toolName !== "script_runner") return undefined;
|
|
206
|
+
if (typeof record.provider !== "string" || record.provider.length === 0) return undefined;
|
|
207
|
+
if (typeof record.model !== "string" || record.model.length === 0) return undefined;
|
|
208
|
+
return { toolName: record.toolName, provider: record.provider, model: record.model };
|
|
204
209
|
}
|
|
205
210
|
|
|
206
|
-
async function reviewToolRequest(
|
|
211
|
+
async function reviewToolRequest(
|
|
212
|
+
ctx: ExtensionContext,
|
|
213
|
+
request: ToolApprovalRequest,
|
|
214
|
+
): Promise<{ review: ToolReview; provider: string; model: string }> {
|
|
207
215
|
const requestJson = JSON.stringify(request);
|
|
208
|
-
const
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
216
|
+
const reviewer = REVIEW_MODELS.find((item) => item.provider === ctx.model?.provider);
|
|
217
|
+
const preferred = reviewer ? [reviewer] : [];
|
|
218
|
+
if (ctx.model) {
|
|
219
|
+
preferred.push({
|
|
220
|
+
provider: ctx.model.provider,
|
|
221
|
+
model: ctx.model.id,
|
|
222
|
+
reasoning: "medium",
|
|
223
|
+
});
|
|
224
|
+
}
|
|
225
|
+
const candidates = await resolveCandidates(ctx, preferred, false);
|
|
226
|
+
const wanted = preferred[0];
|
|
227
|
+
if (
|
|
228
|
+
wanted &&
|
|
229
|
+
!candidates.some((item) => item.model.provider === wanted.provider && item.model.id === wanted.model)
|
|
230
|
+
) {
|
|
231
|
+
ctx.ui.notify(`Tool review skipped ${wanted.provider}/${wanted.model}; trying next model.`, "info");
|
|
232
|
+
}
|
|
233
|
+
const { value, candidate } = await generateToolValidated(
|
|
213
234
|
ctx,
|
|
214
235
|
candidates,
|
|
215
236
|
[REVIEW_SYSTEM_PROMPT, "", "Review this tool request JSON:", requestJson].join("\n"),
|
|
216
237
|
REVIEW_TOOL,
|
|
217
|
-
|
|
218
|
-
if (!Value.Check(REVIEW_SCHEMA, input)) throw new Error("quick reviewer returned an invalid review shape");
|
|
219
|
-
return input;
|
|
220
|
-
},
|
|
238
|
+
reviewFromToolInput,
|
|
221
239
|
(error, output) =>
|
|
222
240
|
[
|
|
223
241
|
`The tool review failed validation: ${error.message}`,
|
|
@@ -226,8 +244,25 @@ async function reviewToolRequest(ctx: ExtensionContext, request: ToolApprovalReq
|
|
|
226
244
|
"Previous response:",
|
|
227
245
|
output,
|
|
228
246
|
].join("\n"),
|
|
229
|
-
{ maxAttempts: 3 },
|
|
247
|
+
{ maxAttempts: 3, notifyOnFallback: true },
|
|
230
248
|
);
|
|
249
|
+
return { review: value, provider: candidate.model.provider, model: candidate.model.id };
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
function reviewFromToolInput(input: unknown): ToolReview {
|
|
253
|
+
if (!input || typeof input !== "object") throw new Error("reviewer returned an invalid review shape");
|
|
254
|
+
const record = input as Record<string, unknown>;
|
|
255
|
+
const decision = record.decision;
|
|
256
|
+
if (decision !== "approved" && decision !== "requires_user_approval") {
|
|
257
|
+
throw new Error("reviewer returned an invalid review shape");
|
|
258
|
+
}
|
|
259
|
+
if (typeof record.summary !== "string") throw new Error("reviewer returned an invalid review shape");
|
|
260
|
+
const summary = truncAt(singleLine(record.summary), 600);
|
|
261
|
+
if (!summary) throw new Error("reviewer returned an invalid review shape");
|
|
262
|
+
const reason = typeof record.reason === "string" ? truncAt(singleLine(record.reason), 300) : "";
|
|
263
|
+
if (decision === "approved") return { decision, summary };
|
|
264
|
+
if (!reason) throw new Error("reviewer returned an invalid review shape");
|
|
265
|
+
return { decision, summary, reason };
|
|
231
266
|
}
|
|
232
267
|
|
|
233
268
|
function formatApproval(summary: string, reason: string): string {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@shanepadgett/tau-agent",
|
|
3
|
-
"version": "0.41.
|
|
3
|
+
"version": "0.41.3",
|
|
4
4
|
"description": "Tau is a custom agentic harness built with pi extensions",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./src/index.ts",
|
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
],
|
|
36
36
|
"dependencies": {
|
|
37
37
|
"@ast-grep/wasm": "0.45.1",
|
|
38
|
-
"@shanepadgett/tau-tui": "0.41.
|
|
38
|
+
"@shanepadgett/tau-tui": "0.41.3",
|
|
39
39
|
"@vscode/tree-sitter-wasm": "0.3.1",
|
|
40
40
|
"image-size": "2.0.2",
|
|
41
41
|
"smol-toml": "1.8.0",
|
|
@@ -71,9 +71,10 @@ export async function generateValidated<T>(
|
|
|
71
71
|
correctionPrompt?: (error: Error, text: string) => string,
|
|
72
72
|
options?: ModelFallbackOptions,
|
|
73
73
|
): Promise<T> {
|
|
74
|
-
|
|
74
|
+
const { value } = await withModelFallback(ctx, candidates, options, (candidate) =>
|
|
75
75
|
requestValidated(ctx, candidate, prompt, validate, correctionPrompt),
|
|
76
76
|
);
|
|
77
|
+
return value;
|
|
77
78
|
}
|
|
78
79
|
|
|
79
80
|
export async function generateToolValidated<T>(
|
|
@@ -84,7 +85,7 @@ export async function generateToolValidated<T>(
|
|
|
84
85
|
validate: (input: unknown) => T,
|
|
85
86
|
correctionPrompt?: (error: Error, output: string) => string,
|
|
86
87
|
options?: ModelFallbackOptions,
|
|
87
|
-
): Promise<T> {
|
|
88
|
+
): Promise<{ value: T; candidate: ModelCandidate }> {
|
|
88
89
|
return withModelFallback(ctx, candidates, options, (candidate) =>
|
|
89
90
|
requestToolValidated(
|
|
90
91
|
ctx,
|
|
@@ -117,7 +118,7 @@ async function withModelFallback<T>(
|
|
|
117
118
|
candidates: readonly ModelCandidate[],
|
|
118
119
|
options: ModelFallbackOptions | undefined,
|
|
119
120
|
request: (candidate: ModelCandidate) => Promise<T>,
|
|
120
|
-
): Promise<T> {
|
|
121
|
+
): Promise<{ value: T; candidate: ModelCandidate }> {
|
|
121
122
|
const failures: string[] = [];
|
|
122
123
|
const statusKey = options?.statusKey;
|
|
123
124
|
|
|
@@ -126,7 +127,7 @@ async function withModelFallback<T>(
|
|
|
126
127
|
if (statusKey) ctx.ui.setStatus(statusKey, `generating (${label})`);
|
|
127
128
|
await options?.onStatus?.(`Generating with ${label}`);
|
|
128
129
|
try {
|
|
129
|
-
return await request(candidate);
|
|
130
|
+
return { value: await request(candidate), candidate };
|
|
130
131
|
} catch (error) {
|
|
131
132
|
if (ctx.signal?.aborted) throw new Error("Cancelled.");
|
|
132
133
|
if (shouldCooldownProvider(error)) await markProviderUnavailable(candidate.model.provider);
|