tau-coding-agent 0.1.6 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -12
- package/extensions/answer.ts +129 -79
- package/extensions/branch-term/README.md +7 -0
- package/extensions/{branch-term.ts → branch-term/index.ts} +113 -104
- package/extensions/btw.ts +17 -2
- package/extensions/caffeinate/README.md +5 -0
- package/extensions/caffeinate/index.ts +144 -0
- package/extensions/fast.ts +292 -0
- package/extensions/ghostty.ts +214 -211
- package/extensions/git-diff-stats.ts +124 -84
- package/extensions/git-pr-status.ts +274 -208
- package/extensions/insights.ts +109 -191
- package/extensions/loop.ts +149 -145
- package/extensions/memory.ts +138 -86
- package/extensions/notify.ts +14 -26
- package/extensions/openai-verbosity.ts +108 -42
- package/extensions/review/fix.ts +15 -6
- package/extensions/review/git.ts +93 -103
- package/extensions/review/index.ts +60 -62
- package/extensions/review/interrupt.ts +117 -0
- package/extensions/review/message-queue.ts +45 -12
- package/extensions/review/models.ts +79 -3
- package/extensions/review/prompts.ts +41 -38
- package/extensions/review/review.ts +161 -166
- package/extensions/review/runner.ts +139 -147
- package/extensions/review/runtime.ts +173 -120
- package/extensions/review/triage.ts +41 -44
- package/extensions/sandbox/bash.ts +775 -0
- package/extensions/sandbox/command.ts +605 -0
- package/extensions/sandbox/config.ts +764 -0
- package/extensions/sandbox/index.ts +132 -3338
- package/extensions/sandbox/permissions/dialog.ts +118 -0
- package/extensions/sandbox/permissions/filesystem.ts +559 -0
- package/extensions/sandbox/permissions/mach-lookup.ts +186 -0
- package/extensions/sandbox/permissions/network.ts +164 -0
- package/extensions/sandbox/permissions/unsandboxed.ts +270 -0
- package/extensions/sandbox/runtime.ts +616 -0
- package/extensions/stash.ts +28 -15
- package/extensions/subagent/README.md +73 -0
- package/extensions/subagent/index.ts +822 -0
- package/extensions/subagent/interrupt.ts +117 -0
- package/extensions/subagent/permissions.ts +101 -0
- package/extensions/subagent/rpc.ts +177 -0
- package/extensions/tool-display-mode.ts +267 -64
- package/extensions/usage/index.ts +249 -356
- package/extensions/usage/openrouter.ts +10 -2
- package/extensions/websearch/README.md +12 -47
- package/extensions/websearch/config.ts +5 -2
- package/extensions/websearch/index.ts +94 -109
- package/extensions/websearch/output.ts +51 -0
- package/extensions/websearch/providers/anthropic.pi.ts +27 -44
- package/extensions/websearch/providers/gemini.browser.ts +68 -77
- package/extensions/websearch/providers/gemini.pi.ts +17 -17
- package/extensions/websearch/providers/openai-codex.pi.ts +145 -7
- package/extensions/websearch/providers/pi-model.shared.ts +33 -39
- package/extensions/websearch/providers/shared.ts +87 -2
- package/extensions/websearch/types.ts +0 -3
- package/extensions/worktree.ts +132 -172
- package/package.json +10 -5
- package/skills/browser-tools/SKILL.md +29 -234
- package/skills/browser-tools/references/cookies.md +36 -0
- package/skills/browser-tools/references/interaction.md +90 -0
- package/skills/browser-tools/references/logging.md +34 -0
- package/skills/git-clean-history/SKILL.md +6 -6
- package/skills/git-commit/SKILL.md +5 -3
- package/skills/github-pull-request/SKILL.md +60 -0
- package/skills/github-pull-request/references/create.md +40 -0
- package/skills/github-pull-request/references/stewardship.md +60 -0
- package/skills/oracle/SKILL.md +3 -3
- package/skills/oracle/scripts/oracle +10 -1
- package/skills/sentry/SKILL.md +15 -185
- package/skills/sentry/references/events.md +79 -0
- package/skills/sentry/references/issues.md +62 -0
- package/skills/sentry/references/logs.md +46 -0
- package/skills/update-changelog/SKILL.md +27 -121
- package/skills/web-design/SKILL.md +16 -105
- package/themes/tau-dark.json +4 -0
- package/extensions/openai-fast.ts +0 -229
- package/extensions/websearch/providers/openai-codex.browser.ts +0 -77
- package/extensions/websearch/providers/openai-codex.shared.ts +0 -123
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
import {
|
|
3
|
+
getKeybindings,
|
|
4
|
+
isKeyRelease,
|
|
5
|
+
type TuiMainScreen,
|
|
6
|
+
type Component,
|
|
7
|
+
type TUI,
|
|
8
|
+
} from "@earendil-works/pi-tui";
|
|
9
|
+
|
|
10
|
+
interface InterruptRequest {
|
|
11
|
+
sessionKey: string;
|
|
12
|
+
cancel?: boolean;
|
|
13
|
+
active: boolean;
|
|
14
|
+
confirming?: boolean;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** Each extension owns its guard; optional peers contribute work through the event bus. */
|
|
18
|
+
export function createInterruptGuard(
|
|
19
|
+
pi: ExtensionAPI,
|
|
20
|
+
ctx: ExtensionContext,
|
|
21
|
+
tui: TUI & Partial<Pick<TuiMainScreen, "getFocusedComponent">>,
|
|
22
|
+
isWorking: () => boolean,
|
|
23
|
+
cancel: () => void,
|
|
24
|
+
isPromptActive: () => boolean,
|
|
25
|
+
) {
|
|
26
|
+
const sessionKey =
|
|
27
|
+
ctx.sessionManager.getSessionFile() ?? `session:${ctx.sessionManager.getSessionId()}`;
|
|
28
|
+
let closed = false;
|
|
29
|
+
let pending:
|
|
30
|
+
| { controller: AbortController; editor: Component; completion: Promise<void> }
|
|
31
|
+
| undefined;
|
|
32
|
+
const unsubscribeWork = pi.events.on("tau:interrupt", (data) => {
|
|
33
|
+
const request = data as InterruptRequest;
|
|
34
|
+
if (request.sessionKey !== sessionKey || closed) return;
|
|
35
|
+
if (pending) request.confirming = true;
|
|
36
|
+
if (!isWorking()) return;
|
|
37
|
+
request.active = true;
|
|
38
|
+
if (request.cancel) cancel();
|
|
39
|
+
});
|
|
40
|
+
const unsubscribeChanged = pi.events.on("tau:work-changed", (data) => {
|
|
41
|
+
if ((data as { sessionKey: string }).sessionKey === sessionKey && pending && !hasWork())
|
|
42
|
+
pending.controller.abort();
|
|
43
|
+
});
|
|
44
|
+
const unsubscribeInput = ctx.ui.onTerminalInput((data) => {
|
|
45
|
+
if (isKeyRelease(data) || !getKeybindings().matches(data, "app.interrupt")) return;
|
|
46
|
+
const editor = tui.getFocusedComponent?.();
|
|
47
|
+
// A dismissed native dialog restores focus before its promise settles. Swallow a rapid
|
|
48
|
+
// extra Escape in that gap instead of accidentally interrupting the parent.
|
|
49
|
+
if (pending) return editor === pending.editor ? { consume: true } : undefined;
|
|
50
|
+
if (
|
|
51
|
+
closed ||
|
|
52
|
+
isPromptActive() ||
|
|
53
|
+
!editor ||
|
|
54
|
+
!("getText" in editor) ||
|
|
55
|
+
!("setText" in editor) ||
|
|
56
|
+
("isShowingAutocomplete" in editor &&
|
|
57
|
+
typeof editor.isShowingAutocomplete === "function" &&
|
|
58
|
+
editor.isShowingAutocomplete()) ||
|
|
59
|
+
!hasWork()
|
|
60
|
+
)
|
|
61
|
+
return;
|
|
62
|
+
const current = {
|
|
63
|
+
controller: new AbortController(),
|
|
64
|
+
editor,
|
|
65
|
+
completion: Promise.resolve(),
|
|
66
|
+
};
|
|
67
|
+
pending = current;
|
|
68
|
+
current.completion = (async () => {
|
|
69
|
+
try {
|
|
70
|
+
const request: {
|
|
71
|
+
run: (signal: AbortSignal) => Promise<void>;
|
|
72
|
+
signal: AbortSignal;
|
|
73
|
+
result?: Promise<unknown>;
|
|
74
|
+
} = {
|
|
75
|
+
signal: current.controller.signal,
|
|
76
|
+
async run(signal) {
|
|
77
|
+
const confirmed = await ctx.ui.confirm("Cancel all ongoing work?", "", { signal });
|
|
78
|
+
if (!confirmed || closed || signal.aborted || !hasWork()) return;
|
|
79
|
+
pi.events.emit("tau:interrupt", { sessionKey, active: false, cancel: true });
|
|
80
|
+
// Reuse the editor's native interrupt, including queue restoration, Bash and retries.
|
|
81
|
+
editor.handleInput?.(data);
|
|
82
|
+
},
|
|
83
|
+
};
|
|
84
|
+
// Keep queued approvals blocked until the confirmed cancellation has been applied.
|
|
85
|
+
pi.events.emit("subagent:permission", request);
|
|
86
|
+
await (request.result ?? request.run(request.signal));
|
|
87
|
+
} catch (error) {
|
|
88
|
+
if (!closed && !current.controller.signal.aborted)
|
|
89
|
+
ctx.ui.notify(`Could not confirm cancellation: ${String(error)}`, "warning");
|
|
90
|
+
} finally {
|
|
91
|
+
if (pending === current) pending = undefined;
|
|
92
|
+
}
|
|
93
|
+
})();
|
|
94
|
+
return { consume: true };
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
refresh() {
|
|
99
|
+
if (!closed) pi.events.emit("tau:work-changed", { sessionKey });
|
|
100
|
+
},
|
|
101
|
+
async dispose() {
|
|
102
|
+
closed = true;
|
|
103
|
+
unsubscribeInput();
|
|
104
|
+
unsubscribeWork();
|
|
105
|
+
unsubscribeChanged();
|
|
106
|
+
pending?.controller.abort();
|
|
107
|
+
pi.events.emit("tau:work-changed", { sessionKey });
|
|
108
|
+
await pending?.completion;
|
|
109
|
+
},
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
function hasWork(): boolean {
|
|
113
|
+
const request: InterruptRequest = { sessionKey, active: false };
|
|
114
|
+
pi.events.emit("tau:interrupt", request);
|
|
115
|
+
return request.active;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
@@ -3,6 +3,7 @@ import {
|
|
|
3
3
|
type ExtensionAPI,
|
|
4
4
|
type ExtensionContext,
|
|
5
5
|
type InputEvent,
|
|
6
|
+
type InputEventResult,
|
|
6
7
|
} from "@earendil-works/pi-coding-agent";
|
|
7
8
|
import { getKeybindings } from "@earendil-works/pi-tui";
|
|
8
9
|
|
|
@@ -20,6 +21,7 @@ type QueuedReviewMessage = {
|
|
|
20
21
|
};
|
|
21
22
|
type QueueState = {
|
|
22
23
|
active: boolean;
|
|
24
|
+
interrupted: boolean;
|
|
23
25
|
messages: QueuedReviewMessage[];
|
|
24
26
|
unsubscribeFollowUpShortcut?: () => void;
|
|
25
27
|
};
|
|
@@ -28,15 +30,27 @@ export type ReviewMessageQueue = ReturnType<typeof createReviewMessageQueue>;
|
|
|
28
30
|
|
|
29
31
|
export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
30
32
|
const states = new Map<string, QueueState>();
|
|
33
|
+
let promptActive = false;
|
|
34
|
+
pi.on("ui_prompt_start", () => {
|
|
35
|
+
promptActive = true;
|
|
36
|
+
});
|
|
37
|
+
pi.on("ui_prompt_end", () => {
|
|
38
|
+
promptActive = false;
|
|
39
|
+
});
|
|
31
40
|
|
|
32
41
|
function start(ctx: ExtensionContext): () => void {
|
|
33
42
|
const sessionKey = getReviewSessionKey(ctx);
|
|
34
43
|
const state = getState(sessionKey);
|
|
35
44
|
state.active = true;
|
|
45
|
+
state.interrupted = false;
|
|
36
46
|
|
|
37
47
|
if (ctx.hasUI && !state.unsubscribeFollowUpShortcut) {
|
|
38
48
|
state.unsubscribeFollowUpShortcut = ctx.ui.onTerminalInput((data) => {
|
|
39
|
-
if (!isActive(sessionKey)
|
|
49
|
+
if (!isActive(sessionKey) || promptActive || matchesConfiguredKey(data, "app.interrupt"))
|
|
50
|
+
return undefined;
|
|
51
|
+
const interrupt = { sessionKey, active: false, confirming: false };
|
|
52
|
+
pi.events.emit("tau:interrupt", interrupt);
|
|
53
|
+
if (interrupt.confirming) return undefined;
|
|
40
54
|
|
|
41
55
|
if (matchesConfiguredKey(data, "app.message.dequeue")) {
|
|
42
56
|
return restoreMessagesToEditor(ctx) ? { consume: true } : undefined;
|
|
@@ -64,24 +78,40 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
|
64
78
|
};
|
|
65
79
|
}
|
|
66
80
|
|
|
67
|
-
function handleInput(event: InputEvent, ctx: ExtensionContext):
|
|
81
|
+
function handleInput(event: InputEvent, ctx: ExtensionContext): InputEventResult | undefined {
|
|
82
|
+
if (event.source === "extension" || isImmediateCommand(event.text)) return;
|
|
83
|
+
if (!event.text.trim() && !event.images?.length) return;
|
|
68
84
|
const sessionKey = getReviewSessionKey(ctx);
|
|
69
|
-
|
|
70
|
-
if (
|
|
85
|
+
const state = states.get(sessionKey);
|
|
86
|
+
if (state?.interrupted) {
|
|
87
|
+
const messages = state.messages;
|
|
88
|
+
state.messages = [];
|
|
89
|
+
state.interrupted = false;
|
|
90
|
+
render(ctx);
|
|
91
|
+
return {
|
|
92
|
+
action: "transform",
|
|
93
|
+
text: [...messages.map((message) => message.text), event.text].filter(Boolean).join("\n\n"),
|
|
94
|
+
images: [...messages.flatMap((message) => message.images ?? []), ...(event.images ?? [])],
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
if (!isActive(sessionKey)) return;
|
|
71
98
|
|
|
72
99
|
// When the main agent is already streaming, Pi's built-in steering/follow-up
|
|
73
100
|
// queues are available. Only provide the review-owned queue while review work
|
|
74
101
|
// is running in the background and the main session is idle.
|
|
75
|
-
if (event.streamingBehavior) return
|
|
76
|
-
if (!ctx.isIdle()) return false;
|
|
77
|
-
if (isImmediateCommand(event.text)) return false;
|
|
78
|
-
if (!event.text.trim() && !event.images?.length) return false;
|
|
102
|
+
if (event.streamingBehavior || !ctx.isIdle()) return;
|
|
79
103
|
|
|
80
104
|
queueMessage(ctx, "steer", {
|
|
81
105
|
text: event.text,
|
|
82
106
|
images: event.images?.length ? [...event.images] : undefined,
|
|
83
107
|
});
|
|
84
|
-
return
|
|
108
|
+
return { action: "handled" };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function retain(ctx: ExtensionContext): void {
|
|
112
|
+
const state = states.get(getReviewSessionKey(ctx));
|
|
113
|
+
if (state) state.interrupted = true;
|
|
114
|
+
render(ctx);
|
|
85
115
|
}
|
|
86
116
|
|
|
87
117
|
function flushSteering(ctx: ExtensionContext, options: FlushOptions = {}): boolean {
|
|
@@ -142,7 +172,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
|
142
172
|
): boolean {
|
|
143
173
|
const sessionKey = getReviewSessionKey(ctx);
|
|
144
174
|
const state = states.get(sessionKey);
|
|
145
|
-
if (!state?.messages.length) return false;
|
|
175
|
+
if (!state?.messages.length || state.interrupted) return false;
|
|
146
176
|
|
|
147
177
|
const selected: QueuedReviewMessage[] = [];
|
|
148
178
|
const remaining: QueuedReviewMessage[] = [];
|
|
@@ -173,7 +203,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
|
173
203
|
function getState(sessionKey: string): QueueState {
|
|
174
204
|
let state = states.get(sessionKey);
|
|
175
205
|
if (!state) {
|
|
176
|
-
state = { active: false, messages: [] };
|
|
206
|
+
state = { active: false, interrupted: false, messages: [] };
|
|
177
207
|
states.set(sessionKey, state);
|
|
178
208
|
}
|
|
179
209
|
return state;
|
|
@@ -199,7 +229,9 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
|
199
229
|
lines.push(
|
|
200
230
|
ctx.ui.theme.fg(
|
|
201
231
|
"dim",
|
|
202
|
-
|
|
232
|
+
states.get(getReviewSessionKey(ctx))?.interrupted
|
|
233
|
+
? "Queued input will be included with your next message"
|
|
234
|
+
: `↳ ${appKeyDisplay("app.message.dequeue")} to edit all queued messages`,
|
|
203
235
|
),
|
|
204
236
|
);
|
|
205
237
|
ctx.ui.setWidget(WIDGET_KEY, lines);
|
|
@@ -210,6 +242,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
|
|
|
210
242
|
handleInput,
|
|
211
243
|
flushSteering,
|
|
212
244
|
flushAll,
|
|
245
|
+
retain,
|
|
213
246
|
clear,
|
|
214
247
|
};
|
|
215
248
|
}
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import { getSupportedThinkingLevels, type Api, type Model } from "@earendil-works/pi-ai";
|
|
2
2
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
3
3
|
|
|
4
|
-
|
|
4
|
+
import { REVIEW_TASK_TIMEOUT_MS } from "./runner.js";
|
|
5
|
+
|
|
6
|
+
const OPENAI_FAST_MODEL_ID = "gpt-5.6-luna";
|
|
5
7
|
const ANTHROPIC_FAST_MODEL_ID = "claude-haiku-4-5";
|
|
6
8
|
|
|
7
9
|
type ModelFamily = "openai" | "anthropic";
|
|
@@ -505,15 +507,89 @@ function resolveUnqualifiedModelPattern(
|
|
|
505
507
|
});
|
|
506
508
|
}
|
|
507
509
|
|
|
510
|
+
function getRequestedProvider(
|
|
511
|
+
modelPattern: string,
|
|
512
|
+
availableModels: Array<Model<Api>>,
|
|
513
|
+
modelRegistry: ExtensionContext["modelRegistry"],
|
|
514
|
+
): string | undefined {
|
|
515
|
+
const { basePattern } = splitModelPatternThinkingSuffix(modelPattern);
|
|
516
|
+
const slash = basePattern.indexOf("/");
|
|
517
|
+
if (slash <= 0) return undefined;
|
|
518
|
+
|
|
519
|
+
const providerPrefix = basePattern.slice(0, slash);
|
|
520
|
+
const provider = modelRegistry
|
|
521
|
+
.getRegisteredProviderIds()
|
|
522
|
+
.find((candidate) => candidate.toLowerCase() === providerPrefix.toLowerCase());
|
|
523
|
+
if (!provider) return undefined;
|
|
524
|
+
if (shouldTreatAsExplicitProviderPattern(basePattern, availableModels, modelRegistry)) {
|
|
525
|
+
return provider;
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
const hasRawModelMatch = availableModels.some(
|
|
529
|
+
(model) => model.id.toLowerCase() === basePattern.toLowerCase(),
|
|
530
|
+
);
|
|
531
|
+
return hasRawModelMatch ? undefined : provider;
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
async function refreshModelCatalog(
|
|
535
|
+
ctx: ExtensionContext,
|
|
536
|
+
requestedModels: string[],
|
|
537
|
+
signal: AbortSignal,
|
|
538
|
+
): Promise<Array<Model<Api>>> {
|
|
539
|
+
signal.throwIfAborted();
|
|
540
|
+
const currentModels = ctx.modelRegistry.getAll();
|
|
541
|
+
const requestedProviders = new Set(
|
|
542
|
+
requestedModels
|
|
543
|
+
.map((pattern) => getRequestedProvider(pattern, currentModels, ctx.modelRegistry))
|
|
544
|
+
.filter((provider): provider is string => Boolean(provider)),
|
|
545
|
+
);
|
|
546
|
+
|
|
547
|
+
if (requestedProviders.size > 0) {
|
|
548
|
+
const result = await ctx.modelRegistry.refresh({
|
|
549
|
+
providers: [...requestedProviders],
|
|
550
|
+
signal: AbortSignal.any([signal, AbortSignal.timeout(REVIEW_TASK_TIMEOUT_MS)]),
|
|
551
|
+
});
|
|
552
|
+
signal.throwIfAborted();
|
|
553
|
+
const failedProvider = [...requestedProviders].find((provider) =>
|
|
554
|
+
[...result.errors.keys()].some(
|
|
555
|
+
(candidate) => candidate.toLowerCase() === provider.toLowerCase(),
|
|
556
|
+
),
|
|
557
|
+
);
|
|
558
|
+
if (result.aborted || failedProvider) {
|
|
559
|
+
const provider =
|
|
560
|
+
failedProvider ?? (requestedProviders.size === 1 ? [...requestedProviders][0] : undefined);
|
|
561
|
+
throw new Error(
|
|
562
|
+
provider
|
|
563
|
+
? `Could not refresh requested provider "${provider}".`
|
|
564
|
+
: "Could not refresh all requested providers.",
|
|
565
|
+
);
|
|
566
|
+
}
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
const hasUnqualifiedRequest =
|
|
570
|
+
requestedModels.length === 0 ||
|
|
571
|
+
requestedModels.some(
|
|
572
|
+
(pattern) => !getRequestedProvider(pattern, currentModels, ctx.modelRegistry),
|
|
573
|
+
);
|
|
574
|
+
if (hasUnqualifiedRequest) {
|
|
575
|
+
await ctx.modelRegistry.refresh({
|
|
576
|
+
signal: AbortSignal.any([signal, AbortSignal.timeout(REVIEW_TASK_TIMEOUT_MS)]),
|
|
577
|
+
});
|
|
578
|
+
signal.throwIfAborted();
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
return ctx.modelRegistry.getAll();
|
|
582
|
+
}
|
|
583
|
+
|
|
508
584
|
export async function resolveModels(
|
|
509
585
|
ctx: ExtensionContext,
|
|
510
586
|
requestedModels: string[],
|
|
511
587
|
currentThinkingLevel: ReviewThinkingLevel,
|
|
588
|
+
signal: AbortSignal,
|
|
512
589
|
): Promise<ResolvedReviewModel[]> {
|
|
513
|
-
await ctx
|
|
590
|
+
const allModels = await refreshModelCatalog(ctx, requestedModels, signal);
|
|
514
591
|
const currentProvider = typeof ctx.model?.provider === "string" ? ctx.model.provider : undefined;
|
|
515
592
|
const currentModelId = ctx.model?.id;
|
|
516
|
-
const allModels = ctx.modelRegistry.getAll();
|
|
517
593
|
|
|
518
594
|
const resolveRequestedModel = (modelPattern: string): ResolvedReviewModel => {
|
|
519
595
|
const { basePattern, thinkingSuffix } = splitModelPatternThinkingSuffix(modelPattern);
|
|
@@ -4,9 +4,7 @@ type FocusDefinition = { suffix: string; qualifier: string; context: string };
|
|
|
4
4
|
|
|
5
5
|
export const REVIEW_RUBRIC_PROMPT = `# Review Guidelines
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
Below are default guidelines for determining what to flag. These are not the final word — if you encounter more specific guidelines elsewhere (in a developer message, user message, file, or project review guidelines appended below), those override these general instructions.
|
|
7
|
+
These are default scope, finding, and severity guidelines for every review focus. More specific custom instructions, user context, or project review guidelines override these defaults, including any limits on follow-up findings.
|
|
10
8
|
|
|
11
9
|
## Determining what to flag
|
|
12
10
|
|
|
@@ -14,14 +12,9 @@ Flag issues that:
|
|
|
14
12
|
1. Meaningfully impact the accuracy, performance, security, or maintainability of the code.
|
|
15
13
|
2. Are discrete and actionable (not general issues or multiple combined issues).
|
|
16
14
|
3. Don't demand rigor inconsistent with the rest of the codebase.
|
|
17
|
-
4.
|
|
18
|
-
5.
|
|
19
|
-
6.
|
|
20
|
-
7. Are clearly not intentional changes by the author.
|
|
21
|
-
8. Call out newly added dependencies explicitly and explain why they're needed.
|
|
22
|
-
9. Apply system-level thinking; flag changes that increase operational risk or on-call burden.
|
|
23
|
-
|
|
24
|
-
If an issue is valid and worth tracking but out of scope for the reviewed change, pre-existing, or merely adjacent, report it only as P3 and clearly frame it as follow-up work. Omit unrelated issues that are speculative, vague, or not worth tracking.
|
|
15
|
+
4. The author would likely fix if aware of them, either in scope or as an explicitly allowed follow-up.
|
|
16
|
+
5. Have provable impact. Identify the affected code and concrete consequences using evidence from the repository or diff. Omit speculative, vague, or purely stylistic issues.
|
|
17
|
+
6. Are clearly not intentional behavior.
|
|
25
18
|
|
|
26
19
|
## Finding field guidelines
|
|
27
20
|
|
|
@@ -35,44 +28,50 @@ If an issue is valid and worth tracking but out of scope for the reviewed change
|
|
|
35
28
|
- P0: critical/blocking.
|
|
36
29
|
- P1: urgent.
|
|
37
30
|
- P2: normal.
|
|
38
|
-
- P3: low/nice-to-have
|
|
31
|
+
- P3: low/nice-to-have.`;
|
|
39
32
|
|
|
40
|
-
|
|
33
|
+
export const REVIEW_DIFF_SCOPE_PROMPT = `In diff reviews, assess issues introduced by the scoped changes and related to their original intent at their actual severity. A concrete, useful issue that is pre-existing, out of scope, or merely adjacent may be reported only as P3 and must be explicitly framed as follow-up work. Do not expand the review into an unrelated audit or opportunistic refactoring.`;
|
|
34
|
+
|
|
35
|
+
export const REVIEW_SNAPSHOT_SCOPE_PROMPT = `In snapshot reviews, assess existing issues in the selected paths at their actual severity. Do not downgrade an issue to P3 merely because it is pre-existing. Keep findings within the selected paths; other code may be inspected as supporting context.`;
|
|
41
36
|
|
|
42
37
|
export const REVIEW_FOCUSES: Record<ReviewFocus, FocusDefinition> = {
|
|
43
38
|
general: {
|
|
44
39
|
suffix: "",
|
|
45
40
|
qualifier: "",
|
|
46
|
-
context:
|
|
41
|
+
context: `Review the scoped code for issues that meaningfully affect accuracy, performance, security, or maintainability.
|
|
42
|
+
|
|
43
|
+
Additional checks:
|
|
44
|
+
1. Examine dependencies explicitly and explain why they're needed when flagging a dependency issue.
|
|
45
|
+
2. Apply system-level thinking; flag code that increases operational risk or on-call burden.`,
|
|
47
46
|
},
|
|
48
47
|
security: {
|
|
49
48
|
suffix: " specializing in security analysis",
|
|
50
49
|
qualifier: " security",
|
|
51
|
-
context: `Review the
|
|
52
|
-
1. Auth and permissions:
|
|
50
|
+
context: `Review the scoped code for potential security issues, such as:
|
|
51
|
+
1. Auth and permissions: routes, commands, jobs, or data access must preserve required authentication, authorization, tenant isolation, and ownership checks.
|
|
53
52
|
2. Untrusted input: SQL or command construction must be parameterized; path, URL, shell, and HTML output must be escaped or encoded for the target context.
|
|
54
53
|
3. Filesystem and process boundaries: user-controlled paths and process arguments must not allow traversal, arbitrary file access, command injection, or unsafe environment changes.
|
|
55
54
|
4. Server-side fetches: server requests to user-controlled URLs must block localhost, private/link-local IP ranges, cloud metadata endpoints, and internal hostnames, including after DNS resolution and redirects.
|
|
56
55
|
5. Redirects and navigation: user-controlled destinations must be same-origin relative paths or explicitly allowlisted origins.
|
|
57
|
-
6. Secrets:
|
|
56
|
+
6. Secrets: logging, errors, telemetry, files, or API responses must not expose tokens, keys, credentials, cookies, or sensitive identifiers.
|
|
58
57
|
7. Serialization and parsing: avoid unsafe deserialization, dynamic code execution, prototype pollution, XML external entities, YAML custom object construction, and parser modes that load external resources.
|
|
59
|
-
8. Dependencies:
|
|
60
|
-
Only flag issues with a concrete exploit path or trust-boundary failure
|
|
58
|
+
8. Dependencies: dependencies that touch input parsing, networking, auth, crypto, secrets, or code execution need an explicit security reason.
|
|
59
|
+
Only flag issues with a concrete exploit path or trust-boundary failure.`,
|
|
61
60
|
},
|
|
62
61
|
reuse: {
|
|
63
62
|
suffix: " specializing in reuse analysis",
|
|
64
63
|
qualifier: " reuse",
|
|
65
|
-
context: `Review the
|
|
66
|
-
1. Search for existing capabilities that could replace
|
|
67
|
-
2. Flag
|
|
64
|
+
context: `Review the scoped code for potential reuse issues, such as:
|
|
65
|
+
1. Search for existing capabilities that could replace custom code: standard library APIs, native platform features, already-installed dependencies, and existing utilities/helpers. Search for relevant names and behavior, then go beyond string matches by inspecting adjacent files, utility files and directories, and shared modules.
|
|
66
|
+
2. Flag functions that duplicate existing functionality. Suggest the existing function, API, or feature to use instead.
|
|
68
67
|
3. Flag any inline logic that could use an existing capability — hand-rolled standard-library behavior, string manipulation, manual path handling, custom environment checks, ad-hoc type guards, native platform features, and similar patterns are common candidates.
|
|
69
|
-
4. Flag
|
|
68
|
+
4. Flag dependencies when the standard library, runtime/platform, or an already-installed dependency provides the same capability or behavior.
|
|
70
69
|
5. Flag duplicate modules, thin pass-through wrappers, and manual registries when they duplicate an existing source of truth or local pattern. Prefer deleting, consolidating, or reusing the existing path.`,
|
|
71
70
|
},
|
|
72
71
|
quality: {
|
|
73
72
|
suffix: " specializing in quality analysis",
|
|
74
73
|
qualifier: " quality",
|
|
75
|
-
context: `Review the
|
|
74
|
+
context: `Review the scoped code for potential quality issues, such as:
|
|
76
75
|
1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls.
|
|
77
76
|
2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones.
|
|
78
77
|
3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction.
|
|
@@ -89,9 +88,9 @@ Only flag issues with a concrete exploit path or trust-boundary failure introduc
|
|
|
89
88
|
testing: {
|
|
90
89
|
suffix: " specializing in test analysis",
|
|
91
90
|
qualifier: " testing",
|
|
92
|
-
context: `Review the
|
|
91
|
+
context: `Review the scoped code for potential testing issues, such as:
|
|
93
92
|
1. High-signal suite: favor a smaller test suite over exhaustive coverage. Treat tests as carrying maintenance cost. Each test should protect important behavior, a realistic failure mode, or a stable shared contract.
|
|
94
|
-
2. Low-value coverage: flag tests
|
|
93
|
+
2. Low-value coverage: flag tests that only cover implementation trivia. Examples include trivial getters/wrappers/constants, exact internal formatting, incidental telemetry/log details or events, timer internals, framework wiring with no behavior of its own, synthetic edge cases with no realistic breakage story, or behavior already covered by a higher-value test.
|
|
95
94
|
3. Test bloat: redundant cases, copy-paste matrices, excessive or repeated setup that should use or extract a fixture/helper, gratuitous snapshots, or unparameterized variations that increase maintenance cost without clear regression signal. Suggest consolidation or deletion in these cases.
|
|
96
95
|
4. Missing coverage: important behavior that can break without a test failing. Only ask for new tests when you can name the public/user-visible contract, security/privacy boundary, data-loss risk, serialization/wire contract, state transition, permission check, concurrency issue, or prior regression being protected.
|
|
97
96
|
5. Weak assertions: tests that do not check observable behavior or invariants.
|
|
@@ -103,10 +102,10 @@ Do not ask for tests just because code changed. Only flag a missing test when yo
|
|
|
103
102
|
efficiency: {
|
|
104
103
|
suffix: " specializing in efficiency analysis",
|
|
105
104
|
qualifier: " efficiency",
|
|
106
|
-
context: `Review the
|
|
105
|
+
context: `Review the scoped code for potential efficiency issues, such as:
|
|
107
106
|
1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns.
|
|
108
107
|
2. Missed concurrency: independent operations run sequentially when they could run in parallel.
|
|
109
|
-
3. Hot-path bloat:
|
|
108
|
+
3. Hot-path bloat: blocking work in startup or per-request/per-render hot paths.
|
|
110
109
|
4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error.
|
|
111
110
|
5. Memory: unbounded data structures, missing cleanup, event listener leaks.
|
|
112
111
|
6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one.
|
|
@@ -143,23 +142,27 @@ export const SUBMIT_TOOL_RETRY_PROMPT = `You did not call {SUBMIT_TOOL} as instr
|
|
|
143
142
|
export const REVIEW_OUTPUT_CONTRACT_PROMPT = `Requirements:
|
|
144
143
|
- Never output findings as text or write them to files.
|
|
145
144
|
- Always call submit_review exactly once as your final action.
|
|
146
|
-
- If no issues are found, pass an empty array of findings to submit_review
|
|
147
|
-
- Omit uncertain or speculative findings.`;
|
|
145
|
+
- If no issues are found, pass an empty array of findings to submit_review.`;
|
|
148
146
|
|
|
149
147
|
export const REVIEW_FOCUS_PROMPT = `You are an expert code reviewer{FOCUS_SUFFIX}.
|
|
150
148
|
|
|
151
149
|
Objective:
|
|
152
|
-
- Find concrete, high-confidence{FOCUS_QUALIFIER} issues
|
|
153
|
-
-
|
|
154
|
-
- Do not flag issues the author would not fix. If there is no finding that a person would definitely want to see and fix, prefer outputting no findings.
|
|
150
|
+
- Find concrete, high-confidence{FOCUS_QUALIFIER} issues under the scope and criteria below.
|
|
151
|
+
- Inspect the full scope and submit every qualifying finding. Do not stop at the first one.
|
|
155
152
|
|
|
156
153
|
{SCOPE_INSTRUCTIONS}
|
|
157
154
|
|
|
155
|
+
{REVIEW_RUBRIC}
|
|
156
|
+
|
|
157
|
+
## Scope policy
|
|
158
|
+
|
|
159
|
+
{SCOPE_POLICY}
|
|
160
|
+
|
|
161
|
+
## Focus
|
|
162
|
+
|
|
158
163
|
{FOCUS_CONTEXT}
|
|
159
164
|
|
|
160
|
-
|
|
161
|
-
- Submit only issues introduced by the scoped changes, locally provable from the repository or diff, discrete, actionable, and likely worth fixing. Do not report speculative, stylistic, or pre-existing issues.
|
|
162
|
-
- This is a read-only review focus. Do not modify files or repository state; do not run mutating commands.
|
|
165
|
+
This is a read-only review focus. Do not modify files or repository state. Do not run mutating commands.
|
|
163
166
|
|
|
164
167
|
{ADDITIONAL_CONTEXT_SECTION}{PROJECT_GUIDELINES_SECTION}
|
|
165
168
|
{OUTPUT_CONTRACT}`;
|
|
@@ -217,7 +220,7 @@ Process:
|
|
|
217
220
|
3) Triage every feedback item exactly once. Do not omit any id.
|
|
218
221
|
4) If a review thread contains back-and-forth, focus on the latest remaining ask.
|
|
219
222
|
5) Resolved or outdated threads often become ignore, but verify before deciding.
|
|
220
|
-
6) This is a read-only triage. Do not modify files or repository state
|
|
223
|
+
6) This is a read-only triage. Do not modify files or repository state. Do not run mutating commands.
|
|
221
224
|
|
|
222
225
|
{SCOPE_INSTRUCTIONS}
|
|
223
226
|
|
|
@@ -259,7 +262,7 @@ Requirements:
|
|
|
259
262
|
- Input findings are already ordered by review priority. The host will keep the lowest id in each group.
|
|
260
263
|
- Keep reason very short.
|
|
261
264
|
- If there are no duplicates, return { "groups": [] }.
|
|
262
|
-
- Before
|
|
265
|
+
- Before returning, check that the output is valid JSON and matches the required structure.`;
|
|
263
266
|
|
|
264
267
|
export const TRIAGE_METADATA_QUERY = `query($owner: String!, $name: String!, $number: Int!) {
|
|
265
268
|
repository(owner: $owner, name: $name) {
|