tau-coding-agent 0.1.6 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +15 -12
  2. package/extensions/answer.ts +129 -79
  3. package/extensions/branch-term/README.md +7 -0
  4. package/extensions/{branch-term.ts → branch-term/index.ts} +113 -104
  5. package/extensions/btw.ts +17 -2
  6. package/extensions/caffeinate/README.md +5 -0
  7. package/extensions/caffeinate/index.ts +144 -0
  8. package/extensions/fast.ts +292 -0
  9. package/extensions/ghostty.ts +214 -211
  10. package/extensions/git-diff-stats.ts +124 -84
  11. package/extensions/git-pr-status.ts +274 -208
  12. package/extensions/insights.ts +109 -191
  13. package/extensions/loop.ts +149 -145
  14. package/extensions/memory.ts +138 -86
  15. package/extensions/notify.ts +14 -26
  16. package/extensions/openai-verbosity.ts +108 -42
  17. package/extensions/review/fix.ts +15 -6
  18. package/extensions/review/git.ts +93 -103
  19. package/extensions/review/index.ts +60 -62
  20. package/extensions/review/interrupt.ts +117 -0
  21. package/extensions/review/message-queue.ts +45 -12
  22. package/extensions/review/models.ts +79 -3
  23. package/extensions/review/prompts.ts +41 -38
  24. package/extensions/review/review.ts +161 -166
  25. package/extensions/review/runner.ts +139 -147
  26. package/extensions/review/runtime.ts +173 -120
  27. package/extensions/review/triage.ts +41 -44
  28. package/extensions/sandbox/bash.ts +775 -0
  29. package/extensions/sandbox/command.ts +605 -0
  30. package/extensions/sandbox/config.ts +764 -0
  31. package/extensions/sandbox/index.ts +132 -3338
  32. package/extensions/sandbox/permissions/dialog.ts +118 -0
  33. package/extensions/sandbox/permissions/filesystem.ts +559 -0
  34. package/extensions/sandbox/permissions/mach-lookup.ts +186 -0
  35. package/extensions/sandbox/permissions/network.ts +164 -0
  36. package/extensions/sandbox/permissions/unsandboxed.ts +270 -0
  37. package/extensions/sandbox/runtime.ts +616 -0
  38. package/extensions/stash.ts +28 -15
  39. package/extensions/subagent/README.md +73 -0
  40. package/extensions/subagent/index.ts +822 -0
  41. package/extensions/subagent/interrupt.ts +117 -0
  42. package/extensions/subagent/permissions.ts +101 -0
  43. package/extensions/subagent/rpc.ts +177 -0
  44. package/extensions/tool-display-mode.ts +267 -64
  45. package/extensions/usage/index.ts +249 -356
  46. package/extensions/usage/openrouter.ts +10 -2
  47. package/extensions/websearch/README.md +12 -47
  48. package/extensions/websearch/config.ts +5 -2
  49. package/extensions/websearch/index.ts +94 -109
  50. package/extensions/websearch/output.ts +51 -0
  51. package/extensions/websearch/providers/anthropic.pi.ts +27 -44
  52. package/extensions/websearch/providers/gemini.browser.ts +68 -77
  53. package/extensions/websearch/providers/gemini.pi.ts +17 -17
  54. package/extensions/websearch/providers/openai-codex.pi.ts +145 -7
  55. package/extensions/websearch/providers/pi-model.shared.ts +33 -39
  56. package/extensions/websearch/providers/shared.ts +87 -2
  57. package/extensions/websearch/types.ts +0 -3
  58. package/extensions/worktree.ts +132 -172
  59. package/package.json +10 -5
  60. package/skills/browser-tools/SKILL.md +29 -234
  61. package/skills/browser-tools/references/cookies.md +36 -0
  62. package/skills/browser-tools/references/interaction.md +90 -0
  63. package/skills/browser-tools/references/logging.md +34 -0
  64. package/skills/git-clean-history/SKILL.md +6 -6
  65. package/skills/git-commit/SKILL.md +5 -3
  66. package/skills/github-pull-request/SKILL.md +60 -0
  67. package/skills/github-pull-request/references/create.md +40 -0
  68. package/skills/github-pull-request/references/stewardship.md +60 -0
  69. package/skills/oracle/SKILL.md +3 -3
  70. package/skills/oracle/scripts/oracle +10 -1
  71. package/skills/sentry/SKILL.md +15 -185
  72. package/skills/sentry/references/events.md +79 -0
  73. package/skills/sentry/references/issues.md +62 -0
  74. package/skills/sentry/references/logs.md +46 -0
  75. package/skills/update-changelog/SKILL.md +27 -121
  76. package/skills/web-design/SKILL.md +16 -105
  77. package/themes/tau-dark.json +4 -0
  78. package/extensions/openai-fast.ts +0 -229
  79. package/extensions/websearch/providers/openai-codex.browser.ts +0 -77
  80. package/extensions/websearch/providers/openai-codex.shared.ts +0 -123
@@ -0,0 +1,117 @@
1
+ import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
2
+ import {
3
+ getKeybindings,
4
+ isKeyRelease,
5
+ type TuiMainScreen,
6
+ type Component,
7
+ type TUI,
8
+ } from "@earendil-works/pi-tui";
9
+
10
+ interface InterruptRequest {
11
+ sessionKey: string;
12
+ cancel?: boolean;
13
+ active: boolean;
14
+ confirming?: boolean;
15
+ }
16
+
17
+ /** Each extension owns its guard; optional peers contribute work through the event bus. */
18
+ export function createInterruptGuard(
19
+ pi: ExtensionAPI,
20
+ ctx: ExtensionContext,
21
+ tui: TUI & Partial<Pick<TuiMainScreen, "getFocusedComponent">>,
22
+ isWorking: () => boolean,
23
+ cancel: () => void,
24
+ isPromptActive: () => boolean,
25
+ ) {
26
+ const sessionKey =
27
+ ctx.sessionManager.getSessionFile() ?? `session:${ctx.sessionManager.getSessionId()}`;
28
+ let closed = false;
29
+ let pending:
30
+ | { controller: AbortController; editor: Component; completion: Promise<void> }
31
+ | undefined;
32
+ const unsubscribeWork = pi.events.on("tau:interrupt", (data) => {
33
+ const request = data as InterruptRequest;
34
+ if (request.sessionKey !== sessionKey || closed) return;
35
+ if (pending) request.confirming = true;
36
+ if (!isWorking()) return;
37
+ request.active = true;
38
+ if (request.cancel) cancel();
39
+ });
40
+ const unsubscribeChanged = pi.events.on("tau:work-changed", (data) => {
41
+ if ((data as { sessionKey: string }).sessionKey === sessionKey && pending && !hasWork())
42
+ pending.controller.abort();
43
+ });
44
+ const unsubscribeInput = ctx.ui.onTerminalInput((data) => {
45
+ if (isKeyRelease(data) || !getKeybindings().matches(data, "app.interrupt")) return;
46
+ const editor = tui.getFocusedComponent?.();
47
+ // A dismissed native dialog restores focus before its promise settles. Swallow a rapid
48
+ // extra Escape in that gap instead of accidentally interrupting the parent.
49
+ if (pending) return editor === pending.editor ? { consume: true } : undefined;
50
+ if (
51
+ closed ||
52
+ isPromptActive() ||
53
+ !editor ||
54
+ !("getText" in editor) ||
55
+ !("setText" in editor) ||
56
+ ("isShowingAutocomplete" in editor &&
57
+ typeof editor.isShowingAutocomplete === "function" &&
58
+ editor.isShowingAutocomplete()) ||
59
+ !hasWork()
60
+ )
61
+ return;
62
+ const current = {
63
+ controller: new AbortController(),
64
+ editor,
65
+ completion: Promise.resolve(),
66
+ };
67
+ pending = current;
68
+ current.completion = (async () => {
69
+ try {
70
+ const request: {
71
+ run: (signal: AbortSignal) => Promise<void>;
72
+ signal: AbortSignal;
73
+ result?: Promise<unknown>;
74
+ } = {
75
+ signal: current.controller.signal,
76
+ async run(signal) {
77
+ const confirmed = await ctx.ui.confirm("Cancel all ongoing work?", "", { signal });
78
+ if (!confirmed || closed || signal.aborted || !hasWork()) return;
79
+ pi.events.emit("tau:interrupt", { sessionKey, active: false, cancel: true });
80
+ // Reuse the editor's native interrupt, including queue restoration, Bash and retries.
81
+ editor.handleInput?.(data);
82
+ },
83
+ };
84
+ // Keep queued approvals blocked until the confirmed cancellation has been applied.
85
+ pi.events.emit("subagent:permission", request);
86
+ await (request.result ?? request.run(request.signal));
87
+ } catch (error) {
88
+ if (!closed && !current.controller.signal.aborted)
89
+ ctx.ui.notify(`Could not confirm cancellation: ${String(error)}`, "warning");
90
+ } finally {
91
+ if (pending === current) pending = undefined;
92
+ }
93
+ })();
94
+ return { consume: true };
95
+ });
96
+
97
+ return {
98
+ refresh() {
99
+ if (!closed) pi.events.emit("tau:work-changed", { sessionKey });
100
+ },
101
+ async dispose() {
102
+ closed = true;
103
+ unsubscribeInput();
104
+ unsubscribeWork();
105
+ unsubscribeChanged();
106
+ pending?.controller.abort();
107
+ pi.events.emit("tau:work-changed", { sessionKey });
108
+ await pending?.completion;
109
+ },
110
+ };
111
+
112
+ function hasWork(): boolean {
113
+ const request: InterruptRequest = { sessionKey, active: false };
114
+ pi.events.emit("tau:interrupt", request);
115
+ return request.active;
116
+ }
117
+ }
@@ -3,6 +3,7 @@ import {
3
3
  type ExtensionAPI,
4
4
  type ExtensionContext,
5
5
  type InputEvent,
6
+ type InputEventResult,
6
7
  } from "@earendil-works/pi-coding-agent";
7
8
  import { getKeybindings } from "@earendil-works/pi-tui";
8
9
 
@@ -20,6 +21,7 @@ type QueuedReviewMessage = {
20
21
  };
21
22
  type QueueState = {
22
23
  active: boolean;
24
+ interrupted: boolean;
23
25
  messages: QueuedReviewMessage[];
24
26
  unsubscribeFollowUpShortcut?: () => void;
25
27
  };
@@ -28,15 +30,27 @@ export type ReviewMessageQueue = ReturnType<typeof createReviewMessageQueue>;
28
30
 
29
31
  export function createReviewMessageQueue(pi: ExtensionAPI) {
30
32
  const states = new Map<string, QueueState>();
33
+ let promptActive = false;
34
+ pi.on("ui_prompt_start", () => {
35
+ promptActive = true;
36
+ });
37
+ pi.on("ui_prompt_end", () => {
38
+ promptActive = false;
39
+ });
31
40
 
32
41
  function start(ctx: ExtensionContext): () => void {
33
42
  const sessionKey = getReviewSessionKey(ctx);
34
43
  const state = getState(sessionKey);
35
44
  state.active = true;
45
+ state.interrupted = false;
36
46
 
37
47
  if (ctx.hasUI && !state.unsubscribeFollowUpShortcut) {
38
48
  state.unsubscribeFollowUpShortcut = ctx.ui.onTerminalInput((data) => {
39
- if (!isActive(sessionKey)) return undefined;
49
+ if (!isActive(sessionKey) || promptActive || matchesConfiguredKey(data, "app.interrupt"))
50
+ return undefined;
51
+ const interrupt = { sessionKey, active: false, confirming: false };
52
+ pi.events.emit("tau:interrupt", interrupt);
53
+ if (interrupt.confirming) return undefined;
40
54
 
41
55
  if (matchesConfiguredKey(data, "app.message.dequeue")) {
42
56
  return restoreMessagesToEditor(ctx) ? { consume: true } : undefined;
@@ -64,24 +78,40 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
64
78
  };
65
79
  }
66
80
 
67
- function handleInput(event: InputEvent, ctx: ExtensionContext): boolean {
81
+ function handleInput(event: InputEvent, ctx: ExtensionContext): InputEventResult | undefined {
82
+ if (event.source === "extension" || isImmediateCommand(event.text)) return;
83
+ if (!event.text.trim() && !event.images?.length) return;
68
84
  const sessionKey = getReviewSessionKey(ctx);
69
- if (!isActive(sessionKey)) return false;
70
- if (event.source === "extension") return false;
85
+ const state = states.get(sessionKey);
86
+ if (state?.interrupted) {
87
+ const messages = state.messages;
88
+ state.messages = [];
89
+ state.interrupted = false;
90
+ render(ctx);
91
+ return {
92
+ action: "transform",
93
+ text: [...messages.map((message) => message.text), event.text].filter(Boolean).join("\n\n"),
94
+ images: [...messages.flatMap((message) => message.images ?? []), ...(event.images ?? [])],
95
+ };
96
+ }
97
+ if (!isActive(sessionKey)) return;
71
98
 
72
99
  // When the main agent is already streaming, Pi's built-in steering/follow-up
73
100
  // queues are available. Only provide the review-owned queue while review work
74
101
  // is running in the background and the main session is idle.
75
- if (event.streamingBehavior) return false;
76
- if (!ctx.isIdle()) return false;
77
- if (isImmediateCommand(event.text)) return false;
78
- if (!event.text.trim() && !event.images?.length) return false;
102
+ if (event.streamingBehavior || !ctx.isIdle()) return;
79
103
 
80
104
  queueMessage(ctx, "steer", {
81
105
  text: event.text,
82
106
  images: event.images?.length ? [...event.images] : undefined,
83
107
  });
84
- return true;
108
+ return { action: "handled" };
109
+ }
110
+
111
+ function retain(ctx: ExtensionContext): void {
112
+ const state = states.get(getReviewSessionKey(ctx));
113
+ if (state) state.interrupted = true;
114
+ render(ctx);
85
115
  }
86
116
 
87
117
  function flushSteering(ctx: ExtensionContext, options: FlushOptions = {}): boolean {
@@ -142,7 +172,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
142
172
  ): boolean {
143
173
  const sessionKey = getReviewSessionKey(ctx);
144
174
  const state = states.get(sessionKey);
145
- if (!state?.messages.length) return false;
175
+ if (!state?.messages.length || state.interrupted) return false;
146
176
 
147
177
  const selected: QueuedReviewMessage[] = [];
148
178
  const remaining: QueuedReviewMessage[] = [];
@@ -173,7 +203,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
173
203
  function getState(sessionKey: string): QueueState {
174
204
  let state = states.get(sessionKey);
175
205
  if (!state) {
176
- state = { active: false, messages: [] };
206
+ state = { active: false, interrupted: false, messages: [] };
177
207
  states.set(sessionKey, state);
178
208
  }
179
209
  return state;
@@ -199,7 +229,9 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
199
229
  lines.push(
200
230
  ctx.ui.theme.fg(
201
231
  "dim",
202
- `↳ ${appKeyDisplay("app.message.dequeue")} to edit all queued messages`,
232
+ states.get(getReviewSessionKey(ctx))?.interrupted
233
+ ? "Queued input will be included with your next message"
234
+ : `↳ ${appKeyDisplay("app.message.dequeue")} to edit all queued messages`,
203
235
  ),
204
236
  );
205
237
  ctx.ui.setWidget(WIDGET_KEY, lines);
@@ -210,6 +242,7 @@ export function createReviewMessageQueue(pi: ExtensionAPI) {
210
242
  handleInput,
211
243
  flushSteering,
212
244
  flushAll,
245
+ retain,
213
246
  clear,
214
247
  };
215
248
  }
@@ -1,7 +1,9 @@
1
1
  import { getSupportedThinkingLevels, type Api, type Model } from "@earendil-works/pi-ai";
2
2
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
3
3
 
4
- const OPENAI_FAST_MODEL_ID = "gpt-5.3-codex-spark";
4
+ import { REVIEW_TASK_TIMEOUT_MS } from "./runner.js";
5
+
6
+ const OPENAI_FAST_MODEL_ID = "gpt-5.6-luna";
5
7
  const ANTHROPIC_FAST_MODEL_ID = "claude-haiku-4-5";
6
8
 
7
9
  type ModelFamily = "openai" | "anthropic";
@@ -505,15 +507,89 @@ function resolveUnqualifiedModelPattern(
505
507
  });
506
508
  }
507
509
 
510
+ function getRequestedProvider(
511
+ modelPattern: string,
512
+ availableModels: Array<Model<Api>>,
513
+ modelRegistry: ExtensionContext["modelRegistry"],
514
+ ): string | undefined {
515
+ const { basePattern } = splitModelPatternThinkingSuffix(modelPattern);
516
+ const slash = basePattern.indexOf("/");
517
+ if (slash <= 0) return undefined;
518
+
519
+ const providerPrefix = basePattern.slice(0, slash);
520
+ const provider = modelRegistry
521
+ .getRegisteredProviderIds()
522
+ .find((candidate) => candidate.toLowerCase() === providerPrefix.toLowerCase());
523
+ if (!provider) return undefined;
524
+ if (shouldTreatAsExplicitProviderPattern(basePattern, availableModels, modelRegistry)) {
525
+ return provider;
526
+ }
527
+
528
+ const hasRawModelMatch = availableModels.some(
529
+ (model) => model.id.toLowerCase() === basePattern.toLowerCase(),
530
+ );
531
+ return hasRawModelMatch ? undefined : provider;
532
+ }
533
+
534
+ async function refreshModelCatalog(
535
+ ctx: ExtensionContext,
536
+ requestedModels: string[],
537
+ signal: AbortSignal,
538
+ ): Promise<Array<Model<Api>>> {
539
+ signal.throwIfAborted();
540
+ const currentModels = ctx.modelRegistry.getAll();
541
+ const requestedProviders = new Set(
542
+ requestedModels
543
+ .map((pattern) => getRequestedProvider(pattern, currentModels, ctx.modelRegistry))
544
+ .filter((provider): provider is string => Boolean(provider)),
545
+ );
546
+
547
+ if (requestedProviders.size > 0) {
548
+ const result = await ctx.modelRegistry.refresh({
549
+ providers: [...requestedProviders],
550
+ signal: AbortSignal.any([signal, AbortSignal.timeout(REVIEW_TASK_TIMEOUT_MS)]),
551
+ });
552
+ signal.throwIfAborted();
553
+ const failedProvider = [...requestedProviders].find((provider) =>
554
+ [...result.errors.keys()].some(
555
+ (candidate) => candidate.toLowerCase() === provider.toLowerCase(),
556
+ ),
557
+ );
558
+ if (result.aborted || failedProvider) {
559
+ const provider =
560
+ failedProvider ?? (requestedProviders.size === 1 ? [...requestedProviders][0] : undefined);
561
+ throw new Error(
562
+ provider
563
+ ? `Could not refresh requested provider "${provider}".`
564
+ : "Could not refresh all requested providers.",
565
+ );
566
+ }
567
+ }
568
+
569
+ const hasUnqualifiedRequest =
570
+ requestedModels.length === 0 ||
571
+ requestedModels.some(
572
+ (pattern) => !getRequestedProvider(pattern, currentModels, ctx.modelRegistry),
573
+ );
574
+ if (hasUnqualifiedRequest) {
575
+ await ctx.modelRegistry.refresh({
576
+ signal: AbortSignal.any([signal, AbortSignal.timeout(REVIEW_TASK_TIMEOUT_MS)]),
577
+ });
578
+ signal.throwIfAborted();
579
+ }
580
+
581
+ return ctx.modelRegistry.getAll();
582
+ }
583
+
508
584
  export async function resolveModels(
509
585
  ctx: ExtensionContext,
510
586
  requestedModels: string[],
511
587
  currentThinkingLevel: ReviewThinkingLevel,
588
+ signal: AbortSignal,
512
589
  ): Promise<ResolvedReviewModel[]> {
513
- await ctx.modelRegistry.refresh();
590
+ const allModels = await refreshModelCatalog(ctx, requestedModels, signal);
514
591
  const currentProvider = typeof ctx.model?.provider === "string" ? ctx.model.provider : undefined;
515
592
  const currentModelId = ctx.model?.id;
516
- const allModels = ctx.modelRegistry.getAll();
517
593
 
518
594
  const resolveRequestedModel = (modelPattern: string): ResolvedReviewModel => {
519
595
  const { basePattern, thinkingSuffix } = splitModelPatternThinkingSuffix(modelPattern);
@@ -4,9 +4,7 @@ type FocusDefinition = { suffix: string; qualifier: string; context: string };
4
4
 
5
5
  export const REVIEW_RUBRIC_PROMPT = `# Review Guidelines
6
6
 
7
- You are acting as a code reviewer for a proposed code change made by another engineer.
8
-
9
- Below are default guidelines for determining what to flag. These are not the final word — if you encounter more specific guidelines elsewhere (in a developer message, user message, file, or project review guidelines appended below), those override these general instructions.
7
+ These are default scope, finding, and severity guidelines for every review focus. More specific custom instructions, user context, or project review guidelines override these defaults, including any limits on follow-up findings.
10
8
 
11
9
  ## Determining what to flag
12
10
 
@@ -14,14 +12,9 @@ Flag issues that:
14
12
  1. Meaningfully impact the accuracy, performance, security, or maintainability of the code.
15
13
  2. Are discrete and actionable (not general issues or multiple combined issues).
16
14
  3. Don't demand rigor inconsistent with the rest of the codebase.
17
- 4. Were introduced in the changes being reviewed and related to the original intent, not adjacent cleanup or opportunistic refactoring.
18
- 5. The author would likely fix if aware of them.
19
- 6. Have provable impact. It is not enough to speculate that a change may disrupt another part, you must identify the parts that are provably affected.
20
- 7. Are clearly not intentional changes by the author.
21
- 8. Call out newly added dependencies explicitly and explain why they're needed.
22
- 9. Apply system-level thinking; flag changes that increase operational risk or on-call burden.
23
-
24
- If an issue is valid and worth tracking but out of scope for the reviewed change, pre-existing, or merely adjacent, report it only as P3 and clearly frame it as follow-up work. Omit unrelated issues that are speculative, vague, or not worth tracking.
15
+ 4. The author would likely fix if aware of them, either in scope or as an explicitly allowed follow-up.
16
+ 5. Have provable impact. Identify the affected code and concrete consequences using evidence from the repository or diff. Omit speculative, vague, or purely stylistic issues.
17
+ 6. Are clearly not intentional behavior.
25
18
 
26
19
  ## Finding field guidelines
27
20
 
@@ -35,44 +28,50 @@ If an issue is valid and worth tracking but out of scope for the reviewed change
35
28
  - P0: critical/blocking.
36
29
  - P1: urgent.
37
30
  - P2: normal.
38
- - P3: low/nice-to-have/out-of-scope.
31
+ - P3: low/nice-to-have.`;
39
32
 
40
- If an issue is valid but out of scope for the reviewed change, pre-existing, or merely adjacent, report it as P3 and frame it as follow-up work.`;
33
+ export const REVIEW_DIFF_SCOPE_PROMPT = `In diff reviews, assess issues introduced by the scoped changes and related to their original intent at their actual severity. A concrete, useful issue that is pre-existing, out of scope, or merely adjacent may be reported only as P3 and must be explicitly framed as follow-up work. Do not expand the review into an unrelated audit or opportunistic refactoring.`;
34
+
35
+ export const REVIEW_SNAPSHOT_SCOPE_PROMPT = `In snapshot reviews, assess existing issues in the selected paths at their actual severity. Do not downgrade an issue to P3 merely because it is pre-existing. Keep findings within the selected paths; other code may be inspected as supporting context.`;
41
36
 
42
37
  export const REVIEW_FOCUSES: Record<ReviewFocus, FocusDefinition> = {
43
38
  general: {
44
39
  suffix: "",
45
40
  qualifier: "",
46
- context: REVIEW_RUBRIC_PROMPT,
41
+ context: `Review the scoped code for issues that meaningfully affect accuracy, performance, security, or maintainability.
42
+
43
+ Additional checks:
44
+ 1. Examine dependencies explicitly and explain why they're needed when flagging a dependency issue.
45
+ 2. Apply system-level thinking; flag code that increases operational risk or on-call burden.`,
47
46
  },
48
47
  security: {
49
48
  suffix: " specializing in security analysis",
50
49
  qualifier: " security",
51
- context: `Review the changes for potential security issues, such as:
52
- 1. Auth and permissions: changed routes, commands, jobs, or data access must preserve required authentication, authorization, tenant isolation, and ownership checks.
50
+ context: `Review the scoped code for potential security issues, such as:
51
+ 1. Auth and permissions: routes, commands, jobs, or data access must preserve required authentication, authorization, tenant isolation, and ownership checks.
53
52
  2. Untrusted input: SQL or command construction must be parameterized; path, URL, shell, and HTML output must be escaped or encoded for the target context.
54
53
  3. Filesystem and process boundaries: user-controlled paths and process arguments must not allow traversal, arbitrary file access, command injection, or unsafe environment changes.
55
54
  4. Server-side fetches: server requests to user-controlled URLs must block localhost, private/link-local IP ranges, cloud metadata endpoints, and internal hostnames, including after DNS resolution and redirects.
56
55
  5. Redirects and navigation: user-controlled destinations must be same-origin relative paths or explicitly allowlisted origins.
57
- 6. Secrets: new logging, errors, telemetry, files, or API responses must not expose tokens, keys, credentials, cookies, or sensitive identifiers.
56
+ 6. Secrets: logging, errors, telemetry, files, or API responses must not expose tokens, keys, credentials, cookies, or sensitive identifiers.
58
57
  7. Serialization and parsing: avoid unsafe deserialization, dynamic code execution, prototype pollution, XML external entities, YAML custom object construction, and parser modes that load external resources.
59
- 8. Dependencies: newly added dependencies that touch input parsing, networking, auth, crypto, secrets, or code execution need an explicit security reason.
60
- Only flag issues with a concrete exploit path or trust-boundary failure introduced by the reviewed changes.`,
58
+ 8. Dependencies: dependencies that touch input parsing, networking, auth, crypto, secrets, or code execution need an explicit security reason.
59
+ Only flag issues with a concrete exploit path or trust-boundary failure.`,
61
60
  },
62
61
  reuse: {
63
62
  suffix: " specializing in reuse analysis",
64
63
  qualifier: " reuse",
65
- context: `Review the changes for potential reuse issues, such as:
66
- 1. Search for existing capabilities that could replace newly written code: standard library APIs, native platform features, already-installed dependencies, and existing utilities/helpers. Search for relevant names and behavior, then go beyond string matches by inspecting adjacent files, utility files and directories, and shared modules.
67
- 2. Flag any new function that duplicates existing functionality. Suggest the existing function, API, or feature to use instead.
64
+ context: `Review the scoped code for potential reuse issues, such as:
65
+ 1. Search for existing capabilities that could replace custom code: standard library APIs, native platform features, already-installed dependencies, and existing utilities/helpers. Search for relevant names and behavior, then go beyond string matches by inspecting adjacent files, utility files and directories, and shared modules.
66
+ 2. Flag functions that duplicate existing functionality. Suggest the existing function, API, or feature to use instead.
68
67
  3. Flag any inline logic that could use an existing capability — hand-rolled standard-library behavior, string manipulation, manual path handling, custom environment checks, ad-hoc type guards, native platform features, and similar patterns are common candidates.
69
- 4. Flag new dependencies when the standard library, runtime/platform, or an already-installed dependency provides the same capability or behavior.
68
+ 4. Flag dependencies when the standard library, runtime/platform, or an already-installed dependency provides the same capability or behavior.
70
69
  5. Flag duplicate modules, thin pass-through wrappers, and manual registries when they duplicate an existing source of truth or local pattern. Prefer deleting, consolidating, or reusing the existing path.`,
71
70
  },
72
71
  quality: {
73
72
  suffix: " specializing in quality analysis",
74
73
  qualifier: " quality",
75
- context: `Review the changes for potential quality issues, such as:
74
+ context: `Review the scoped code for potential quality issues, such as:
76
75
  1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls.
77
76
  2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones.
78
77
  3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction.
@@ -89,9 +88,9 @@ Only flag issues with a concrete exploit path or trust-boundary failure introduc
89
88
  testing: {
90
89
  suffix: " specializing in test analysis",
91
90
  qualifier: " testing",
92
- context: `Review the changes for potential testing issues, such as:
91
+ context: `Review the scoped code for potential testing issues, such as:
93
92
  1. High-signal suite: favor a smaller test suite over exhaustive coverage. Treat tests as carrying maintenance cost. Each test should protect important behavior, a realistic failure mode, or a stable shared contract.
94
- 2. Low-value coverage: flag tests added only to cover implementation trivia. Examples include trivial getters/wrappers/constants, exact internal formatting, incidental telemetry/log details or events, timer internals, framework wiring with no behavior of its own, synthetic edge cases with no realistic breakage story, or behavior already covered by a higher-value test.
93
+ 2. Low-value coverage: flag tests that only cover implementation trivia. Examples include trivial getters/wrappers/constants, exact internal formatting, incidental telemetry/log details or events, timer internals, framework wiring with no behavior of its own, synthetic edge cases with no realistic breakage story, or behavior already covered by a higher-value test.
95
94
  3. Test bloat: redundant cases, copy-paste matrices, excessive or repeated setup that should use or extract a fixture/helper, gratuitous snapshots, or unparameterized variations that increase maintenance cost without clear regression signal. Suggest consolidation or deletion in these cases.
96
95
  4. Missing coverage: important behavior that can break without a test failing. Only ask for new tests when you can name the public/user-visible contract, security/privacy boundary, data-loss risk, serialization/wire contract, state transition, permission check, concurrency issue, or prior regression being protected.
97
96
  5. Weak assertions: tests that do not check observable behavior or invariants.
@@ -103,10 +102,10 @@ Do not ask for tests just because code changed. Only flag a missing test when yo
103
102
  efficiency: {
104
103
  suffix: " specializing in efficiency analysis",
105
104
  qualifier: " efficiency",
106
- context: `Review the changes for potential efficiency issues, such as:
105
+ context: `Review the scoped code for potential efficiency issues, such as:
107
106
  1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns.
108
107
  2. Missed concurrency: independent operations run sequentially when they could run in parallel.
109
- 3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths.
108
+ 3. Hot-path bloat: blocking work in startup or per-request/per-render hot paths.
110
109
  4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error.
111
110
  5. Memory: unbounded data structures, missing cleanup, event listener leaks.
112
111
  6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one.
@@ -143,23 +142,27 @@ export const SUBMIT_TOOL_RETRY_PROMPT = `You did not call {SUBMIT_TOOL} as instr
143
142
  export const REVIEW_OUTPUT_CONTRACT_PROMPT = `Requirements:
144
143
  - Never output findings as text or write them to files.
145
144
  - Always call submit_review exactly once as your final action.
146
- - If no issues are found, pass an empty array of findings to submit_review.
147
- - Omit uncertain or speculative findings.`;
145
+ - If no issues are found, pass an empty array of findings to submit_review.`;
148
146
 
149
147
  export const REVIEW_FOCUS_PROMPT = `You are an expert code reviewer{FOCUS_SUFFIX}.
150
148
 
151
149
  Objective:
152
- - Find concrete, high-confidence{FOCUS_QUALIFIER} issues introduced by the scoped changes.
153
- - Submit every finding the author would fix if they were made aware of it. Do not stop at the first qualifying finding — continue until you have listed every qualifying finding.
154
- - Do not flag issues the author would not fix. If there is no finding that a person would definitely want to see and fix, prefer outputting no findings.
150
+ - Find concrete, high-confidence{FOCUS_QUALIFIER} issues under the scope and criteria below.
151
+ - Inspect the full scope and submit every qualifying finding. Do not stop at the first one.
155
152
 
156
153
  {SCOPE_INSTRUCTIONS}
157
154
 
155
+ {REVIEW_RUBRIC}
156
+
157
+ ## Scope policy
158
+
159
+ {SCOPE_POLICY}
160
+
161
+ ## Focus
162
+
158
163
  {FOCUS_CONTEXT}
159
164
 
160
- Important:
161
- - Submit only issues introduced by the scoped changes, locally provable from the repository or diff, discrete, actionable, and likely worth fixing. Do not report speculative, stylistic, or pre-existing issues.
162
- - This is a read-only review focus. Do not modify files or repository state; do not run mutating commands.
165
+ This is a read-only review focus. Do not modify files or repository state. Do not run mutating commands.
163
166
 
164
167
  {ADDITIONAL_CONTEXT_SECTION}{PROJECT_GUIDELINES_SECTION}
165
168
  {OUTPUT_CONTRACT}`;
@@ -217,7 +220,7 @@ Process:
217
220
  3) Triage every feedback item exactly once. Do not omit any id.
218
221
  4) If a review thread contains back-and-forth, focus on the latest remaining ask.
219
222
  5) Resolved or outdated threads often become ignore, but verify before deciding.
220
- 6) This is a read-only triage. Do not modify files or repository state; do not run mutating commands.
223
+ 6) This is a read-only triage. Do not modify files or repository state. Do not run mutating commands.
221
224
 
222
225
  {SCOPE_INSTRUCTIONS}
223
226
 
@@ -259,7 +262,7 @@ Requirements:
259
262
  - Input findings are already ordered by review priority. The host will keep the lowest id in each group.
260
263
  - Keep reason very short.
261
264
  - If there are no duplicates, return { "groups": [] }.
262
- - Before sending, self-check that JSON.parse(output) would succeed.`;
265
+ - Before returning, check that the output is valid JSON and matches the required structure.`;
263
266
 
264
267
  export const TRIAGE_METADATA_QUERY = `query($owner: String!, $name: String!, $number: Int!) {
265
268
  repository(owner: $owner, name: $name) {