@doist/doistbot-cli 1.0.7 → 1.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/dist/actions/auth.js +19 -5
  2. package/dist/actions/doctor.js +3 -1
  3. package/dist/actions/review.js +9 -4
  4. package/dist/auth.js +79 -30
  5. package/dist/config.js +12 -7
  6. package/dist/input.js +11 -11
  7. package/package.json +5 -4
  8. package/sandbox/dist/core/pi.js +10 -4
  9. package/sandbox/dist/core/shared.js +9 -9
  10. package/sandbox/dist/core/span-test-helpers.js +13 -0
  11. package/sandbox/dist/main.js +3 -0
  12. package/sandbox/dist/tasks/chat/chat.js +7 -6
  13. package/sandbox/dist/tasks/issue-triage/autofix-capabilities.js +22 -0
  14. package/sandbox/dist/tasks/issue-triage/fix-attempt.js +7 -0
  15. package/sandbox/dist/tasks/issue-triage/routing.js +124 -15
  16. package/sandbox/dist/tasks/issue-triage/triage.js +2 -0
  17. package/sandbox/dist/tasks/review/engines/multi-focus.js +16 -8
  18. package/sandbox/dist/tasks/review/multi-focus-prompt.js +2 -0
  19. package/sandbox/dist/tasks/review/review-summary.js +1 -1
  20. package/sandbox/dist/tasks/review-slice-plan/context.js +69 -0
  21. package/sandbox/dist/tasks/review-slice-plan/plan.js +209 -0
  22. package/sandbox/dist/tasks/review-slice-plan/prompt.js +59 -0
  23. package/sandbox/dist/tasks/review-slice-plan/review-slice-plan.js +470 -0
  24. package/sandbox/src/tasks/review/prompts/review-focus-prompts/efficiency.md +10 -12
  25. package/sandbox/src/tasks/review/prompts/review-focus-prompts/general.md +24 -7
  26. package/sandbox/src/tasks/review/prompts/review-focus-prompts/quality.md +14 -22
  27. package/sandbox/src/tasks/review/prompts/review-focus-prompts/reuse.md +13 -0
  28. package/sandbox/src/tasks/review/prompts/review-focus-prompts/security.md +18 -0
  29. package/sandbox/src/tasks/review/prompts/review-focus-prompts/tests.md +12 -22
@@ -0,0 +1,470 @@
1
+ import { spawnSync } from 'child_process';
2
+ import { SpanStatusCode } from '@opentelemetry/api';
3
+ import { buildReviewHygieneSlicePlanComment, encodeReviewSlicePlanBaseBranch, findReviewSlicePlanSlot, normalizeReviewSlicePlanMarkerValue, replaceReviewSlicePlanSlot, REVIEW_HYGIENE_COMMENT_MARKER, ReviewSlicePlanState, } from 'doistbot-async-routing-contracts';
4
+ import { isKnownDoistbotLogin } from '../../core/bot-logins.js';
5
+ import { recordTaskMetrics } from '../../core/datadog-metrics.js';
6
+ import { logger } from '../../core/logging.js';
7
+ import { DEFAULT_GEMINI_FLASH_MODEL, invokePiPrompt } from '../../core/pi.js';
8
+ import { cloneRepository, createGitHubClients, WORKSPACE_DIR } from '../../core/shared.js';
9
+ import { genAiProviderName, recordEvent, startToolSpan, withPhase } from '../../core/tracing.js';
10
+ import { parseReviewSlicePlanContext } from './context.js';
11
+ import { parseReviewSlicePlanOutput, renderReviewSlicePlan } from './plan.js';
12
+ import { buildReviewSlicePlanPrompt, REVIEW_SLICE_PLAN_SYSTEM_PROMPT } from './prompt.js';
13
+ const REVIEW_SLICE_PLAN_TIMEOUT_MS = 7 * 60 * 1_000;
14
+ const REVIEW_SLICE_PLAN_RETRY_DELAY_MS = 1_000;
15
+ const MAX_COMMENT_BODY_CHARS = 50_000;
16
+ class StaleReviewSlicePlanError extends Error {
17
+ }
18
+ export async function main() {
19
+ const startedAt = performance.now();
20
+ const ctx = parseReviewSlicePlanContext();
21
+ const repository = `${ctx.owner}/${ctx.repo}`;
22
+ logger.setBaseContext({
23
+ deliveryId: ctx.deliveryId ?? 'unknown',
24
+ repository,
25
+ prNumber: ctx.prNumber,
26
+ });
27
+ logger.info('Review slice plan task started', {
28
+ dryRun: ctx.dryRun,
29
+ model: DEFAULT_GEMINI_FLASH_MODEL,
30
+ headSha: ctx.headSha,
31
+ commentId: ctx.commentId,
32
+ });
33
+ let exitCode = 0;
34
+ let completedSuccessfully = false;
35
+ let githubClients;
36
+ await withPhase('review-slice-plan', {
37
+ task: 'review-slice-plan',
38
+ 'gen_ai.operation.name': 'invoke_agent',
39
+ 'gen_ai.agent.name': 'doistbot.review-slice-plan',
40
+ 'pr.number': ctx.prNumber,
41
+ dry_run: ctx.dryRun,
42
+ }, async (span) => {
43
+ try {
44
+ const { app, readOctokit, writeOctokit } = await createGitHubClients(ctx);
45
+ githubClients = { readOctokit, writeOctokit };
46
+ if (!ctx.dryRun &&
47
+ !(await ownsPendingSlicePlanSlot({ ctx, octokit: readOctokit }))) {
48
+ logger.info('Review slice plan task no longer owns the pending comment slot');
49
+ completedSuccessfully = true;
50
+ return;
51
+ }
52
+ const source = await loadReviewSlicePlanSource({ ctx, octokit: readOctokit });
53
+ assertExpectedRevision(ctx, source);
54
+ await cloneRepository({
55
+ app,
56
+ owner: ctx.owner,
57
+ repo: ctx.repo,
58
+ prNumber: ctx.prNumber,
59
+ installationId: ctx.installationId,
60
+ });
61
+ assertWorkspaceHead(ctx.headSha);
62
+ const prompt = buildReviewSlicePlanPrompt({
63
+ repository,
64
+ prNumber: ctx.prNumber,
65
+ title: source.title,
66
+ body: source.body,
67
+ author: source.author,
68
+ baseBranch: source.baseBranch,
69
+ headSha: source.headSha,
70
+ reviewHygiene: ctx.reviewHygiene,
71
+ files: source.files,
72
+ commits: source.commits,
73
+ });
74
+ source.files = withoutPatches(source.files);
75
+ const modelOutput = await invokeSlicePlanner({ ctx, prompt });
76
+ let plan;
77
+ try {
78
+ plan = parseReviewSlicePlanOutput(modelOutput, source.files);
79
+ }
80
+ catch (error) {
81
+ logger.warn('Gemini slice planner output failed validation; retrying once', {
82
+ error: error instanceof Error ? error.message : String(error),
83
+ });
84
+ const repairedOutput = await invokeSlicePlanner({
85
+ ctx,
86
+ prompt: `${prompt}\n\nYour previous response failed schema or changed-file coverage validation. Re-evaluate the evidence and return one corrected JSON object only.`,
87
+ maxAttempts: 1,
88
+ });
89
+ plan = parseReviewSlicePlanOutput(repairedOutput, source.files);
90
+ }
91
+ const slicePlanContent = renderPlanWithinCommentLimit(plan, ctx.baseBranch);
92
+ if (ctx.dryRun) {
93
+ printDryRunResult({ ctx, source, plan, slicePlanContent });
94
+ }
95
+ else {
96
+ const latestPr = await readOctokit.rest.pulls.get({
97
+ owner: ctx.owner,
98
+ repo: ctx.repo,
99
+ pull_number: ctx.prNumber,
100
+ });
101
+ assertExpectedRevision(ctx, {
102
+ ...source,
103
+ baseBranch: latestPr.data.base.ref,
104
+ headSha: latestPr.data.head.sha,
105
+ });
106
+ const updated = await updateOwnedSlicePlanSlot({
107
+ ctx,
108
+ readOctokit,
109
+ writeOctokit,
110
+ state: ReviewSlicePlanState.Complete,
111
+ content: slicePlanContent,
112
+ });
113
+ if (!updated) {
114
+ logger.info('Review slice plan result discarded because the comment slot changed');
115
+ }
116
+ else {
117
+ logger.info('Review hygiene comment updated with contextual slice plan', {
118
+ commentId: ctx.commentId,
119
+ slices: plan.slices.length,
120
+ strategy: plan.strategy,
121
+ });
122
+ }
123
+ }
124
+ completedSuccessfully = true;
125
+ }
126
+ catch (error) {
127
+ exitCode = 1;
128
+ const message = error instanceof Error ? error.message : String(error);
129
+ span.setStatus({ code: SpanStatusCode.ERROR, message });
130
+ logger.error('Review slice plan task failed', { error: message });
131
+ if (!ctx.dryRun && githubClients) {
132
+ try {
133
+ await updateOwnedSlicePlanSlot({
134
+ ctx,
135
+ ...githubClients,
136
+ state: ReviewSlicePlanState.Failed,
137
+ content: buildFailureMessage(error),
138
+ });
139
+ }
140
+ catch (updateError) {
141
+ logger.warn('Failed to replace the pending slice plan after an error', {
142
+ error: updateError instanceof Error
143
+ ? updateError.message
144
+ : String(updateError),
145
+ });
146
+ }
147
+ }
148
+ }
149
+ });
150
+ await recordTaskMetrics({
151
+ type: 'slice-plan',
152
+ repository,
153
+ model: 'gemini',
154
+ outcome: completedSuccessfully ? 'success' : 'failure',
155
+ durationMs: performance.now() - startedAt,
156
+ ...(completedSuccessfully ? {} : { reason: 'exception' }),
157
+ });
158
+ if (exitCode !== 0) {
159
+ process.exitCode = exitCode;
160
+ }
161
+ }
162
+ async function loadReviewSlicePlanSource({ ctx, octokit, }) {
163
+ const [pullRequest, files, commits] = await Promise.all([
164
+ octokit.rest.pulls.get({
165
+ owner: ctx.owner,
166
+ repo: ctx.repo,
167
+ pull_number: ctx.prNumber,
168
+ }),
169
+ octokit.paginate(octokit.rest.pulls.listFiles, {
170
+ owner: ctx.owner,
171
+ repo: ctx.repo,
172
+ pull_number: ctx.prNumber,
173
+ per_page: 100,
174
+ }),
175
+ octokit.paginate(octokit.rest.pulls.listCommits, {
176
+ owner: ctx.owner,
177
+ repo: ctx.repo,
178
+ pull_number: ctx.prNumber,
179
+ per_page: 100,
180
+ }),
181
+ ]);
182
+ if (files.length < pullRequest.data.changed_files) {
183
+ throw new Error(`GitHub returned only ${files.length} of ${pullRequest.data.changed_files} changed files; refusing to generate an incomplete plan`);
184
+ }
185
+ return {
186
+ title: pullRequest.data.title,
187
+ body: pullRequest.data.body,
188
+ author: pullRequest.data.user?.login ?? 'unknown',
189
+ baseBranch: pullRequest.data.base.ref,
190
+ headSha: pullRequest.data.head.sha,
191
+ files: files.map((file) => ({
192
+ filename: file.filename,
193
+ status: file.status,
194
+ additions: file.additions,
195
+ deletions: file.deletions,
196
+ ...(file.previous_filename ? { previousFilename: file.previous_filename } : {}),
197
+ ...(file.patch ? { patch: file.patch } : {}),
198
+ })),
199
+ commits: commits.map((commit) => ({
200
+ sha: commit.sha,
201
+ message: commit.commit.message,
202
+ })),
203
+ };
204
+ }
205
+ function assertExpectedRevision(ctx, source) {
206
+ if (source.headSha !== ctx.headSha || source.baseBranch !== ctx.baseBranch) {
207
+ throw new StaleReviewSlicePlanError(`PR revision changed while planning (expected ${ctx.baseBranch}@${ctx.headSha}, got ${source.baseBranch}@${source.headSha})`);
208
+ }
209
+ }
210
+ function assertWorkspaceHead(expectedHeadSha) {
211
+ const result = spawnSync('git', ['-C', WORKSPACE_DIR, 'rev-parse', 'HEAD'], {
212
+ encoding: 'utf8',
213
+ stdio: 'pipe',
214
+ });
215
+ const actualHeadSha = typeof result.stdout === 'string' ? result.stdout.trim() : '';
216
+ if (result.error || result.status !== 0 || actualHeadSha !== expectedHeadSha) {
217
+ throw new StaleReviewSlicePlanError(`Cloned PR head did not match the requested revision (${actualHeadSha || 'unknown'})`);
218
+ }
219
+ }
220
+ async function invokeSlicePlanner({ ctx, prompt, maxAttempts = 2, }) {
221
+ return await withPhase('review-slice-plan.model', {
222
+ 'gen_ai.operation.name': 'chat',
223
+ 'gen_ai.provider.name': genAiProviderName('gemini'),
224
+ 'gen_ai.request.model': DEFAULT_GEMINI_FLASH_MODEL,
225
+ }, async () => {
226
+ let lastError = 'Gemini did not return a response';
227
+ for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
228
+ const progress = createSlicePlannerProgressHandler();
229
+ const result = await invokePiPrompt({
230
+ provider: 'gemini',
231
+ apiKey: ctx.geminiApiKey,
232
+ modelName: DEFAULT_GEMINI_FLASH_MODEL,
233
+ prompt,
234
+ cwd: WORKSPACE_DIR,
235
+ timeoutMs: REVIEW_SLICE_PLAN_TIMEOUT_MS,
236
+ capabilityPreset: 'read-only',
237
+ thinkingLevel: 'high',
238
+ systemPrompt: REVIEW_SLICE_PLAN_SYSTEM_PROMPT,
239
+ noContextFiles: true,
240
+ onProgress: progress.onProgress,
241
+ }).finally(progress.close);
242
+ if (result.ok) {
243
+ logger.info('Gemini slice planner completed', {
244
+ attempt,
245
+ durationMs: result.durationMs,
246
+ outputChars: result.output.length,
247
+ stopReason: result.stopReason,
248
+ toolUsage: result.toolUsage,
249
+ });
250
+ return result.output;
251
+ }
252
+ lastError = result.error;
253
+ logger.warn('Gemini slice planner request failed', {
254
+ attempt,
255
+ durationMs: result.durationMs,
256
+ timedOut: result.timedOut,
257
+ stopReason: result.stopReason,
258
+ error: result.error,
259
+ });
260
+ if (attempt < maxAttempts) {
261
+ const delayMs = REVIEW_SLICE_PLAN_RETRY_DELAY_MS * attempt;
262
+ recordEvent('review-slice-plan.model.retry', { attempt, delay_ms: delayMs });
263
+ await new Promise((resolve) => setTimeout(resolve, delayMs));
264
+ }
265
+ }
266
+ throw new Error(`Gemini slice planner failed after ${maxAttempts} attempt(s): ${lastError}`);
267
+ });
268
+ }
269
+ function createSlicePlannerProgressHandler() {
270
+ const inFlightTools = new Map();
271
+ function toolKey(event) {
272
+ return event.toolCallId ?? event.toolName ?? '';
273
+ }
274
+ function onProgress(event) {
275
+ switch (event.type) {
276
+ case 'tool_start': {
277
+ const key = toolKey(event);
278
+ const displaced = inFlightTools.get(key);
279
+ if (displaced) {
280
+ displaced.setStatus({
281
+ code: SpanStatusCode.ERROR,
282
+ message: 'replaced by overlapping tool progress',
283
+ });
284
+ displaced.end();
285
+ }
286
+ const toolSpan = startToolSpan(event.toolName, {
287
+ 'gen_ai.operation.name': 'execute_tool',
288
+ 'gen_ai.tool.name': event.toolName,
289
+ ...(event.toolCallId ? { 'gen_ai.tool.call.id': event.toolCallId } : {}),
290
+ });
291
+ inFlightTools.set(key, toolSpan);
292
+ return;
293
+ }
294
+ case 'tool_end': {
295
+ const key = toolKey(event);
296
+ const toolSpan = inFlightTools.get(key);
297
+ if (toolSpan) {
298
+ if (event.isError) {
299
+ toolSpan.setStatus({
300
+ code: SpanStatusCode.ERROR,
301
+ message: 'tool reported error',
302
+ });
303
+ }
304
+ toolSpan.end();
305
+ inFlightTools.delete(key);
306
+ }
307
+ return;
308
+ }
309
+ case 'thinking':
310
+ case 'responding':
311
+ recordEvent('model.heartbeat', { state: event.type });
312
+ }
313
+ }
314
+ function close() {
315
+ for (const toolSpan of inFlightTools.values()) {
316
+ toolSpan.setStatus({
317
+ code: SpanStatusCode.ERROR,
318
+ message: 'tool span did not receive matching end',
319
+ });
320
+ toolSpan.end();
321
+ }
322
+ inFlightTools.clear();
323
+ }
324
+ return { onProgress, close };
325
+ }
326
+ function withoutPatches(files) {
327
+ return files.map((file) => ({
328
+ filename: file.filename,
329
+ status: file.status,
330
+ additions: file.additions,
331
+ deletions: file.deletions,
332
+ ...(file.previousFilename ? { previousFilename: file.previousFilename } : {}),
333
+ }));
334
+ }
335
+ function renderPlanWithinCommentLimit(plan, baseBranch) {
336
+ const full = renderReviewSlicePlan(plan, { baseBranch });
337
+ if (full.length <= MAX_COMMENT_BODY_CHARS) {
338
+ return full;
339
+ }
340
+ const compact = renderReviewSlicePlan(plan, { baseBranch, compact: true });
341
+ if (compact.length > MAX_COMMENT_BODY_CHARS) {
342
+ throw new Error('Validated slice plan is too large to fit in a GitHub comment');
343
+ }
344
+ return compact;
345
+ }
346
+ async function ownsPendingSlicePlanSlot({ ctx, octokit, }) {
347
+ return (await fetchOwnedPendingSlot({ ctx, octokit })) !== undefined;
348
+ }
349
+ async function fetchOwnedPendingSlot({ ctx, octokit, }) {
350
+ const commentId = requireCommentId(ctx);
351
+ const response = await octokit.rest.issues.getComment({
352
+ owner: ctx.owner,
353
+ repo: ctx.repo,
354
+ comment_id: commentId,
355
+ });
356
+ const body = response.data.body;
357
+ const slot = findReviewSlicePlanSlot(body);
358
+ if (!isKnownDoistbotLogin(response.data.user?.login) ||
359
+ !body?.includes(REVIEW_HYGIENE_COMMENT_MARKER) ||
360
+ slot?.state !== ReviewSlicePlanState.Pending ||
361
+ slot.headSha !== normalizeReviewSlicePlanMarkerValue(ctx.headSha) ||
362
+ slot.baseBranch !== encodeReviewSlicePlanBaseBranch(ctx.baseBranch) ||
363
+ slot.runId !== normalizeReviewSlicePlanMarkerValue(ctx.runId)) {
364
+ return undefined;
365
+ }
366
+ return { body };
367
+ }
368
+ export async function updateOwnedSlicePlanSlot({ ctx, readOctokit, writeOctokit, state, content, }) {
369
+ const commentId = requireCommentId(ctx);
370
+ const initial = await fetchOwnedPendingSlot({ ctx, octokit: readOctokit });
371
+ if (!initial) {
372
+ return false;
373
+ }
374
+ const latest = await fetchOwnedPendingSlot({ ctx, octokit: readOctokit });
375
+ if (latest?.body !== initial.body) {
376
+ return false;
377
+ }
378
+ const updatedBody = replaceReviewSlicePlanSlot(latest.body, {
379
+ state: ReviewSlicePlanState.Pending,
380
+ headSha: ctx.headSha,
381
+ baseBranch: ctx.baseBranch,
382
+ runId: ctx.runId,
383
+ }, { state, headSha: ctx.headSha, baseBranch: ctx.baseBranch, runId: ctx.runId }, content);
384
+ if (!updatedBody) {
385
+ return false;
386
+ }
387
+ await updateIssueCommentWithFallback({
388
+ ctx,
389
+ commentId,
390
+ body: updatedBody,
391
+ readOctokit,
392
+ writeOctokit,
393
+ });
394
+ return true;
395
+ }
396
+ async function updateIssueCommentWithFallback({ ctx, commentId, body, readOctokit, writeOctokit, }) {
397
+ function update(octokit) {
398
+ return octokit.rest.issues.updateComment({
399
+ owner: ctx.owner,
400
+ repo: ctx.repo,
401
+ comment_id: commentId,
402
+ body,
403
+ });
404
+ }
405
+ try {
406
+ await update(writeOctokit);
407
+ }
408
+ catch (error) {
409
+ const status = error?.status;
410
+ if (writeOctokit !== readOctokit && (status === 401 || status === 403)) {
411
+ logger.warn('GitHub token could not update slice plan comment; retrying as app');
412
+ await update(readOctokit);
413
+ return;
414
+ }
415
+ throw error;
416
+ }
417
+ }
418
+ function printDryRunResult({ ctx, source, plan, slicePlanContent, }) {
419
+ const pendingComment = buildReviewHygieneSlicePlanComment(ctx.reviewHygiene, {
420
+ baseBranch: ctx.baseBranch,
421
+ headSha: ctx.headSha,
422
+ runId: ctx.runId,
423
+ prAuthorLogin: source.author,
424
+ }).body;
425
+ const finalComment = replaceReviewSlicePlanSlot(pendingComment, {
426
+ state: ReviewSlicePlanState.Pending,
427
+ headSha: ctx.headSha,
428
+ baseBranch: ctx.baseBranch,
429
+ runId: ctx.runId,
430
+ }, {
431
+ state: ReviewSlicePlanState.Complete,
432
+ headSha: ctx.headSha,
433
+ baseBranch: ctx.baseBranch,
434
+ runId: ctx.runId,
435
+ }, slicePlanContent);
436
+ if (!finalComment) {
437
+ throw new Error('Failed to render the final dry-run hygiene comment');
438
+ }
439
+ process.stdout.write([
440
+ '--- Initial hygiene comment ---',
441
+ pendingComment,
442
+ '--- Final hygiene comment ---',
443
+ finalComment,
444
+ '--- Structured slice plan ---',
445
+ JSON.stringify({
446
+ dryRun: true,
447
+ repository: `${ctx.owner}/${ctx.repo}`,
448
+ prNumber: ctx.prNumber,
449
+ model: DEFAULT_GEMINI_FLASH_MODEL,
450
+ decision: ctx.reviewHygiene,
451
+ plan,
452
+ }, null, 2),
453
+ '',
454
+ ].join('\n'));
455
+ }
456
+ function buildFailureMessage(error) {
457
+ if (error instanceof StaleReviewSlicePlanError) {
458
+ return 'The PR changed while I was preparing this plan, so I discarded the stale result. Run `@doistbot /review` again on the latest revision to regenerate it.';
459
+ }
460
+ return "I couldn't generate a reliable contextual slicing plan for this revision. Please use the generic stacking instructions below or retry the review.";
461
+ }
462
+ function requireCommentId(ctx) {
463
+ if (ctx.commentId === undefined) {
464
+ throw new Error('COMMENT_ID is required outside dry-run mode');
465
+ }
466
+ return ctx.commentId;
467
+ }
468
+ if (process.env.VITEST !== 'true') {
469
+ await main();
470
+ }
@@ -1,18 +1,16 @@
1
1
  ## Review Focus: Efficiency
2
2
 
3
- For this pass, focus only on unnecessary work, hot-path regressions, and avoidable performance or resource overhead introduced by the diff.
3
+ For this pass, focus only on efficiency issues introduced by the diff.
4
4
 
5
- Flag findings for issues such as:
5
+ Review the changes for potential efficiency issues, such as:
6
6
 
7
- - Unnecessary work: redundant computations, repeated reads, or redundant calls, N+1 patterns.
8
- - Missed concurrency: independent work is done sequentially when batching or concurrency is clearly available
9
- - Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths.
10
- - Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error.
11
- - Memory: unbounded data structures, missing cleanup, event listener leaks.
12
- - Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one.
13
- - Accidental indirection: wrapper chains, adapters, or registries that add repeated runtime work without hiding real complexity. Prefer deletion or consolidation when the local code shows the extra work.
14
- - Backpressure: treat backpressure handling as critical to system stability; flag unbounded queues, missing flow control, or producer-consumer imbalances.
15
-
16
- Prefer concrete, evidenced concerns over speculative micro-optimizations. Do not report theoretical overhead unless the changed path is plausibly repeated, user-facing, queue-backed, or otherwise non-trivial at the PR's expected scale. Avoid theoretical speedups for tiny or one-time work.
7
+ 1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns.
8
+ 2. Missed concurrency: independent operations run sequentially when they could run in parallel.
9
+ 3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths.
10
+ 4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error.
11
+ 5. Memory: unbounded data structures, missing cleanup, event listener leaks.
12
+ 6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one.
13
+ 7. Accidental indirection: wrapper chains, adapters, or registries that add repeated runtime work without hiding real complexity. Prefer deletion or consolidation when the local code shows the extra work.
14
+ 8. Backpressure: treat backpressure handling as critical to system stability; flag unbounded queues, missing flow control, or producer-consumer imbalances.
17
15
 
18
16
  If the diff does not introduce any findings for this focus, return an empty comments array.
@@ -1,15 +1,32 @@
1
1
  ## Review Focus: General
2
2
 
3
- For this pass, focus only on broad correctness, behavior, and maintainability issues introduced by the diff.
3
+ For this pass, focus only on general issues introduced by the diff.
4
4
 
5
- Flag findings when:
5
+ You are acting as a code reviewer for a proposed code change made by another engineer.
6
6
 
7
- - the change can cause incorrect behavior, crashes, or broken contracts
8
- - a new edge case or error path is left unhandled
9
- - the diff introduces a maintainability problem that is directly tied to correctness
7
+ Below are default guidelines for determining what to flag. These are not the final word — if you encounter more specific guidelines elsewhere (in a developer message, user message, file, or project review guidelines appended below), those override these general instructions.
10
8
 
11
- Only report a finding when you can point to the concrete scenario, input, caller, or contract affected by the changed code. If the concern is a possible risk without evidence in the diff or nearby code, omit it.
9
+ ## Determining what to flag
12
10
 
13
- Do not spend time on reuse, efficiency, or test-quality concerns unless they are central to a real correctness issue.
11
+ Flag issues that:
12
+
13
+ 1. Meaningfully impact the accuracy, performance, security, or maintainability of the code.
14
+ 2. Are discrete and actionable (not general issues or multiple combined issues).
15
+ 3. Don't demand rigor inconsistent with the rest of the codebase.
16
+ 4. Were introduced in the changes being reviewed and related to the original intent, not adjacent cleanup or opportunistic refactoring.
17
+ 5. The author would likely fix if aware of them.
18
+ 6. Have provable impact. It is not enough to speculate that a change may disrupt another part, you must identify the parts that are provably affected.
19
+ 7. Are clearly not intentional changes by the author.
20
+ 8. Call out newly added dependencies explicitly and explain why they're needed.
21
+ 9. Apply system-level thinking; flag changes that increase operational risk or on-call burden.
22
+
23
+ If an issue introduced by the reviewed changes is valid and worth tracking but outside the PR's original intent or merely adjacent, report it only as P3 and clearly frame it as follow-up work. Do not report pre-existing issues unless the changed code makes them newly incorrect. Omit unrelated issues that are speculative, vague, or not worth tracking.
24
+
25
+ ## Finding field guidelines
26
+
27
+ 1. Explain why the issue matters and the concrete scenario/environment where it fails.
28
+ 2. Keep each finding brief, matter-of-fact, and easy to understand.
29
+ 3. Keep suggestions specific and actionable.
30
+ 4. Avoid flattery or filler phrases like "Great job...".
14
31
 
15
32
  If the diff does not introduce any findings for this focus, return an empty comments array.
@@ -1,28 +1,20 @@
1
1
  ## Review Focus: Quality
2
2
 
3
- For this pass, focus only on design quality, architecture, type safety, missed reuse, duplicated logic, and consistency with established patterns.
3
+ For this pass, focus only on quality issues introduced by the diff.
4
4
 
5
- Flag findings when:
5
+ Review the changes for potential quality issues, such as:
6
6
 
7
- - logic lives in the wrong layer or breaks an existing abstraction boundary
8
- - the diff introduces redundant state, brittle APIs, or stringly-typed code
9
- - the change deviates from surrounding patterns or documented guidance in a way that creates ongoing maintenance cost
10
- - a concrete type-safety or design issue should be fixed even if the code “works”
11
- - the diff reinvents a helper, utility, abstraction, or repeated block that already exists nearby
12
- - state duplicates existing state, cached values could be derived, or observers/effects could be direct calls
13
- - the diff adds new parameters to a function instead of generalizing or restructuring existing ones
14
- - near-duplicate code blocks are copied with slight variation and should be unified with a shared abstraction
15
- - internal details are exposed that should be encapsulated, or existing abstraction boundaries are broken
16
- - raw strings are used where constants, enums (string unions), or branded types already exist in the codebase
17
- - the diff adds wrappers or abstractions without clear reuse value instead of using a simple, direct solution
18
- - ternary chains, deeply nested if/else blocks, or nested switches obscure distinct cases, duplicate branches, or make error/edge paths easy to miss
19
- - broad try/catch blocks, fallback/null guard/logging paths, or safe wrappers are added without a real trust boundary or documented failure mode
20
- - logging-and-continue patterns hide errors where explicit failures or predictable failure modes would be better
21
- - errors are checked against message strings instead of codes or stable identifiers
22
- - broad any/type-ignore casts, sleeps/timeouts, fake success returns, removed checks, or path mutation hide a real failure
23
-
24
- Only report a design-quality finding when the maintenance cost is concrete: name the abstraction boundary, existing pattern, type contract, caller impact, helper, module, or repeated changed block that makes the change costly. Do not turn a naming, formatting, or preference nit into a [P1] or [P2].
25
-
26
- Do not repeat generic correctness bugs already covered by the general pass unless the design concern is distinct.
7
+ 1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls.
8
+ 2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones.
9
+ 3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction.
10
+ 4. Layering and leaky abstractions: logic that lives in the wrong layer, exposes internal details that should be encapsulated, or breaks existing abstraction boundaries.
11
+ 5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase.
12
+ 6. Simplicity/YAGNI: prefer simple, direct solutions over wrappers, abstractions, configuration, options, extensibility, or scaffolding without clear reuse value or explicit need. Prefer deletion or direct code until the second use appears.
13
+ 7. Shrinkage: flag code that preserves behavior with fewer branches, lines, moving parts, or custom helpers. Do not shrink away input validation at trust boundaries, data-loss error handling, security measures, or accessibility basics.
14
+ 8. Nested conditionals: ternary chains, deeply nested if/else blocks, or nested switches should be simplified when they obscure distinct cases, duplicate branches, or make error/edge paths easy to miss.
15
+ 9. Over-defensive code: broad try/catch blocks, fallback/null guard/logging paths, or safe wrappers that are not tied to a real trust boundary or documented failure mode.
16
+ 10. Fail-fast: favor explicit failures over logging-and-continue patterns that hide errors. Prefer predictable failure modes over silent degradation.
17
+ 11. Error classification: ensure errors are checked against codes or stable identifiers, never error message strings.
18
+ 12. Band-aid code: broad any/type-ignore casts, sleeps/timeouts, fake success returns, removed checks, or path mutation that hides a real failure.
27
19
 
28
20
  If the diff does not introduce any findings for this focus, return an empty comments array.
@@ -0,0 +1,13 @@
1
+ ## Review Focus: Reuse
2
+
3
+ For this pass, focus only on reuse issues introduced by the diff.
4
+
5
+ Review the changes for potential reuse issues, such as:
6
+
7
+ 1. Search for existing capabilities that could replace newly written code: standard library APIs, native platform features, already-installed dependencies, and existing utilities/helpers. Search for relevant names and behavior, then go beyond string matches by inspecting adjacent files, utility files and directories, and shared modules.
8
+ 2. Flag any new function that duplicates existing functionality. Suggest the existing function, API, or feature to use instead.
9
+ 3. Flag any inline logic that could use an existing capability — hand-rolled standard-library behavior, string manipulation, manual path handling, custom environment checks, ad-hoc type guards, native platform features, and similar patterns are common candidates.
10
+ 4. Flag new dependencies when the standard library, runtime/platform, or an already-installed dependency provides the same capability or behavior.
11
+ 5. Flag duplicate modules, thin pass-through wrappers, and manual registries when they duplicate an existing source of truth or local pattern. Prefer deleting, consolidating, or reusing the existing path.
12
+
13
+ If the diff does not introduce any findings for this focus, return an empty comments array.
@@ -0,0 +1,18 @@
1
+ ## Review Focus: Security
2
+
3
+ For this pass, focus only on security issues introduced by the diff.
4
+
5
+ Review the changes for potential security issues, such as:
6
+
7
+ 1. Auth and permissions: changed routes, commands, jobs, or data access must preserve required authentication, authorization, tenant isolation, and ownership checks.
8
+ 2. Untrusted input: SQL or command construction must be parameterized; path, URL, shell, and HTML output must be escaped or encoded for the target context.
9
+ 3. Filesystem and process boundaries: user-controlled paths and process arguments must not allow traversal, arbitrary file access, command injection, or unsafe environment changes.
10
+ 4. Server-side fetches: server requests to user-controlled URLs must block localhost, private/link-local IP ranges, cloud metadata endpoints, and internal hostnames, including after DNS resolution and redirects.
11
+ 5. Redirects and navigation: user-controlled destinations must be same-origin relative paths or explicitly allowlisted origins.
12
+ 6. Secrets: new logging, errors, telemetry, files, or API responses must not expose tokens, keys, credentials, cookies, or sensitive identifiers.
13
+ 7. Serialization and parsing: avoid unsafe deserialization, dynamic code execution, prototype pollution, XML external entities, YAML custom object construction, and parser modes that load external resources.
14
+ 8. Dependencies: newly added dependencies that touch input parsing, networking, auth, crypto, secrets, or code execution need an explicit security reason.
15
+
16
+ Only flag issues with a concrete exploit path or trust-boundary failure introduced by the reviewed changes.
17
+
18
+ If the diff does not introduce any findings for this focus, return an empty comments array.