co-maintainer 0.4.13 → 0.5.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +57 -38
  3. package/dist/main.js +2 -3
  4. package/dist/package.json +21 -15
  5. package/dist/src/ai/batch.d.ts +24 -1
  6. package/dist/src/ai/batch.js +75 -23
  7. package/dist/src/ai/estimate.d.ts +40 -0
  8. package/dist/src/ai/estimate.js +113 -0
  9. package/dist/src/ai/fake.d.ts +1 -1
  10. package/dist/src/ai/fake.js +1 -1
  11. package/dist/src/ai/hetzner.js +1 -1
  12. package/dist/src/ai/openrouter.d.ts +13 -0
  13. package/dist/src/ai/openrouter.js +66 -12
  14. package/dist/src/ai/pricing.d.ts +14 -0
  15. package/dist/src/ai/pricing.js +77 -0
  16. package/dist/src/ai/provider.js +8 -0
  17. package/dist/src/ai/verify.d.ts +25 -0
  18. package/dist/src/ai/verify.js +78 -0
  19. package/dist/src/cli/args.js +25 -29
  20. package/dist/src/cli/commands/config.d.ts +44 -0
  21. package/dist/src/cli/commands/config.js +220 -0
  22. package/dist/src/cli/commands/probe.js +70 -103
  23. package/dist/src/cli/commands/registry.d.ts +57 -0
  24. package/dist/src/cli/commands/registry.js +713 -0
  25. package/dist/src/cli/commands/review.js +66 -43
  26. package/dist/src/cli/commands/serve.d.ts +4 -2
  27. package/dist/src/cli/commands/serve.js +14 -10
  28. package/dist/src/cli/commands/set.d.ts +2 -1
  29. package/dist/src/cli/commands/set.js +63 -17
  30. package/dist/src/cli/commands/view.d.ts +1 -0
  31. package/dist/src/cli/commands/view.js +180 -0
  32. package/dist/src/cli/error.d.ts +39 -0
  33. package/dist/src/cli/error.js +64 -0
  34. package/dist/src/cli/main.d.ts +7 -0
  35. package/dist/src/cli/main.js +82 -1
  36. package/dist/src/cli/prompt.js +21 -7
  37. package/dist/src/cli/review_args.d.ts +4 -0
  38. package/dist/src/cli/review_args.js +22 -9
  39. package/dist/src/cli/review_output.d.ts +2 -2
  40. package/dist/src/cli/review_output.js +7 -6
  41. package/dist/src/cli/review_result.d.ts +59 -8
  42. package/dist/src/cli/review_result.js +145 -103
  43. package/dist/src/config.d.ts +15 -0
  44. package/dist/src/config.js +31 -0
  45. package/dist/src/github/app.d.ts +13 -0
  46. package/dist/src/github/app.js +23 -0
  47. package/dist/src/github/app_manifest.d.ts +50 -0
  48. package/dist/src/github/app_manifest.js +138 -0
  49. package/dist/src/github/client.js +7 -1
  50. package/dist/src/github/collect.js +1 -1
  51. package/dist/src/github/gh.js +75 -18
  52. package/dist/src/knowledge/facts.js +8 -3
  53. package/dist/src/knowledge/guide.d.ts +8 -0
  54. package/dist/src/knowledge/guide.js +32 -8
  55. package/dist/src/knowledge/probe.js +13 -3
  56. package/dist/src/knowledge/sections.d.ts +9 -0
  57. package/dist/src/knowledge/sections.js +16 -0
  58. package/dist/src/knowledge/skill.d.ts +5 -0
  59. package/dist/src/knowledge/skill.js +68 -21
  60. package/dist/src/knowledge/synthesis.d.ts +17 -1
  61. package/dist/src/knowledge/synthesis.js +70 -15
  62. package/dist/src/knowledge/types.d.ts +5 -0
  63. package/dist/src/local/codegraph_prepare.d.ts +3 -2
  64. package/dist/src/local/codegraph_prepare.js +4 -2
  65. package/dist/src/local/git_ops.d.ts +5 -4
  66. package/dist/src/local/git_ops.js +8 -9
  67. package/dist/src/local/review_local.js +36 -14
  68. package/dist/src/pr/checkout.js +14 -3
  69. package/dist/src/pr/codegraph_tools.js +3 -3
  70. package/dist/src/pr/diff_summary.js +3 -3
  71. package/dist/src/pr/findings.js +16 -5
  72. package/dist/src/pr/findings_json.d.ts +26 -0
  73. package/dist/src/pr/findings_json.js +187 -0
  74. package/dist/src/pr/review_copy.d.ts +20 -0
  75. package/dist/src/pr/review_copy.js +40 -0
  76. package/dist/src/pr/reviewer.d.ts +2 -1
  77. package/dist/src/pr/reviewer.js +84 -91
  78. package/dist/src/remote/client.js +42 -42
  79. package/dist/src/remote/http.d.ts +5 -0
  80. package/dist/src/remote/http.js +49 -0
  81. package/dist/src/remote/server/guides.d.ts +22 -0
  82. package/dist/src/remote/server/guides.js +81 -0
  83. package/dist/src/remote/server/routes.js +24 -0
  84. package/dist/src/review/blocking.d.ts +35 -0
  85. package/dist/src/review/blocking.js +45 -0
  86. package/dist/src/review/carry_over.d.ts +12 -1
  87. package/dist/src/review/carry_over.js +35 -9
  88. package/dist/src/review/engine.d.ts +1 -0
  89. package/dist/src/review/guides.js +28 -1
  90. package/dist/src/server/api/installations.js +1 -1
  91. package/dist/src/server/api/repos.js +86 -3
  92. package/dist/src/server/api/settings.js +3 -3
  93. package/dist/src/server/app.d.ts +3 -0
  94. package/dist/src/server/app.js +12 -5
  95. package/dist/src/server/pages/activity.d.ts +1 -0
  96. package/dist/src/server/pages/activity.js +2 -2
  97. package/dist/src/server/pages/add_repo.js +73 -5
  98. package/dist/src/server/pages/client.d.ts +1 -1
  99. package/dist/src/server/pages/client.js +1 -1
  100. package/dist/src/server/pages/home.d.ts +6 -1
  101. package/dist/src/server/pages/home.js +25 -2
  102. package/dist/src/server/pages/knowledge.js +1 -1
  103. package/dist/src/server/pages/layout.js +2 -2
  104. package/dist/src/server/pages/repo.d.ts +1 -1
  105. package/dist/src/server/pages/repo.js +21 -2
  106. package/dist/src/server/pages/repo_prs.d.ts +5 -1
  107. package/dist/src/server/pages/repo_prs.js +36 -1
  108. package/dist/src/server/pages/repo_remote.js +2 -2
  109. package/dist/src/server/pages/repo_settings.js +2 -2
  110. package/dist/src/server/pages/router.d.ts +6 -0
  111. package/dist/src/server/pages/router.js +129 -9
  112. package/dist/src/server/pages/settings.d.ts +1 -1
  113. package/dist/src/server/pages/settings.js +87 -9
  114. package/dist/src/server/pages/styles.d.ts +1 -1
  115. package/dist/src/server/pages/styles.js +1 -1
  116. package/dist/src/services/probe.d.ts +28 -0
  117. package/dist/src/services/probe.js +133 -0
  118. package/dist/src/services/remake_cron.d.ts +1 -1
  119. package/dist/src/services/remake_cron.js +3 -3
  120. package/dist/src/services/remote_review.js +14 -4
  121. package/dist/src/services/review.d.ts +11 -0
  122. package/dist/src/services/review.js +39 -11
  123. package/dist/src/services/setup.js +37 -19
  124. package/dist/src/services/setup_checklist.d.ts +10 -0
  125. package/dist/src/services/setup_checklist.js +87 -0
  126. package/dist/src/store/app_db.js +5 -2
  127. package/dist/src/store/cache_db.d.ts +4 -0
  128. package/dist/src/store/cache_db.js +27 -4
  129. package/dist/src/store/deliveries.d.ts +7 -0
  130. package/dist/src/store/deliveries.js +15 -0
  131. package/dist/src/tools/codegraph.d.ts +30 -0
  132. package/dist/src/tools/codegraph.js +49 -12
  133. package/dist/src/types.d.ts +10 -0
  134. package/dist/src/util/log.d.ts +3 -0
  135. package/dist/src/util/log.js +12 -0
  136. package/dist/src/util/run_summary.d.ts +33 -0
  137. package/dist/src/util/run_summary.js +47 -0
  138. package/dist/src/util/webhook_reachability.d.ts +11 -0
  139. package/dist/src/util/webhook_reachability.js +64 -0
  140. package/package.json +21 -15
@@ -0,0 +1,187 @@
1
+ const SEVERITIES = ["P0", "P1", "P2", "P3"];
2
+ /** OpenRouter `response_format` for a review. Providers that ignore it still
3
+ * get the prompt's instruction to return a fenced JSON block. */
4
+ export const FINDINGS_JSON_SCHEMA = {
5
+ type: "json_schema",
6
+ json_schema: {
7
+ name: "review_findings",
8
+ strict: false,
9
+ schema: {
10
+ type: "object",
11
+ properties: {
12
+ findings: {
13
+ type: "array",
14
+ items: {
15
+ type: "object",
16
+ properties: {
17
+ severity: { type: "string", enum: [...SEVERITIES] },
18
+ blocking: { type: "boolean" },
19
+ path: { type: "string" },
20
+ lineFrom: { type: "integer" },
21
+ lineTo: { type: "integer" },
22
+ symbol: { type: "string" },
23
+ title: { type: "string" },
24
+ body: { type: "string" },
25
+ suggestion: { type: "string" },
26
+ },
27
+ required: ["severity", "path", "lineFrom", "title", "body"],
28
+ },
29
+ },
30
+ previousFindings: {
31
+ type: "array",
32
+ items: {
33
+ type: "object",
34
+ properties: {
35
+ id: { type: "string" },
36
+ state: { type: "string", enum: ["open", "closed"] },
37
+ path: { type: "string" },
38
+ lineFrom: { type: "integer" },
39
+ lineTo: { type: "integer" },
40
+ },
41
+ required: ["id", "state"],
42
+ },
43
+ },
44
+ },
45
+ required: ["findings"],
46
+ },
47
+ },
48
+ };
49
+ /** The prompt's contract, kept next to the schema so the two cannot drift. */
50
+ export const FINDINGS_JSON_INSTRUCTIONS = `Return a single JSON object, and nothing else, with this shape:
51
+ {"findings":[{"severity":"P1","blocking":true,"path":"src/a.ts","lineFrom":42,"lineTo":42,"symbol":"helper()","title":"The helper ignores its argument","body":"One sentence on what is wrong and its impact, then a short evidence paragraph.","suggestion":"the exact replacement lines, or omit this field"}],"previousFindings":[{"id":"F1","state":"open"}]}
52
+
53
+ Rules for the JSON:
54
+ - "severity" is exactly one of P0, P1, P2, P3. "blocking" is true or false.
55
+ - "lineFrom" and "lineTo" are numbers in the new file, copied from the DIFF
56
+ column, never counted from the @@ header. "lineTo" may equal "lineFrom".
57
+ - "title" is one sentence naming the defect, without the severity or the path.
58
+ - "body" is the explanation. Do not repeat the title, the path, or the location
59
+ line inside it; our code renders those. Do not add a closing sentence asking
60
+ whether to explain more, and do not use the words Mechanism, Symptom,
61
+ Scenario, Verified, Repro, Options, or Scope as labels.
62
+ - "suggestion" holds only the replacement source lines, with their original
63
+ indentation, when the fix replaces the exact lines lineFrom-lineTo in one
64
+ hunk of the same file. Omit it otherwise. Never wrap it in a code fence.
65
+ - When a fix needs removed lines, another file, or more than one hunk, omit
66
+ "suggestion".
67
+ - Return every independently actionable finding, including none: use
68
+ {"findings":[]} when the change is clean. Never invent a finding to fill the
69
+ array.`;
70
+ function asNumber(value) {
71
+ const number = typeof value === "string" ? Number(value) : value;
72
+ return typeof number === "number" && Number.isFinite(number)
73
+ ? Math.trunc(number)
74
+ : undefined;
75
+ }
76
+ function asText(value) {
77
+ return typeof value === "string" ? value.trim() : "";
78
+ }
79
+ /** Like {@link asText} but keeps the leading indentation. A suggestion is
80
+ * source lines with their original indentation, so trimming the left edge
81
+ * changes the code it proposes (CORE-40). Only the outer blank lines go. */
82
+ function asSource(value) {
83
+ return typeof value === "string"
84
+ ? value.replace(/^\s*\n/, "").replace(/\s+$/, "")
85
+ : "";
86
+ }
87
+ /** Strips a Markdown fence and any prose around the JSON object. */
88
+ function extractJson(text) {
89
+ const cleaned = text
90
+ .trim()
91
+ .replace(/^```(?:json)?\s*/i, "")
92
+ .replace(/\s*```$/, "")
93
+ .trim();
94
+ try {
95
+ return JSON.parse(cleaned);
96
+ }
97
+ catch {
98
+ const start = cleaned.indexOf("{");
99
+ const end = cleaned.lastIndexOf("}");
100
+ if (start < 0 || end <= start)
101
+ return undefined;
102
+ try {
103
+ return JSON.parse(cleaned.slice(start, end + 1));
104
+ }
105
+ catch {
106
+ return undefined;
107
+ }
108
+ }
109
+ }
110
+ /** One JSON finding as Markdown, in the exact shape `parseFindings` reads.
111
+ * Rendering it here is the point: the model never writes this text itself. */
112
+ function findingMarkdown(raw) {
113
+ const path = asText(raw.path);
114
+ const title = asText(raw.title);
115
+ const lineFrom = asNumber(raw.lineFrom);
116
+ if (!path || !title || lineFrom === undefined || lineFrom <= 0) {
117
+ return undefined;
118
+ }
119
+ const severity = asText(raw.severity).toUpperCase();
120
+ const level = SEVERITIES.includes(severity)
121
+ ? severity
122
+ : "P2";
123
+ const impact = raw.blocking === true ? "blocking" : "non-blocking";
124
+ const symbol = asText(raw.symbol);
125
+ const lineTo = asNumber(raw.lineTo) ?? lineFrom;
126
+ const span = lineTo === lineFrom ? `${lineFrom}` : `${lineFrom}-${lineTo}`;
127
+ const suggestion = asSource(raw.suggestion);
128
+ const lines = [
129
+ `### [${level} · ${impact}] \`${path}\`${symbol ? `: \`${symbol}\`` : ""}`,
130
+ `Location: \`${path}:${span}\``,
131
+ "",
132
+ asText(raw.body),
133
+ ];
134
+ if (suggestion) {
135
+ lines.push("", `Suggestion: \`${path}:${span}\``, "```suggestion", suggestion, "```");
136
+ }
137
+ return lines.join("\n");
138
+ }
139
+ /** Reads the structured output and renders it as Markdown. `undefined` means
140
+ * the text was not usable JSON, so the caller falls back to the legacy
141
+ * Markdown parser rather than dropping the review. */
142
+ export function findingsMarkdownFromJson(text) {
143
+ const parsed = extractJson(text);
144
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
145
+ return undefined;
146
+ }
147
+ const raw = Array.isArray(parsed.findings) ? parsed.findings : undefined;
148
+ if (!raw)
149
+ return undefined;
150
+ const blocks = raw
151
+ .map((item) => item && typeof item === "object"
152
+ ? findingMarkdown(item)
153
+ : undefined)
154
+ .filter((block) => block !== undefined);
155
+ // A JSON object whose findings were all missing a path or a title is a
156
+ // malformed answer, not a clean review: falling back keeps it visible.
157
+ if (raw.length > 0 && blocks.length === 0)
158
+ return undefined;
159
+ const parts = ["## Findings", ""];
160
+ parts.push(blocks.length ? blocks.join("\n\n") : "No actionable findings.");
161
+ const previous = Array.isArray(parsed.previousFindings)
162
+ ? parsed.previousFindings
163
+ : [];
164
+ const verdicts = previous
165
+ .map((item) => {
166
+ if (!item || typeof item !== "object")
167
+ return undefined;
168
+ const value = item;
169
+ const id = asText(value.id);
170
+ const state = asText(value.state).toLowerCase();
171
+ if (!id || (state !== "open" && state !== "closed"))
172
+ return undefined;
173
+ if (state === "closed")
174
+ return `- ${id}: closed`;
175
+ const path = asText(value.path);
176
+ const from = asNumber(value.lineFrom);
177
+ const to = asNumber(value.lineTo) ?? from;
178
+ if (!path || from === undefined)
179
+ return `- ${id}: open`;
180
+ return `- ${id}: open ${path}:${from}${to !== from ? `-${to}` : ""}`;
181
+ })
182
+ .filter((line) => line !== undefined);
183
+ if (verdicts.length) {
184
+ parts.push("", "## Previous findings", "", verdicts.join("\n"));
185
+ }
186
+ return parts.join("\n");
187
+ }
@@ -0,0 +1,20 @@
1
+ /** The severity legend and the review-copy rule (CORE-83 / D11).
2
+ *
3
+ * Product text avoids the em dash and the semicolon: the docs linter rejects
4
+ * them, and the finding parser had to accept both a `—` and a `:` heading
5
+ * separator only because the prompt itself modelled the em dash. The colon is
6
+ * the one separator now. The legacy parser still reads an em dash heading so
7
+ * comments posted before 0.5.0 stay readable. */
8
+ /** `[P2] path: symbol` headings and `Summary: 3 new, 1 open, ...` lines are
9
+ * built from this legend, so the four levels read the same everywhere. */
10
+ export declare const SEVERITY_LEGEND: ReadonlyArray<{
11
+ level: string;
12
+ name: string;
13
+ description: string;
14
+ }>;
15
+ /** The `## Severity` block the review text starts with. The colon keeps it
16
+ * free of the em dash the old prompt used. */
17
+ export declare function severitySection(): string;
18
+ /** What the model is told about prose (CORE-83). The review body is rendered
19
+ * by our code, not the model, so this only has to keep the fields clean. */
20
+ export declare const REVIEW_COPY_RULE = "Write prose without the em dash or the semicolon. Use a colon or a comma instead.";
@@ -0,0 +1,40 @@
1
+ /** The severity legend and the review-copy rule (CORE-83 / D11).
2
+ *
3
+ * Product text avoids the em dash and the semicolon: the docs linter rejects
4
+ * them, and the finding parser had to accept both a `—` and a `:` heading
5
+ * separator only because the prompt itself modelled the em dash. The colon is
6
+ * the one separator now. The legacy parser still reads an em dash heading so
7
+ * comments posted before 0.5.0 stay readable. */
8
+ /** `[P2] path: symbol` headings and `Summary: 3 new, 1 open, ...` lines are
9
+ * built from this legend, so the four levels read the same everywhere. */
10
+ export const SEVERITY_LEGEND = [
11
+ {
12
+ level: "P0",
13
+ name: "Critical",
14
+ description: "production outage, data loss, or security issue",
15
+ },
16
+ {
17
+ level: "P1",
18
+ name: "High",
19
+ description: "major behavior is broken and should be fixed before merge",
20
+ },
21
+ {
22
+ level: "P2",
23
+ name: "Medium",
24
+ description: "important correctness or maintainability issue",
25
+ },
26
+ {
27
+ level: "P3",
28
+ name: "Low",
29
+ description: "minor, non-blocking improvement or edge case",
30
+ },
31
+ ];
32
+ /** The `## Severity` block the review text starts with. The colon keeps it
33
+ * free of the em dash the old prompt used. */
34
+ export function severitySection() {
35
+ const items = SEVERITY_LEGEND.map((row) => `- ${row.level}: ${row.name}: ${row.description}.`);
36
+ return `## Severity\n\n${items.join("\n")}\n`;
37
+ }
38
+ /** What the model is told about prose (CORE-83). The review body is rendered
39
+ * by our code, not the model, so this only has to keep the fields clean. */
40
+ export const REVIEW_COPY_RULE = "Write prose without the em dash or the semicolon. Use a colon or a comma instead.";
@@ -16,7 +16,7 @@ export declare const NO_DIAGRAM_RULES = "Do not use Mermaid or any other diagram
16
16
  export declare function reviewSystemPrompt(diagrams: boolean): string;
17
17
  /** Shared PR and local/remote workspace review instructions: confirm claims
18
18
  * against the indexed graph, not only the diff slice. */
19
- export declare const CODEGRAPH_DIFF_VERIFICATION = "Examine the changes line by line, not just file by file \u2014 a single file can\ncontain more than one independent defect, and a change that looks fine in\nisolation can be wrong once you trace what calls it or what else it affects.\nWhen a finding depends on behavior outside the changed lines, confirm it with\ncodegraph-node, codegraph-callers, codegraph-callees, codegraph-impact, and\ncodegraph-affected before reporting it. Drop or correct findings that only seem\nplausible from the diff but contradict unchanged callers, callees, or the same\npattern elsewhere in the repo. Use those tools to check blast radius and whether\na test reaches the path \u2014 do not guess coverage or impact from the diff alone\nwhen a tool can answer. A missing regression test is not a substitute for\nidentifying the concrete input or code path that misbehaves when you can.";
19
+ export declare const CODEGRAPH_DIFF_VERIFICATION = "Examine the changes line by line, not just file by file. A single file can\ncontain more than one independent defect, and a change that looks fine in\nisolation can be wrong once you trace what calls it or what else it affects.\nWhen a finding depends on behavior outside the changed lines, confirm it with\ncodegraph-node, codegraph-callers, codegraph-callees, codegraph-impact, and\ncodegraph-affected before reporting it. Drop or correct findings that only seem\nplausible from the diff but contradict unchanged callers, callees, or the same\npattern elsewhere in the repo. Use those tools to check blast radius and whether\na test reaches the path. Do not guess coverage or impact from the diff alone\nwhen a tool can answer. A missing regression test is not a substitute for\nidentifying the concrete input or code path that misbehaves when you can.";
20
20
  /** GitHub omits `patch` entirely for files it considers too large, and the file
21
21
  * still appears in the compare response with only its counts. Rendering that as
22
22
  * an empty body reads as "this file did not change", and a reviewer then
@@ -31,6 +31,7 @@ export type { Snapshot };
31
31
  export declare function reviewPullRequest(client: GitHubClient, options: Options, usage?: UsageSink, snapshot?: Snapshot, ai?: AiProvider, progress?: ProgressSink, extras?: ReviewExtras): Promise<AiResponse & {
32
32
  visiblePaths: string[];
33
33
  guideBuiltAt: string | null;
34
+ codegraphState: "used" | "disabled" | "unavailable";
34
35
  }>;
35
36
  /** Local workspace review — same prompt loop as PR review without GitHub. */
36
37
  export declare function reviewWorkspaceRevision(revision: Revision, options: Options, headSha: string, usage?: UsageSink, ai?: AiProvider, progress?: ProgressSink, extras?: ReviewExtras): Promise<AiResponse & {
@@ -3,8 +3,10 @@ import { completeWithMermaidTools, } from "../ai/mermaid_loop.js";
3
3
  import { loadGuides } from "../review/guides.js";
4
4
  import { computeScope } from "./scope.js";
5
5
  import { prepareCodegraphTools } from "./codegraph_tools.js";
6
+ import { REVIEW_COPY_RULE, severitySection } from "./review_copy.js";
6
7
  import { needsSummary, READ_FULL_DIFF_TOOL, readFullDiff, summarizeDiff, } from "./diff_summary.js";
7
8
  import { numberPatch } from "./hunks.js";
9
+ import { FINDINGS_JSON_INSTRUCTIONS, FINDINGS_JSON_SCHEMA, findingsMarkdownFromJson, } from "./findings_json.js";
8
10
  const MAX_REVIEW_DIFF_CHARS = 240_000;
9
11
  export const MERMAID_GUIDANCE = `Mermaid selection and minimal syntax:
10
12
  flowchart = decisions, branches, pipelines, and fallback paths;
@@ -30,7 +32,7 @@ concise by default.`;
30
32
  const REVIEW_DIAGRAM_RULES = `People generally find it easier to understand the
31
33
  problem you've identified when it's presented in diagrams. When a multi-step
32
34
  flow, lifecycle, dependency, data model, protocol, or architecture change is
33
- part of the finding, you are expected to draw it — do not skip the diagram
35
+ part of the finding, you are expected to draw it. Do not skip the diagram
34
36
  just to avoid the extra tool call. During the initial review, use at most one
35
37
  diagram, so spend it on the finding that benefits most. If a user later asks
36
38
  for detailed reasoning in a reply, that reply may use up to five diagrams, but
@@ -46,20 +48,20 @@ const DIAGRAM_PROMPT_RULES = `People generally find it easier to understand the
46
48
  problem you've identified when it's presented in diagrams. When a finding
47
49
  involves a multi-step flow, lifecycle, dependency, data model, protocol, or
48
50
  architecture change, you are expected to include a Mermaid fenced code block
49
- for it — reading the syntax with read-mermaid-syntaxes first is a small cost,
51
+ for it. Reading the syntax with read-mermaid-syntaxes first is a small cost,
50
52
  not a reason to skip the diagram. This review may contain at most one diagram
51
53
  in total, so if more than one finding qualifies, pick the one the diagram
52
54
  clarifies most. Its type must be one of the types named in the system
53
55
  instructions. Keep labels short and grounded in the supplied evidence. A
54
56
  one-line fix or an obvious, single-step cause and effect genuinely needs no
55
- diagram — that is the only reason to omit one. Close every fenced block.`;
57
+ diagram. That is the only reason to omit one. Close every fenced block.`;
56
58
  export function reviewSystemPrompt(diagrams) {
57
59
  return `${REVIEW_ROLE}
58
60
  ${diagrams ? REVIEW_DIAGRAM_RULES : NO_DIAGRAM_RULES}`;
59
61
  }
60
62
  /** Shared PR and local/remote workspace review instructions: confirm claims
61
63
  * against the indexed graph, not only the diff slice. */
62
- export const CODEGRAPH_DIFF_VERIFICATION = `Examine the changes line by line, not just file by file — a single file can
64
+ export const CODEGRAPH_DIFF_VERIFICATION = `Examine the changes line by line, not just file by file. A single file can
63
65
  contain more than one independent defect, and a change that looks fine in
64
66
  isolation can be wrong once you trace what calls it or what else it affects.
65
67
  When a finding depends on behavior outside the changed lines, confirm it with
@@ -67,7 +69,7 @@ codegraph-node, codegraph-callers, codegraph-callees, codegraph-impact, and
67
69
  codegraph-affected before reporting it. Drop or correct findings that only seem
68
70
  plausible from the diff but contradict unchanged callers, callees, or the same
69
71
  pattern elsewhere in the repo. Use those tools to check blast radius and whether
70
- a test reaches the path — do not guess coverage or impact from the diff alone
72
+ a test reaches the path. Do not guess coverage or impact from the diff alone
71
73
  when a tool can answer. A missing regression test is not a substitute for
72
74
  identifying the concrete input or code path that misbehaves when you can.`;
73
75
  function text(value, limit = 20_000) {
@@ -76,6 +78,33 @@ function text(value, limit = 20_000) {
76
78
  ? `${result.slice(0, limit)}\n[truncated]`
77
79
  : result;
78
80
  }
81
+ /** Turns a model reply into the Markdown every consumer already parses
82
+ * (CORE-40 / F02). A reply that is not usable JSON is retried once with an
83
+ * explicit reminder; if that also fails, the raw text is returned so the
84
+ * legacy Markdown parser can still read it. This keeps a provider that ignores
85
+ * `response_format` working, and never drops a review on the floor. */
86
+ async function normalizeReviewResponse(provider, request, response, extraTools, maxToolRounds, usage, report, options) {
87
+ const direct = findingsMarkdownFromJson(response.text);
88
+ if (direct !== undefined)
89
+ return direct;
90
+ report("AI response was not valid JSON, asking once more");
91
+ const retry = await completeWithMermaidTools(provider, {
92
+ ...request,
93
+ job: "review_pull_request_retry",
94
+ prompt: `${request.prompt}\n\nThe previous answer was not valid JSON. Return only the JSON object described above, with no prose and no code fence.`,
95
+ }, 1, extraTools, maxToolRounds);
96
+ if (usage)
97
+ await usage(retry);
98
+ if (options.debug) {
99
+ console.log(`[debug] JSON retry response · input=${retry.tokensIn} tokens · output=${retry.tokensOut} tokens`);
100
+ console.log(`\n----- JSON RETRY -----\n${retry.text}\n`);
101
+ }
102
+ const afterRetry = findingsMarkdownFromJson(retry.text);
103
+ if (afterRetry !== undefined)
104
+ return afterRetry;
105
+ report("AI response was still not JSON, falling back to Markdown parsing");
106
+ return response.text.trim();
107
+ }
79
108
  const MAX_FILE_PATCH_CHARS = 12_000;
80
109
  /** GitHub omits `patch` entirely for files it considers too large, and the file
81
110
  * still appears in the compare response with only its counts. Rendering that as
@@ -93,12 +122,12 @@ export function filePatch(file) {
93
122
  const counts = Number.isFinite(changes) && changes > 0
94
123
  ? `${changes} changed lines (+${additions} -${deletions})`
95
124
  : "an unreported number of changed lines";
96
- return (`[${status}; ${counts}; diff withheld by GitHub, not shown here. ` +
125
+ return (`[${status}: ${counts}, diff withheld by GitHub, not shown here. ` +
97
126
  `Do not treat this file as unchanged and do not report its contents.]`);
98
127
  }
99
128
  if (patch.length > MAX_FILE_PATCH_CHARS) {
100
- return (`${numberPatch(patch.slice(0, MAX_FILE_PATCH_CHARS))}\n[${status}; ${changes} ` +
101
- `changed lines total; this file's diff is cut off here, later hunks are ` +
129
+ return (`${numberPatch(patch.slice(0, MAX_FILE_PATCH_CHARS))}\n[${status}: ${changes} ` +
130
+ `changed lines total. This file's diff is cut off here, later hunks are ` +
102
131
  `not shown.]`);
103
132
  }
104
133
  return numberPatch(patch);
@@ -153,7 +182,7 @@ export async function reviewPullRequest(client, options, usage, snapshot, ai, pr
153
182
  : await client.pages(`repos/${options.repo}/pulls/${number}/files`);
154
183
  report(`diff files loaded · ${files.length} files`);
155
184
  if (!guide) {
156
- throw new Error(`repos/${options.repo}/PR_REVIEW_GUIDE.md was not found; run init first`);
185
+ throw new Error(`repos/${options.repo}/PR_REVIEW_GUIDE.md was not found. Run init first`);
157
186
  }
158
187
  if (options.debug) {
159
188
  console.log(`[debug] guides · short=${guide.length} chars · detailed=${detailed.length} chars · codebase=${codebase.length} chars`);
@@ -208,7 +237,7 @@ export async function reviewPullRequest(client, options, usage, snapshot, ai, pr
208
237
  patchByPath.set(path, patch);
209
238
  const description = await summarizeDiff(path, patch, lowProvider, usage);
210
239
  report(`summarized large diff · ${path} · ${changes} changed lines`);
211
- return (`FILE: ${path}\n[${changes} changed lines — summarized below; ` +
240
+ return (`FILE: ${path}\n[${changes} changed lines, summarized below. ` +
212
241
  `call read-full-diff("${path}") for the complete diff if this is not ` +
213
242
  `enough]\n${description}`);
214
243
  })),
@@ -242,7 +271,7 @@ export async function reviewPullRequest(client, options, usage, snapshot, ai, pr
242
271
  const diffWasTruncated = ownPatch.length > MAX_REVIEW_DIFF_CHARS;
243
272
  const ownDiff = text(ownPatch, MAX_REVIEW_DIFF_CHARS);
244
273
  const unchangedListing = extras?.unchangedPaths?.length
245
- ? `\nUNCHANGED SINCE LAST REVIEW (paths only — do not re-report findings here):\n${extras.unchangedPaths
274
+ ? `\nUNCHANGED SINCE LAST REVIEW (paths only, do not re-report findings here):\n${extras.unchangedPaths
246
275
  .map((path) => `- ${path}`)
247
276
  .join("\n")}`
248
277
  : "";
@@ -253,8 +282,8 @@ export async function reviewPullRequest(client, options, usage, snapshot, ai, pr
253
282
  ? `${ownDiff}${unchangedListing}`
254
283
  : `${ownDiff}
255
284
 
256
- UPSTREAM CONTEXT — arrived via a merge this round, not authored by this pull
257
- request. Do not raise a finding located only in this code; only note an
285
+ UPSTREAM CONTEXT: arrived via a merge this round, not authored by this pull
286
+ request. Do not raise a finding located only in this code. Only note an
258
287
  interaction if the pull request's own change above relies on or conflicts with
259
288
  one of these files, and never mark that finding blocking:
260
289
  ${upstreamListing}${unchangedListing}`;
@@ -265,66 +294,35 @@ ${upstreamListing}${unchangedListing}`;
265
294
  const carryBlock = extras?.carryPrompt ? `${extras.carryPrompt}\n` : "";
266
295
  const prompt = `Review this pull request against the repository's review guide and
267
296
  codebase conventions. Find only actionable code-level violations supported by
268
- the diff and either the guide or the codebase conventions — a pull request
297
+ the diff and either the guide or the codebase conventions. A pull request
269
298
  that departs from how this repository's own code is actually written is a
270
299
  valid finding even when the review guide has no matching rule.
271
300
  Do not repeat existing review comments unless the diff still contains the issue.
272
301
  Do not invent requirements. Ignore bot noise and historical PR identities.
273
- Reason thoroughly, then return concise Markdown only with either:
274
- "## Findings" followed by findings, or "## Findings\\n\\nNo actionable findings."
275
- There is no fixed number of findings. Return every independently actionable
276
- finding supported by the diff and guide, including zero findings when appropriate.
277
- Do not stop early; inspect all supplied diff text first and return the natural
278
- count. If the diff contains a truncation marker, limit claims to the supplied
279
- text and do not imply that omitted files were reviewed.
302
+ Reason thoroughly, then answer with the JSON object described below and nothing
303
+ else. There is no fixed number of findings. Return every independently
304
+ actionable finding supported by the diff and guide, including zero findings
305
+ when appropriate. Do not stop early; inspect all supplied diff text first and
306
+ return the natural count. If the diff contains a truncation marker, limit
307
+ claims to the supplied text and do not imply that omitted files were reviewed.
280
308
  Do not invent low-value findings.
281
309
  When the DIFF section below has an UPSTREAM CONTEXT part, that code arrived
282
310
  through a merge and was not authored by this pull request; do not raise a
283
311
  finding located only there, and never mark blocking a finding whose only
284
312
  support is upstream context.
285
313
  ${CODEGRAPH_DIFF_VERIFICATION}
286
- Each finding must use this exact structure, keeping the default finding under
287
- 120 words excluding an optional diagram and an optional suggestion:
288
-
289
- ### [P1 · blocking] \`path/to/file.ts\` — \`symbol()\`
290
- Location: \`path/to/file.ts:42\`
291
-
292
- One sentence describing what is wrong and its impact.
293
-
294
- Add one short evidence paragraph explaining the mechanism or reproduction.
295
- Do not add labels such as Mechanism, Symptom, Scenario, Verified, Repro,
296
- Options, or Scope unless that detail is necessary to understand a complex
297
- finding. End every finding with this exact sentence on its own line:
298
- "If you'd like me to explain it in more detail, please ask." No finding may
299
- omit it and nothing may follow it.
300
-
301
- Use P0-P3 severity and exactly either "blocking" or "non-blocking".
302
- Keep the Location line machine-readable; it is removed from user-facing
303
- review copies. Use Markdown backticks around paths and symbols.
314
+ Keep the default finding's "body" under 120 words excluding an optional
315
+ suggestion. Write prose only: do not emit Markdown headings, a Location line,
316
+ backticks around the path, or the sentence "If you'd like me to explain it in
317
+ more detail, please ask." Our code renders the heading, the location, and the
318
+ suggestion from your JSON fields. Use P0-P3 severity, and true for blocking.
319
+ ${REVIEW_COPY_RULE}
304
320
  Every diff line in the DIFF section starts with its line number in the new
305
- file. Copy Location numbers from that column instead of counting from the @@
306
- header, and use \`path:from-to\` when the finding spans several lines. Removed
307
- lines have no number, so anchor a finding about removed code to the nearest
308
- numbered line.
309
-
310
- Include one GitHub suggestion when the fix is a direct replacement of
311
- consecutive numbered lines from a single hunk of the same file and you are
312
- confident in the exact replacement text — this is the common case for
313
- single-line and small multi-line fixes. Keep Location as the full span of
314
- the problem, and put the suggestion right before the closing sentence:
315
-
316
- Suggestion: \`path/to/file.ts:42-43\`
317
- \`\`\`suggestion
318
- every line of 42-43 as it should read, with its original indentation
319
- \`\`\`
320
-
321
- The Suggestion range must sit inside the Location range and cover only the
322
- lines the fix changes. The block replaces that whole range, so write every
323
- line of it, not just the edited part, and write nothing else inside the block.
324
- When the replacement itself contains three backticks, open and close the block
325
- with four. Leave the suggestion out when the fix needs removed lines, another
326
- file, or more than one hunk, or when you are not sure of the exact code.
321
+ file. Copy "lineFrom" and "lineTo" from that column instead of counting from
322
+ the @@ header. Removed lines have no number, so anchor a finding about removed
323
+ code to the nearest numbered line.
327
324
  ${diagrams ? DIAGRAM_PROMPT_RULES : NO_DIAGRAM_RULES}
325
+ ${FINDINGS_JSON_INSTRUCTIONS}
328
326
 
329
327
  REVIEW GUIDE:
330
328
  ${guide}
@@ -355,6 +353,7 @@ ${diff}`;
355
353
  prompt,
356
354
  maxTokens: 24_000 * matrix,
357
355
  reasoningEffort: "high",
356
+ responseFormat: FINDINGS_JSON_SCHEMA,
358
357
  };
359
358
  report(`AI request · model=${options.highModel ?? "openrouter default"} · ` +
360
359
  `prompt=${prompt.length} chars · maxTokens=${request.maxTokens}`);
@@ -371,9 +370,12 @@ ${diff}`;
371
370
  console.log(`\n----- INITIAL REVIEW -----\n${response.text}\n`);
372
371
  }
373
372
  if (!response.text.trim()) {
374
- throw new Error("OpenRouter returned an empty review; the reasoning budget may have been exhausted");
373
+ throw new Error("OpenRouter returned an empty review. The reasoning budget may have been exhausted");
375
374
  }
376
- let reviewText = response.text.trim();
375
+ // The model returns JSON; we render the Markdown (CORE-40 / F02). A reply
376
+ // that is not usable JSON is retried once and then falls back to the legacy
377
+ // Markdown parser, so a provider that ignores the schema still works.
378
+ let reviewText = await normalizeReviewResponse(provider, request, response, extraTools, maxToolRounds, usage, report, options);
377
379
  for (let pass = 2; pass <= matrix; pass++) {
378
380
  const improvementRequest = {
379
381
  ...request,
@@ -381,13 +383,13 @@ ${diff}`;
381
383
  prompt: `Audit the draft review below against the complete pull-request diff
382
384
  and the supplied review guides. Preserve valid findings, correct inaccurate ones,
383
385
  remove duplicate or unsupported ones, and add every missing actionable finding.
384
- Do not stop early and do not invent requirements. Keep the existing finding
385
- structure unchanged. Return only the complete revised review in the same format.
386
+ Do not stop early and do not invent requirements. Return the complete revised
387
+ review as the same JSON object the instructions above describe, and nothing else.
386
388
 
387
389
  ORIGINAL REVIEW CONTEXT:
388
390
  ${prompt}
389
391
 
390
- DRAFT REVIEW:
392
+ DRAFT REVIEW (Markdown rendering of the previous JSON answer):
391
393
  ${reviewText}`,
392
394
  };
393
395
  if (options.debug) {
@@ -402,7 +404,7 @@ ${reviewText}`,
402
404
  if (!response.text.trim()) {
403
405
  throw new Error(`OpenRouter returned an empty review improvement at pass ${pass - 1}`);
404
406
  }
405
- reviewText = response.text.trim();
407
+ reviewText = await normalizeReviewResponse(provider, improvementRequest, response, extraTools, maxToolRounds, usage, report, options);
406
408
  if (options.debug) {
407
409
  console.log(`[debug] improvement ${pass - 1} response · input=${response.tokensIn} tokens · output=${response.tokensOut} tokens`);
408
410
  console.log(`\n----- IMPROVED REVIEW ${pass - 1} -----\n${reviewText}\n`);
@@ -411,16 +413,14 @@ ${reviewText}`,
411
413
  const visiblePaths = ownFiles.map(({ path }) => path);
412
414
  return {
413
415
  ...response,
414
- text: `## Severity
415
-
416
- - P0 — Critical: production outage, data loss, or security issue.
417
- - P1 — High: major behavior is broken and should be fixed before merge.
418
- - P2 — Medium: important correctness or maintainability issue.
419
- - P3 — Low: minor, non-blocking improvement or edge case.
420
-
421
- ${reviewText}`,
416
+ text: `${severitySection()}\n${reviewText}`,
422
417
  visiblePaths,
423
418
  guideBuiltAt: guides.guideBuiltAt,
419
+ codegraphState: !options.useCodegraph
420
+ ? "disabled"
421
+ : codegraphTools.length > 0
422
+ ? "used"
423
+ : "unavailable",
424
424
  };
425
425
  }
426
426
  function revisionToGithubFiles(revision) {
@@ -441,7 +441,7 @@ export async function reviewWorkspaceRevision(revision, options, headSha, usage,
441
441
  const { shortGuide, detailed, codebase, skill } = guides;
442
442
  const guide = shortGuide || skill;
443
443
  if (!guide) {
444
- throw new Error(`repos/${options.repo}/PR_REVIEW_GUIDE.md was not found; run init first`);
444
+ throw new Error(`repos/${options.repo}/PR_REVIEW_GUIDE.md was not found. Run init first`);
445
445
  }
446
446
  const files = revisionToGithubFiles(revision);
447
447
  const unchanged = new Set(extras?.unchangedPaths ?? []);
@@ -463,7 +463,7 @@ export async function reviewWorkspaceRevision(revision, options, headSha, usage,
463
463
  return `FILE: ${path}\n${filePatch(file)}`;
464
464
  patchByPath.set(path, patch);
465
465
  const description = await summarizeDiff(path, patch, lowProvider, usage);
466
- return `FILE: ${path}\n[${changes} changed lines — summarized below]\n${description}`;
466
+ return `FILE: ${path}\n[${changes} changed lines, summarized below]\n${description}`;
467
467
  }));
468
468
  const codegraphTools = options.useCodegraph
469
469
  ? extras?.prepareCodegraphTools
@@ -497,13 +497,12 @@ export async function reviewWorkspaceRevision(revision, options, headSha, usage,
497
497
  const prompt = `Review these local changes against the repository's review guide and
498
498
  codebase conventions. Find only actionable code-level violations supported by
499
499
  the diff and either the guide or the codebase conventions.
500
- Return concise Markdown with either "## Findings" and findings, or
501
- "## Findings\\n\\nNo actionable findings."
502
500
  Do not invent low-value findings. If the diff contains a truncation marker or a
503
501
  file is summarized, limit claims to the supplied text and use read-full-diff or
504
502
  codegraph tools before asserting behavior outside what was shown.
505
503
  ${CODEGRAPH_DIFF_VERIFICATION}
506
504
  ${diagrams ? DIAGRAM_PROMPT_RULES : NO_DIAGRAM_RULES}
505
+ ${FINDINGS_JSON_INSTRUCTIONS}
507
506
 
508
507
  REVIEW GUIDE:
509
508
  ${guide}
@@ -529,6 +528,7 @@ ${ownDiff}${unchangedListing}`;
529
528
  prompt,
530
529
  maxTokens: 24_000 * matrix,
531
530
  reasoningEffort: "high",
531
+ responseFormat: FINDINGS_JSON_SCHEMA,
532
532
  };
533
533
  report(`AI request · prompt=${prompt.length} chars`);
534
534
  let response = await completeWithMermaidTools(provider, request, 1, extraTools, maxToolRounds);
@@ -537,7 +537,7 @@ ${ownDiff}${unchangedListing}`;
537
537
  if (!response.text.trim()) {
538
538
  throw new Error("OpenRouter returned an empty review");
539
539
  }
540
- let reviewText = response.text.trim();
540
+ let reviewText = await normalizeReviewResponse(provider, request, response, extraTools, maxToolRounds, usage, report, options);
541
541
  for (let pass = 2; pass <= matrix; pass++) {
542
542
  const improvementRequest = {
543
543
  ...request,
@@ -547,18 +547,18 @@ Preserve valid findings, correct inaccurate ones, remove duplicate or
547
547
  unsupported ones, and add every missing actionable finding. Re-check each
548
548
  finding with codegraph when it depends on behavior outside the diff; remove
549
549
  findings that only looked plausible from the diff slice.
550
- Return only the complete revised review in the same format.
550
+ Return the complete revised review as the same JSON object, and nothing else.
551
551
 
552
552
  ORIGINAL REVIEW CONTEXT:
553
553
  ${prompt}
554
554
 
555
- DRAFT REVIEW:
555
+ DRAFT REVIEW (Markdown rendering of the previous JSON answer):
556
556
  ${reviewText}`,
557
557
  };
558
558
  response = await completeWithMermaidTools(provider, improvementRequest, 1, extraTools, maxToolRounds);
559
559
  if (usage)
560
560
  await usage(response);
561
- reviewText = response.text.trim();
561
+ reviewText = await normalizeReviewResponse(provider, improvementRequest, response, extraTools, maxToolRounds, usage, report, options);
562
562
  }
563
563
  const codegraphState = !options.useCodegraph
564
564
  ? "disabled"
@@ -567,14 +567,7 @@ ${reviewText}`,
567
567
  : "unavailable";
568
568
  return {
569
569
  ...response,
570
- text: `## Severity
571
-
572
- - P0 — Critical: production outage, data loss, or security issue.
573
- - P1 — High: major behavior is broken and should be fixed before merge.
574
- - P2 — Medium: important correctness or maintainability issue.
575
- - P3 — Low: minor, non-blocking improvement or edge case.
576
-
577
- ${reviewText}`,
570
+ text: `${severitySection()}\n${reviewText}`,
578
571
  visiblePaths: ownFiles.map(({ path }) => path),
579
572
  guideBuiltAt: guides.guideBuiltAt,
580
573
  codegraphState,