codecartographer-pi 0.23.0 → 0.24.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +44 -10
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +5 -2
- package/agent-skill/codecartographer/references/broadside.md +5 -1
- package/dist/core/amendment.d.ts +6 -3
- package/dist/core/amendment.js +21 -10
- package/dist/core/broadside-verify.d.ts +95 -0
- package/dist/core/broadside-verify.js +433 -0
- package/dist/core/broadside.d.ts +45 -10
- package/dist/core/broadside.js +61 -16
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/status.d.ts +14 -0
- package/dist/core/status.js +104 -13
- package/dist/extensions/codecarto/agent-state.d.ts +7 -1
- package/dist/extensions/codecarto/agent-state.js +9 -1
- package/dist/extensions/codecarto/auto-runner.js +3 -2
- package/dist/extensions/codecarto/broadside-flags.d.ts +4 -2
- package/dist/extensions/codecarto/broadside-flags.js +24 -6
- package/dist/extensions/codecarto/index.js +65 -7
- package/dist/mcp-server/server.d.ts +2 -1
- package/dist/mcp-server/server.js +50 -9
- package/package.json +1 -1
|
@@ -0,0 +1,433 @@
|
|
|
1
|
+
// Broad-Side verification pass (#143): the hybrid the roadmap kept coming
|
|
2
|
+
// back to. The batch sweep is cheap and reads files without being able to
|
|
3
|
+
// look anything up, and its measured weakness is precision, not coverage —
|
|
4
|
+
// on this repository the top twelve findings by severity were two real
|
|
5
|
+
// defects and ten claims that a look at the guard, the caller, or the
|
|
6
|
+
// tsconfig would have dismissed. So after collect, one sync-priced call per
|
|
7
|
+
// finding, with three read-only tools confined to the repository, reads the
|
|
8
|
+
// cited code and says whether the failure is reachable.
|
|
9
|
+
//
|
|
10
|
+
// Measured on 2026-09-13 (the #143 comparison run): twelve findings, the
|
|
11
|
+
// rubric below, `google/gemini-3.7-flash` at low reasoning effort —
|
|
12
|
+
// twelve-for-twelve agreement with a reviewer's ground truth, precision of
|
|
13
|
+
// the confirmed set from 17% to 100%, both true findings kept, $0.13 in all
|
|
14
|
+
// (about a cent a finding). The rubric mattered more than the model: a first
|
|
15
|
+
// draft without the `not-a-defect` verdict "confirmed" two type casts that
|
|
16
|
+
// every caller satisfies, because they were literally true of the code.
|
|
17
|
+
//
|
|
18
|
+
// Read-only on purpose. A tool-using pass that could mutate would need the
|
|
19
|
+
// headless-agent retry rule (retry only before the first tool call); one
|
|
20
|
+
// that only reads keeps the batch property that re-running is always safe.
|
|
21
|
+
import { readdir, readFile, stat, writeFile } from "node:fs/promises";
|
|
22
|
+
import { isAbsolute, join, relative, resolve } from "node:path";
|
|
23
|
+
import { BROADSIDE_DIR, BROADSIDE_LENS_IDS, broadsideDirFor, defaultReasoningFor, isSlurpable, listRepoFiles, loadBroadsideState, loadSavedLensResults, parseLensJson, persistBroadsideRunMerging, } from "./broadside.js";
|
|
24
|
+
export const BROADSIDE_CHAT_URL = "https://openrouter.ai/api/v1/chat/completions";
|
|
25
|
+
/** How many findings `verify` reads by default, most severe first. */
|
|
26
|
+
export const BROADSIDE_VERIFY_DEFAULT_TOP = 10;
|
|
27
|
+
/** Tool calls one finding may spend before it must answer. */
|
|
28
|
+
export const BROADSIDE_VERIFY_MAX_TOOL_CALLS = 8;
|
|
29
|
+
/** The lenses whose findings carry a file:line and a claim to check. */
|
|
30
|
+
export const BROADSIDE_VERIFIABLE_LENSES = ["defect", "security"];
|
|
31
|
+
const READ_FILE_MAX_LINES = 200;
|
|
32
|
+
const GREP_MAX_MATCHES = 40;
|
|
33
|
+
const GREP_MAX_FILE_BYTES = 1_000_000;
|
|
34
|
+
const TOOL_OUTPUT_MAX_CHARS = 12_000;
|
|
35
|
+
const SEVERITY_RANK = { critical: 0, high: 1, medium: 2, low: 3 };
|
|
36
|
+
/** The sync-priced id behind a `:batch` model id (`vendor/name:batch` → `vendor/name`). */
|
|
37
|
+
export function syncModelFor(batchModel) {
|
|
38
|
+
return batchModel.replace(/:batch$/, "");
|
|
39
|
+
}
|
|
40
|
+
/** The findings a run's saved lens results carry, most severe first. */
|
|
41
|
+
export function rankVerifiableFindings(stored) {
|
|
42
|
+
const out = [];
|
|
43
|
+
for (const result of stored) {
|
|
44
|
+
if (!BROADSIDE_VERIFIABLE_LENSES.includes(result.lensId) || result.truncated)
|
|
45
|
+
continue;
|
|
46
|
+
const parsed = parseLensJson(result.content);
|
|
47
|
+
for (const finding of parsed?.findings ?? []) {
|
|
48
|
+
const location = typeof finding.location === "string" ? finding.location : "";
|
|
49
|
+
const title = typeof finding.title === "string" ? finding.title : "";
|
|
50
|
+
if (!location || !title)
|
|
51
|
+
continue;
|
|
52
|
+
out.push({
|
|
53
|
+
lensId: result.lensId,
|
|
54
|
+
customId: result.customId,
|
|
55
|
+
severity: typeof finding.severity === "string" ? finding.severity.toLowerCase() : "unknown",
|
|
56
|
+
title,
|
|
57
|
+
location,
|
|
58
|
+
description: typeof finding.description === "string" ? finding.description : "",
|
|
59
|
+
pattern: typeof finding.pattern === "string" ? finding.pattern : typeof finding.category === "string" ? finding.category : "",
|
|
60
|
+
});
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
// Stable: severity, then lens order, then the order the lens listed them.
|
|
64
|
+
return out
|
|
65
|
+
.map((finding, order) => ({ finding, order }))
|
|
66
|
+
.sort((a, b) => (SEVERITY_RANK[a.finding.severity] ?? 9) - (SEVERITY_RANK[b.finding.severity] ?? 9)
|
|
67
|
+
|| BROADSIDE_LENS_IDS.indexOf(a.finding.lensId) - BROADSIDE_LENS_IDS.indexOf(b.finding.lensId)
|
|
68
|
+
|| a.order - b.order)
|
|
69
|
+
.map(({ finding }) => finding);
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Three read-only tools over the repository's own file list — the same
|
|
73
|
+
* listing the lenses scan (tracked and untracked, ignore rules applied) minus
|
|
74
|
+
* everything {@link isSlurpable} keeps out of a lens: credential stores, build
|
|
75
|
+
* output, binaries. A path outside the repository, or one the listing does
|
|
76
|
+
* not contain, is an error the model sees, not a read.
|
|
77
|
+
*/
|
|
78
|
+
export async function createRepoReader(cwd) {
|
|
79
|
+
const { files } = await listRepoFiles(cwd);
|
|
80
|
+
const readable = new Set(files.filter(isSlurpable));
|
|
81
|
+
const confine = (path) => {
|
|
82
|
+
const abs = resolve(cwd, path);
|
|
83
|
+
const rel = relative(cwd, abs).split("\\").join("/");
|
|
84
|
+
if (!rel || rel.startsWith("..") || isAbsolute(rel))
|
|
85
|
+
throw new Error(`path is outside the repository: ${path}`);
|
|
86
|
+
return rel;
|
|
87
|
+
};
|
|
88
|
+
return {
|
|
89
|
+
async readFile(path, startLine, endLine) {
|
|
90
|
+
const rel = confine(path);
|
|
91
|
+
if (!readable.has(rel))
|
|
92
|
+
return `error: ${rel} is not a readable source file of this repository`;
|
|
93
|
+
const lines = (await readFile(join(cwd, rel), "utf8")).split("\n");
|
|
94
|
+
const start = Math.max(1, Math.floor(Number(startLine ?? 1)) || 1);
|
|
95
|
+
const requestedEnd = Math.floor(Number(endLine ?? start + READ_FILE_MAX_LINES - 1)) || start;
|
|
96
|
+
const end = Math.min(lines.length, requestedEnd, start + READ_FILE_MAX_LINES - 1);
|
|
97
|
+
if (start > lines.length)
|
|
98
|
+
return `error: ${rel} has ${lines.length} lines`;
|
|
99
|
+
return lines.slice(start - 1, end).map((line, i) => `${start + i}: ${line}`).join("\n");
|
|
100
|
+
},
|
|
101
|
+
async grep(pattern, pathPrefix) {
|
|
102
|
+
let regex;
|
|
103
|
+
try {
|
|
104
|
+
regex = new RegExp(pattern);
|
|
105
|
+
}
|
|
106
|
+
catch (error) {
|
|
107
|
+
return `error: invalid pattern (${error instanceof Error ? error.message : String(error)})`;
|
|
108
|
+
}
|
|
109
|
+
const prefix = pathPrefix ? confine(pathPrefix) : "";
|
|
110
|
+
const matches = [];
|
|
111
|
+
for (const rel of readable) {
|
|
112
|
+
if (prefix && rel !== prefix && !rel.startsWith(`${prefix}/`) && !rel.startsWith(prefix))
|
|
113
|
+
continue;
|
|
114
|
+
let text;
|
|
115
|
+
try {
|
|
116
|
+
if ((await stat(join(cwd, rel))).size > GREP_MAX_FILE_BYTES)
|
|
117
|
+
continue;
|
|
118
|
+
text = await readFile(join(cwd, rel), "utf8");
|
|
119
|
+
}
|
|
120
|
+
catch {
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
const lines = text.split("\n");
|
|
124
|
+
for (let i = 0; i < lines.length && matches.length < GREP_MAX_MATCHES; i++) {
|
|
125
|
+
if (regex.test(lines[i]))
|
|
126
|
+
matches.push(`${rel}:${i + 1}: ${lines[i]}`);
|
|
127
|
+
}
|
|
128
|
+
if (matches.length >= GREP_MAX_MATCHES)
|
|
129
|
+
break;
|
|
130
|
+
}
|
|
131
|
+
return matches.length > 0 ? matches.join("\n") : "(no matches)";
|
|
132
|
+
},
|
|
133
|
+
async listDir(path) {
|
|
134
|
+
const rel = path === "." || path === "" ? "" : confine(path);
|
|
135
|
+
try {
|
|
136
|
+
const entries = await readdir(join(cwd, rel), { withFileTypes: true });
|
|
137
|
+
return entries
|
|
138
|
+
.filter((entry) => {
|
|
139
|
+
const child = rel ? `${rel}/${entry.name}` : entry.name;
|
|
140
|
+
return entry.isDirectory() ? [...readable].some((f) => f.startsWith(`${child}/`)) : readable.has(child);
|
|
141
|
+
})
|
|
142
|
+
.map((entry) => (entry.isDirectory() ? `${entry.name}/` : entry.name))
|
|
143
|
+
.sort()
|
|
144
|
+
.join("\n") || "(empty)";
|
|
145
|
+
}
|
|
146
|
+
catch (error) {
|
|
147
|
+
return `error: ${error instanceof Error ? error.message : String(error)}`;
|
|
148
|
+
}
|
|
149
|
+
},
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
const TOOLS = [
|
|
153
|
+
{ type: "function", function: { name: "read_file", description: "Read a range of lines (1-based, inclusive) from a source file of the repository; at most 200 lines per call.", parameters: { type: "object", properties: { path: { type: "string" }, start_line: { type: "integer" }, end_line: { type: "integer" } }, required: ["path"] } } },
|
|
154
|
+
{ type: "function", function: { name: "grep", description: "Search the repository's source files for a regular expression; returns at most 40 matching lines as path:line: text.", parameters: { type: "object", properties: { pattern: { type: "string" }, path_prefix: { type: "string", description: "optional directory or file to search under" } }, required: ["pattern"] } } },
|
|
155
|
+
{ type: "function", function: { name: "list_dir", description: "List a directory of the repository.", parameters: { type: "object", properties: { path: { type: "string" } }, required: ["path"] } } },
|
|
156
|
+
];
|
|
157
|
+
const VERDICT_SCHEMA = {
|
|
158
|
+
name: "broadside_verification",
|
|
159
|
+
strict: true,
|
|
160
|
+
schema: {
|
|
161
|
+
type: "object",
|
|
162
|
+
properties: {
|
|
163
|
+
verdict: { type: "string", enum: ["confirmed", "not-a-defect", "discarded", "unclear"] },
|
|
164
|
+
confidence: { type: "string", enum: ["high", "medium", "low"] },
|
|
165
|
+
evidence: {
|
|
166
|
+
type: "array",
|
|
167
|
+
items: { type: "object", properties: { file: { type: "string" }, lines: { type: "string" }, note: { type: "string" } }, required: ["file", "lines", "note"], additionalProperties: false },
|
|
168
|
+
},
|
|
169
|
+
reasoning: { type: "string" },
|
|
170
|
+
},
|
|
171
|
+
required: ["verdict", "confidence", "evidence", "reasoning"],
|
|
172
|
+
additionalProperties: false,
|
|
173
|
+
},
|
|
174
|
+
};
|
|
175
|
+
/**
|
|
176
|
+
* The rubric. The order is deliberate — it is the order a reviewer settles a
|
|
177
|
+
* claim in — and `not-a-defect` is the verdict that separates "the code does
|
|
178
|
+
* what the claim says" from "and that is a bug": without it, a cast every
|
|
179
|
+
* caller satisfies gets confirmed because it is literally there.
|
|
180
|
+
*/
|
|
181
|
+
export const BROADSIDE_VERIFY_SYSTEM_PROMPT = "You verify a scouting finding produced by a one-shot batch scan against the real source code. " +
|
|
182
|
+
"The scan saw files without being able to look anything up; you can. Use the tools to read the cited location " +
|
|
183
|
+
"and whatever else the claim depends on (callers, the definition of a helper, error handling around it). " +
|
|
184
|
+
"Then decide, in this order: " +
|
|
185
|
+
"'discarded' — the claim is wrong about the code (the guard exists, the value cannot be what the claim assumes, " +
|
|
186
|
+
"the cited line does something else, the condition it fears is ruled out by the project's config or runtime); " +
|
|
187
|
+
"'not-a-defect' — the claim is literally true of the code but no caller, input, or state can reach the failure it " +
|
|
188
|
+
"describes: a TypeScript cast every caller satisfies, a hypothetical about an environment the project does not " +
|
|
189
|
+
"target, a style or type-hygiene observation; " +
|
|
190
|
+
"'confirmed' — the failure is reachable: name the concrete input, call site, or sequence that triggers it, and what " +
|
|
191
|
+
"then goes wrong; " +
|
|
192
|
+
"'unclear' — settling it needs runtime behaviour or specification knowledge the code does not contain. " +
|
|
193
|
+
"Be strict: a real but different problem than the one claimed is 'discarded' with the difference noted, and " +
|
|
194
|
+
"'confirmed' without a trigger you found in the code is not allowed. " +
|
|
195
|
+
"Cite line ranges you actually read. When you are done, reply with only the JSON verdict object.";
|
|
196
|
+
function findingPrompt(index, finding) {
|
|
197
|
+
return (`Finding ${index} (${finding.lensId} lens, severity ${finding.severity}):\n` +
|
|
198
|
+
`Title: ${finding.title}\nLocation: ${finding.location}\nPattern/category: ${finding.pattern || "-"}\n` +
|
|
199
|
+
`Description: ${finding.description}\n\n` +
|
|
200
|
+
"Verify it. Reply with a JSON object {verdict, confidence, evidence:[{file,lines,note}], reasoning}.");
|
|
201
|
+
}
|
|
202
|
+
async function chat(fetcher, apiKey, model, body) {
|
|
203
|
+
const resp = await fetcher(BROADSIDE_CHAT_URL, {
|
|
204
|
+
method: "POST",
|
|
205
|
+
headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json" },
|
|
206
|
+
body: JSON.stringify({ model, reasoning: defaultReasoningFor(), usage: { include: true }, ...body }),
|
|
207
|
+
signal: AbortSignal.timeout(120_000),
|
|
208
|
+
});
|
|
209
|
+
const data = (await resp.json());
|
|
210
|
+
if (!resp.ok || data.error) {
|
|
211
|
+
const detail = data.error?.message ?? JSON.stringify(data.error ?? data).slice(0, 300);
|
|
212
|
+
throw new Error(`OpenRouter chat: HTTP ${resp.status}: ${detail}`);
|
|
213
|
+
}
|
|
214
|
+
return data;
|
|
215
|
+
}
|
|
216
|
+
function parseVerdict(text) {
|
|
217
|
+
const trimmed = text.trim();
|
|
218
|
+
const fenced = /```(?:json)?\s*([\s\S]*?)```/.exec(trimmed);
|
|
219
|
+
try {
|
|
220
|
+
const parsed = JSON.parse(fenced ? fenced[1] : trimmed);
|
|
221
|
+
return typeof parsed?.verdict === "string" ? parsed : null;
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
return null;
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
/** Verify one finding: up to the tool budget, then a verdict. */
|
|
228
|
+
export async function verifyFinding(finding, index, reader, apiKey, model, fetcher) {
|
|
229
|
+
const messages = [
|
|
230
|
+
{ role: "system", content: BROADSIDE_VERIFY_SYSTEM_PROMPT },
|
|
231
|
+
{ role: "user", content: findingPrompt(index, finding) },
|
|
232
|
+
];
|
|
233
|
+
let toolCalls = 0;
|
|
234
|
+
let cost = 0;
|
|
235
|
+
const usageCost = (data) => {
|
|
236
|
+
const usage = data.usage;
|
|
237
|
+
return typeof usage?.cost === "number" ? usage.cost : 0;
|
|
238
|
+
};
|
|
239
|
+
try {
|
|
240
|
+
// The +2 leaves room for the final schema-forced call after the budget.
|
|
241
|
+
for (let step = 0; step < BROADSIDE_VERIFY_MAX_TOOL_CALLS + 2; step++) {
|
|
242
|
+
const budgetSpent = toolCalls >= BROADSIDE_VERIFY_MAX_TOOL_CALLS;
|
|
243
|
+
if (budgetSpent)
|
|
244
|
+
messages.push({ role: "user", content: "Tool budget spent. Emit the verdict JSON object now from what you have read." });
|
|
245
|
+
const data = await chat(fetcher, apiKey, model, budgetSpent
|
|
246
|
+
? { messages, response_format: { type: "json_schema", json_schema: VERDICT_SCHEMA }, max_tokens: 2000 }
|
|
247
|
+
: { messages, tools: TOOLS, tool_choice: "auto", max_tokens: 4000 });
|
|
248
|
+
cost += usageCost(data);
|
|
249
|
+
const choice = data.choices?.[0];
|
|
250
|
+
const message = choice?.message ?? { role: "assistant", content: "" };
|
|
251
|
+
messages.push(message);
|
|
252
|
+
const calls = message.tool_calls;
|
|
253
|
+
if (calls && calls.length > 0 && !budgetSpent) {
|
|
254
|
+
for (const call of calls) {
|
|
255
|
+
toolCalls += 1;
|
|
256
|
+
let args = {};
|
|
257
|
+
try {
|
|
258
|
+
args = JSON.parse(call.function.arguments || "{}");
|
|
259
|
+
}
|
|
260
|
+
catch {
|
|
261
|
+
// Malformed arguments: the model sees the error and can retry.
|
|
262
|
+
}
|
|
263
|
+
let output;
|
|
264
|
+
try {
|
|
265
|
+
output = call.function.name === "read_file"
|
|
266
|
+
? await reader.readFile(String(args.path ?? ""), args.start_line, args.end_line)
|
|
267
|
+
: call.function.name === "grep"
|
|
268
|
+
? await reader.grep(String(args.pattern ?? ""), typeof args.path_prefix === "string" ? args.path_prefix : undefined)
|
|
269
|
+
: call.function.name === "list_dir"
|
|
270
|
+
? await reader.listDir(String(args.path ?? "."))
|
|
271
|
+
: `error: unknown tool ${call.function.name}`;
|
|
272
|
+
}
|
|
273
|
+
catch (error) {
|
|
274
|
+
output = `error: ${error instanceof Error ? error.message : String(error)}`;
|
|
275
|
+
}
|
|
276
|
+
messages.push({ role: "tool", tool_call_id: call.id, content: output.slice(0, TOOL_OUTPUT_MAX_CHARS) });
|
|
277
|
+
}
|
|
278
|
+
continue;
|
|
279
|
+
}
|
|
280
|
+
const verdict = parseVerdict(typeof message.content === "string" ? message.content : "");
|
|
281
|
+
if (verdict)
|
|
282
|
+
return finish(verdict, toolCalls, cost);
|
|
283
|
+
if (budgetSpent)
|
|
284
|
+
break;
|
|
285
|
+
// Prose instead of JSON: one schema-forced call, no tools.
|
|
286
|
+
const final = await chat(fetcher, apiKey, model, {
|
|
287
|
+
messages: [...messages, { role: "user", content: "Emit the verdict JSON object now." }],
|
|
288
|
+
response_format: { type: "json_schema", json_schema: VERDICT_SCHEMA },
|
|
289
|
+
max_tokens: 2000,
|
|
290
|
+
});
|
|
291
|
+
cost += usageCost(final);
|
|
292
|
+
const forced = parseVerdict(String((final.choices?.[0]?.message?.content) ?? ""));
|
|
293
|
+
if (forced)
|
|
294
|
+
return finish(forced, toolCalls, cost);
|
|
295
|
+
break;
|
|
296
|
+
}
|
|
297
|
+
return { verdict: "error", confidence: "low", evidence: [], reasoning: "no verdict within the tool budget", toolCalls, cost };
|
|
298
|
+
}
|
|
299
|
+
catch (error) {
|
|
300
|
+
return { verdict: "error", confidence: "low", evidence: [], reasoning: error instanceof Error ? error.message : String(error), toolCalls, cost };
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
function finish(verdict, toolCalls, cost) {
|
|
304
|
+
const known = ["confirmed", "not-a-defect", "discarded", "unclear"];
|
|
305
|
+
const value = String(verdict.verdict);
|
|
306
|
+
const evidence = Array.isArray(verdict.evidence)
|
|
307
|
+
? verdict.evidence.map((e) => ({ file: String(e.file ?? ""), lines: String(e.lines ?? ""), note: String(e.note ?? "") }))
|
|
308
|
+
: [];
|
|
309
|
+
return {
|
|
310
|
+
verdict: (known.includes(value) ? value : "unclear"),
|
|
311
|
+
confidence: typeof verdict.confidence === "string" ? verdict.confidence : "low",
|
|
312
|
+
evidence,
|
|
313
|
+
reasoning: typeof verdict.reasoning === "string" ? verdict.reasoning : "",
|
|
314
|
+
toolCalls,
|
|
315
|
+
cost,
|
|
316
|
+
};
|
|
317
|
+
}
|
|
318
|
+
/**
|
|
319
|
+
* Verify the top findings of a collected run against the repository.
|
|
320
|
+
*
|
|
321
|
+
* `maxCost` is a running cap, not a pre-flight estimate: a sync call's cost
|
|
322
|
+
* is only known when it returns, so the pass stops *before* starting the next
|
|
323
|
+
* finding once the cap is reached and reports `partial`. On the default
|
|
324
|
+
* model a finding costs about a cent.
|
|
325
|
+
*/
|
|
326
|
+
export async function runBroadsideVerify(cwd, apiKey, opts = {}) {
|
|
327
|
+
const broadsideDir = broadsideDirFor(cwd);
|
|
328
|
+
const state = await loadBroadsideState(broadsideDir);
|
|
329
|
+
const run = opts.runId ? state.runs.find((candidate) => candidate.id === opts.runId) : state.runs[state.runs.length - 1];
|
|
330
|
+
if (!run) {
|
|
331
|
+
throw new Error(opts.runId ? `No Broad-Side run with id ${opts.runId}.` : "No Broad-Side run recorded. Submit and collect one first.");
|
|
332
|
+
}
|
|
333
|
+
const runDir = join(broadsideDir, run.outputDir);
|
|
334
|
+
const stored = await loadSavedLensResults(runDir, run.lenses);
|
|
335
|
+
const ranked = rankVerifiableFindings(stored);
|
|
336
|
+
if (ranked.length === 0) {
|
|
337
|
+
throw new Error(`Run ${run.id} has no verifiable findings on disk: the defect and security lenses either did not run, are not collected yet, or found nothing. Collect the run first.`);
|
|
338
|
+
}
|
|
339
|
+
const top = Math.max(1, Math.floor(opts.top ?? BROADSIDE_VERIFY_DEFAULT_TOP));
|
|
340
|
+
// The run's model is a `:batch` variant; the sync endpoint wants the base id.
|
|
341
|
+
const model = opts.model ?? syncModelFor(run.model);
|
|
342
|
+
const maxCost = opts.maxCost ?? 0;
|
|
343
|
+
const fetcher = opts.fetcher ?? fetch;
|
|
344
|
+
const reader = await createRepoReader(cwd);
|
|
345
|
+
const selected = ranked.slice(0, top);
|
|
346
|
+
const findings = [];
|
|
347
|
+
let totalCost = 0;
|
|
348
|
+
let stoppedByCost = false;
|
|
349
|
+
for (const [i, candidate] of selected.entries()) {
|
|
350
|
+
if (opts.signal?.aborted)
|
|
351
|
+
break;
|
|
352
|
+
if (maxCost > 0 && totalCost >= maxCost) {
|
|
353
|
+
stoppedByCost = true;
|
|
354
|
+
break;
|
|
355
|
+
}
|
|
356
|
+
const outcome = await verifyFinding(candidate, i + 1, reader, apiKey, model, fetcher);
|
|
357
|
+
const finding = {
|
|
358
|
+
index: i + 1,
|
|
359
|
+
lensId: candidate.lensId,
|
|
360
|
+
customId: candidate.customId,
|
|
361
|
+
severity: candidate.severity,
|
|
362
|
+
title: candidate.title,
|
|
363
|
+
location: candidate.location,
|
|
364
|
+
...outcome,
|
|
365
|
+
};
|
|
366
|
+
findings.push(finding);
|
|
367
|
+
totalCost += outcome.cost;
|
|
368
|
+
opts.onProgress?.(finding);
|
|
369
|
+
}
|
|
370
|
+
const status = findings.length === selected.length ? "completed" : "partial";
|
|
371
|
+
const entry = {
|
|
372
|
+
status,
|
|
373
|
+
model,
|
|
374
|
+
top,
|
|
375
|
+
verified: findings.length,
|
|
376
|
+
confirmed: findings.filter((f) => f.verdict === "confirmed").length,
|
|
377
|
+
cost: totalCost,
|
|
378
|
+
at: new Date().toISOString(),
|
|
379
|
+
};
|
|
380
|
+
const result = {
|
|
381
|
+
runId: run.id,
|
|
382
|
+
outputDir: join(".codecarto", BROADSIDE_DIR, run.id),
|
|
383
|
+
model,
|
|
384
|
+
status,
|
|
385
|
+
candidates: ranked.length,
|
|
386
|
+
findings,
|
|
387
|
+
totalCost,
|
|
388
|
+
...(stoppedByCost && { stoppedByCost: true }),
|
|
389
|
+
};
|
|
390
|
+
await writeFile(join(runDir, "verified.json"), `${JSON.stringify({ ...entry, run_id: run.id, candidates: ranked.length, findings }, null, "\t")}\n`, "utf8");
|
|
391
|
+
await writeFile(join(runDir, "verified.md"), renderVerifiedMarkdown(result), "utf8");
|
|
392
|
+
run.verify = entry;
|
|
393
|
+
await persistBroadsideRunMerging(broadsideDir, run);
|
|
394
|
+
return result;
|
|
395
|
+
}
|
|
396
|
+
const VERDICT_MARK = { confirmed: "✓", "not-a-defect": "–", discarded: "✗", unclear: "?", error: "!" };
|
|
397
|
+
export function renderVerifiedMarkdown(result) {
|
|
398
|
+
const lines = [
|
|
399
|
+
`# Verified findings — run ${result.runId}`,
|
|
400
|
+
"",
|
|
401
|
+
`${result.findings.length} of ${result.candidates} verifiable finding(s) read against the source on \`${result.model}\` (most severe first), $${result.totalCost.toFixed(4)}.` +
|
|
402
|
+
(result.stoppedByCost ? " Stopped by the cost cap before the rest." : ""),
|
|
403
|
+
"",
|
|
404
|
+
"A **confirmed** finding names the input, call site, or sequence that reaches the failure. **not-a-defect** means the claim is",
|
|
405
|
+
"literally true of the code but nothing can reach the failure it describes; **discarded** means the claim is wrong about the",
|
|
406
|
+
"code; **unclear** needs runtime or specification knowledge. Every verdict is still a model's reading — a confirmed finding",
|
|
407
|
+
"is a lead worth a human's next look, not a validated claim.",
|
|
408
|
+
"",
|
|
409
|
+
];
|
|
410
|
+
for (const f of result.findings) {
|
|
411
|
+
lines.push(`## ${VERDICT_MARK[f.verdict] ?? "?"} ${f.index}. [${f.severity}] ${f.title}`, "", `- **verdict**: ${f.verdict} (${f.confidence})`, `- **location**: ${f.location}`, `- **lens**: ${f.lensId} (${f.customId})`);
|
|
412
|
+
if (f.evidence.length > 0)
|
|
413
|
+
lines.push(`- **evidence**: ${f.evidence.map((e) => `${e.file}:${e.lines} — ${e.note}`).join("; ")}`);
|
|
414
|
+
lines.push(`- **reasoning**: ${f.reasoning.replace(/\s+/g, " ").trim()}`, `- **cost**: $${f.cost.toFixed(4)} (${f.toolCalls} tool call(s))`, "");
|
|
415
|
+
}
|
|
416
|
+
return lines.join("\n");
|
|
417
|
+
}
|
|
418
|
+
export function verifyResultText(result) {
|
|
419
|
+
const counts = { confirmed: 0, "not-a-defect": 0, discarded: 0, unclear: 0, error: 0 };
|
|
420
|
+
for (const f of result.findings)
|
|
421
|
+
counts[f.verdict] = (counts[f.verdict] ?? 0) + 1;
|
|
422
|
+
const lines = [
|
|
423
|
+
`Broad-Side verify — run ${result.runId}: ${result.status}`,
|
|
424
|
+
` ${result.findings.length} of ${result.candidates} verifiable finding(s) read on ${result.model} | cost: $${result.totalCost.toFixed(4)}` +
|
|
425
|
+
(result.stoppedByCost ? " (stopped by the cost cap)" : ""),
|
|
426
|
+
` confirmed ${counts.confirmed} · not-a-defect ${counts["not-a-defect"]} · discarded ${counts.discarded} · unclear ${counts.unclear}` + (counts.error ? ` · error ${counts.error}` : ""),
|
|
427
|
+
];
|
|
428
|
+
for (const f of result.findings) {
|
|
429
|
+
lines.push(` ${VERDICT_MARK[f.verdict] ?? "?"} [${f.severity}] ${f.title} @ ${f.location} — ${f.verdict}`);
|
|
430
|
+
}
|
|
431
|
+
lines.push(`Details in ${result.outputDir}/verified.md. A confirmed finding is a lead for a human's next look, not a validated claim.`);
|
|
432
|
+
return lines.join("\n");
|
|
433
|
+
}
|
package/dist/core/broadside.d.ts
CHANGED
|
@@ -244,6 +244,17 @@ export type BroadsideTriageEntry = {
|
|
|
244
244
|
cost?: number;
|
|
245
245
|
error?: string;
|
|
246
246
|
};
|
|
247
|
+
/** Recorded on the run once a verification pass has run (#143); see core/broadside-verify.ts. */
|
|
248
|
+
export type BroadsideVerifyEntry = {
|
|
249
|
+
/** `completed`: every selected finding got a verdict; `partial`: the cost cap or an abort stopped it early. */
|
|
250
|
+
status: "completed" | "partial";
|
|
251
|
+
model: string;
|
|
252
|
+
top: number;
|
|
253
|
+
verified: number;
|
|
254
|
+
confirmed: number;
|
|
255
|
+
cost: number;
|
|
256
|
+
at: string;
|
|
257
|
+
};
|
|
247
258
|
/** The truncation retry pass of one run: one batch per model (#206). */
|
|
248
259
|
export type BroadsideRetryEntry = {
|
|
249
260
|
status: "submitted" | "completed" | "failed";
|
|
@@ -275,6 +286,8 @@ export type BroadsideRun = {
|
|
|
275
286
|
* run cannot both submit it (#322). Absent until a collect claims it.
|
|
276
287
|
*/
|
|
277
288
|
retry?: BroadsideRetryEntry;
|
|
289
|
+
/** The verification pass over the top findings, when one has run (#143). */
|
|
290
|
+
verify?: BroadsideVerifyEntry;
|
|
278
291
|
totalCost?: number;
|
|
279
292
|
pricing?: ModelPricing;
|
|
280
293
|
maxCost?: number;
|
|
@@ -505,12 +518,16 @@ type LensDefinition = {
|
|
|
505
518
|
skipTestFiles?: boolean;
|
|
506
519
|
globsFor: (info: RepoInfo) => string[];
|
|
507
520
|
/**
|
|
508
|
-
* Where to look when `globsFor` matches
|
|
509
|
-
* api lenses target server/, auth, and middleware paths
|
|
510
|
-
* where the trust boundary usually lives; a service whose
|
|
511
|
-
* `src/server.js` matched none of them and got no security
|
|
512
|
-
*
|
|
513
|
-
*
|
|
521
|
+
* Where to look when `globsFor` matches no source file (#319). The
|
|
522
|
+
* security and api lenses target server/, auth, and middleware paths
|
|
523
|
+
* because that is where the trust boundary usually lives; a service whose
|
|
524
|
+
* server is `src/server.js` matched none of them and got no security
|
|
525
|
+
* review at all. A match that is only documents is the same starvation:
|
|
526
|
+
* `SECURITY.md` satisfied the security lens on CodeCartographer itself,
|
|
527
|
+
* which then reviewed a policy and reported zero findings. The fallback
|
|
528
|
+
* is the language's whole source set, added to whatever did match —
|
|
529
|
+
* priced as such, and said so in the estimate, the run record, and the
|
|
530
|
+
* prompt.
|
|
514
531
|
*/
|
|
515
532
|
fallbackGlobsFor?: (info: RepoInfo) => string[];
|
|
516
533
|
systemPrompt: (info: RepoInfo) => string;
|
|
@@ -520,18 +537,36 @@ export declare function getLens(lensId: BroadsideLensId): LensDefinition;
|
|
|
520
537
|
export declare function listLenses(): LensDefinition[];
|
|
521
538
|
/** The languages Broad-Side can scan; anything else is refused at submit. */
|
|
522
539
|
export declare const BROADSIDE_LANGUAGES: readonly ["go", "python", "rust", "typescript", "javascript"];
|
|
540
|
+
/**
|
|
541
|
+
* The files a run scans, and where they came from. Contents are always read
|
|
542
|
+
* from the working tree, so the list is the working tree's too: tracked files
|
|
543
|
+
* plus untracked ones git does not ignore, minus files deleted on disk. The
|
|
544
|
+
* list used to come from `git ls-tree HEAD`, so a run mixed the committed
|
|
545
|
+
* file list with uncommitted contents and never saw an untracked file (#248).
|
|
546
|
+
* A target that is not a git repository gets a bounded walk.
|
|
547
|
+
*/
|
|
548
|
+
export declare function listRepoFiles(targetDir: string): Promise<{
|
|
549
|
+
files: string[];
|
|
550
|
+
snapshot: RepoSnapshotSource;
|
|
551
|
+
}>;
|
|
523
552
|
export declare function collectRepoInfo(targetDir: string, opts?: {
|
|
524
553
|
redact?: boolean;
|
|
525
554
|
}): Promise<RepoInfo>;
|
|
555
|
+
export declare function isSlurpable(relPath: string): boolean;
|
|
526
556
|
type CollectedFile = {
|
|
527
557
|
relPath: string;
|
|
528
558
|
moduleName: string;
|
|
529
559
|
};
|
|
530
560
|
/**
|
|
531
|
-
* The files a lens will read: its targeted globs, or — when those match
|
|
532
|
-
*
|
|
533
|
-
* sentence saying so (#319). The sentence
|
|
534
|
-
* batch entry, and the prompt, so a fallback
|
|
561
|
+
* The files a lens will read: its targeted globs, or — when those match no
|
|
562
|
+
* source file and the lens declares a fallback — the fallback globs on top
|
|
563
|
+
* of whatever did match, with a sentence saying so (#319). The sentence
|
|
564
|
+
* travels to the estimate, the batch entry, and the prompt, so a fallback
|
|
565
|
+
* scan is never a silent one.
|
|
566
|
+
*
|
|
567
|
+
* "No source file" rather than "no file": a policy document or a config
|
|
568
|
+
* file under a targeted path satisfies the globs and leaves the lens with
|
|
569
|
+
* nothing to review, and the coverage note it writes back is the only sign.
|
|
535
570
|
*/
|
|
536
571
|
export declare function selectLensFiles(allFiles: string[], lens: LensDefinition, info: RepoInfo): {
|
|
537
572
|
files: CollectedFile[];
|
package/dist/core/broadside.js
CHANGED
|
@@ -749,7 +749,19 @@ const LENSES = {
|
|
|
749
749
|
"Return a JSON object following the defect_scan_report schema. " +
|
|
750
750
|
"Cite file:line for every finding. List which patterns you checked. " +
|
|
751
751
|
"If the code looks clean for a pattern, say so rather than staying silent. " +
|
|
752
|
-
"Prefer precision over volume — 3 solid findings beat 15 vague ones
|
|
752
|
+
"Prefer precision over volume — 3 solid findings beat 15 vague ones.\n\n" +
|
|
753
|
+
// The verification pass (#143) confirmed 2 of the 12 top findings a
|
|
754
|
+
// scan produced with the paragraph above alone; the other ten were
|
|
755
|
+
// casts and assertions every caller satisfied, guards that lived one
|
|
756
|
+
// call away, or environments the project does not target. The rubric
|
|
757
|
+
// the verifier applies is asked of the scan itself, up front.
|
|
758
|
+
"A finding is a reachable failure: name in the description the concrete input, call site, or sequence " +
|
|
759
|
+
"that reaches it and what then goes wrong. A cast, assertion, `any`, or non-null `!` that every caller " +
|
|
760
|
+
"you can see satisfies, a hypothetical about a runtime or environment the project does not target, or a " +
|
|
761
|
+
"style or type-hygiene observation is not a defect — leave it out, or if it is worth a note, report it " +
|
|
762
|
+
"at severity low under the pattern name `type-hygiene` so it ranks apart from reachable failures. " +
|
|
763
|
+
"When the guard you looked for may live in another module, say which check you could not find " +
|
|
764
|
+
"rather than asserting it is absent; severity high or medium is for failures you traced to a trigger.");
|
|
753
765
|
},
|
|
754
766
|
userPrompt: (info, source, moduleName) => `Scan this ${info.language} module for mechanical defects.\n\n` +
|
|
755
767
|
`Module: ${moduleName}\n\n` +
|
|
@@ -920,7 +932,7 @@ const SOURCE_SPECS = {
|
|
|
920
932
|
* file list with uncommitted contents and never saw an untracked file (#248).
|
|
921
933
|
* A target that is not a git repository gets a bounded walk.
|
|
922
934
|
*/
|
|
923
|
-
async function listRepoFiles(targetDir) {
|
|
935
|
+
export async function listRepoFiles(targetDir) {
|
|
924
936
|
try {
|
|
925
937
|
const listed = await execFileAsync("git", ["-C", targetDir, "ls-files", "-z", "--cached", "--others", "--exclude-standard"], { maxBuffer: 64 * 1024 * 1024, timeout: GIT_TIMEOUT_MS });
|
|
926
938
|
const deleted = await execFileAsync("git", ["-C", targetDir, "ls-files", "-z", "--deleted"], {
|
|
@@ -1200,7 +1212,7 @@ function matchesAnyGlob(path, globs) {
|
|
|
1200
1212
|
}
|
|
1201
1213
|
return false;
|
|
1202
1214
|
}
|
|
1203
|
-
function isSlurpable(relPath) {
|
|
1215
|
+
export function isSlurpable(relPath) {
|
|
1204
1216
|
// A credential store is never a lens input, whatever its globs say (#252).
|
|
1205
1217
|
if (isSecretFile(relPath))
|
|
1206
1218
|
return false;
|
|
@@ -1254,25 +1266,51 @@ function collectFilesMatching(allFiles, lens, globs) {
|
|
|
1254
1266
|
}
|
|
1255
1267
|
return out;
|
|
1256
1268
|
}
|
|
1269
|
+
/** Code in any language Broad-Side scans as, whatever this repo's is. */
|
|
1270
|
+
const SOURCE_EXTENSIONS = new Set(Object.values(SOURCE_SPECS).flatMap((spec) => spec.exts));
|
|
1271
|
+
function isSourceFile(relPath) {
|
|
1272
|
+
const dot = relPath.lastIndexOf(".");
|
|
1273
|
+
return dot > relPath.lastIndexOf("/") && SOURCE_EXTENSIONS.has(relPath.slice(dot).toLowerCase());
|
|
1274
|
+
}
|
|
1275
|
+
/** `a, b, c and 4 more` — a matched-file list short enough for a status line. */
|
|
1276
|
+
function listSome(paths, max = 3) {
|
|
1277
|
+
if (paths.length <= max)
|
|
1278
|
+
return paths.join(", ");
|
|
1279
|
+
return `${paths.slice(0, max).join(", ")} and ${paths.length - max} more`;
|
|
1280
|
+
}
|
|
1257
1281
|
/**
|
|
1258
|
-
* The files a lens will read: its targeted globs, or — when those match
|
|
1259
|
-
*
|
|
1260
|
-
* sentence saying so (#319). The sentence
|
|
1261
|
-
* batch entry, and the prompt, so a fallback
|
|
1282
|
+
* The files a lens will read: its targeted globs, or — when those match no
|
|
1283
|
+
* source file and the lens declares a fallback — the fallback globs on top
|
|
1284
|
+
* of whatever did match, with a sentence saying so (#319). The sentence
|
|
1285
|
+
* travels to the estimate, the batch entry, and the prompt, so a fallback
|
|
1286
|
+
* scan is never a silent one.
|
|
1287
|
+
*
|
|
1288
|
+
* "No source file" rather than "no file": a policy document or a config
|
|
1289
|
+
* file under a targeted path satisfies the globs and leaves the lens with
|
|
1290
|
+
* nothing to review, and the coverage note it writes back is the only sign.
|
|
1262
1291
|
*/
|
|
1263
1292
|
export function selectLensFiles(allFiles, lens, info) {
|
|
1264
1293
|
const globs = lens.globsFor(info).filter(Boolean);
|
|
1265
1294
|
const targeted = collectFilesMatching(allFiles, lens, globs);
|
|
1266
|
-
if (
|
|
1295
|
+
if (globs.length === 0 || !lens.fallbackGlobsFor)
|
|
1296
|
+
return { files: targeted };
|
|
1297
|
+
if (targeted.some((f) => isSourceFile(f.relPath)))
|
|
1267
1298
|
return { files: targeted };
|
|
1268
1299
|
const fallbackGlobs = lens.fallbackGlobsFor(info).filter(Boolean);
|
|
1269
|
-
const
|
|
1270
|
-
|
|
1271
|
-
|
|
1300
|
+
const matched = new Set(targeted.map((f) => f.relPath));
|
|
1301
|
+
const sources = collectFilesMatching(allFiles, lens, fallbackGlobs).filter((f) => !matched.has(f.relPath));
|
|
1302
|
+
if (sources.length === 0)
|
|
1303
|
+
return { files: targeted };
|
|
1304
|
+
const excluded = lens.skipTestFiles ? "test files excluded" : "";
|
|
1305
|
+
const scanned = `scanned all ${info.language} sources (${fallbackGlobs.join(", ")})`;
|
|
1272
1306
|
return {
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1307
|
+
// What did match rides first: the policy the model is about to check
|
|
1308
|
+
// the code against, ahead of the code.
|
|
1309
|
+
files: [...targeted, ...sources],
|
|
1310
|
+
fallback: targeted.length === 0
|
|
1311
|
+
? `no files matched ${globs.join(", ")}${excluded ? ` (${excluded})` : ""}; ${scanned} instead`
|
|
1312
|
+
: `no source files matched ${globs.join(", ")} (only ${listSome(targeted.map((f) => f.relPath))}` +
|
|
1313
|
+
`${excluded ? `; ${excluded}` : ""}); ${scanned} as well`,
|
|
1276
1314
|
};
|
|
1277
1315
|
}
|
|
1278
1316
|
function collectLensFiles(allFiles, lens, info) {
|
|
@@ -1396,8 +1434,8 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1396
1434
|
// so the model judges the trust boundary wherever it appears
|
|
1397
1435
|
// and does not report the missing server/ as a finding (#319).
|
|
1398
1436
|
(slice.fallback
|
|
1399
|
-
? `NOTE: this repository has no files under the paths this lens usually reads (${slice.fallback}). ` +
|
|
1400
|
-
"What follows is every source file it has; locate the trust boundary and the request-handling code wherever they live.\n\n"
|
|
1437
|
+
? `NOTE: this repository has no source files under the paths this lens usually reads (${slice.fallback}). ` +
|
|
1438
|
+
"What follows is every source file it has, after anything those paths did match; locate the trust boundary and the request-handling code wherever they live.\n\n"
|
|
1401
1439
|
: "") + lens.userPrompt(info, slice.content, slice.moduleName),
|
|
1402
1440
|
},
|
|
1403
1441
|
],
|
|
@@ -1624,6 +1662,10 @@ export async function persistBroadsideRunMerging(broadsideDir, run) {
|
|
|
1624
1662
|
run.triage = onDisk.triage;
|
|
1625
1663
|
if (retryEntryRank(onDisk.retry) > retryEntryRank(run.retry))
|
|
1626
1664
|
run.retry = onDisk.retry;
|
|
1665
|
+
// A verification pass another process recorded is never dropped by
|
|
1666
|
+
// a collect that never knew about it; a newer pass replaces an older.
|
|
1667
|
+
if (onDisk.verify && (!run.verify || onDisk.verify.at > run.verify.at))
|
|
1668
|
+
run.verify = onDisk.verify;
|
|
1627
1669
|
for (const [lensId, theirs] of Object.entries(onDisk.batches)) {
|
|
1628
1670
|
if (theirs && batchEntryRank(theirs) > batchEntryRank(run.batches[lensId]))
|
|
1629
1671
|
run.batches[lensId] = theirs;
|
|
@@ -3545,6 +3587,9 @@ export function statusText(state) {
|
|
|
3545
3587
|
}
|
|
3546
3588
|
lines.push(` synthesis: ${run.synthesis.status}`);
|
|
3547
3589
|
lines.push(` triage: ${run.triage?.status ?? "pending"}`);
|
|
3590
|
+
if (run.verify) {
|
|
3591
|
+
lines.push(` verify: ${run.verify.status} — ${run.verify.confirmed} confirmed of ${run.verify.verified} read on ${run.verify.model}, $${run.verify.cost.toFixed(4)}`);
|
|
3592
|
+
}
|
|
3548
3593
|
if (run.totalCost !== undefined)
|
|
3549
3594
|
lines.push(` total cost: $${run.totalCost.toFixed(6)}`);
|
|
3550
3595
|
}
|