codecartographer-pi 0.22.3 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +44 -1
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +5 -2
- package/agent-skill/codecartographer/references/broadside.md +5 -1
- package/dist/core/broadside-verify.d.ts +95 -0
- package/dist/core/broadside-verify.js +433 -0
- package/dist/core/broadside.d.ts +63 -0
- package/dist/core/broadside.js +79 -23
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/extensions/codecarto/broadside-flags.d.ts +4 -2
- package/dist/extensions/codecarto/broadside-flags.js +24 -6
- package/dist/extensions/codecarto/index.js +36 -4
- package/dist/mcp-server/server.d.ts +2 -1
- package/dist/mcp-server/server.js +45 -9
- package/package.json +1 -1
|
@@ -0,0 +1,433 @@
|
|
|
1
|
+
// Broad-Side verification pass (#143): the hybrid the roadmap kept coming
|
|
2
|
+
// back to. The batch sweep is cheap and reads files without being able to
|
|
3
|
+
// look anything up, and its measured weakness is precision, not coverage —
|
|
4
|
+
// on this repository the top twelve findings by severity were two real
|
|
5
|
+
// defects and ten claims that a look at the guard, the caller, or the
|
|
6
|
+
// tsconfig would have dismissed. So after collect, one sync-priced call per
|
|
7
|
+
// finding, with three read-only tools confined to the repository, reads the
|
|
8
|
+
// cited code and says whether the failure is reachable.
|
|
9
|
+
//
|
|
10
|
+
// Measured on 2026-09-13 (the #143 comparison run): twelve findings, the
|
|
11
|
+
// rubric below, `google/gemini-3.7-flash` at low reasoning effort —
|
|
12
|
+
// twelve-for-twelve agreement with a reviewer's ground truth, precision of
|
|
13
|
+
// the confirmed set from 17% to 100%, both true findings kept, $0.13 in all
|
|
14
|
+
// (about a cent a finding). The rubric mattered more than the model: a first
|
|
15
|
+
// draft without the `not-a-defect` verdict "confirmed" two type casts that
|
|
16
|
+
// every caller satisfies, because they were literally true of the code.
|
|
17
|
+
//
|
|
18
|
+
// Read-only on purpose. A tool-using pass that could mutate would need the
|
|
19
|
+
// headless-agent retry rule (retry only before the first tool call); one
|
|
20
|
+
// that only reads keeps the batch property that re-running is always safe.
|
|
21
|
+
import { readdir, readFile, stat, writeFile } from "node:fs/promises";
|
|
22
|
+
import { isAbsolute, join, relative, resolve } from "node:path";
|
|
23
|
+
import { BROADSIDE_DIR, BROADSIDE_LENS_IDS, broadsideDirFor, defaultReasoningFor, isSlurpable, listRepoFiles, loadBroadsideState, loadSavedLensResults, parseLensJson, persistBroadsideRunMerging, } from "./broadside.js";
|
|
24
|
+
export const BROADSIDE_CHAT_URL = "https://openrouter.ai/api/v1/chat/completions";
|
|
25
|
+
/** How many findings `verify` reads by default, most severe first. */
|
|
26
|
+
export const BROADSIDE_VERIFY_DEFAULT_TOP = 10;
|
|
27
|
+
/** Tool calls one finding may spend before it must answer. */
|
|
28
|
+
export const BROADSIDE_VERIFY_MAX_TOOL_CALLS = 8;
|
|
29
|
+
/** The lenses whose findings carry a file:line and a claim to check. */
|
|
30
|
+
export const BROADSIDE_VERIFIABLE_LENSES = ["defect", "security"];
|
|
31
|
+
const READ_FILE_MAX_LINES = 200;
|
|
32
|
+
const GREP_MAX_MATCHES = 40;
|
|
33
|
+
const GREP_MAX_FILE_BYTES = 1_000_000;
|
|
34
|
+
const TOOL_OUTPUT_MAX_CHARS = 12_000;
|
|
35
|
+
const SEVERITY_RANK = { critical: 0, high: 1, medium: 2, low: 3 };
|
|
36
|
+
/** The sync-priced id behind a `:batch` model id (`vendor/name:batch` → `vendor/name`). */
|
|
37
|
+
export function syncModelFor(batchModel) {
|
|
38
|
+
return batchModel.replace(/:batch$/, "");
|
|
39
|
+
}
|
|
40
|
+
/** The findings a run's saved lens results carry, most severe first. */
|
|
41
|
+
export function rankVerifiableFindings(stored) {
|
|
42
|
+
const out = [];
|
|
43
|
+
for (const result of stored) {
|
|
44
|
+
if (!BROADSIDE_VERIFIABLE_LENSES.includes(result.lensId) || result.truncated)
|
|
45
|
+
continue;
|
|
46
|
+
const parsed = parseLensJson(result.content);
|
|
47
|
+
for (const finding of parsed?.findings ?? []) {
|
|
48
|
+
const location = typeof finding.location === "string" ? finding.location : "";
|
|
49
|
+
const title = typeof finding.title === "string" ? finding.title : "";
|
|
50
|
+
if (!location || !title)
|
|
51
|
+
continue;
|
|
52
|
+
out.push({
|
|
53
|
+
lensId: result.lensId,
|
|
54
|
+
customId: result.customId,
|
|
55
|
+
severity: typeof finding.severity === "string" ? finding.severity.toLowerCase() : "unknown",
|
|
56
|
+
title,
|
|
57
|
+
location,
|
|
58
|
+
description: typeof finding.description === "string" ? finding.description : "",
|
|
59
|
+
pattern: typeof finding.pattern === "string" ? finding.pattern : typeof finding.category === "string" ? finding.category : "",
|
|
60
|
+
});
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
// Stable: severity, then lens order, then the order the lens listed them.
|
|
64
|
+
return out
|
|
65
|
+
.map((finding, order) => ({ finding, order }))
|
|
66
|
+
.sort((a, b) => (SEVERITY_RANK[a.finding.severity] ?? 9) - (SEVERITY_RANK[b.finding.severity] ?? 9)
|
|
67
|
+
|| BROADSIDE_LENS_IDS.indexOf(a.finding.lensId) - BROADSIDE_LENS_IDS.indexOf(b.finding.lensId)
|
|
68
|
+
|| a.order - b.order)
|
|
69
|
+
.map(({ finding }) => finding);
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Three read-only tools over the repository's own file list — the same
|
|
73
|
+
* listing the lenses scan (tracked and untracked, ignore rules applied) minus
|
|
74
|
+
* everything {@link isSlurpable} keeps out of a lens: credential stores, build
|
|
75
|
+
* output, binaries. A path outside the repository, or one the listing does
|
|
76
|
+
* not contain, is an error the model sees, not a read.
|
|
77
|
+
*/
|
|
78
|
+
export async function createRepoReader(cwd) {
|
|
79
|
+
const { files } = await listRepoFiles(cwd);
|
|
80
|
+
const readable = new Set(files.filter(isSlurpable));
|
|
81
|
+
const confine = (path) => {
|
|
82
|
+
const abs = resolve(cwd, path);
|
|
83
|
+
const rel = relative(cwd, abs).split("\\").join("/");
|
|
84
|
+
if (!rel || rel.startsWith("..") || isAbsolute(rel))
|
|
85
|
+
throw new Error(`path is outside the repository: ${path}`);
|
|
86
|
+
return rel;
|
|
87
|
+
};
|
|
88
|
+
return {
|
|
89
|
+
async readFile(path, startLine, endLine) {
|
|
90
|
+
const rel = confine(path);
|
|
91
|
+
if (!readable.has(rel))
|
|
92
|
+
return `error: ${rel} is not a readable source file of this repository`;
|
|
93
|
+
const lines = (await readFile(join(cwd, rel), "utf8")).split("\n");
|
|
94
|
+
const start = Math.max(1, Math.floor(Number(startLine ?? 1)) || 1);
|
|
95
|
+
const requestedEnd = Math.floor(Number(endLine ?? start + READ_FILE_MAX_LINES - 1)) || start;
|
|
96
|
+
const end = Math.min(lines.length, requestedEnd, start + READ_FILE_MAX_LINES - 1);
|
|
97
|
+
if (start > lines.length)
|
|
98
|
+
return `error: ${rel} has ${lines.length} lines`;
|
|
99
|
+
return lines.slice(start - 1, end).map((line, i) => `${start + i}: ${line}`).join("\n");
|
|
100
|
+
},
|
|
101
|
+
async grep(pattern, pathPrefix) {
|
|
102
|
+
let regex;
|
|
103
|
+
try {
|
|
104
|
+
regex = new RegExp(pattern);
|
|
105
|
+
}
|
|
106
|
+
catch (error) {
|
|
107
|
+
return `error: invalid pattern (${error instanceof Error ? error.message : String(error)})`;
|
|
108
|
+
}
|
|
109
|
+
const prefix = pathPrefix ? confine(pathPrefix) : "";
|
|
110
|
+
const matches = [];
|
|
111
|
+
for (const rel of readable) {
|
|
112
|
+
if (prefix && rel !== prefix && !rel.startsWith(`${prefix}/`) && !rel.startsWith(prefix))
|
|
113
|
+
continue;
|
|
114
|
+
let text;
|
|
115
|
+
try {
|
|
116
|
+
if ((await stat(join(cwd, rel))).size > GREP_MAX_FILE_BYTES)
|
|
117
|
+
continue;
|
|
118
|
+
text = await readFile(join(cwd, rel), "utf8");
|
|
119
|
+
}
|
|
120
|
+
catch {
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
const lines = text.split("\n");
|
|
124
|
+
for (let i = 0; i < lines.length && matches.length < GREP_MAX_MATCHES; i++) {
|
|
125
|
+
if (regex.test(lines[i]))
|
|
126
|
+
matches.push(`${rel}:${i + 1}: ${lines[i]}`);
|
|
127
|
+
}
|
|
128
|
+
if (matches.length >= GREP_MAX_MATCHES)
|
|
129
|
+
break;
|
|
130
|
+
}
|
|
131
|
+
return matches.length > 0 ? matches.join("\n") : "(no matches)";
|
|
132
|
+
},
|
|
133
|
+
async listDir(path) {
|
|
134
|
+
const rel = path === "." || path === "" ? "" : confine(path);
|
|
135
|
+
try {
|
|
136
|
+
const entries = await readdir(join(cwd, rel), { withFileTypes: true });
|
|
137
|
+
return entries
|
|
138
|
+
.filter((entry) => {
|
|
139
|
+
const child = rel ? `${rel}/${entry.name}` : entry.name;
|
|
140
|
+
return entry.isDirectory() ? [...readable].some((f) => f.startsWith(`${child}/`)) : readable.has(child);
|
|
141
|
+
})
|
|
142
|
+
.map((entry) => (entry.isDirectory() ? `${entry.name}/` : entry.name))
|
|
143
|
+
.sort()
|
|
144
|
+
.join("\n") || "(empty)";
|
|
145
|
+
}
|
|
146
|
+
catch (error) {
|
|
147
|
+
return `error: ${error instanceof Error ? error.message : String(error)}`;
|
|
148
|
+
}
|
|
149
|
+
},
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
const TOOLS = [
|
|
153
|
+
{ type: "function", function: { name: "read_file", description: "Read a range of lines (1-based, inclusive) from a source file of the repository; at most 200 lines per call.", parameters: { type: "object", properties: { path: { type: "string" }, start_line: { type: "integer" }, end_line: { type: "integer" } }, required: ["path"] } } },
|
|
154
|
+
{ type: "function", function: { name: "grep", description: "Search the repository's source files for a regular expression; returns at most 40 matching lines as path:line: text.", parameters: { type: "object", properties: { pattern: { type: "string" }, path_prefix: { type: "string", description: "optional directory or file to search under" } }, required: ["pattern"] } } },
|
|
155
|
+
{ type: "function", function: { name: "list_dir", description: "List a directory of the repository.", parameters: { type: "object", properties: { path: { type: "string" } }, required: ["path"] } } },
|
|
156
|
+
];
|
|
157
|
+
const VERDICT_SCHEMA = {
|
|
158
|
+
name: "broadside_verification",
|
|
159
|
+
strict: true,
|
|
160
|
+
schema: {
|
|
161
|
+
type: "object",
|
|
162
|
+
properties: {
|
|
163
|
+
verdict: { type: "string", enum: ["confirmed", "not-a-defect", "discarded", "unclear"] },
|
|
164
|
+
confidence: { type: "string", enum: ["high", "medium", "low"] },
|
|
165
|
+
evidence: {
|
|
166
|
+
type: "array",
|
|
167
|
+
items: { type: "object", properties: { file: { type: "string" }, lines: { type: "string" }, note: { type: "string" } }, required: ["file", "lines", "note"], additionalProperties: false },
|
|
168
|
+
},
|
|
169
|
+
reasoning: { type: "string" },
|
|
170
|
+
},
|
|
171
|
+
required: ["verdict", "confidence", "evidence", "reasoning"],
|
|
172
|
+
additionalProperties: false,
|
|
173
|
+
},
|
|
174
|
+
};
|
|
175
|
+
/**
|
|
176
|
+
* The rubric. The order is deliberate — it is the order a reviewer settles a
|
|
177
|
+
* claim in — and `not-a-defect` is the verdict that separates "the code does
|
|
178
|
+
* what the claim says" from "and that is a bug": without it, a cast every
|
|
179
|
+
* caller satisfies gets confirmed because it is literally there.
|
|
180
|
+
*/
|
|
181
|
+
export const BROADSIDE_VERIFY_SYSTEM_PROMPT = "You verify a scouting finding produced by a one-shot batch scan against the real source code. " +
|
|
182
|
+
"The scan saw files without being able to look anything up; you can. Use the tools to read the cited location " +
|
|
183
|
+
"and whatever else the claim depends on (callers, the definition of a helper, error handling around it). " +
|
|
184
|
+
"Then decide, in this order: " +
|
|
185
|
+
"'discarded' — the claim is wrong about the code (the guard exists, the value cannot be what the claim assumes, " +
|
|
186
|
+
"the cited line does something else, the condition it fears is ruled out by the project's config or runtime); " +
|
|
187
|
+
"'not-a-defect' — the claim is literally true of the code but no caller, input, or state can reach the failure it " +
|
|
188
|
+
"describes: a TypeScript cast every caller satisfies, a hypothetical about an environment the project does not " +
|
|
189
|
+
"target, a style or type-hygiene observation; " +
|
|
190
|
+
"'confirmed' — the failure is reachable: name the concrete input, call site, or sequence that triggers it, and what " +
|
|
191
|
+
"then goes wrong; " +
|
|
192
|
+
"'unclear' — settling it needs runtime behaviour or specification knowledge the code does not contain. " +
|
|
193
|
+
"Be strict: a real but different problem than the one claimed is 'discarded' with the difference noted, and " +
|
|
194
|
+
"'confirmed' without a trigger you found in the code is not allowed. " +
|
|
195
|
+
"Cite line ranges you actually read. When you are done, reply with only the JSON verdict object.";
|
|
196
|
+
function findingPrompt(index, finding) {
|
|
197
|
+
return (`Finding ${index} (${finding.lensId} lens, severity ${finding.severity}):\n` +
|
|
198
|
+
`Title: ${finding.title}\nLocation: ${finding.location}\nPattern/category: ${finding.pattern || "-"}\n` +
|
|
199
|
+
`Description: ${finding.description}\n\n` +
|
|
200
|
+
"Verify it. Reply with a JSON object {verdict, confidence, evidence:[{file,lines,note}], reasoning}.");
|
|
201
|
+
}
|
|
202
|
+
async function chat(fetcher, apiKey, model, body) {
|
|
203
|
+
const resp = await fetcher(BROADSIDE_CHAT_URL, {
|
|
204
|
+
method: "POST",
|
|
205
|
+
headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json" },
|
|
206
|
+
body: JSON.stringify({ model, reasoning: defaultReasoningFor(), usage: { include: true }, ...body }),
|
|
207
|
+
signal: AbortSignal.timeout(120_000),
|
|
208
|
+
});
|
|
209
|
+
const data = (await resp.json());
|
|
210
|
+
if (!resp.ok || data.error) {
|
|
211
|
+
const detail = data.error?.message ?? JSON.stringify(data.error ?? data).slice(0, 300);
|
|
212
|
+
throw new Error(`OpenRouter chat: HTTP ${resp.status}: ${detail}`);
|
|
213
|
+
}
|
|
214
|
+
return data;
|
|
215
|
+
}
|
|
216
|
+
function parseVerdict(text) {
|
|
217
|
+
const trimmed = text.trim();
|
|
218
|
+
const fenced = /```(?:json)?\s*([\s\S]*?)```/.exec(trimmed);
|
|
219
|
+
try {
|
|
220
|
+
const parsed = JSON.parse(fenced ? fenced[1] : trimmed);
|
|
221
|
+
return typeof parsed?.verdict === "string" ? parsed : null;
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
return null;
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
/** Verify one finding: up to the tool budget, then a verdict. */
|
|
228
|
+
export async function verifyFinding(finding, index, reader, apiKey, model, fetcher) {
|
|
229
|
+
const messages = [
|
|
230
|
+
{ role: "system", content: BROADSIDE_VERIFY_SYSTEM_PROMPT },
|
|
231
|
+
{ role: "user", content: findingPrompt(index, finding) },
|
|
232
|
+
];
|
|
233
|
+
let toolCalls = 0;
|
|
234
|
+
let cost = 0;
|
|
235
|
+
const usageCost = (data) => {
|
|
236
|
+
const usage = data.usage;
|
|
237
|
+
return typeof usage?.cost === "number" ? usage.cost : 0;
|
|
238
|
+
};
|
|
239
|
+
try {
|
|
240
|
+
// The +2 leaves room for the final schema-forced call after the budget.
|
|
241
|
+
for (let step = 0; step < BROADSIDE_VERIFY_MAX_TOOL_CALLS + 2; step++) {
|
|
242
|
+
const budgetSpent = toolCalls >= BROADSIDE_VERIFY_MAX_TOOL_CALLS;
|
|
243
|
+
if (budgetSpent)
|
|
244
|
+
messages.push({ role: "user", content: "Tool budget spent. Emit the verdict JSON object now from what you have read." });
|
|
245
|
+
const data = await chat(fetcher, apiKey, model, budgetSpent
|
|
246
|
+
? { messages, response_format: { type: "json_schema", json_schema: VERDICT_SCHEMA }, max_tokens: 2000 }
|
|
247
|
+
: { messages, tools: TOOLS, tool_choice: "auto", max_tokens: 4000 });
|
|
248
|
+
cost += usageCost(data);
|
|
249
|
+
const choice = data.choices?.[0];
|
|
250
|
+
const message = choice?.message ?? { role: "assistant", content: "" };
|
|
251
|
+
messages.push(message);
|
|
252
|
+
const calls = message.tool_calls;
|
|
253
|
+
if (calls && calls.length > 0 && !budgetSpent) {
|
|
254
|
+
for (const call of calls) {
|
|
255
|
+
toolCalls += 1;
|
|
256
|
+
let args = {};
|
|
257
|
+
try {
|
|
258
|
+
args = JSON.parse(call.function.arguments || "{}");
|
|
259
|
+
}
|
|
260
|
+
catch {
|
|
261
|
+
// Malformed arguments: the model sees the error and can retry.
|
|
262
|
+
}
|
|
263
|
+
let output;
|
|
264
|
+
try {
|
|
265
|
+
output = call.function.name === "read_file"
|
|
266
|
+
? await reader.readFile(String(args.path ?? ""), args.start_line, args.end_line)
|
|
267
|
+
: call.function.name === "grep"
|
|
268
|
+
? await reader.grep(String(args.pattern ?? ""), typeof args.path_prefix === "string" ? args.path_prefix : undefined)
|
|
269
|
+
: call.function.name === "list_dir"
|
|
270
|
+
? await reader.listDir(String(args.path ?? "."))
|
|
271
|
+
: `error: unknown tool ${call.function.name}`;
|
|
272
|
+
}
|
|
273
|
+
catch (error) {
|
|
274
|
+
output = `error: ${error instanceof Error ? error.message : String(error)}`;
|
|
275
|
+
}
|
|
276
|
+
messages.push({ role: "tool", tool_call_id: call.id, content: output.slice(0, TOOL_OUTPUT_MAX_CHARS) });
|
|
277
|
+
}
|
|
278
|
+
continue;
|
|
279
|
+
}
|
|
280
|
+
const verdict = parseVerdict(typeof message.content === "string" ? message.content : "");
|
|
281
|
+
if (verdict)
|
|
282
|
+
return finish(verdict, toolCalls, cost);
|
|
283
|
+
if (budgetSpent)
|
|
284
|
+
break;
|
|
285
|
+
// Prose instead of JSON: one schema-forced call, no tools.
|
|
286
|
+
const final = await chat(fetcher, apiKey, model, {
|
|
287
|
+
messages: [...messages, { role: "user", content: "Emit the verdict JSON object now." }],
|
|
288
|
+
response_format: { type: "json_schema", json_schema: VERDICT_SCHEMA },
|
|
289
|
+
max_tokens: 2000,
|
|
290
|
+
});
|
|
291
|
+
cost += usageCost(final);
|
|
292
|
+
const forced = parseVerdict(String((final.choices?.[0]?.message?.content) ?? ""));
|
|
293
|
+
if (forced)
|
|
294
|
+
return finish(forced, toolCalls, cost);
|
|
295
|
+
break;
|
|
296
|
+
}
|
|
297
|
+
return { verdict: "error", confidence: "low", evidence: [], reasoning: "no verdict within the tool budget", toolCalls, cost };
|
|
298
|
+
}
|
|
299
|
+
catch (error) {
|
|
300
|
+
return { verdict: "error", confidence: "low", evidence: [], reasoning: error instanceof Error ? error.message : String(error), toolCalls, cost };
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
function finish(verdict, toolCalls, cost) {
|
|
304
|
+
const known = ["confirmed", "not-a-defect", "discarded", "unclear"];
|
|
305
|
+
const value = String(verdict.verdict);
|
|
306
|
+
const evidence = Array.isArray(verdict.evidence)
|
|
307
|
+
? verdict.evidence.map((e) => ({ file: String(e.file ?? ""), lines: String(e.lines ?? ""), note: String(e.note ?? "") }))
|
|
308
|
+
: [];
|
|
309
|
+
return {
|
|
310
|
+
verdict: (known.includes(value) ? value : "unclear"),
|
|
311
|
+
confidence: typeof verdict.confidence === "string" ? verdict.confidence : "low",
|
|
312
|
+
evidence,
|
|
313
|
+
reasoning: typeof verdict.reasoning === "string" ? verdict.reasoning : "",
|
|
314
|
+
toolCalls,
|
|
315
|
+
cost,
|
|
316
|
+
};
|
|
317
|
+
}
|
|
318
|
+
/**
|
|
319
|
+
* Verify the top findings of a collected run against the repository.
|
|
320
|
+
*
|
|
321
|
+
* `maxCost` is a running cap, not a pre-flight estimate: a sync call's cost
|
|
322
|
+
* is only known when it returns, so the pass stops *before* starting the next
|
|
323
|
+
* finding once the cap is reached and reports `partial`. On the default
|
|
324
|
+
* model a finding costs about a cent.
|
|
325
|
+
*/
|
|
326
|
+
export async function runBroadsideVerify(cwd, apiKey, opts = {}) {
|
|
327
|
+
const broadsideDir = broadsideDirFor(cwd);
|
|
328
|
+
const state = await loadBroadsideState(broadsideDir);
|
|
329
|
+
const run = opts.runId ? state.runs.find((candidate) => candidate.id === opts.runId) : state.runs[state.runs.length - 1];
|
|
330
|
+
if (!run) {
|
|
331
|
+
throw new Error(opts.runId ? `No Broad-Side run with id ${opts.runId}.` : "No Broad-Side run recorded. Submit and collect one first.");
|
|
332
|
+
}
|
|
333
|
+
const runDir = join(broadsideDir, run.outputDir);
|
|
334
|
+
const stored = await loadSavedLensResults(runDir, run.lenses);
|
|
335
|
+
const ranked = rankVerifiableFindings(stored);
|
|
336
|
+
if (ranked.length === 0) {
|
|
337
|
+
throw new Error(`Run ${run.id} has no verifiable findings on disk: the defect and security lenses either did not run, are not collected yet, or found nothing. Collect the run first.`);
|
|
338
|
+
}
|
|
339
|
+
const top = Math.max(1, Math.floor(opts.top ?? BROADSIDE_VERIFY_DEFAULT_TOP));
|
|
340
|
+
// The run's model is a `:batch` variant; the sync endpoint wants the base id.
|
|
341
|
+
const model = opts.model ?? syncModelFor(run.model);
|
|
342
|
+
const maxCost = opts.maxCost ?? 0;
|
|
343
|
+
const fetcher = opts.fetcher ?? fetch;
|
|
344
|
+
const reader = await createRepoReader(cwd);
|
|
345
|
+
const selected = ranked.slice(0, top);
|
|
346
|
+
const findings = [];
|
|
347
|
+
let totalCost = 0;
|
|
348
|
+
let stoppedByCost = false;
|
|
349
|
+
for (const [i, candidate] of selected.entries()) {
|
|
350
|
+
if (opts.signal?.aborted)
|
|
351
|
+
break;
|
|
352
|
+
if (maxCost > 0 && totalCost >= maxCost) {
|
|
353
|
+
stoppedByCost = true;
|
|
354
|
+
break;
|
|
355
|
+
}
|
|
356
|
+
const outcome = await verifyFinding(candidate, i + 1, reader, apiKey, model, fetcher);
|
|
357
|
+
const finding = {
|
|
358
|
+
index: i + 1,
|
|
359
|
+
lensId: candidate.lensId,
|
|
360
|
+
customId: candidate.customId,
|
|
361
|
+
severity: candidate.severity,
|
|
362
|
+
title: candidate.title,
|
|
363
|
+
location: candidate.location,
|
|
364
|
+
...outcome,
|
|
365
|
+
};
|
|
366
|
+
findings.push(finding);
|
|
367
|
+
totalCost += outcome.cost;
|
|
368
|
+
opts.onProgress?.(finding);
|
|
369
|
+
}
|
|
370
|
+
const status = findings.length === selected.length ? "completed" : "partial";
|
|
371
|
+
const entry = {
|
|
372
|
+
status,
|
|
373
|
+
model,
|
|
374
|
+
top,
|
|
375
|
+
verified: findings.length,
|
|
376
|
+
confirmed: findings.filter((f) => f.verdict === "confirmed").length,
|
|
377
|
+
cost: totalCost,
|
|
378
|
+
at: new Date().toISOString(),
|
|
379
|
+
};
|
|
380
|
+
const result = {
|
|
381
|
+
runId: run.id,
|
|
382
|
+
outputDir: join(".codecarto", BROADSIDE_DIR, run.id),
|
|
383
|
+
model,
|
|
384
|
+
status,
|
|
385
|
+
candidates: ranked.length,
|
|
386
|
+
findings,
|
|
387
|
+
totalCost,
|
|
388
|
+
...(stoppedByCost && { stoppedByCost: true }),
|
|
389
|
+
};
|
|
390
|
+
await writeFile(join(runDir, "verified.json"), `${JSON.stringify({ ...entry, run_id: run.id, candidates: ranked.length, findings }, null, "\t")}\n`, "utf8");
|
|
391
|
+
await writeFile(join(runDir, "verified.md"), renderVerifiedMarkdown(result), "utf8");
|
|
392
|
+
run.verify = entry;
|
|
393
|
+
await persistBroadsideRunMerging(broadsideDir, run);
|
|
394
|
+
return result;
|
|
395
|
+
}
|
|
396
|
+
const VERDICT_MARK = { confirmed: "✓", "not-a-defect": "–", discarded: "✗", unclear: "?", error: "!" };
|
|
397
|
+
export function renderVerifiedMarkdown(result) {
|
|
398
|
+
const lines = [
|
|
399
|
+
`# Verified findings — run ${result.runId}`,
|
|
400
|
+
"",
|
|
401
|
+
`${result.findings.length} of ${result.candidates} verifiable finding(s) read against the source on \`${result.model}\` (most severe first), $${result.totalCost.toFixed(4)}.` +
|
|
402
|
+
(result.stoppedByCost ? " Stopped by the cost cap before the rest." : ""),
|
|
403
|
+
"",
|
|
404
|
+
"A **confirmed** finding names the input, call site, or sequence that reaches the failure. **not-a-defect** means the claim is",
|
|
405
|
+
"literally true of the code but nothing can reach the failure it describes; **discarded** means the claim is wrong about the",
|
|
406
|
+
"code; **unclear** needs runtime or specification knowledge. Every verdict is still a model's reading — a confirmed finding",
|
|
407
|
+
"is a lead worth a human's next look, not a validated claim.",
|
|
408
|
+
"",
|
|
409
|
+
];
|
|
410
|
+
for (const f of result.findings) {
|
|
411
|
+
lines.push(`## ${VERDICT_MARK[f.verdict] ?? "?"} ${f.index}. [${f.severity}] ${f.title}`, "", `- **verdict**: ${f.verdict} (${f.confidence})`, `- **location**: ${f.location}`, `- **lens**: ${f.lensId} (${f.customId})`);
|
|
412
|
+
if (f.evidence.length > 0)
|
|
413
|
+
lines.push(`- **evidence**: ${f.evidence.map((e) => `${e.file}:${e.lines} — ${e.note}`).join("; ")}`);
|
|
414
|
+
lines.push(`- **reasoning**: ${f.reasoning.replace(/\s+/g, " ").trim()}`, `- **cost**: $${f.cost.toFixed(4)} (${f.toolCalls} tool call(s))`, "");
|
|
415
|
+
}
|
|
416
|
+
return lines.join("\n");
|
|
417
|
+
}
|
|
418
|
+
export function verifyResultText(result) {
|
|
419
|
+
const counts = { confirmed: 0, "not-a-defect": 0, discarded: 0, unclear: 0, error: 0 };
|
|
420
|
+
for (const f of result.findings)
|
|
421
|
+
counts[f.verdict] = (counts[f.verdict] ?? 0) + 1;
|
|
422
|
+
const lines = [
|
|
423
|
+
`Broad-Side verify — run ${result.runId}: ${result.status}`,
|
|
424
|
+
` ${result.findings.length} of ${result.candidates} verifiable finding(s) read on ${result.model} | cost: $${result.totalCost.toFixed(4)}` +
|
|
425
|
+
(result.stoppedByCost ? " (stopped by the cost cap)" : ""),
|
|
426
|
+
` confirmed ${counts.confirmed} · not-a-defect ${counts["not-a-defect"]} · discarded ${counts.discarded} · unclear ${counts.unclear}` + (counts.error ? ` · error ${counts.error}` : ""),
|
|
427
|
+
];
|
|
428
|
+
for (const f of result.findings) {
|
|
429
|
+
lines.push(` ${VERDICT_MARK[f.verdict] ?? "?"} [${f.severity}] ${f.title} @ ${f.location} — ${f.verdict}`);
|
|
430
|
+
}
|
|
431
|
+
lines.push(`Details in ${result.outputDir}/verified.md. A confirmed finding is a lead for a human's next look, not a validated claim.`);
|
|
432
|
+
return lines.join("\n");
|
|
433
|
+
}
|
package/dist/core/broadside.d.ts
CHANGED
|
@@ -121,6 +121,12 @@ export type FileSlice = {
|
|
|
121
121
|
redactedValues?: number;
|
|
122
122
|
/** The files in this slice that had at least one value redacted. */
|
|
123
123
|
redactedFiles?: string[];
|
|
124
|
+
/**
|
|
125
|
+
* Set when the lens's targeted globs matched nothing and the slice was
|
|
126
|
+
* built from its fallback globs instead (#319). The estimate, the batch
|
|
127
|
+
* entry, and the prompt all say so.
|
|
128
|
+
*/
|
|
129
|
+
fallback?: string;
|
|
124
130
|
};
|
|
125
131
|
/**
|
|
126
132
|
* OpenRouter's unified `reasoning` control, as sent on a lens request.
|
|
@@ -203,6 +209,12 @@ export type BroadsideBatchEntry = {
|
|
|
203
209
|
error?: unknown;
|
|
204
210
|
/** Why a `skipped` lens had nothing to submit: the globs that matched no file. */
|
|
205
211
|
reason?: string;
|
|
212
|
+
/**
|
|
213
|
+
* Set when the lens scanned its fallback scope because its targeted globs
|
|
214
|
+
* matched nothing (#319): "no files matched …; scanned all javascript
|
|
215
|
+
* sources instead". Absent for a targeted scan.
|
|
216
|
+
*/
|
|
217
|
+
fallback?: string;
|
|
206
218
|
/** Set when this lens used a model other than the run default. */
|
|
207
219
|
model?: string;
|
|
208
220
|
/** The completion ceiling of this lens's model; bounds the truncation retry. */
|
|
@@ -232,6 +244,17 @@ export type BroadsideTriageEntry = {
|
|
|
232
244
|
cost?: number;
|
|
233
245
|
error?: string;
|
|
234
246
|
};
|
|
247
|
+
/** Recorded on the run once a verification pass has run (#143); see core/broadside-verify.ts. */
|
|
248
|
+
export type BroadsideVerifyEntry = {
|
|
249
|
+
/** `completed`: every selected finding got a verdict; `partial`: the cost cap or an abort stopped it early. */
|
|
250
|
+
status: "completed" | "partial";
|
|
251
|
+
model: string;
|
|
252
|
+
top: number;
|
|
253
|
+
verified: number;
|
|
254
|
+
confirmed: number;
|
|
255
|
+
cost: number;
|
|
256
|
+
at: string;
|
|
257
|
+
};
|
|
235
258
|
/** The truncation retry pass of one run: one batch per model (#206). */
|
|
236
259
|
export type BroadsideRetryEntry = {
|
|
237
260
|
status: "submitted" | "completed" | "failed";
|
|
@@ -263,6 +286,8 @@ export type BroadsideRun = {
|
|
|
263
286
|
* run cannot both submit it (#322). Absent until a collect claims it.
|
|
264
287
|
*/
|
|
265
288
|
retry?: BroadsideRetryEntry;
|
|
289
|
+
/** The verification pass over the top findings, when one has run (#143). */
|
|
290
|
+
verify?: BroadsideVerifyEntry;
|
|
266
291
|
totalCost?: number;
|
|
267
292
|
pricing?: ModelPricing;
|
|
268
293
|
maxCost?: number;
|
|
@@ -348,6 +373,8 @@ export type BroadsideEstimate = {
|
|
|
348
373
|
/** The model this lens would use — `model` unless a per-lens override applies. */
|
|
349
374
|
model: string;
|
|
350
375
|
pricing: ModelPricing;
|
|
376
|
+
/** Set when this lens is priced on its fallback scope (#319); see BroadsideBatchEntry.fallback. */
|
|
377
|
+
fallback?: string;
|
|
351
378
|
}>;
|
|
352
379
|
/** True when at least one lens uses a model other than the run default. */
|
|
353
380
|
mixedModels: boolean;
|
|
@@ -490,6 +517,15 @@ type LensDefinition = {
|
|
|
490
517
|
reasoning?: BroadsideReasoning;
|
|
491
518
|
skipTestFiles?: boolean;
|
|
492
519
|
globsFor: (info: RepoInfo) => string[];
|
|
520
|
+
/**
|
|
521
|
+
* Where to look when `globsFor` matches nothing (#319). The security and
|
|
522
|
+
* api lenses target server/, auth, and middleware paths because that is
|
|
523
|
+
* where the trust boundary usually lives; a service whose server is
|
|
524
|
+
* `src/server.js` matched none of them and got no security review at all.
|
|
525
|
+
* The fallback is the language's whole source set — priced as such, and
|
|
526
|
+
* said so in the estimate, the run record, and the prompt.
|
|
527
|
+
*/
|
|
528
|
+
fallbackGlobsFor?: (info: RepoInfo) => string[];
|
|
493
529
|
systemPrompt: (info: RepoInfo) => string;
|
|
494
530
|
userPrompt: (info: RepoInfo, source: string, moduleName: string) => string;
|
|
495
531
|
};
|
|
@@ -497,9 +533,36 @@ export declare function getLens(lensId: BroadsideLensId): LensDefinition;
|
|
|
497
533
|
export declare function listLenses(): LensDefinition[];
|
|
498
534
|
/** The languages Broad-Side can scan; anything else is refused at submit. */
|
|
499
535
|
export declare const BROADSIDE_LANGUAGES: readonly ["go", "python", "rust", "typescript", "javascript"];
|
|
536
|
+
/**
|
|
537
|
+
* The files a run scans, and where they came from. Contents are always read
|
|
538
|
+
* from the working tree, so the list is the working tree's too: tracked files
|
|
539
|
+
* plus untracked ones git does not ignore, minus files deleted on disk. The
|
|
540
|
+
* list used to come from `git ls-tree HEAD`, so a run mixed the committed
|
|
541
|
+
* file list with uncommitted contents and never saw an untracked file (#248).
|
|
542
|
+
* A target that is not a git repository gets a bounded walk.
|
|
543
|
+
*/
|
|
544
|
+
export declare function listRepoFiles(targetDir: string): Promise<{
|
|
545
|
+
files: string[];
|
|
546
|
+
snapshot: RepoSnapshotSource;
|
|
547
|
+
}>;
|
|
500
548
|
export declare function collectRepoInfo(targetDir: string, opts?: {
|
|
501
549
|
redact?: boolean;
|
|
502
550
|
}): Promise<RepoInfo>;
|
|
551
|
+
export declare function isSlurpable(relPath: string): boolean;
|
|
552
|
+
type CollectedFile = {
|
|
553
|
+
relPath: string;
|
|
554
|
+
moduleName: string;
|
|
555
|
+
};
|
|
556
|
+
/**
|
|
557
|
+
* The files a lens will read: its targeted globs, or — when those match
|
|
558
|
+
* nothing and the lens declares a fallback — the fallback globs, with a
|
|
559
|
+
* sentence saying so (#319). The sentence travels to the estimate, the
|
|
560
|
+
* batch entry, and the prompt, so a fallback scan is never a silent one.
|
|
561
|
+
*/
|
|
562
|
+
export declare function selectLensFiles(allFiles: string[], lens: LensDefinition, info: RepoInfo): {
|
|
563
|
+
files: CollectedFile[];
|
|
564
|
+
fallback?: string;
|
|
565
|
+
};
|
|
503
566
|
export declare function gatherSlices(targetDir: string, lens: LensDefinition, info: RepoInfo, opts?: {
|
|
504
567
|
redact?: boolean;
|
|
505
568
|
}): Promise<FileSlice[]>;
|