codecartographer-pi 0.24.1 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +20 -1
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +5 -4
- package/agent-skill/codecartographer/references/broadside.md +5 -1
- package/dist/core/broadside/client.d.ts +56 -0
- package/dist/core/broadside/client.js +200 -0
- package/dist/core/broadside/collect.d.ts +68 -0
- package/dist/core/broadside/collect.js +676 -0
- package/dist/core/broadside/constants.d.ts +51 -0
- package/dist/core/broadside/constants.js +74 -0
- package/dist/core/broadside/lenses.d.ts +31 -0
- package/dist/core/broadside/lenses.js +312 -0
- package/dist/core/broadside/models.d.ts +46 -0
- package/dist/core/broadside/models.js +321 -0
- package/dist/core/broadside/render.d.ts +20 -0
- package/dist/core/broadside/render.js +285 -0
- package/dist/core/broadside/repo.d.ts +58 -0
- package/dist/core/broadside/repo.js +592 -0
- package/dist/core/broadside/requests.d.ts +23 -0
- package/dist/core/broadside/requests.js +71 -0
- package/dist/core/broadside/results.d.ts +36 -0
- package/dist/core/broadside/results.js +163 -0
- package/dist/core/broadside/schemas.d.ts +2 -0
- package/dist/core/broadside/schemas.js +342 -0
- package/dist/core/broadside/state.d.ts +99 -0
- package/dist/core/broadside/state.js +384 -0
- package/dist/core/broadside/submit.d.ts +30 -0
- package/dist/core/broadside/submit.js +350 -0
- package/dist/core/broadside/types.d.ts +491 -0
- package/dist/core/broadside/types.js +107 -0
- package/dist/core/{broadside-verify.d.ts → broadside/verify.d.ts} +23 -2
- package/dist/core/{broadside-verify.js → broadside/verify.js} +43 -5
- package/dist/core/broadside.d.ts +14 -890
- package/dist/core/broadside.js +25 -3564
- package/dist/core/completion.js +91 -72
- package/dist/core/dashboard-writer.js +9 -1
- package/dist/core/index.d.ts +0 -1
- package/dist/core/index.js +0 -1
- package/dist/core/library.d.ts +24 -1
- package/dist/core/library.js +46 -15
- package/dist/core/orchestrator-config.js +22 -8
- package/dist/core/status.d.ts +42 -23
- package/dist/core/status.js +163 -137
- package/dist/core/workspace.d.ts +2 -0
- package/dist/core/workspace.js +49 -25
- package/dist/core/yaml.js +9 -3
- package/dist/extensions/codecarto/auto-runner.d.ts +7 -0
- package/dist/extensions/codecarto/auto-runner.js +54 -23
- package/dist/extensions/codecarto/broadside-flags.d.ts +3 -1
- package/dist/extensions/codecarto/broadside-flags.js +13 -0
- package/dist/extensions/codecarto/index.js +13 -7
- package/dist/extensions/codecarto/phase-compaction.js +6 -2
- package/dist/mcp-server/server.d.ts +1 -0
- package/dist/mcp-server/server.js +28 -5
- package/package.json +1 -1
|
@@ -0,0 +1,676 @@
|
|
|
1
|
+
// runBroadsideCollect and runBroadsideStatus: polling, saving, the truncation retry, and the synthesis and triage post-passes (verdict-aware).
|
|
2
|
+
//
|
|
3
|
+
// Split out of core/broadside.ts (#339); the barrel there re-exports every
|
|
4
|
+
// name, so `core/index.ts` and the tests see one module as before.
|
|
5
|
+
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
6
|
+
import { join } from "node:path";
|
|
7
|
+
import { pathExists } from "../utils.js";
|
|
8
|
+
import { BROADSIDE_DEAD_BATCH_STATUSES, BROADSIDE_DEFAULT_POLL_BUDGET_MS, BROADSIDE_TERMINAL_ENTRY_STATUSES } from "./constants.js";
|
|
9
|
+
import { retryReasoningFor } from "./types.js";
|
|
10
|
+
import { SCHEMAS } from "./schemas.js";
|
|
11
|
+
import { getLens } from "./lenses.js";
|
|
12
|
+
import { sanitizeId } from "./repo.js";
|
|
13
|
+
import { broadsideDirFor, claimRunSlot, loadBroadsideState, persistBroadsideRunMerging, resetRunPostPasses } from "./state.js";
|
|
14
|
+
import { explainBatchError, pollBatchesConcurrently, submitBatch } from "./client.js";
|
|
15
|
+
import { extractContent, loadSavedLensResults, loadStoredRequests, parseLensJson, renderFindingsMarkdown, saveLensResults } from "./results.js";
|
|
16
|
+
import { parseSynthesisTopFindings } from "./render.js";
|
|
17
|
+
/**
|
|
18
|
+
* The verdicts a verify pass left in the run directory, or null when none
|
|
19
|
+
* has run (#338). A file that does not parse is treated as absent: the
|
|
20
|
+
* post-passes then run from the findings alone, which is what they did
|
|
21
|
+
* before verdicts existed, and `status` shows the pass carried no verdicts.
|
|
22
|
+
*/
|
|
23
|
+
export async function loadPostPassVerdicts(runDir) {
|
|
24
|
+
const path = join(runDir, "verified.json");
|
|
25
|
+
if (!(await pathExists(path)))
|
|
26
|
+
return null;
|
|
27
|
+
try {
|
|
28
|
+
const parsed = JSON.parse(await readFile(path, "utf8"));
|
|
29
|
+
const findings = Array.isArray(parsed.findings) ? parsed.findings : [];
|
|
30
|
+
const verdicts = findings
|
|
31
|
+
.filter((f) => typeof f.title === "string" && typeof f.verdict === "string")
|
|
32
|
+
.map((f) => ({
|
|
33
|
+
lensId: String(f.lensId ?? ""),
|
|
34
|
+
customId: String(f.customId ?? ""),
|
|
35
|
+
severity: String(f.severity ?? ""),
|
|
36
|
+
title: String(f.title),
|
|
37
|
+
location: String(f.location ?? ""),
|
|
38
|
+
verdict: String(f.verdict),
|
|
39
|
+
confidence: String(f.confidence ?? ""),
|
|
40
|
+
evidence: Array.isArray(f.evidence)
|
|
41
|
+
? f.evidence.map((e) => ({ file: String(e.file ?? ""), lines: String(e.lines ?? ""), note: String(e.note ?? "") }))
|
|
42
|
+
: [],
|
|
43
|
+
reasoning: String(f.reasoning ?? ""),
|
|
44
|
+
}));
|
|
45
|
+
return verdicts.length > 0 ? verdicts : null;
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* The verdicts as a section of the post-pass user message: one line per
|
|
53
|
+
* finding with the verdict, the evidence the verifier cited, and its
|
|
54
|
+
* reasoning, so the pass can rank on them rather than on the batch model's
|
|
55
|
+
* own severities (#338).
|
|
56
|
+
*/
|
|
57
|
+
export function renderPostPassVerdicts(verdicts) {
|
|
58
|
+
const counts = new Map();
|
|
59
|
+
for (const v of verdicts)
|
|
60
|
+
counts.set(v.verdict, (counts.get(v.verdict) ?? 0) + 1);
|
|
61
|
+
const tally = [...counts.entries()].map(([verdict, n]) => `${n} ${verdict}`).join(", ");
|
|
62
|
+
const lines = [
|
|
63
|
+
"",
|
|
64
|
+
`## Verification verdicts (${verdicts.length} finding(s) read against the source by a read-only-tools pass: ${tally})`,
|
|
65
|
+
"",
|
|
66
|
+
"A verdict outranks the batch severity of the finding it names. `confirmed` means the verifier found a reachable " +
|
|
67
|
+
"failure and named its trigger; `not-a-defect` means the claim is literally true of the code but nothing reaches the " +
|
|
68
|
+
"failure it describes; `discarded` means the claim is wrong about the code; `unclear` means the code alone could not " +
|
|
69
|
+
"settle it; `error` means the pass could not read it — treat that finding as unverified. Findings not listed here " +
|
|
70
|
+
"were not read and stay unverified leads.",
|
|
71
|
+
"",
|
|
72
|
+
];
|
|
73
|
+
for (const v of verdicts) {
|
|
74
|
+
const evidence = v.evidence.map((e) => `${e.file}${e.lines ? `:${e.lines}` : ""}${e.note ? ` (${e.note})` : ""}`).join("; ");
|
|
75
|
+
lines.push(`- [${v.verdict}${v.confidence ? `, ${v.confidence} confidence` : ""}] ${v.lensId}/${v.customId} — [${v.severity}] ${v.title}` +
|
|
76
|
+
`${v.location ? ` @ ${v.location}` : ""}` +
|
|
77
|
+
`${v.reasoning ? `\n Reasoning: ${v.reasoning.replace(/\s+/g, " ").trim()}` : ""}` +
|
|
78
|
+
`${evidence ? `\n Evidence: ${evidence}` : ""}`);
|
|
79
|
+
}
|
|
80
|
+
lines.push("");
|
|
81
|
+
return lines.join("\n");
|
|
82
|
+
}
|
|
83
|
+
const SYNTHESIS_VERDICT_INSTRUCTIONS = " A verification pass has read some of the findings against the source; its verdicts follow the reports. " +
|
|
84
|
+
"Lead top_findings with the confirmed findings and begin each such summary with 'verified: confirmed — ' and the " +
|
|
85
|
+
"trigger the verifier named; keep an unclear one with 'verified: unclear — '. A discarded or not-a-defect finding " +
|
|
86
|
+
"does not appear in top_findings and is not counted in severity_summary. Say in the executive summary how many " +
|
|
87
|
+
"findings were verified and how the verdicts split; findings the pass did not read remain unverified, and the " +
|
|
88
|
+
"summary says so of them, not of the confirmed ones.";
|
|
89
|
+
const TRIAGE_VERDICT_INSTRUCTIONS = " A verification pass has read some of the findings against the source; its verdicts follow the findings. " +
|
|
90
|
+
"A confirmed finding ranks above every unverified finding of the same or lower severity: put the confirmed " +
|
|
91
|
+
"findings at the top of the queue and begin each one's rationale with 'verified: confirmed — ' and the trigger " +
|
|
92
|
+
"the verifier named. Keep an unclear finding in the queue with 'verified: unclear — ' in its rationale. Do not " +
|
|
93
|
+
"queue a discarded or not-a-defect finding: list each in omitted, beginning with 'verified: discarded — ' or " +
|
|
94
|
+
"'verified: not a defect — ' and the reason the pass gave. Findings the pass did not read stay unverified leads, " +
|
|
95
|
+
"and the summary says how many verdicts the queue was built from.";
|
|
96
|
+
function buildSynthesisRequest(findingsText, truncatedNote, model, verdicts = null) {
|
|
97
|
+
return {
|
|
98
|
+
custom_id: "synthesis",
|
|
99
|
+
body: {
|
|
100
|
+
model,
|
|
101
|
+
messages: [
|
|
102
|
+
{
|
|
103
|
+
role: "system",
|
|
104
|
+
content: "You are a technical editor synthesizing multiple analysis reports about a single " +
|
|
105
|
+
"codebase into one coherent summary. The reports come from different lenses — " +
|
|
106
|
+
"architecture, API surface, security review, defect scanning, convention extraction, " +
|
|
107
|
+
"and porting assessment. Cross-reference findings across lenses: if a security issue " +
|
|
108
|
+
"also appears as a defect, merge them. Produce a JSON object following the " +
|
|
109
|
+
"synthesis_report schema. Prioritize the most actionable findings. " +
|
|
110
|
+
"Be honest about gaps — if a lens found nothing, say 'no issues found' rather than " +
|
|
111
|
+
"inventing problems. These are scouting signals from a batch model, not verified " +
|
|
112
|
+
"claims; note that in the summary." +
|
|
113
|
+
(verdicts ? SYNTHESIS_VERDICT_INSTRUCTIONS : ""),
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
role: "user",
|
|
117
|
+
content: "Synthesize these analysis reports into a single summary.\n\n" +
|
|
118
|
+
findingsText +
|
|
119
|
+
truncatedNote +
|
|
120
|
+
(verdicts ? renderPostPassVerdicts(verdicts) : "") +
|
|
121
|
+
"\nReturn the synthesis_report JSON schema.",
|
|
122
|
+
},
|
|
123
|
+
],
|
|
124
|
+
response_format: { type: "json_schema", json_schema: SCHEMAS.synthesis },
|
|
125
|
+
max_tokens: 12_000,
|
|
126
|
+
},
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
function buildTriageRequest(findingsText, truncatedNote, model, verdicts = null) {
|
|
130
|
+
return {
|
|
131
|
+
custom_id: "triage",
|
|
132
|
+
body: {
|
|
133
|
+
model,
|
|
134
|
+
messages: [
|
|
135
|
+
{
|
|
136
|
+
role: "system",
|
|
137
|
+
content: "You are a senior engineering lead turning unverified scouting findings into a " +
|
|
138
|
+
"prioritized work order. Given the findings below, produce a JSON object following " +
|
|
139
|
+
"the triage_report schema. Score every lead by impact and fix difficulty, assign a " +
|
|
140
|
+
"priority (P0 urgent/safety-critical to P3 nice-to-have), give a rough effort " +
|
|
141
|
+
"estimate, group the queue by module where sensible, and justify each call in the " +
|
|
142
|
+
"rationale. Merge duplicate leads instead of listing them twice. Drop leads that are " +
|
|
143
|
+
"too vague to act on and record each drop in omitted with the reason. These findings " +
|
|
144
|
+
"are UNVERIFIED scouting signals from a cheap batch model: the queue is a starting " +
|
|
145
|
+
"point for re-verification, not a commitment — say so in the summary, and never " +
|
|
146
|
+
"inflate a severity you cannot see evidence for." +
|
|
147
|
+
(verdicts ? TRIAGE_VERDICT_INSTRUCTIONS : ""),
|
|
148
|
+
},
|
|
149
|
+
{
|
|
150
|
+
role: "user",
|
|
151
|
+
content: "Triage these scouting findings into a prioritized work order.\n\n" +
|
|
152
|
+
findingsText +
|
|
153
|
+
truncatedNote +
|
|
154
|
+
(verdicts ? renderPostPassVerdicts(verdicts) : "") +
|
|
155
|
+
"\nReturn the triage_report JSON schema.",
|
|
156
|
+
},
|
|
157
|
+
],
|
|
158
|
+
response_format: { type: "json_schema", json_schema: SCHEMAS.triage },
|
|
159
|
+
max_tokens: 10_000,
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
function parseTriageItems(content) {
|
|
164
|
+
try {
|
|
165
|
+
const parsed = (parseLensJson(content) ?? {});
|
|
166
|
+
const items = Array.isArray(parsed.items) ? parsed.items : [];
|
|
167
|
+
return items
|
|
168
|
+
.filter((item) => typeof item.title === "string")
|
|
169
|
+
.map((item) => ({
|
|
170
|
+
title: String(item.title),
|
|
171
|
+
severity: String(item.severity ?? "unknown"),
|
|
172
|
+
module: String(item.module ?? "unknown"),
|
|
173
|
+
impact: (["high", "medium", "low"].includes(String(item.impact)) ? String(item.impact) : "medium"),
|
|
174
|
+
difficulty: (["high", "medium", "low"].includes(String(item.difficulty)) ? String(item.difficulty) : "medium"),
|
|
175
|
+
priority: String(item.priority ?? "?"),
|
|
176
|
+
effort_estimate: String(item.effort_estimate ?? ""),
|
|
177
|
+
rationale: String(item.rationale ?? ""),
|
|
178
|
+
}));
|
|
179
|
+
}
|
|
180
|
+
catch {
|
|
181
|
+
return [];
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
185
|
+
const broadsideDir = broadsideDirFor(cwd);
|
|
186
|
+
const state = await loadBroadsideState(broadsideDir);
|
|
187
|
+
const run = opts.runId ? state.runs.find((candidate) => candidate.id === opts.runId) : state.runs[state.runs.length - 1];
|
|
188
|
+
if (!run) {
|
|
189
|
+
if (opts.runId) {
|
|
190
|
+
const known = state.runs.map((candidate) => candidate.id);
|
|
191
|
+
throw new Error(`No Broad-Side run with id ${opts.runId}. ` +
|
|
192
|
+
(known.length > 0 ? `Recorded runs: ${known.join(", ")}.` : "No runs are recorded; call codecarto_broadside with action 'submit' first."));
|
|
193
|
+
}
|
|
194
|
+
throw new Error("No Broad-Side run recorded. Call codecarto_broadside with action 'submit' first.");
|
|
195
|
+
}
|
|
196
|
+
const runDir = join(broadsideDir, run.outputDir);
|
|
197
|
+
await mkdir(runDir, { recursive: true });
|
|
198
|
+
// The spending slots this collect has claimed (#322); only a claimed slot
|
|
199
|
+
// is ever submitted from here. Every write-back merges with the file, so a
|
|
200
|
+
// slot another collect has moved further along is never overwritten.
|
|
201
|
+
const owned = new Set();
|
|
202
|
+
const persist = () => persistBroadsideRunMerging(broadsideDir, run);
|
|
203
|
+
const aborted = () => opts.signal?.aborted === true;
|
|
204
|
+
const deadline = Date.now() + (opts.waitMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
|
|
205
|
+
let totalCost = 0;
|
|
206
|
+
let resultCount = 0;
|
|
207
|
+
let truncatedCount = 0;
|
|
208
|
+
const lensOutcomes = {};
|
|
209
|
+
const allLensResults = [];
|
|
210
|
+
// Terminal entries are settled already; everything else polls in parallel
|
|
211
|
+
// against one shared deadline (#136), then results save in lens order so
|
|
212
|
+
// output layout stays deterministic.
|
|
213
|
+
const inFlight = [];
|
|
214
|
+
for (const lensId of run.lenses) {
|
|
215
|
+
const entry = run.batches[lensId];
|
|
216
|
+
if (!entry || !entry.batchId) {
|
|
217
|
+
lensOutcomes[lensId] = { status: entry?.status ?? "failed", resultCount: 0 };
|
|
218
|
+
continue;
|
|
219
|
+
}
|
|
220
|
+
if (BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status)) {
|
|
221
|
+
totalCost += entry.cost ?? 0;
|
|
222
|
+
resultCount += entry.resultCount ?? 0;
|
|
223
|
+
lensOutcomes[lensId] = { status: entry.status, cost: entry.cost, resultCount: entry.resultCount };
|
|
224
|
+
continue;
|
|
225
|
+
}
|
|
226
|
+
inFlight.push({ lensId, batchId: entry.batchId });
|
|
227
|
+
}
|
|
228
|
+
const polled = await pollBatchesConcurrently(inFlight, apiKey, {
|
|
229
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
230
|
+
fetcher: opts.fetcher,
|
|
231
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
232
|
+
signal: opts.signal,
|
|
233
|
+
onStatus: opts.onStatus,
|
|
234
|
+
});
|
|
235
|
+
for (const { lensId } of inFlight) {
|
|
236
|
+
const entry = run.batches[lensId];
|
|
237
|
+
if (!entry)
|
|
238
|
+
continue;
|
|
239
|
+
const batch = polled.get(entry.batchId) ?? { id: entry.batchId, status: "timeout" };
|
|
240
|
+
const status = String(batch.status ?? "unknown");
|
|
241
|
+
entry.status = status;
|
|
242
|
+
if (status === "completed") {
|
|
243
|
+
const usage = (batch.usage ?? {});
|
|
244
|
+
const cost = typeof usage.cost === "number" ? usage.cost : undefined;
|
|
245
|
+
entry.cost = cost;
|
|
246
|
+
entry.completedAt = new Date().toISOString();
|
|
247
|
+
const stored = await saveLensResults(runDir, lensId, batch);
|
|
248
|
+
entry.resultCount = stored.length;
|
|
249
|
+
const truncated = stored.filter((s) => s.truncated).length;
|
|
250
|
+
allLensResults.push(...stored);
|
|
251
|
+
resultCount += stored.length;
|
|
252
|
+
truncatedCount += truncated;
|
|
253
|
+
totalCost += cost ?? 0;
|
|
254
|
+
await writeFile(join(runDir, `raw-${lensId}.json`), `${JSON.stringify(batch, null, "\t")}\n`, "utf8");
|
|
255
|
+
// A batch can complete with every request failed — the account's
|
|
256
|
+
// concurrent-job quota filling after acceptance does exactly this.
|
|
257
|
+
// The per-request errors are on disk as `<id>.error.json`, but a
|
|
258
|
+
// lens reporting "completed, 0 result(s)" with the reason buried
|
|
259
|
+
// there read as an empty repository rather than a refused run.
|
|
260
|
+
const results = Array.isArray(batch.results) ? batch.results : [];
|
|
261
|
+
const failed = results.filter((r) => r.error && extractContent(r) === null);
|
|
262
|
+
const allFailed = stored.length === 0 && failed.length > 0
|
|
263
|
+
? `all ${failed.length} request(s) failed: ${explainBatchError(failed[0].error)}`
|
|
264
|
+
: null;
|
|
265
|
+
if (allFailed)
|
|
266
|
+
entry.error = allFailed;
|
|
267
|
+
lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, truncated, ...(allFailed && { error: allFailed }) };
|
|
268
|
+
}
|
|
269
|
+
else {
|
|
270
|
+
// Every non-completed outcome still has to reach the report.
|
|
271
|
+
// `lensOutcomes` is what the caller renders, and this branch used to
|
|
272
|
+
// require `batch.error` — but the commonest failure here is the
|
|
273
|
+
// synthetic `{ status: "timeout" }` the poll returns when its budget
|
|
274
|
+
// expires with the batch still in flight, and that carries no error.
|
|
275
|
+
// A lens that never came back was therefore omitted entirely,
|
|
276
|
+
// indistinguishable in the output from one that was never requested.
|
|
277
|
+
if (batch.error)
|
|
278
|
+
entry.error = batch.error;
|
|
279
|
+
const error = explainBatchError(batch.error);
|
|
280
|
+
lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, ...(error && { error }) };
|
|
281
|
+
}
|
|
282
|
+
await persist();
|
|
283
|
+
}
|
|
284
|
+
// #133: re-submit truncated slices once with a bumped output cap and low
|
|
285
|
+
// reasoning effort. Batch requests are pure, so re-running is always safe;
|
|
286
|
+
// the aim is to recover coverage the first pass lost to a max_tokens
|
|
287
|
+
// cutoff, not to loop forever. Low effort because the cutoff is usually
|
|
288
|
+
// thinking, and a doubled budget doubled the thinking where a token cap
|
|
289
|
+
// was ignored (see retryReasoningFor).
|
|
290
|
+
//
|
|
291
|
+
// All bumped requests for one model go out as ONE batch, and the batches
|
|
292
|
+
// (one per model, since a batch carries a single model) are polled
|
|
293
|
+
// together against the shared deadline. Each truncated slice used to be
|
|
294
|
+
// submitted and polled to terminal before the next was submitted, so a
|
|
295
|
+
// model that truncated 11 of 13 slices turned a five-minute collect into
|
|
296
|
+
// eleven sequential round trips — the serialization #136 removed from the
|
|
297
|
+
// lens pass, still present here (#206). Grouping also keeps the retry to
|
|
298
|
+
// one job per model against OpenRouter's 16-concurrent-job quota.
|
|
299
|
+
let retriedCount = 0;
|
|
300
|
+
let retryElsewhere = false;
|
|
301
|
+
// A collect that polled nothing — every lens already terminal — still owes
|
|
302
|
+
// the retry if the collect that saved the results never got to it (it
|
|
303
|
+
// died, or its client did: #322). Read the saved results back and let the
|
|
304
|
+
// claim decide; a recovered slice re-parses clean, so this costs nothing
|
|
305
|
+
// once the retry has run.
|
|
306
|
+
if (opts.retryTruncated !== false && allLensResults.length === 0 && !aborted()) {
|
|
307
|
+
const everyLensTerminal = run.lenses.every((lensId) => {
|
|
308
|
+
const entry = run.batches[lensId];
|
|
309
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
310
|
+
});
|
|
311
|
+
if (everyLensTerminal) {
|
|
312
|
+
const restored = await loadSavedLensResults(runDir, run.lenses);
|
|
313
|
+
if (restored.some((s) => s.truncated)) {
|
|
314
|
+
allLensResults.push(...restored);
|
|
315
|
+
truncatedCount = restored.filter((s) => s.truncated).length;
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
}
|
|
319
|
+
if (opts.retryTruncated !== false && truncatedCount > 0 && !aborted()) {
|
|
320
|
+
// Claim the pass before spending: a second collect on this run finds the
|
|
321
|
+
// claim and leaves the retry to the first (#322). A retry another
|
|
322
|
+
// collect has already settled is not run again — its truncation is
|
|
323
|
+
// what it is.
|
|
324
|
+
if (await claimRunSlot(broadsideDir, run, "retry"))
|
|
325
|
+
owned.add("retry");
|
|
326
|
+
else if (run.retry?.status === "submitted")
|
|
327
|
+
retryElsewhere = true;
|
|
328
|
+
}
|
|
329
|
+
if (opts.retryTruncated !== false && truncatedCount > 0 && owned.has("retry")) {
|
|
330
|
+
const requestsByCustomId = await loadStoredRequests(runDir);
|
|
331
|
+
const byModel = new Map();
|
|
332
|
+
for (const stored of allLensResults) {
|
|
333
|
+
if (!stored.truncated)
|
|
334
|
+
continue;
|
|
335
|
+
const original = requestsByCustomId[stored.customId];
|
|
336
|
+
if (!original)
|
|
337
|
+
continue;
|
|
338
|
+
const lensEntry = run.batches[stored.lensId];
|
|
339
|
+
// A lens may have run on its own model (config `lens_models`), with its
|
|
340
|
+
// own completion ceiling. Re-submitting against the run default would
|
|
341
|
+
// change the model mid-run and could exceed that lens's real ceiling.
|
|
342
|
+
const lensModel = lensEntry?.model ?? run.model;
|
|
343
|
+
const lensCap = lensEntry?.outputCap ?? run.outputCap;
|
|
344
|
+
const previousMax = original.body.max_tokens ?? getLens(stored.lensId).maxTokens;
|
|
345
|
+
const bumpedMax = lensCap ? Math.min(previousMax * 2, lensCap) : previousMax * 2;
|
|
346
|
+
if (bumpedMax <= previousMax)
|
|
347
|
+
continue; // already at the ceiling
|
|
348
|
+
const group = byModel.get(lensModel) ?? { requests: [], slices: new Map() };
|
|
349
|
+
group.requests.push({
|
|
350
|
+
...original,
|
|
351
|
+
body: { ...original.body, max_tokens: bumpedMax, reasoning: retryReasoningFor(original.body.reasoning) },
|
|
352
|
+
});
|
|
353
|
+
group.slices.set(stored.customId, stored);
|
|
354
|
+
byModel.set(lensModel, group);
|
|
355
|
+
}
|
|
356
|
+
// Submit every group, then poll whatever was accepted, together.
|
|
357
|
+
const submitted = [];
|
|
358
|
+
const refusals = [];
|
|
359
|
+
for (const [model, group] of byModel) {
|
|
360
|
+
if (aborted())
|
|
361
|
+
break;
|
|
362
|
+
try {
|
|
363
|
+
const { batchId, error } = await submitBatch(group.requests, apiKey, opts.fetcher, model);
|
|
364
|
+
if (!error && batchId)
|
|
365
|
+
submitted.push({ model, batchId });
|
|
366
|
+
else
|
|
367
|
+
refusals.push(`${model}: ${explainBatchError(error) ?? "no batch id returned"}`);
|
|
368
|
+
}
|
|
369
|
+
catch (error) {
|
|
370
|
+
// A retry batch that fails to submit leaves its slices' original
|
|
371
|
+
// truncated results in place — nothing is lost, and the reason
|
|
372
|
+
// travels on the entry rather than vanishing here (#370).
|
|
373
|
+
refusals.push(`${model}: ${explainBatchError(error) ?? String(error)}`);
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
// Record the ids under the claim so a later collect can see what was
|
|
377
|
+
// paid for, even if this one never returns. No group at all means every
|
|
378
|
+
// truncated slice was already at its model's ceiling: nothing to retry.
|
|
379
|
+
run.retry = {
|
|
380
|
+
...run.retry,
|
|
381
|
+
batches: submitted,
|
|
382
|
+
status: submitted.length > 0 ? "submitted" : byModel.size === 0 ? "completed" : "failed",
|
|
383
|
+
...(refusals.length > 0 && { error: refusals.join("; ") }),
|
|
384
|
+
};
|
|
385
|
+
await persist();
|
|
386
|
+
const polled = await pollBatchesConcurrently(submitted.map(({ model, batchId }) => ({ lensId: `retry:${model}`, batchId })), apiKey, {
|
|
387
|
+
// Share the caller's deadline. Each of these polls used to start a
|
|
388
|
+
// fresh 25-minute budget, so `wait_seconds` bounded only the lens
|
|
389
|
+
// poll and a collect could run for the caller's budget plus fifty
|
|
390
|
+
// minutes.
|
|
391
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
392
|
+
fetcher: opts.fetcher,
|
|
393
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
394
|
+
signal: opts.signal,
|
|
395
|
+
onStatus: opts.onStatus,
|
|
396
|
+
});
|
|
397
|
+
for (const { model, batchId } of submitted) {
|
|
398
|
+
const batch = polled.get(batchId);
|
|
399
|
+
if (!batch || batch.status !== "completed")
|
|
400
|
+
continue;
|
|
401
|
+
const group = byModel.get(model);
|
|
402
|
+
const usage = (batch.usage ?? {});
|
|
403
|
+
// Kept on the entry, not just added to this collect's running total:
|
|
404
|
+
// a later collect on the run used to report a total without it.
|
|
405
|
+
if (typeof usage.cost === "number")
|
|
406
|
+
run.retry = { ...run.retry, cost: (run.retry?.cost ?? 0) + usage.cost };
|
|
407
|
+
const results = Array.isArray(batch.results) ? batch.results : [];
|
|
408
|
+
for (const result of results) {
|
|
409
|
+
const stored = group.slices.get(String(result.custom_id ?? ""));
|
|
410
|
+
if (!stored)
|
|
411
|
+
continue;
|
|
412
|
+
const content = extractContent(result);
|
|
413
|
+
if (content === null)
|
|
414
|
+
continue;
|
|
415
|
+
const parsed = parseLensJson(content);
|
|
416
|
+
if (parsed === null)
|
|
417
|
+
continue; // still no good
|
|
418
|
+
await writeFile(join(runDir, `${sanitizeId(stored.customId)}.json`), `${JSON.stringify(parsed, null, "\t")}\n`, "utf8");
|
|
419
|
+
await writeFile(join(runDir, `${sanitizeId(stored.customId)}.md`), renderFindingsMarkdown(content), "utf8");
|
|
420
|
+
stored.content = content;
|
|
421
|
+
stored.truncated = false;
|
|
422
|
+
retriedCount += 1;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
// Every retry batch reached a terminal status, or the poll ran out.
|
|
426
|
+
if (submitted.length > 0 && submitted.every(({ batchId }) => polled.get(batchId)?.status === "completed")) {
|
|
427
|
+
run.retry = { ...run.retry, status: "completed" };
|
|
428
|
+
}
|
|
429
|
+
truncatedCount = allLensResults.filter((s) => s.truncated).length;
|
|
430
|
+
for (const [lensId, outcome] of Object.entries(lensOutcomes)) {
|
|
431
|
+
if (outcome.truncated !== undefined) {
|
|
432
|
+
outcome.truncated = allLensResults.filter((s) => s.lensId === lensId && s.truncated).length;
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
await persist();
|
|
436
|
+
}
|
|
437
|
+
// Synthesis + triage: cross-lens post-passes, only after every lens batch
|
|
438
|
+
// is terminal. Triage turns the leads into a prioritized work order.
|
|
439
|
+
run.triage ??= { status: "pending" };
|
|
440
|
+
let topFindings = [];
|
|
441
|
+
let topTriageItems = [];
|
|
442
|
+
const wantSynthesis = opts.includeSynthesis !== false;
|
|
443
|
+
const wantTriage = opts.includeTriage !== false;
|
|
444
|
+
// A regenerate resets the wanted, settled passes to pending on disk first —
|
|
445
|
+
// the merging persist keeps whatever is further along on disk, so an
|
|
446
|
+
// in-memory reset alone would be undone by the next persist (#338).
|
|
447
|
+
let regenerated = [];
|
|
448
|
+
if (opts.regeneratePostPasses) {
|
|
449
|
+
if (!wantSynthesis && !wantTriage) {
|
|
450
|
+
throw new Error("Nothing to regenerate: both post-passes are disabled for this collect.");
|
|
451
|
+
}
|
|
452
|
+
const lensesSettled = run.lenses.every((lensId) => {
|
|
453
|
+
const entry = run.batches[lensId];
|
|
454
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
455
|
+
});
|
|
456
|
+
if (!lensesSettled) {
|
|
457
|
+
throw new Error(`Cannot regenerate the post-passes of run ${run.id}: its lens batches are still running — collect them first.`);
|
|
458
|
+
}
|
|
459
|
+
regenerated = await resetRunPostPasses(broadsideDir, run, { synthesis: wantSynthesis, triage: wantTriage });
|
|
460
|
+
}
|
|
461
|
+
// A resumed collect polls nothing — every lens is already terminal — so the
|
|
462
|
+
// findings the post-passes need have to come back off disk, or a run whose
|
|
463
|
+
// first collect was interrupted could never produce its executive report
|
|
464
|
+
// and work order, however many times it was re-run.
|
|
465
|
+
const postPassUnfinished = (entry) => entry.status === "pending" || entry.status === "submitted";
|
|
466
|
+
if ((wantSynthesis || wantTriage) && allLensResults.length === 0
|
|
467
|
+
&& (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
|
|
468
|
+
const restored = await loadSavedLensResults(runDir, run.lenses);
|
|
469
|
+
if (restored.length > 0) {
|
|
470
|
+
allLensResults.push(...restored);
|
|
471
|
+
truncatedCount = restored.filter((s) => s.truncated).length;
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
if ((wantSynthesis || wantTriage) && allLensResults.length > 0) {
|
|
475
|
+
const allTerminal = run.lenses.every((lensId) => {
|
|
476
|
+
const entry = run.batches[lensId];
|
|
477
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
478
|
+
});
|
|
479
|
+
if (allTerminal && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
|
|
480
|
+
const findingsText = allLensResults
|
|
481
|
+
.map((r) => `## ${r.lensId} — ${r.customId}\n\n${r.content}\n`)
|
|
482
|
+
.join("\n");
|
|
483
|
+
// A verify pass that ran before this point leaves its verdicts in the
|
|
484
|
+
// run directory; the post-passes rank on them when present (#338).
|
|
485
|
+
const verdicts = await loadPostPassVerdicts(runDir);
|
|
486
|
+
const truncatedNote = truncatedCount > 0
|
|
487
|
+
? `\n\nNOTE: ${truncatedCount} lens result(s) were truncated at the output token limit and are ` +
|
|
488
|
+
"not included above. Any gap they would have covered is unrepresented — do not treat " +
|
|
489
|
+
"silence on a module as a clean bill.\n"
|
|
490
|
+
: "";
|
|
491
|
+
// Both post-passes consume the same findings; they run as two
|
|
492
|
+
// batches (different response_format schemas cannot share one)
|
|
493
|
+
// submitted together and polled in turn.
|
|
494
|
+
// Claim each wanted, still-pending pass before building its request:
|
|
495
|
+
// a second collect on this run adopts the first one's entry instead
|
|
496
|
+
// of submitting its own (#322). An abort submits nothing further.
|
|
497
|
+
const passes = [];
|
|
498
|
+
for (const kind of ["synthesis", "triage"]) {
|
|
499
|
+
const want = kind === "synthesis" ? wantSynthesis : wantTriage;
|
|
500
|
+
if (!want || aborted())
|
|
501
|
+
continue;
|
|
502
|
+
if ((kind === "synthesis" ? run.synthesis : run.triage).status !== "pending")
|
|
503
|
+
continue;
|
|
504
|
+
if (!(await claimRunSlot(broadsideDir, run, kind)))
|
|
505
|
+
continue;
|
|
506
|
+
owned.add(kind);
|
|
507
|
+
const entry = kind === "synthesis" ? run.synthesis : run.triage;
|
|
508
|
+
if (verdicts)
|
|
509
|
+
entry.verdicts = verdicts.length;
|
|
510
|
+
else
|
|
511
|
+
delete entry.verdicts;
|
|
512
|
+
passes.push({
|
|
513
|
+
kind,
|
|
514
|
+
request: kind === "synthesis"
|
|
515
|
+
? buildSynthesisRequest(findingsText, truncatedNote, run.model, verdicts)
|
|
516
|
+
: buildTriageRequest(findingsText, truncatedNote, run.model, verdicts),
|
|
517
|
+
entry,
|
|
518
|
+
});
|
|
519
|
+
}
|
|
520
|
+
const submitted = new Map();
|
|
521
|
+
// A pass can be left at "submitted" when an earlier collect returned
|
|
522
|
+
// before its batch reached a terminal status — the batch still runs
|
|
523
|
+
// and is still charged, so the result exists and is simply unclaimed.
|
|
524
|
+
// Nothing above would ever look at it again: the pass list is built
|
|
525
|
+
// from "pending" entries only. Poll those regardless of the want
|
|
526
|
+
// flags, because the spend already happened and discarding a
|
|
527
|
+
// finished result is worse than saving one the caller opted out of.
|
|
528
|
+
for (const kind of ["synthesis", "triage"]) {
|
|
529
|
+
const entry = kind === "synthesis" ? run.synthesis : run.triage;
|
|
530
|
+
if (entry.status !== "submitted" || !entry.batchId)
|
|
531
|
+
continue;
|
|
532
|
+
if (submitted.has(entry.batchId))
|
|
533
|
+
continue;
|
|
534
|
+
submitted.set(entry.batchId, {
|
|
535
|
+
batchId: entry.batchId,
|
|
536
|
+
pass: { kind, request: undefined, entry },
|
|
537
|
+
});
|
|
538
|
+
}
|
|
539
|
+
await Promise.allSettled(passes.map(async (pass) => {
|
|
540
|
+
pass.entry.status = "submitted";
|
|
541
|
+
try {
|
|
542
|
+
const { batchId, error } = await submitBatch([pass.request], apiKey, opts.fetcher, run.model);
|
|
543
|
+
if (error) {
|
|
544
|
+
pass.entry.status = "failed";
|
|
545
|
+
return;
|
|
546
|
+
}
|
|
547
|
+
pass.entry.batchId = batchId;
|
|
548
|
+
submitted.set(batchId, { batchId, pass });
|
|
549
|
+
}
|
|
550
|
+
catch {
|
|
551
|
+
pass.entry.status = "failed";
|
|
552
|
+
}
|
|
553
|
+
}));
|
|
554
|
+
await persist();
|
|
555
|
+
// Poll both passes together against the shared deadline. Polled in
|
|
556
|
+
// turn, the first pass could spend the whole budget and leave the
|
|
557
|
+
// second a single poll (0.22.1 live run: triage settled, synthesis
|
|
558
|
+
// left running though it had been submitted at the same moment).
|
|
559
|
+
// A pass whose poll runs out stays `submitted`, so the batch is
|
|
560
|
+
// already paid for and a later collect claims its result.
|
|
561
|
+
const polledPasses = await pollBatchesConcurrently([...submitted.values()].map(({ batchId, pass }) => ({ lensId: pass.kind, batchId })), apiKey, {
|
|
562
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
563
|
+
fetcher: opts.fetcher,
|
|
564
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
565
|
+
signal: opts.signal,
|
|
566
|
+
onStatus: opts.onStatus,
|
|
567
|
+
});
|
|
568
|
+
for (const { batchId, pass } of submitted.values()) {
|
|
569
|
+
const batch = polledPasses.get(batchId) ?? { id: batchId, status: "timeout" };
|
|
570
|
+
if (batch.status === "completed") {
|
|
571
|
+
const usage = (batch.usage ?? {});
|
|
572
|
+
const cost = typeof usage.cost === "number" ? usage.cost : undefined;
|
|
573
|
+
pass.entry.status = "completed";
|
|
574
|
+
pass.entry.cost = cost;
|
|
575
|
+
const results = Array.isArray(batch.results) ? batch.results : [];
|
|
576
|
+
const content = results.length > 0 ? extractContent(results[0]) : null;
|
|
577
|
+
// The same tolerance the lens path has (#366): a reply wrapped in
|
|
578
|
+
// a code fence is JSON; one that is not JSON at all — cut off at
|
|
579
|
+
// the output cap, or prose — is a failed pass that says so, not a
|
|
580
|
+
// completed one with nothing in it.
|
|
581
|
+
const parsed = content !== null ? parseLensJson(content) : null;
|
|
582
|
+
if (content !== null && parsed !== null) {
|
|
583
|
+
const normalized = JSON.stringify(parsed, null, "\t");
|
|
584
|
+
await writeFile(join(runDir, `${pass.kind}.json`), `${normalized}\n`, "utf8");
|
|
585
|
+
await writeFile(join(runDir, `${pass.kind}.md`), renderFindingsMarkdown(normalized), "utf8");
|
|
586
|
+
if (pass.kind === "synthesis") {
|
|
587
|
+
topFindings = parseSynthesisTopFindings(normalized);
|
|
588
|
+
}
|
|
589
|
+
else {
|
|
590
|
+
topTriageItems = parseTriageItems(normalized);
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
else {
|
|
594
|
+
pass.entry.status = "failed";
|
|
595
|
+
pass.entry.error = content === null
|
|
596
|
+
? "the batch completed without a result body"
|
|
597
|
+
: "the reply was not a JSON object — cut off at the output cap, or prose; the raw text is saved beside the run";
|
|
598
|
+
if (content !== null)
|
|
599
|
+
await writeFile(join(runDir, `${pass.kind}.raw.txt`), `${content}\n`, "utf8");
|
|
600
|
+
}
|
|
601
|
+
}
|
|
602
|
+
else if (BROADSIDE_DEAD_BATCH_STATUSES.includes(String(batch.status))) {
|
|
603
|
+
// The batch will never produce a result, so retire the pass.
|
|
604
|
+
// This used to require `batch.error`, leaving an expired or
|
|
605
|
+
// cancelled batch parked at "submitted" forever — and since a
|
|
606
|
+
// resumed collect re-polls anything still "submitted", it
|
|
607
|
+
// would re-poll a dead batch on every future run.
|
|
608
|
+
pass.entry.status = "failed";
|
|
609
|
+
if (batch.error)
|
|
610
|
+
pass.entry.error = batch.error instanceof Error ? batch.error.message : String(batch.error);
|
|
611
|
+
}
|
|
612
|
+
// A "timeout" is deliberately left at "submitted": the batch is
|
|
613
|
+
// still running server-side and has already been paid for, so a
|
|
614
|
+
// later collect should claim its result rather than discard it.
|
|
615
|
+
await persist();
|
|
616
|
+
}
|
|
617
|
+
}
|
|
618
|
+
}
|
|
619
|
+
const terminal = run.lenses.every((lensId) => {
|
|
620
|
+
const entry = run.batches[lensId];
|
|
621
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
622
|
+
});
|
|
623
|
+
run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
|
|
624
|
+
// The run's total is the sum of what its entries record, not of what this
|
|
625
|
+
// collect happened to poll: a repeat collect used to report — and persist
|
|
626
|
+
// — a total without the post-passes and the retry an earlier collect had
|
|
627
|
+
// settled, so the recorded cost of a run went down each time it was read.
|
|
628
|
+
totalCost += (run.retry?.cost ?? 0) + (run.synthesis.cost ?? 0) + (run.triage.cost ?? 0) + (run.retiredCost ?? 0);
|
|
629
|
+
run.totalCost = totalCost;
|
|
630
|
+
await persist();
|
|
631
|
+
await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
|
|
632
|
+
experimental: true,
|
|
633
|
+
method: "Broad-Side (OpenRouter Batch API)",
|
|
634
|
+
model: run.model,
|
|
635
|
+
pricing: run.pricing,
|
|
636
|
+
max_cost: run.maxCost,
|
|
637
|
+
run_id: run.id,
|
|
638
|
+
created_at: run.createdAt,
|
|
639
|
+
status: run.status,
|
|
640
|
+
total_cost: totalCost,
|
|
641
|
+
result_count: resultCount,
|
|
642
|
+
truncated_count: truncatedCount,
|
|
643
|
+
retried_count: retriedCount,
|
|
644
|
+
synthesis: run.synthesis,
|
|
645
|
+
triage: run.triage,
|
|
646
|
+
lenses: run.lenses,
|
|
647
|
+
// Which lens ran on which model. Absent means the run default —
|
|
648
|
+
// a reader comparing two runs needs to know a lens changed model.
|
|
649
|
+
lens_models: Object.fromEntries(Object.entries(run.batches)
|
|
650
|
+
.filter(([, batch]) => batch?.model)
|
|
651
|
+
.map(([lensId, batch]) => [lensId, batch.model])),
|
|
652
|
+
disclaimer: "Findings are unverified scouting signals from a batch model, not validated claims. " +
|
|
653
|
+
"Re-verify every file:line lead with the interactive pipeline or by hand.",
|
|
654
|
+
}, null, "\t")}\n`, "utf8");
|
|
655
|
+
return {
|
|
656
|
+
runId: run.id,
|
|
657
|
+
status: run.status,
|
|
658
|
+
totalCost,
|
|
659
|
+
resultCount,
|
|
660
|
+
truncatedCount,
|
|
661
|
+
retriedCount,
|
|
662
|
+
...(run.retry?.error && { retryError: run.retry.error }),
|
|
663
|
+
...(retryElsewhere && { retryElsewhere: true }),
|
|
664
|
+
lensOutcomes,
|
|
665
|
+
synthesis: run.synthesis,
|
|
666
|
+
triage: run.triage,
|
|
667
|
+
topFindings,
|
|
668
|
+
topTriageItems,
|
|
669
|
+
...(regenerated.length > 0 && { regenerated }),
|
|
670
|
+
};
|
|
671
|
+
}
|
|
672
|
+
export async function runBroadsideStatus(cwd) {
|
|
673
|
+
const broadsideDir = broadsideDirFor(cwd);
|
|
674
|
+
const state = await loadBroadsideState(broadsideDir);
|
|
675
|
+
return { state };
|
|
676
|
+
}
|