codecartographer-pi 0.24.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +31 -9
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +4 -3
- package/agent-skill/codecartographer/references/broadside.md +5 -1
- package/dist/core/amendment.d.ts +6 -3
- package/dist/core/amendment.js +21 -10
- package/dist/core/broadside.d.ts +87 -16
- package/dist/core/broadside.js +230 -30
- package/dist/core/status.d.ts +14 -0
- package/dist/core/status.js +104 -13
- package/dist/extensions/codecarto/agent-state.d.ts +7 -1
- package/dist/extensions/codecarto/agent-state.js +9 -1
- package/dist/extensions/codecarto/auto-runner.d.ts +7 -0
- package/dist/extensions/codecarto/auto-runner.js +16 -2
- package/dist/extensions/codecarto/broadside-flags.d.ts +3 -1
- package/dist/extensions/codecarto/broadside-flags.js +13 -0
- package/dist/extensions/codecarto/index.js +38 -8
- package/dist/mcp-server/server.d.ts +1 -0
- package/dist/mcp-server/server.js +18 -1
- package/package.json +1 -1
package/dist/core/broadside.js
CHANGED
|
@@ -749,7 +749,19 @@ const LENSES = {
|
|
|
749
749
|
"Return a JSON object following the defect_scan_report schema. " +
|
|
750
750
|
"Cite file:line for every finding. List which patterns you checked. " +
|
|
751
751
|
"If the code looks clean for a pattern, say so rather than staying silent. " +
|
|
752
|
-
"Prefer precision over volume — 3 solid findings beat 15 vague ones
|
|
752
|
+
"Prefer precision over volume — 3 solid findings beat 15 vague ones.\n\n" +
|
|
753
|
+
// The verification pass (#143) confirmed 2 of the 12 top findings a
|
|
754
|
+
// scan produced with the paragraph above alone; the other ten were
|
|
755
|
+
// casts and assertions every caller satisfied, guards that lived one
|
|
756
|
+
// call away, or environments the project does not target. The rubric
|
|
757
|
+
// the verifier applies is asked of the scan itself, up front.
|
|
758
|
+
"A finding is a reachable failure: name in the description the concrete input, call site, or sequence " +
|
|
759
|
+
"that reaches it and what then goes wrong. A cast, assertion, `any`, or non-null `!` that every caller " +
|
|
760
|
+
"you can see satisfies, a hypothetical about a runtime or environment the project does not target, or a " +
|
|
761
|
+
"style or type-hygiene observation is not a defect — leave it out, or if it is worth a note, report it " +
|
|
762
|
+
"at severity low under the pattern name `type-hygiene` so it ranks apart from reachable failures. " +
|
|
763
|
+
"When the guard you looked for may live in another module, say which check you could not find " +
|
|
764
|
+
"rather than asserting it is absent; severity high or medium is for failures you traced to a trigger.");
|
|
753
765
|
},
|
|
754
766
|
userPrompt: (info, source, moduleName) => `Scan this ${info.language} module for mechanical defects.\n\n` +
|
|
755
767
|
`Module: ${moduleName}\n\n` +
|
|
@@ -1254,25 +1266,51 @@ function collectFilesMatching(allFiles, lens, globs) {
|
|
|
1254
1266
|
}
|
|
1255
1267
|
return out;
|
|
1256
1268
|
}
|
|
1269
|
+
/** Code in any language Broad-Side scans as, whatever this repo's is. */
|
|
1270
|
+
const SOURCE_EXTENSIONS = new Set(Object.values(SOURCE_SPECS).flatMap((spec) => spec.exts));
|
|
1271
|
+
function isSourceFile(relPath) {
|
|
1272
|
+
const dot = relPath.lastIndexOf(".");
|
|
1273
|
+
return dot > relPath.lastIndexOf("/") && SOURCE_EXTENSIONS.has(relPath.slice(dot).toLowerCase());
|
|
1274
|
+
}
|
|
1275
|
+
/** `a, b, c and 4 more` — a matched-file list short enough for a status line. */
|
|
1276
|
+
function listSome(paths, max = 3) {
|
|
1277
|
+
if (paths.length <= max)
|
|
1278
|
+
return paths.join(", ");
|
|
1279
|
+
return `${paths.slice(0, max).join(", ")} and ${paths.length - max} more`;
|
|
1280
|
+
}
|
|
1257
1281
|
/**
|
|
1258
|
-
* The files a lens will read: its targeted globs, or — when those match
|
|
1259
|
-
*
|
|
1260
|
-
* sentence saying so (#319). The sentence
|
|
1261
|
-
* batch entry, and the prompt, so a fallback
|
|
1282
|
+
* The files a lens will read: its targeted globs, or — when those match no
|
|
1283
|
+
* source file and the lens declares a fallback — the fallback globs on top
|
|
1284
|
+
* of whatever did match, with a sentence saying so (#319). The sentence
|
|
1285
|
+
* travels to the estimate, the batch entry, and the prompt, so a fallback
|
|
1286
|
+
* scan is never a silent one.
|
|
1287
|
+
*
|
|
1288
|
+
* "No source file" rather than "no file": a policy document or a config
|
|
1289
|
+
* file under a targeted path satisfies the globs and leaves the lens with
|
|
1290
|
+
* nothing to review, and the coverage note it writes back is the only sign.
|
|
1262
1291
|
*/
|
|
1263
1292
|
export function selectLensFiles(allFiles, lens, info) {
|
|
1264
1293
|
const globs = lens.globsFor(info).filter(Boolean);
|
|
1265
1294
|
const targeted = collectFilesMatching(allFiles, lens, globs);
|
|
1266
|
-
if (
|
|
1295
|
+
if (globs.length === 0 || !lens.fallbackGlobsFor)
|
|
1296
|
+
return { files: targeted };
|
|
1297
|
+
if (targeted.some((f) => isSourceFile(f.relPath)))
|
|
1267
1298
|
return { files: targeted };
|
|
1268
1299
|
const fallbackGlobs = lens.fallbackGlobsFor(info).filter(Boolean);
|
|
1269
|
-
const
|
|
1270
|
-
|
|
1271
|
-
|
|
1300
|
+
const matched = new Set(targeted.map((f) => f.relPath));
|
|
1301
|
+
const sources = collectFilesMatching(allFiles, lens, fallbackGlobs).filter((f) => !matched.has(f.relPath));
|
|
1302
|
+
if (sources.length === 0)
|
|
1303
|
+
return { files: targeted };
|
|
1304
|
+
const excluded = lens.skipTestFiles ? "test files excluded" : "";
|
|
1305
|
+
const scanned = `scanned all ${info.language} sources (${fallbackGlobs.join(", ")})`;
|
|
1272
1306
|
return {
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1307
|
+
// What did match rides first: the policy the model is about to check
|
|
1308
|
+
// the code against, ahead of the code.
|
|
1309
|
+
files: [...targeted, ...sources],
|
|
1310
|
+
fallback: targeted.length === 0
|
|
1311
|
+
? `no files matched ${globs.join(", ")}${excluded ? ` (${excluded})` : ""}; ${scanned} instead`
|
|
1312
|
+
: `no source files matched ${globs.join(", ")} (only ${listSome(targeted.map((f) => f.relPath))}` +
|
|
1313
|
+
`${excluded ? `; ${excluded}` : ""}); ${scanned} as well`,
|
|
1276
1314
|
};
|
|
1277
1315
|
}
|
|
1278
1316
|
function collectLensFiles(allFiles, lens, info) {
|
|
@@ -1396,8 +1434,8 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1396
1434
|
// so the model judges the trust boundary wherever it appears
|
|
1397
1435
|
// and does not report the missing server/ as a finding (#319).
|
|
1398
1436
|
(slice.fallback
|
|
1399
|
-
? `NOTE: this repository has no files under the paths this lens usually reads (${slice.fallback}). ` +
|
|
1400
|
-
"What follows is every source file it has; locate the trust boundary and the request-handling code wherever they live.\n\n"
|
|
1437
|
+
? `NOTE: this repository has no source files under the paths this lens usually reads (${slice.fallback}). ` +
|
|
1438
|
+
"What follows is every source file it has, after anything those paths did match; locate the trust boundary and the request-handling code wherever they live.\n\n"
|
|
1401
1439
|
: "") + lens.userPrompt(info, slice.content, slice.moduleName),
|
|
1402
1440
|
},
|
|
1403
1441
|
],
|
|
@@ -1680,6 +1718,39 @@ export async function claimRunSlot(broadsideDir, run, slot) {
|
|
|
1680
1718
|
});
|
|
1681
1719
|
return owned;
|
|
1682
1720
|
}
|
|
1721
|
+
/**
|
|
1722
|
+
* Put a run's settled post-passes back to `pending` on disk so the next
|
|
1723
|
+
* claim re-runs them (#338). A pass another collect has in flight is left
|
|
1724
|
+
* alone — its result is still coming. The replaced results' cost moves to
|
|
1725
|
+
* `retiredCost`, so the run's total keeps counting money it spent. Returns
|
|
1726
|
+
* the passes that were reset, in the order they will be re-run.
|
|
1727
|
+
*/
|
|
1728
|
+
export async function resetRunPostPasses(broadsideDir, run, wanted) {
|
|
1729
|
+
const reset = [];
|
|
1730
|
+
await updateBroadsideStateAtomically(broadsideDir, (state) => {
|
|
1731
|
+
const index = state.runs.findIndex((candidate) => candidate.id === run.id);
|
|
1732
|
+
const onDisk = index === -1 ? run : state.runs[index];
|
|
1733
|
+
for (const kind of ["synthesis", "triage"]) {
|
|
1734
|
+
if (!wanted[kind])
|
|
1735
|
+
continue;
|
|
1736
|
+
const theirs = onDisk[kind] ?? { status: "pending" };
|
|
1737
|
+
if (theirs.status !== "completed" && theirs.status !== "failed") {
|
|
1738
|
+
// pending: nothing to reset; submitted: in flight elsewhere.
|
|
1739
|
+
run[kind] = theirs;
|
|
1740
|
+
continue;
|
|
1741
|
+
}
|
|
1742
|
+
if (theirs.cost)
|
|
1743
|
+
onDisk.retiredCost = (onDisk.retiredCost ?? 0) + theirs.cost;
|
|
1744
|
+
onDisk[kind] = { status: "pending" };
|
|
1745
|
+
run[kind] = onDisk[kind];
|
|
1746
|
+
run.retiredCost = onDisk.retiredCost;
|
|
1747
|
+
reset.push(kind);
|
|
1748
|
+
}
|
|
1749
|
+
if (index === -1)
|
|
1750
|
+
state.runs.push(run);
|
|
1751
|
+
});
|
|
1752
|
+
return reset;
|
|
1753
|
+
}
|
|
1683
1754
|
/** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
|
|
1684
1755
|
function parseReasoningConfig(raw) {
|
|
1685
1756
|
if (raw === false)
|
|
@@ -2684,8 +2755,86 @@ async function loadStoredRequests(runDir) {
|
|
|
2684
2755
|
return {};
|
|
2685
2756
|
}
|
|
2686
2757
|
}
|
|
2687
|
-
|
|
2688
|
-
|
|
2758
|
+
/**
|
|
2759
|
+
* The verdicts a verify pass left in the run directory, or null when none
|
|
2760
|
+
* has run (#338). A file that does not parse is treated as absent: the
|
|
2761
|
+
* post-passes then run from the findings alone, which is what they did
|
|
2762
|
+
* before verdicts existed, and `status` shows the pass carried no verdicts.
|
|
2763
|
+
*/
|
|
2764
|
+
export async function loadPostPassVerdicts(runDir) {
|
|
2765
|
+
const path = join(runDir, "verified.json");
|
|
2766
|
+
if (!(await pathExists(path)))
|
|
2767
|
+
return null;
|
|
2768
|
+
try {
|
|
2769
|
+
const parsed = JSON.parse(await readFile(path, "utf8"));
|
|
2770
|
+
const findings = Array.isArray(parsed.findings) ? parsed.findings : [];
|
|
2771
|
+
const verdicts = findings
|
|
2772
|
+
.filter((f) => typeof f.title === "string" && typeof f.verdict === "string")
|
|
2773
|
+
.map((f) => ({
|
|
2774
|
+
lensId: String(f.lensId ?? ""),
|
|
2775
|
+
customId: String(f.customId ?? ""),
|
|
2776
|
+
severity: String(f.severity ?? ""),
|
|
2777
|
+
title: String(f.title),
|
|
2778
|
+
location: String(f.location ?? ""),
|
|
2779
|
+
verdict: String(f.verdict),
|
|
2780
|
+
confidence: String(f.confidence ?? ""),
|
|
2781
|
+
evidence: Array.isArray(f.evidence)
|
|
2782
|
+
? f.evidence.map((e) => ({ file: String(e.file ?? ""), lines: String(e.lines ?? ""), note: String(e.note ?? "") }))
|
|
2783
|
+
: [],
|
|
2784
|
+
reasoning: String(f.reasoning ?? ""),
|
|
2785
|
+
}));
|
|
2786
|
+
return verdicts.length > 0 ? verdicts : null;
|
|
2787
|
+
}
|
|
2788
|
+
catch {
|
|
2789
|
+
return null;
|
|
2790
|
+
}
|
|
2791
|
+
}
|
|
2792
|
+
/**
|
|
2793
|
+
* The verdicts as a section of the post-pass user message: one line per
|
|
2794
|
+
* finding with the verdict, the evidence the verifier cited, and its
|
|
2795
|
+
* reasoning, so the pass can rank on them rather than on the batch model's
|
|
2796
|
+
* own severities (#338).
|
|
2797
|
+
*/
|
|
2798
|
+
export function renderPostPassVerdicts(verdicts) {
|
|
2799
|
+
const counts = new Map();
|
|
2800
|
+
for (const v of verdicts)
|
|
2801
|
+
counts.set(v.verdict, (counts.get(v.verdict) ?? 0) + 1);
|
|
2802
|
+
const tally = [...counts.entries()].map(([verdict, n]) => `${n} ${verdict}`).join(", ");
|
|
2803
|
+
const lines = [
|
|
2804
|
+
"",
|
|
2805
|
+
`## Verification verdicts (${verdicts.length} finding(s) read against the source by a read-only-tools pass: ${tally})`,
|
|
2806
|
+
"",
|
|
2807
|
+
"A verdict outranks the batch severity of the finding it names. `confirmed` means the verifier found a reachable " +
|
|
2808
|
+
"failure and named its trigger; `not-a-defect` means the claim is literally true of the code but nothing reaches the " +
|
|
2809
|
+
"failure it describes; `discarded` means the claim is wrong about the code; `unclear` means the code alone could not " +
|
|
2810
|
+
"settle it; `error` means the pass could not read it — treat that finding as unverified. Findings not listed here " +
|
|
2811
|
+
"were not read and stay unverified leads.",
|
|
2812
|
+
"",
|
|
2813
|
+
];
|
|
2814
|
+
for (const v of verdicts) {
|
|
2815
|
+
const evidence = v.evidence.map((e) => `${e.file}${e.lines ? `:${e.lines}` : ""}${e.note ? ` (${e.note})` : ""}`).join("; ");
|
|
2816
|
+
lines.push(`- [${v.verdict}${v.confidence ? `, ${v.confidence} confidence` : ""}] ${v.lensId}/${v.customId} — [${v.severity}] ${v.title}` +
|
|
2817
|
+
`${v.location ? ` @ ${v.location}` : ""}` +
|
|
2818
|
+
`${v.reasoning ? `\n Reasoning: ${v.reasoning.replace(/\s+/g, " ").trim()}` : ""}` +
|
|
2819
|
+
`${evidence ? `\n Evidence: ${evidence}` : ""}`);
|
|
2820
|
+
}
|
|
2821
|
+
lines.push("");
|
|
2822
|
+
return lines.join("\n");
|
|
2823
|
+
}
|
|
2824
|
+
const SYNTHESIS_VERDICT_INSTRUCTIONS = " A verification pass has read some of the findings against the source; its verdicts follow the reports. " +
|
|
2825
|
+
"Lead top_findings with the confirmed findings and begin each such summary with 'verified: confirmed — ' and the " +
|
|
2826
|
+
"trigger the verifier named; keep an unclear one with 'verified: unclear — '. A discarded or not-a-defect finding " +
|
|
2827
|
+
"does not appear in top_findings and is not counted in severity_summary. Say in the executive summary how many " +
|
|
2828
|
+
"findings were verified and how the verdicts split; findings the pass did not read remain unverified, and the " +
|
|
2829
|
+
"summary says so of them, not of the confirmed ones.";
|
|
2830
|
+
const TRIAGE_VERDICT_INSTRUCTIONS = " A verification pass has read some of the findings against the source; its verdicts follow the findings. " +
|
|
2831
|
+
"A confirmed finding ranks above every unverified finding of the same or lower severity: put the confirmed " +
|
|
2832
|
+
"findings at the top of the queue and begin each one's rationale with 'verified: confirmed — ' and the trigger " +
|
|
2833
|
+
"the verifier named. Keep an unclear finding in the queue with 'verified: unclear — ' in its rationale. Do not " +
|
|
2834
|
+
"queue a discarded or not-a-defect finding: list each in omitted, beginning with 'verified: discarded — ' or " +
|
|
2835
|
+
"'verified: not a defect — ' and the reason the pass gave. Findings the pass did not read stay unverified leads, " +
|
|
2836
|
+
"and the summary says how many verdicts the queue was built from.";
|
|
2837
|
+
function buildSynthesisRequest(findingsText, truncatedNote, model, verdicts = null) {
|
|
2689
2838
|
return {
|
|
2690
2839
|
custom_id: "synthesis",
|
|
2691
2840
|
body: {
|
|
@@ -2701,13 +2850,15 @@ function buildSynthesisRequest(findingsText, truncatedNote, model) {
|
|
|
2701
2850
|
"synthesis_report schema. Prioritize the most actionable findings. " +
|
|
2702
2851
|
"Be honest about gaps — if a lens found nothing, say 'no issues found' rather than " +
|
|
2703
2852
|
"inventing problems. These are scouting signals from a batch model, not verified " +
|
|
2704
|
-
"claims; note that in the summary."
|
|
2853
|
+
"claims; note that in the summary." +
|
|
2854
|
+
(verdicts ? SYNTHESIS_VERDICT_INSTRUCTIONS : ""),
|
|
2705
2855
|
},
|
|
2706
2856
|
{
|
|
2707
2857
|
role: "user",
|
|
2708
2858
|
content: "Synthesize these analysis reports into a single summary.\n\n" +
|
|
2709
2859
|
findingsText +
|
|
2710
2860
|
truncatedNote +
|
|
2861
|
+
(verdicts ? renderPostPassVerdicts(verdicts) : "") +
|
|
2711
2862
|
"\nReturn the synthesis_report JSON schema.",
|
|
2712
2863
|
},
|
|
2713
2864
|
],
|
|
@@ -2716,7 +2867,7 @@ function buildSynthesisRequest(findingsText, truncatedNote, model) {
|
|
|
2716
2867
|
},
|
|
2717
2868
|
};
|
|
2718
2869
|
}
|
|
2719
|
-
function buildTriageRequest(findingsText, truncatedNote, model) {
|
|
2870
|
+
function buildTriageRequest(findingsText, truncatedNote, model, verdicts = null) {
|
|
2720
2871
|
return {
|
|
2721
2872
|
custom_id: "triage",
|
|
2722
2873
|
body: {
|
|
@@ -2733,13 +2884,15 @@ function buildTriageRequest(findingsText, truncatedNote, model) {
|
|
|
2733
2884
|
"too vague to act on and record each drop in omitted with the reason. These findings " +
|
|
2734
2885
|
"are UNVERIFIED scouting signals from a cheap batch model: the queue is a starting " +
|
|
2735
2886
|
"point for re-verification, not a commitment — say so in the summary, and never " +
|
|
2736
|
-
"inflate a severity you cannot see evidence for."
|
|
2887
|
+
"inflate a severity you cannot see evidence for." +
|
|
2888
|
+
(verdicts ? TRIAGE_VERDICT_INSTRUCTIONS : ""),
|
|
2737
2889
|
},
|
|
2738
2890
|
{
|
|
2739
2891
|
role: "user",
|
|
2740
2892
|
content: "Triage these scouting findings into a prioritized work order.\n\n" +
|
|
2741
2893
|
findingsText +
|
|
2742
2894
|
truncatedNote +
|
|
2895
|
+
(verdicts ? renderPostPassVerdicts(verdicts) : "") +
|
|
2743
2896
|
"\nReturn the triage_report JSON schema.",
|
|
2744
2897
|
},
|
|
2745
2898
|
],
|
|
@@ -2982,7 +3135,10 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2982
3135
|
continue;
|
|
2983
3136
|
const group = byModel.get(model);
|
|
2984
3137
|
const usage = (batch.usage ?? {});
|
|
2985
|
-
|
|
3138
|
+
// Kept on the entry, not just added to this collect's running total:
|
|
3139
|
+
// a later collect on the run used to report a total without it.
|
|
3140
|
+
if (typeof usage.cost === "number")
|
|
3141
|
+
run.retry = { ...run.retry, cost: (run.retry?.cost ?? 0) + usage.cost };
|
|
2986
3142
|
const results = Array.isArray(batch.results) ? batch.results : [];
|
|
2987
3143
|
for (const result of results) {
|
|
2988
3144
|
const stored = group.slices.get(String(result.custom_id ?? ""));
|
|
@@ -3020,6 +3176,23 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3020
3176
|
let topTriageItems = [];
|
|
3021
3177
|
const wantSynthesis = opts.includeSynthesis !== false;
|
|
3022
3178
|
const wantTriage = opts.includeTriage !== false;
|
|
3179
|
+
// A regenerate resets the wanted, settled passes to pending on disk first —
|
|
3180
|
+
// the merging persist keeps whatever is further along on disk, so an
|
|
3181
|
+
// in-memory reset alone would be undone by the next persist (#338).
|
|
3182
|
+
let regenerated = [];
|
|
3183
|
+
if (opts.regeneratePostPasses) {
|
|
3184
|
+
if (!wantSynthesis && !wantTriage) {
|
|
3185
|
+
throw new Error("Nothing to regenerate: both post-passes are disabled for this collect.");
|
|
3186
|
+
}
|
|
3187
|
+
const lensesSettled = run.lenses.every((lensId) => {
|
|
3188
|
+
const entry = run.batches[lensId];
|
|
3189
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
3190
|
+
});
|
|
3191
|
+
if (!lensesSettled) {
|
|
3192
|
+
throw new Error(`Cannot regenerate the post-passes of run ${run.id}: its lens batches are still running — collect them first.`);
|
|
3193
|
+
}
|
|
3194
|
+
regenerated = await resetRunPostPasses(broadsideDir, run, { synthesis: wantSynthesis, triage: wantTriage });
|
|
3195
|
+
}
|
|
3023
3196
|
// A resumed collect polls nothing — every lens is already terminal — so the
|
|
3024
3197
|
// findings the post-passes need have to come back off disk, or a run whose
|
|
3025
3198
|
// first collect was interrupted could never produce its executive report
|
|
@@ -3042,6 +3215,9 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3042
3215
|
const findingsText = allLensResults
|
|
3043
3216
|
.map((r) => `## ${r.lensId} — ${r.customId}\n\n${r.content}\n`)
|
|
3044
3217
|
.join("\n");
|
|
3218
|
+
// A verify pass that ran before this point leaves its verdicts in the
|
|
3219
|
+
// run directory; the post-passes rank on them when present (#338).
|
|
3220
|
+
const verdicts = await loadPostPassVerdicts(runDir);
|
|
3045
3221
|
const truncatedNote = truncatedCount > 0
|
|
3046
3222
|
? `\n\nNOTE: ${truncatedCount} lens result(s) were truncated at the output token limit and are ` +
|
|
3047
3223
|
"not included above. Any gap they would have covered is unrepresented — do not treat " +
|
|
@@ -3063,12 +3239,17 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3063
3239
|
if (!(await claimRunSlot(broadsideDir, run, kind)))
|
|
3064
3240
|
continue;
|
|
3065
3241
|
owned.add(kind);
|
|
3242
|
+
const entry = kind === "synthesis" ? run.synthesis : run.triage;
|
|
3243
|
+
if (verdicts)
|
|
3244
|
+
entry.verdicts = verdicts.length;
|
|
3245
|
+
else
|
|
3246
|
+
delete entry.verdicts;
|
|
3066
3247
|
passes.push({
|
|
3067
3248
|
kind,
|
|
3068
3249
|
request: kind === "synthesis"
|
|
3069
|
-
? buildSynthesisRequest(findingsText, truncatedNote, run.model)
|
|
3070
|
-
: buildTriageRequest(findingsText, truncatedNote, run.model),
|
|
3071
|
-
entry
|
|
3250
|
+
? buildSynthesisRequest(findingsText, truncatedNote, run.model, verdicts)
|
|
3251
|
+
: buildTriageRequest(findingsText, truncatedNote, run.model, verdicts),
|
|
3252
|
+
entry,
|
|
3072
3253
|
});
|
|
3073
3254
|
}
|
|
3074
3255
|
const submitted = new Map();
|
|
@@ -3126,7 +3307,6 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3126
3307
|
const cost = typeof usage.cost === "number" ? usage.cost : undefined;
|
|
3127
3308
|
pass.entry.status = "completed";
|
|
3128
3309
|
pass.entry.cost = cost;
|
|
3129
|
-
totalCost += cost ?? 0;
|
|
3130
3310
|
const results = Array.isArray(batch.results) ? batch.results : [];
|
|
3131
3311
|
const content = results.length > 0 ? extractContent(results[0]) : null;
|
|
3132
3312
|
if (content !== null) {
|
|
@@ -3162,6 +3342,11 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3162
3342
|
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
3163
3343
|
});
|
|
3164
3344
|
run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
|
|
3345
|
+
// The run's total is the sum of what its entries record, not of what this
|
|
3346
|
+
// collect happened to poll: a repeat collect used to report — and persist
|
|
3347
|
+
// — a total without the post-passes and the retry an earlier collect had
|
|
3348
|
+
// settled, so the recorded cost of a run went down each time it was read.
|
|
3349
|
+
totalCost += (run.retry?.cost ?? 0) + (run.synthesis.cost ?? 0) + (run.triage.cost ?? 0) + (run.retiredCost ?? 0);
|
|
3165
3350
|
run.totalCost = totalCost;
|
|
3166
3351
|
await persist();
|
|
3167
3352
|
await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
|
|
@@ -3201,6 +3386,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
3201
3386
|
triage: run.triage,
|
|
3202
3387
|
topFindings,
|
|
3203
3388
|
topTriageItems,
|
|
3389
|
+
...(regenerated.length > 0 && { regenerated }),
|
|
3204
3390
|
};
|
|
3205
3391
|
}
|
|
3206
3392
|
export async function runBroadsideStatus(cwd) {
|
|
@@ -3483,10 +3669,17 @@ export function collectResultText(result) {
|
|
|
3483
3669
|
lines.push(` ${kind}: failed${entry.error ? ` — ${explainBatchError(entry.error)}` : ""}`);
|
|
3484
3670
|
}
|
|
3485
3671
|
};
|
|
3672
|
+
// Whether a pass was built from a verify pass's verdicts is part of what
|
|
3673
|
+
// it is: a work order that ranked on batch severities alone is the one
|
|
3674
|
+
// that put two dismissed casts above the confirmed finding (#338).
|
|
3675
|
+
const builtFrom = (entry) => entry.verdicts ? ` (built from ${entry.verdicts} verdict${entry.verdicts === 1 ? "" : "s"})` : " (no verdicts)";
|
|
3676
|
+
if (result.regenerated && result.regenerated.length > 0) {
|
|
3677
|
+
lines.push(` regenerated: ${result.regenerated.join(", ")}`);
|
|
3678
|
+
}
|
|
3486
3679
|
if (result.synthesis.status === "completed") {
|
|
3487
|
-
lines.push(` synthesis: completed, $${(result.synthesis.cost ?? 0).toFixed(6)}`);
|
|
3680
|
+
lines.push(` synthesis: completed, $${(result.synthesis.cost ?? 0).toFixed(6)}${builtFrom(result.synthesis)}`);
|
|
3488
3681
|
if (result.topFindings.length > 0) {
|
|
3489
|
-
lines.push("", "Top findings (unverified leads):");
|
|
3682
|
+
lines.push("", result.synthesis.verdicts ? "Top findings (verdicts applied; unread ones are unverified leads):" : "Top findings (unverified leads):");
|
|
3490
3683
|
for (const f of result.topFindings.slice(0, 10)) {
|
|
3491
3684
|
lines.push(` [${f.severity}] ${f.title}`);
|
|
3492
3685
|
}
|
|
@@ -3496,9 +3689,11 @@ export function collectResultText(result) {
|
|
|
3496
3689
|
passInFlight("synthesis", result.synthesis);
|
|
3497
3690
|
}
|
|
3498
3691
|
if (result.triage.status === "completed") {
|
|
3499
|
-
lines.push(` triage: completed, $${(result.triage.cost ?? 0).toFixed(6)}`);
|
|
3692
|
+
lines.push(` triage: completed, $${(result.triage.cost ?? 0).toFixed(6)}${builtFrom(result.triage)}`);
|
|
3500
3693
|
if (result.topTriageItems.length > 0) {
|
|
3501
|
-
lines.push("",
|
|
3694
|
+
lines.push("", result.triage.verdicts
|
|
3695
|
+
? "Triage — prioritized work order (confirmed findings first; re-verify the unread ones before acting):"
|
|
3696
|
+
: "Triage — prioritized work order (re-verify before acting):");
|
|
3502
3697
|
for (const item of result.topTriageItems.slice(0, 10)) {
|
|
3503
3698
|
lines.push(` ${item.priority} [${item.severity}/${item.module}] ${item.title}` +
|
|
3504
3699
|
(item.effort_estimate ? ` (${item.effort_estimate})` : ""));
|
|
@@ -3508,6 +3703,10 @@ export function collectResultText(result) {
|
|
|
3508
3703
|
else {
|
|
3509
3704
|
passInFlight("triage", result.triage);
|
|
3510
3705
|
}
|
|
3706
|
+
if (result.status === "completed" && !result.synthesis.verdicts && !result.triage.verdicts
|
|
3707
|
+
&& (result.synthesis.status === "completed" || result.triage.status === "completed")) {
|
|
3708
|
+
lines.push("", "Run verify, then collect --regenerate, to rebuild the report and the work order from verdicts.");
|
|
3709
|
+
}
|
|
3511
3710
|
lines.push("", "Disclaimer: Broad-Side findings are unverified scouting signals from a batch model, not validated claims.");
|
|
3512
3711
|
return lines.join("\n");
|
|
3513
3712
|
}
|
|
@@ -3547,8 +3746,9 @@ export function statusText(state) {
|
|
|
3547
3746
|
lines.push(` ${lensId}: ${entry.status}${entry.batchId ? ` (${entry.batchId})` : ""}${entry.cost !== undefined ? `, $${entry.cost.toFixed(6)}` : ""}` +
|
|
3548
3747
|
(entry.status === "skipped" && entry.reason ? ` — ${entry.reason}` : entry.fallback ? ` — ${entry.fallback}` : ""));
|
|
3549
3748
|
}
|
|
3550
|
-
|
|
3551
|
-
lines.push(`
|
|
3749
|
+
const builtFrom = (entry) => entry?.status === "completed" ? (entry.verdicts ? ` (built from ${entry.verdicts} verdict${entry.verdicts === 1 ? "" : "s"})` : " (no verdicts)") : "";
|
|
3750
|
+
lines.push(` synthesis: ${run.synthesis.status}${builtFrom(run.synthesis)}`);
|
|
3751
|
+
lines.push(` triage: ${run.triage?.status ?? "pending"}${builtFrom(run.triage)}`);
|
|
3552
3752
|
if (run.verify) {
|
|
3553
3753
|
lines.push(` verify: ${run.verify.status} — ${run.verify.confirmed} confirmed of ${run.verify.verified} read on ${run.verify.model}, $${run.verify.cost.toFixed(4)}`);
|
|
3554
3754
|
}
|
package/dist/core/status.d.ts
CHANGED
|
@@ -2,6 +2,12 @@ import type { ClosureEntry, NormalizedStatus, OpenQuestionEntry, PostPipelineEnt
|
|
|
2
2
|
export declare const LOCK_RETRY_MS = 125;
|
|
3
3
|
export declare const LOCK_TIMEOUT_MS = 5000;
|
|
4
4
|
export declare const STALE_LOCK_MS = 60000;
|
|
5
|
+
/**
|
|
6
|
+
* How old the removal lock (`<lock>.break`, see {@link withRemovalLock}) may
|
|
7
|
+
* be before it is treated as left behind by a crashed process. It is held
|
|
8
|
+
* across one stat and one rm, so anything this old was abandoned.
|
|
9
|
+
*/
|
|
10
|
+
export declare const BREAK_LOCK_STALE_MS = 5000;
|
|
5
11
|
export declare function assertSafePhaseId(phaseId: string): void;
|
|
6
12
|
/**
|
|
7
13
|
* A YAML scalar as text: strings as written, numbers and booleans spelled
|
|
@@ -73,5 +79,13 @@ export interface LockHandle {
|
|
|
73
79
|
* the token, release removed whoever's lock was there: after a stale break
|
|
74
80
|
* the previous holder's release deleted the new holder's lock, and a third
|
|
75
81
|
* writer walked straight in (#227).
|
|
82
|
+
*
|
|
83
|
+
* Every removal — a release or a stale break — happens under the removal
|
|
84
|
+
* lock (`<lock>.break`) and re-checks what it is about to remove there.
|
|
85
|
+
* Two waiters that both saw a stale lock used to both `rm` it: the second
|
|
86
|
+
* `rm` landed after the first waiter had re-created the file, so both held
|
|
87
|
+
* the lock (#342). A file can only be created while the path is free, and
|
|
88
|
+
* only a removal-lock holder removes, so what a holder verified is what it
|
|
89
|
+
* removes.
|
|
76
90
|
*/
|
|
77
91
|
export declare function acquireLock(lockPath: string): Promise<LockHandle>;
|
package/dist/core/status.js
CHANGED
|
@@ -8,6 +8,12 @@ import { loadYamlFile } from "./yaml.js";
|
|
|
8
8
|
export const LOCK_RETRY_MS = 125;
|
|
9
9
|
export const LOCK_TIMEOUT_MS = 5000;
|
|
10
10
|
export const STALE_LOCK_MS = 60_000;
|
|
11
|
+
/**
|
|
12
|
+
* How old the removal lock (`<lock>.break`, see {@link withRemovalLock}) may
|
|
13
|
+
* be before it is treated as left behind by a crashed process. It is held
|
|
14
|
+
* across one stat and one rm, so anything this old was abandoned.
|
|
15
|
+
*/
|
|
16
|
+
export const BREAK_LOCK_STALE_MS = LOCK_TIMEOUT_MS;
|
|
11
17
|
export function assertSafePhaseId(phaseId) {
|
|
12
18
|
if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(phaseId)) {
|
|
13
19
|
throw new Error(`Invalid phase id: ${phaseId}`);
|
|
@@ -434,6 +440,14 @@ export function applyHandoff(status, handoff) {
|
|
|
434
440
|
* the token, release removed whoever's lock was there: after a stale break
|
|
435
441
|
* the previous holder's release deleted the new holder's lock, and a third
|
|
436
442
|
* writer walked straight in (#227).
|
|
443
|
+
*
|
|
444
|
+
* Every removal — a release or a stale break — happens under the removal
|
|
445
|
+
* lock (`<lock>.break`) and re-checks what it is about to remove there.
|
|
446
|
+
* Two waiters that both saw a stale lock used to both `rm` it: the second
|
|
447
|
+
* `rm` landed after the first waiter had re-created the file, so both held
|
|
448
|
+
* the lock (#342). A file can only be created while the path is free, and
|
|
449
|
+
* only a removal-lock holder removes, so what a holder verified is what it
|
|
450
|
+
* removes.
|
|
437
451
|
*/
|
|
438
452
|
export async function acquireLock(lockPath) {
|
|
439
453
|
const startedAt = Date.now();
|
|
@@ -464,9 +478,13 @@ export async function acquireLock(lockPath) {
|
|
|
464
478
|
try {
|
|
465
479
|
const lockStat = await stat(lockPath);
|
|
466
480
|
if (Date.now() - lockStat.mtimeMs > STALE_LOCK_MS) {
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
481
|
+
const broken = await breakStaleLock(lockPath);
|
|
482
|
+
if (broken) {
|
|
483
|
+
brokeStale = broken;
|
|
484
|
+
continue;
|
|
485
|
+
}
|
|
486
|
+
// Another waiter is breaking it, or already has: fall
|
|
487
|
+
// through to a wait and try the open again.
|
|
470
488
|
}
|
|
471
489
|
}
|
|
472
490
|
catch {
|
|
@@ -479,6 +497,70 @@ export async function acquireLock(lockPath) {
|
|
|
479
497
|
}
|
|
480
498
|
}
|
|
481
499
|
}
|
|
500
|
+
/**
|
|
501
|
+
* Run `remove` while holding `<lockPath>.break`, the lock that serializes
|
|
502
|
+
* removals of `lockPath`. Waits up to {@link LOCK_TIMEOUT_MS}; a removal lock
|
|
503
|
+
* older than {@link BREAK_LOCK_STALE_MS} is a crashed remover's and is
|
|
504
|
+
* cleared. Resolves to `undefined` when the removal lock could not be had
|
|
505
|
+
* in time — the caller decides what that means.
|
|
506
|
+
*/
|
|
507
|
+
async function withRemovalLock(lockPath, remove) {
|
|
508
|
+
const breakPath = `${lockPath}.break`;
|
|
509
|
+
const startedAt = Date.now();
|
|
510
|
+
while (true) {
|
|
511
|
+
try {
|
|
512
|
+
const handle = await open(breakPath, "wx");
|
|
513
|
+
await handle.close();
|
|
514
|
+
break;
|
|
515
|
+
}
|
|
516
|
+
catch (error) {
|
|
517
|
+
if (error.code !== "EEXIST")
|
|
518
|
+
throw error;
|
|
519
|
+
try {
|
|
520
|
+
const breakStat = await stat(breakPath);
|
|
521
|
+
if (Date.now() - breakStat.mtimeMs > BREAK_LOCK_STALE_MS) {
|
|
522
|
+
await rm(breakPath, { force: true }).catch(() => undefined);
|
|
523
|
+
continue;
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
catch {
|
|
527
|
+
continue;
|
|
528
|
+
}
|
|
529
|
+
if (Date.now() - startedAt > LOCK_TIMEOUT_MS)
|
|
530
|
+
return undefined;
|
|
531
|
+
await sleep(LOCK_RETRY_MS);
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
try {
|
|
535
|
+
return await remove();
|
|
536
|
+
}
|
|
537
|
+
finally {
|
|
538
|
+
await rm(breakPath, { force: true }).catch(() => undefined);
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
/**
|
|
542
|
+
* Remove a lock older than {@link STALE_LOCK_MS}, under the removal lock and
|
|
543
|
+
* only if it is still that old there: the holder may have released and a
|
|
544
|
+
* new one acquired between the caller's stat and this one. Resolves to the
|
|
545
|
+
* broken lock's holder, or null when nothing was removed.
|
|
546
|
+
*/
|
|
547
|
+
async function breakStaleLock(lockPath) {
|
|
548
|
+
const broken = await withRemovalLock(lockPath, async () => {
|
|
549
|
+
let lockStat;
|
|
550
|
+
try {
|
|
551
|
+
lockStat = await stat(lockPath);
|
|
552
|
+
}
|
|
553
|
+
catch {
|
|
554
|
+
return null;
|
|
555
|
+
}
|
|
556
|
+
if (Date.now() - lockStat.mtimeMs <= STALE_LOCK_MS)
|
|
557
|
+
return null;
|
|
558
|
+
const holder = await describeLockHolder(lockPath);
|
|
559
|
+
await rm(lockPath, { force: true }).catch(() => undefined);
|
|
560
|
+
return holder;
|
|
561
|
+
});
|
|
562
|
+
return broken ?? null;
|
|
563
|
+
}
|
|
482
564
|
/**
|
|
483
565
|
* Remove the lock at `lockPath` only if it is still ours. A lock that vanished
|
|
484
566
|
* (someone broke it as stale) or that now carries another holder's token is
|
|
@@ -486,16 +568,25 @@ export async function acquireLock(lockPath) {
|
|
|
486
568
|
* than removed unverified.
|
|
487
569
|
*/
|
|
488
570
|
async function releaseOwnedLock(lockPath, token) {
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
571
|
+
const removeIfOwned = async () => {
|
|
572
|
+
let content;
|
|
573
|
+
try {
|
|
574
|
+
content = await readFile(lockPath, "utf8");
|
|
575
|
+
}
|
|
576
|
+
catch {
|
|
577
|
+
return true;
|
|
578
|
+
}
|
|
579
|
+
if (content.split(/\r?\n/)[2] !== token)
|
|
580
|
+
return true;
|
|
581
|
+
await rm(lockPath, { force: true }).catch(() => undefined);
|
|
582
|
+
return true;
|
|
583
|
+
};
|
|
584
|
+
// Serialized with stale breaks so a break in progress cannot land on a
|
|
585
|
+
// lock this release has already replaced (#342). A removal lock that
|
|
586
|
+
// cannot be had in time falls back to the token-checked removal alone —
|
|
587
|
+
// the guarantee before #342, never less.
|
|
588
|
+
if ((await withRemovalLock(lockPath, removeIfOwned)) === undefined)
|
|
589
|
+
await removeIfOwned();
|
|
499
590
|
}
|
|
500
591
|
async function describeLockHolder(lockPath) {
|
|
501
592
|
try {
|
|
@@ -32,5 +32,11 @@ export declare function finishPhase(phaseId: string, outcome: {
|
|
|
32
32
|
* Clear a phase from the activity map. The widget (M2) will linger finished
|
|
33
33
|
* phases for a turn or two before calling this; for now (M1) we clear after
|
|
34
34
|
* a fixed timeout so the orchestrator notification stays meaningful.
|
|
35
|
+
*
|
|
36
|
+
* With `activity`, clear only while the map still holds that very entry.
|
|
37
|
+
* The runner's linger timer fires 30 s after a phase ends; a re-run of the
|
|
38
|
+
* same phase inside that window used to lose its live entry to the earlier
|
|
39
|
+
* run's timer — the widget dropped it and the re-entry guard let a second
|
|
40
|
+
* sub-agent start on the phase (#343).
|
|
35
41
|
*/
|
|
36
|
-
export declare function clearPhase(phaseId: string): void;
|
|
42
|
+
export declare function clearPhase(phaseId: string, activity?: PhaseActivity): void;
|
|
@@ -42,7 +42,15 @@ export function finishPhase(phaseId, outcome) {
|
|
|
42
42
|
* Clear a phase from the activity map. The widget (M2) will linger finished
|
|
43
43
|
* phases for a turn or two before calling this; for now (M1) we clear after
|
|
44
44
|
* a fixed timeout so the orchestrator notification stays meaningful.
|
|
45
|
+
*
|
|
46
|
+
* With `activity`, clear only while the map still holds that very entry.
|
|
47
|
+
* The runner's linger timer fires 30 s after a phase ends; a re-run of the
|
|
48
|
+
* same phase inside that window used to lose its live entry to the earlier
|
|
49
|
+
* run's timer — the widget dropped it and the re-entry guard let a second
|
|
50
|
+
* sub-agent start on the phase (#343).
|
|
45
51
|
*/
|
|
46
|
-
export function clearPhase(phaseId) {
|
|
52
|
+
export function clearPhase(phaseId, activity) {
|
|
53
|
+
if (activity && phaseActivity.get(phaseId) !== activity)
|
|
54
|
+
return;
|
|
47
55
|
phaseActivity.delete(phaseId);
|
|
48
56
|
}
|
|
@@ -97,4 +97,11 @@ export type AutoDecision = {
|
|
|
97
97
|
};
|
|
98
98
|
export declare function decideAfterPhase(phaseStatus: SinglePhaseResult["status"], phaseError: string | undefined, validation: ValidationResult | null, strict: boolean): AutoDecision;
|
|
99
99
|
export declare function runAuto(ctx: ExtensionContext, pi: ExtensionAPI, initialState: WorkspaceState, options: AutoRunOptions): Promise<AutoRunResult>;
|
|
100
|
+
/**
|
|
101
|
+
* The one-line notification for an auto run's end. A run that stopped short
|
|
102
|
+
* carries its reason: the auto-summary message and the widget carry it too,
|
|
103
|
+
* but under `pi -p` neither is rendered, and "stopped: 0/1 phases" alone
|
|
104
|
+
* sent a reader back to the code to find out why (#347).
|
|
105
|
+
*/
|
|
106
|
+
export declare function describeAutoOutcome(result: AutoRunResult): string;
|
|
100
107
|
export declare function buildAutoSummary(result: AutoRunResult, availableSkills?: string[]): string;
|