codecartographer-pi 0.19.0 → 0.19.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -31,11 +31,12 @@
31
31
  // executable surfaces (Pi and MCP), not the pure template. What the template does
32
32
  // carry is the reading guide for its output — `.codecarto/broadside/SKILL.md`,
33
33
  // served by codecarto_skill under the name `broadside` (see readBroadsideSkill).
34
- import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
34
+ import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
35
35
  import { execFile } from "node:child_process";
36
36
  import { promisify } from "node:util";
37
- import { join } from "node:path";
37
+ import { join, relative } from "node:path";
38
38
  import { pathExists, sleep } from "./utils.js";
39
+ import { acquireLock } from "./status.js";
39
40
  import { loadYamlFile } from "./yaml.js";
40
41
  import { packagedWorkspaceDir } from "./workspace.js";
41
42
  const execFileAsync = promisify(execFile);
@@ -49,8 +50,14 @@ export const BROADSIDE_STATE_FILE = "state.json";
49
50
  export const BROADSIDE_CONFIG_FILE = "config.yaml";
50
51
  export const BROADSIDE_STATE_SCHEMA_VERSION = 1;
51
52
  // Per-token pricing in USD (OpenRouter, google/gemini-3.7-flash:batch).
52
- export const BROADSIDE_INPUT_PRICE_PER_M = 0.1875;
53
- export const BROADSIDE_OUTPUT_PRICE_PER_M = 0.9375;
53
+ // OpenRouter's listed rates for the `:batch` variant, which already carry the
54
+ // batch discount — the sync model is $0.75/$3.75. These were half these values
55
+ // until a live run compared them against the catalog: the batch discount had
56
+ // been applied a second time by hand, so every estimate for the default model
57
+ // came out at half its true cost and `max_cost` bound at twice what the user
58
+ // asked for. They are the offline fallback only; the live catalog wins.
59
+ export const BROADSIDE_INPUT_PRICE_PER_M = 0.375;
60
+ export const BROADSIDE_OUTPUT_PRICE_PER_M = 1.875;
54
61
  // OpenRouter's public model catalog; pricing, context, and capabilities live
55
62
  // per model id. The benchmarks endpoint adds coding/intelligence indices.
56
63
  export const BROADSIDE_MODELS_URL = "https://openrouter.ai/api/v1/models";
@@ -67,6 +74,34 @@ export const BROADSIDE_LENS_IDS = [
67
74
  ];
68
75
  export const BROADSIDE_POLL_INTERVAL_MS = 15_000;
69
76
  export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
77
+ /**
78
+ * The share of a lens's output budget reasoning may spend.
79
+ *
80
+ * `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
81
+ * at the remaining quarter makes that assumption true by construction and
82
+ * guarantees the answer has room. A floor keeps the cap sane for a small lens.
83
+ */
84
+ export const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
85
+ export const BROADSIDE_MIN_REASONING_TOKENS = 512;
86
+ /**
87
+ * Cap reasoning for a lens request — deliberately a cap, not an off switch.
88
+ *
89
+ * Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
90
+ * the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
91
+ * disabled"*, turning a partial result into none at all. Capping works whether
92
+ * or not a provider allows reasoning to be switched off.
93
+ *
94
+ * The failure this prevents is the budget being spent thinking rather than
95
+ * answering. Measured on one run: 5,758 of a 6,000-token budget went to
96
+ * reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
97
+ * those tokens bill at the full output rate. The shipped default model does the
98
+ * same thing less consistently (reasoning tokens from 0 to 5,757 across 13
99
+ * slices, three of them cut off at `finish_reason: length`), so this is not a
100
+ * multi-model concern.
101
+ */
102
+ export function defaultReasoningFor(maxTokens) {
103
+ return { max_tokens: Math.max(BROADSIDE_MIN_REASONING_TOKENS, Math.floor(maxTokens * BROADSIDE_REASONING_BUDGET_FRACTION)) };
104
+ }
70
105
  /** Thrown when a confirm hook declines a run. Nothing was submitted. */
71
106
  export class BroadsideCancelledError extends Error {
72
107
  constructor(message = "Broad-Side submission cancelled. Nothing was submitted.") {
@@ -848,7 +883,10 @@ async function walkFiles(rootDir, dir, depth, remaining) {
848
883
  out = out.concat(children);
849
884
  }
850
885
  else if (entry.isFile()) {
851
- const rel = join(dir, entry.name).slice(rootDir.length + 1).split("\\").join("/");
886
+ // relative() rather than slice(rootDir.length + 1): the hand-rolled
887
+ // slice cut one character too many whenever rootDir carried a trailing
888
+ // separator, and mangled every path outright when rootDir was "/".
889
+ const rel = relative(rootDir, join(dir, entry.name)).split("\\").join("/");
852
890
  out.push(rel);
853
891
  }
854
892
  }
@@ -1126,7 +1164,7 @@ async function sumFileSizes(targetDir, files) {
1126
1164
  return total;
1127
1165
  }
1128
1166
  // ---------- request building ----------
1129
- export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride) {
1167
+ export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride, reasoningOverride) {
1130
1168
  const moduleTag = sanitizeId(slice.moduleName);
1131
1169
  const customId = sliceCount > 1 ? `${lens.id}-${moduleTag}-${index + 1}` : `${lens.id}-${moduleTag}`;
1132
1170
  return {
@@ -1139,12 +1177,36 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
1139
1177
  ],
1140
1178
  response_format: { type: "json_schema", json_schema: SCHEMAS[lens.schemaName] },
1141
1179
  max_tokens: maxTokensOverride ?? lens.maxTokens,
1180
+ // Always sent, never inherited: an absent field means the model's
1181
+ // own default, and that default is what truncated the JSON.
1182
+ reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(maxTokensOverride ?? lens.maxTokens),
1142
1183
  },
1143
1184
  };
1144
1185
  }
1145
- export function estimateCost(lens, slices, pricing, maxTokensOverride) {
1146
- const inputTokens = Math.ceil(slices.reduce((sum, s) => sum + (lens.maxChars === 0 ? 6000 : s.chars), 0) / 4);
1147
- const outputTokens = Math.ceil((maxTokensOverride ?? lens.maxTokens) * 0.75);
1186
+ /**
1187
+ * Pre-flight cost estimate for one lens.
1188
+ *
1189
+ * Every slice is its own batch request, so both halves scale with the slice
1190
+ * count. The output half used to be a single `maxTokens * 0.75` for the whole
1191
+ * lens no matter how many requests it sent — on a repository that sliced into
1192
+ * 13 modules that budgeted one request's output and shipped thirteen, and a
1193
+ * live run came in at roughly 3x its estimate. Since this number is what
1194
+ * `max_cost` binds against, under-counting it lets a run outspend the cap the
1195
+ * user set.
1196
+ *
1197
+ * @param info - Repo info, when the caller has it: lets the estimate include
1198
+ * the system prompt and JSON schema each request carries. Omitted, the
1199
+ * estimate covers slice content only, which is what the old signature did.
1200
+ */
1201
+ export function estimateCost(lens, slices, pricing, maxTokensOverride, info) {
1202
+ const sliceChars = slices.reduce((sum, s) => sum + (lens.maxChars === 0 ? 6000 : s.chars), 0);
1203
+ // The system prompt and the response schema ride on every request, so they
1204
+ // are paid once per slice rather than once per lens.
1205
+ const perRequestOverhead = info
1206
+ ? (lens.systemPrompt(info)?.length ?? 0) + JSON.stringify(SCHEMAS[lens.schemaName] ?? {}).length
1207
+ : 0;
1208
+ const inputTokens = Math.ceil((sliceChars + perRequestOverhead * slices.length) / 4);
1209
+ const outputTokens = slices.length * Math.ceil((maxTokensOverride ?? lens.maxTokens) * 0.75);
1148
1210
  const cost = (inputTokens / 1_000_000) * pricing.inputPerM +
1149
1211
  (outputTokens / 1_000_000) * pricing.outputPerM;
1150
1212
  return { inputTokens, outputTokens, cost };
@@ -1195,9 +1257,94 @@ export async function loadBroadsideState(broadsideDir) {
1195
1257
  return defaultBroadsideState();
1196
1258
  }
1197
1259
  }
1260
+ /**
1261
+ * Overwrite `state.json` wholesale with `state`.
1262
+ *
1263
+ * Prefer {@link persistBroadsideRun} anywhere a live operation is recording its
1264
+ * own progress — this entry point replaces the file, so any run a concurrent
1265
+ * process recorded in the meantime is erased. It remains the right call for
1266
+ * seeding a fresh workspace and for test fixtures, where "make the file exactly
1267
+ * this" is the intent.
1268
+ */
1198
1269
  export async function saveBroadsideState(broadsideDir, state) {
1199
1270
  await mkdir(broadsideDir, { recursive: true });
1200
- await writeFile(join(broadsideDir, BROADSIDE_STATE_FILE), `${JSON.stringify(state, null, "\t")}\n`, "utf8");
1271
+ const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
1272
+ const lock = await acquireLock(`${statePath}.lock`);
1273
+ try {
1274
+ await writeBroadsideStateFile(statePath, state);
1275
+ }
1276
+ finally {
1277
+ await lock.release();
1278
+ }
1279
+ }
1280
+ /** Serialize through a temp file so a crash mid-write cannot truncate state.json. */
1281
+ async function writeBroadsideStateFile(statePath, state) {
1282
+ const tempPath = `${statePath}.${process.pid}.${Date.now()}.tmp`;
1283
+ await writeFile(tempPath, `${JSON.stringify(state, null, "\t")}\n`, "utf8");
1284
+ await rename(tempPath, statePath);
1285
+ }
1286
+ /**
1287
+ * Read-modify-write `state.json` under a lock.
1288
+ *
1289
+ * The lock is held only for the read-modify-write, never for the surrounding
1290
+ * operation: a `collect` can poll for the better part of an hour, and holding
1291
+ * the lock across that would push every concurrent caller past the 5s lock
1292
+ * timeout.
1293
+ */
1294
+ export async function updateBroadsideStateAtomically(broadsideDir, mutate) {
1295
+ await mkdir(broadsideDir, { recursive: true });
1296
+ const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
1297
+ const lock = await acquireLock(`${statePath}.lock`);
1298
+ try {
1299
+ const state = await loadBroadsideState(broadsideDir);
1300
+ await mutate(state);
1301
+ await writeBroadsideStateFile(statePath, state);
1302
+ return state;
1303
+ }
1304
+ finally {
1305
+ await lock.release();
1306
+ }
1307
+ }
1308
+ /**
1309
+ * Record one run's current shape, merged into whatever is on disk *now*.
1310
+ *
1311
+ * Broad-Side operations are long-lived and hold their state in memory while
1312
+ * they poll. Writing that snapshot back wholesale silently erased any run a
1313
+ * concurrent operation had recorded since it was loaded, orphaning that run's
1314
+ * paid results on disk — present as files, invisible to `list`, and unreachable
1315
+ * by `collect`, which finds its run by position in `state.runs`. Observed live:
1316
+ * a submit at 23:35 was erased by a collect that had loaded state before it and
1317
+ * wrote back at 00:08.
1318
+ *
1319
+ * Merging by run id also self-heals: a run erased by an older writer is
1320
+ * restored the next time its own operation checkpoints.
1321
+ */
1322
+ export async function persistBroadsideRun(broadsideDir, run) {
1323
+ return updateBroadsideStateAtomically(broadsideDir, (state) => {
1324
+ const index = state.runs.findIndex((candidate) => candidate.id === run.id);
1325
+ if (index === -1)
1326
+ state.runs.push(run);
1327
+ else
1328
+ state.runs[index] = run;
1329
+ });
1330
+ }
1331
+ /** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
1332
+ function parseReasoningConfig(raw) {
1333
+ if (raw === false)
1334
+ return { enabled: false };
1335
+ if (raw === true)
1336
+ return { enabled: true };
1337
+ if (!raw || typeof raw !== "object")
1338
+ return null;
1339
+ const value = raw;
1340
+ const out = {};
1341
+ if (typeof value.enabled === "boolean")
1342
+ out.enabled = value.enabled;
1343
+ if (value.effort === "minimal" || value.effort === "low" || value.effort === "medium" || value.effort === "high")
1344
+ out.effort = value.effort;
1345
+ if (typeof value.max_tokens === "number" && value.max_tokens > 0)
1346
+ out.max_tokens = value.max_tokens;
1347
+ return Object.keys(out).length > 0 ? out : null;
1201
1348
  }
1202
1349
  export async function loadBroadsideConfig(broadsideDir) {
1203
1350
  const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
@@ -1239,6 +1386,10 @@ export async function loadBroadsideConfig(broadsideDir) {
1239
1386
  ? { inputPerM: inputOverride, outputPerM: outputOverride }
1240
1387
  : null,
1241
1388
  lensModels,
1389
+ // An escape hatch, not a knob to reach for: a model whose reasoning is
1390
+ // worth paying for needs its lens maxTokens raised to cover both the
1391
+ // thinking and the answer, or the JSON truncates exactly as before.
1392
+ reasoning: parseReasoningConfig(raw.reasoning),
1242
1393
  incremental: flag("incremental", false),
1243
1394
  retryTruncated: flag("retry_truncated", true),
1244
1395
  includeSynthesis: flag("include_synthesis", true),
@@ -1332,10 +1483,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
1332
1483
  },
1333
1484
  };
1334
1485
  }
1335
- const builtIn = builtInCatalogEntry(model);
1336
- if (builtIn)
1337
- return { model, source: "built-in", entry: builtIn };
1338
- // Unknown model: on-disk cache first, then the live catalog.
1486
+ // On-disk cache first, then the live catalog — for the default model too.
1487
+ // Hardcoded rates used to short-circuit here, which meant a stale constant
1488
+ // could never self-correct even though the catalog was already being
1489
+ // fetched for every other model. The authoritative source wins; the
1490
+ // constants below are what we fall back to when the network is unavailable.
1339
1491
  const cache = await readCatalogCache(broadsideDir);
1340
1492
  const cached = cache?.models[model];
1341
1493
  if (cached && Date.now() - new Date(cache.fetched_at).getTime() < BROADSIDE_CATALOG_CACHE_TTL_MS) {
@@ -1366,6 +1518,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
1366
1518
  await writeCatalogCache(broadsideDir, updated);
1367
1519
  return { model, source: "live", entry: live };
1368
1520
  }
1521
+ // Offline fallback: the default model's rates and capabilities are known at
1522
+ // compile time, so a network failure does not have to stop a run.
1523
+ const builtIn = builtInCatalogEntry(model);
1524
+ if (builtIn)
1525
+ return { model, source: "built-in", entry: builtIn };
1369
1526
  throw new Error(`Could not resolve per-token pricing for batch model "${model}". ` +
1370
1527
  "Set pricing.input_per_m and pricing.output_per_m in .codecarto/broadside/config.yaml " +
1371
1528
  "(USD per million tokens), or check the model id against https://openrouter.ai/models?variant=batch.");
@@ -1469,6 +1626,14 @@ export async function fetchBatch(batchId, apiKey, fetcher = fetch) {
1469
1626
  data.http_status = resp.status;
1470
1627
  return data;
1471
1628
  }
1629
+ /**
1630
+ * Batch statuses that will never produce a result.
1631
+ *
1632
+ * Deliberately excludes the synthetic `timeout` this module returns when a poll
1633
+ * budget expires: that batch is still running server-side and has already been
1634
+ * charged, so callers must come back for it rather than retire it.
1635
+ */
1636
+ export const BROADSIDE_DEAD_BATCH_STATUSES = ["failed", "expired", "cancelled", "auth-failed"];
1472
1637
  export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
1473
1638
  const deadline = Date.now() + (opts.deadlineMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
1474
1639
  const intervalMs = opts.pollIntervalMs ?? BROADSIDE_POLL_INTERVAL_MS;
@@ -1491,7 +1656,7 @@ export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
1491
1656
  const status = String(batch.status ?? "unknown");
1492
1657
  const counts = (batch.request_counts ?? {});
1493
1658
  opts.onStatus?.(status, counts);
1494
- if (["completed", "failed", "expired", "cancelled", "auth-failed"].includes(status))
1659
+ if (status === "completed" || BROADSIDE_DEAD_BATCH_STATUSES.includes(status))
1495
1660
  return batch;
1496
1661
  if (Date.now() >= deadline)
1497
1662
  return { id: batchId, status: "timeout" };
@@ -1563,14 +1728,30 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1563
1728
  const sourceDirty = await gitDirty(cwd);
1564
1729
  let baseHead = null;
1565
1730
  let changed = null;
1731
+ const incrementalOutcome = {
1732
+ requested: opts.incremental === true,
1733
+ applied: false,
1734
+ baseHead: null,
1735
+ };
1566
1736
  if (opts.incremental) {
1567
1737
  const state = await loadBroadsideState(broadsideDir);
1568
1738
  // The baseline is the most recent run that recorded a HEAD — a
1569
1739
  // submit-only run (never collected) is still a valid committed base.
1570
1740
  const previous = [...state.runs].reverse().find((r) => r.sourceHead);
1571
- if (previous?.sourceHead && !sourceDirty) {
1741
+ if (sourceDirty) {
1742
+ incrementalOutcome.reason = "dirty-worktree";
1743
+ }
1744
+ else if (!previous?.sourceHead) {
1745
+ incrementalOutcome.reason = "no-baseline";
1746
+ }
1747
+ else {
1572
1748
  baseHead = previous.sourceHead;
1573
1749
  changed = await changedFilesSince(cwd, baseHead);
1750
+ incrementalOutcome.baseHead = baseHead;
1751
+ if (changed)
1752
+ incrementalOutcome.applied = true;
1753
+ else
1754
+ incrementalOutcome.reason = "diff-failed";
1574
1755
  }
1575
1756
  }
1576
1757
  // Slice offline first so the estimate covers every request we would send.
@@ -1591,7 +1772,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1591
1772
  const lensModel = modelForLens(lensId);
1592
1773
  const { pricing: lensPricing, outputCap: lensOutputCap } = resolved.get(lensModel);
1593
1774
  const maxTokens = lensOutputCap ? Math.min(lens.maxTokens, lensOutputCap) : lens.maxTokens;
1594
- const estimate = estimateCost(lens, slices, lensPricing, maxTokens);
1775
+ const estimate = estimateCost(lens, slices, lensPricing, maxTokens, info);
1595
1776
  estimatedInputTokens += estimate.inputTokens;
1596
1777
  estimatedOutputTokens += estimate.outputTokens;
1597
1778
  estimatedTotalCost += estimate.cost;
@@ -1626,6 +1807,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1626
1807
  exceedsLimit,
1627
1808
  baseHead,
1628
1809
  sourceDirty,
1810
+ incremental: incrementalOutcome,
1629
1811
  ...(outputCap !== undefined && { outputCap }),
1630
1812
  });
1631
1813
  if (!approved)
@@ -1659,7 +1841,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1659
1841
  baseHead,
1660
1842
  };
1661
1843
  state.runs.push(run);
1662
- await saveBroadsideState(broadsideDir, state);
1844
+ await persistBroadsideRun(broadsideDir, run);
1663
1845
  const requestsByCustomId = {};
1664
1846
  const submissions = [];
1665
1847
  // Submit from the estimate rather than recomputing: the user approved that
@@ -1668,7 +1850,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1668
1850
  const { lens, maxTokens, lensModel, lensOutputCap } = priced;
1669
1851
  const lensId = lens.id;
1670
1852
  const slices = slicesByLens.get(lensId) ?? [];
1671
- const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens));
1853
+ const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens, config.reasoning ?? undefined));
1672
1854
  for (const request of requests)
1673
1855
  requestsByCustomId[request.custom_id] = request;
1674
1856
  const entry = {
@@ -1707,7 +1889,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1707
1889
  })());
1708
1890
  }
1709
1891
  await Promise.allSettled(submissions);
1710
- await saveBroadsideState(broadsideDir, state);
1892
+ await persistBroadsideRun(broadsideDir, run);
1711
1893
  // Persist the exact request bodies so collect can re-submit a truncated
1712
1894
  // slice (bumped output cap) without re-walking the repo (#133). The run
1713
1895
  // dir is created here rather than waiting for collect so a crash between
@@ -1734,6 +1916,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1734
1916
  : resolved.get(model).supportsStructuredOutputs,
1735
1917
  expirationDate: defaultEntry.expirationDate ?? null,
1736
1918
  },
1919
+ incremental: incrementalOutcome,
1737
1920
  };
1738
1921
  }
1739
1922
  function extractContent(result) {
@@ -1799,6 +1982,46 @@ export async function saveLensResults(runDir, lensId, batch) {
1799
1982
  }
1800
1983
  return out;
1801
1984
  }
1985
+ /**
1986
+ * Rebuild lens results from what a previous collect already wrote to disk.
1987
+ *
1988
+ * The post-passes are gated on having lens findings in hand, and a collect
1989
+ * only holds the ones *it* polled. When an earlier collect saved every lens
1990
+ * and then died before synthesis and triage ran — the batch window is long
1991
+ * and a poll can easily be interrupted — the next collect finds every lens
1992
+ * already terminal, skips them all, and would otherwise reach the post-pass
1993
+ * gate with nothing to hand it. Reading the saved results back is what makes
1994
+ * "a resumed collect can finish whichever is still pending" true.
1995
+ */
1996
+ export async function loadSavedLensResults(runDir, lenses) {
1997
+ if (!(await pathExists(runDir)))
1998
+ return [];
1999
+ const reserved = new Set(["requests.json", "run-meta.json", "synthesis.json", "triage.json"]);
2000
+ const out = [];
2001
+ // Longest lens id first: no id is a prefix of another today, but ordering
2002
+ // keeps that from becoming a silent misattribution if one ever is.
2003
+ const ordered = [...lenses].sort((a, b) => b.length - a.length);
2004
+ for (const name of (await readdir(runDir)).sort()) {
2005
+ if (!name.endsWith(".json") || name.endsWith(".error.json") || reserved.has(name) || name.startsWith("raw-"))
2006
+ continue;
2007
+ const customId = name.slice(0, -".json".length);
2008
+ const lensId = ordered.find((id) => customId === id || customId.startsWith(`${id}-`));
2009
+ if (!lensId)
2010
+ continue;
2011
+ const content = await readFile(join(runDir, name), "utf8").catch(() => null);
2012
+ if (content === null)
2013
+ continue;
2014
+ out.push({
2015
+ lensId,
2016
+ customId,
2017
+ moduleName: customId.replace(/^[a-z]+-/, ""),
2018
+ content,
2019
+ raw: {},
2020
+ truncated: parseLensJson(content) === null,
2021
+ });
2022
+ }
2023
+ return out;
2024
+ }
1802
2025
  async function loadStoredRequests(runDir) {
1803
2026
  const path = join(runDir, "requests.json");
1804
2027
  if (!(await pathExists(path)))
@@ -1956,11 +2179,19 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
1956
2179
  await writeFile(join(runDir, `raw-${lensId}.json`), `${JSON.stringify(batch, null, "\t")}\n`, "utf8");
1957
2180
  lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, truncated };
1958
2181
  }
1959
- else if (batch.error) {
1960
- entry.error = batch.error;
2182
+ else {
2183
+ // Every non-completed outcome still has to reach the report.
2184
+ // `lensOutcomes` is what the caller renders, and this branch used to
2185
+ // require `batch.error` — but the commonest failure here is the
2186
+ // synthetic `{ status: "timeout" }` the poll returns when its budget
2187
+ // expires with the batch still in flight, and that carries no error.
2188
+ // A lens that never came back was therefore omitted entirely,
2189
+ // indistinguishable in the output from one that was never requested.
2190
+ if (batch.error)
2191
+ entry.error = batch.error;
1961
2192
  lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount };
1962
2193
  }
1963
- await saveBroadsideState(broadsideDir, state);
2194
+ await persistBroadsideRun(broadsideDir, run);
1964
2195
  }
1965
2196
  // #133: re-submit truncated slices once with a bumped output cap. Batch
1966
2197
  // requests are pure, so re-running is always safe; the aim is to recover
@@ -1993,7 +2224,11 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
1993
2224
  if (error)
1994
2225
  continue;
1995
2226
  const batch = await pollBatchUntilTerminal(batchId, apiKey, {
1996
- deadlineMs: BROADSIDE_DEFAULT_POLL_BUDGET_MS,
2227
+ // Share the caller's deadline. Each of these polls used to
2228
+ // start a fresh 25-minute budget, so `wait_seconds` bounded
2229
+ // only the lens poll and a collect could run for the caller's
2230
+ // budget plus fifty minutes.
2231
+ deadlineMs: Math.max(0, deadline - Date.now()),
1997
2232
  onStatus: (status, counts) => opts.onStatus?.(`${stored.lensId}:retry`, status, counts),
1998
2233
  fetcher: opts.fetcher,
1999
2234
  });
@@ -2023,7 +2258,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2023
2258
  outcome.truncated = allLensResults.filter((s) => s.lensId === lensId && s.truncated).length;
2024
2259
  }
2025
2260
  }
2026
- await saveBroadsideState(broadsideDir, state);
2261
+ await persistBroadsideRun(broadsideDir, run);
2027
2262
  }
2028
2263
  // Synthesis + triage: cross-lens post-passes, only after every lens batch
2029
2264
  // is terminal. Triage turns the leads into a prioritized work order.
@@ -2032,12 +2267,25 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2032
2267
  let topTriageItems = [];
2033
2268
  const wantSynthesis = opts.includeSynthesis !== false;
2034
2269
  const wantTriage = opts.includeTriage !== false;
2270
+ // A resumed collect polls nothing — every lens is already terminal — so the
2271
+ // findings the post-passes need have to come back off disk, or a run whose
2272
+ // first collect was interrupted could never produce its executive report
2273
+ // and work order, however many times it was re-run.
2274
+ const postPassUnfinished = (entry) => entry.status === "pending" || entry.status === "submitted";
2275
+ if ((wantSynthesis || wantTriage) && allLensResults.length === 0
2276
+ && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
2277
+ const restored = await loadSavedLensResults(runDir, run.lenses);
2278
+ if (restored.length > 0) {
2279
+ allLensResults.push(...restored);
2280
+ truncatedCount = restored.filter((s) => s.truncated).length;
2281
+ }
2282
+ }
2035
2283
  if ((wantSynthesis || wantTriage) && allLensResults.length > 0) {
2036
2284
  const allTerminal = run.lenses.every((lensId) => {
2037
2285
  const entry = run.batches[lensId];
2038
2286
  return entry && ["completed", "failed", "expired", "cancelled", "auth-failed", "skipped", "rejected"].includes(entry.status);
2039
2287
  });
2040
- if (allTerminal && (run.synthesis.status === "pending" || run.triage.status === "pending")) {
2288
+ if (allTerminal && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
2041
2289
  const findingsText = allLensResults
2042
2290
  .map((r) => `## ${r.lensId} — ${r.customId}\n\n${r.content}\n`)
2043
2291
  .join("\n");
@@ -2066,6 +2314,24 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2066
2314
  : []),
2067
2315
  ];
2068
2316
  const submitted = new Map();
2317
+ // A pass can be left at "submitted" when an earlier collect returned
2318
+ // before its batch reached a terminal status — the batch still runs
2319
+ // and is still charged, so the result exists and is simply unclaimed.
2320
+ // Nothing above would ever look at it again: the pass list is built
2321
+ // from "pending" entries only. Poll those regardless of the want
2322
+ // flags, because the spend already happened and discarding a
2323
+ // finished result is worse than saving one the caller opted out of.
2324
+ for (const kind of ["synthesis", "triage"]) {
2325
+ const entry = kind === "synthesis" ? run.synthesis : run.triage;
2326
+ if (entry.status !== "submitted" || !entry.batchId)
2327
+ continue;
2328
+ if (submitted.has(entry.batchId))
2329
+ continue;
2330
+ submitted.set(entry.batchId, {
2331
+ batchId: entry.batchId,
2332
+ pass: { kind, request: undefined, entry },
2333
+ });
2334
+ }
2069
2335
  await Promise.allSettled(passes.map(async (pass) => {
2070
2336
  pass.entry.status = "submitted";
2071
2337
  try {
@@ -2081,10 +2347,13 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2081
2347
  pass.entry.status = "failed";
2082
2348
  }
2083
2349
  }));
2084
- await saveBroadsideState(broadsideDir, state);
2350
+ await persistBroadsideRun(broadsideDir, run);
2085
2351
  for (const { batchId, pass } of submitted.values()) {
2086
2352
  const batch = await pollBatchUntilTerminal(batchId, apiKey, {
2087
- deadlineMs: BROADSIDE_DEFAULT_POLL_BUDGET_MS,
2353
+ // Shares the caller's deadline, as the retry poll above does.
2354
+ // A pass whose poll runs out stays `submitted`, so the batch
2355
+ // is already paid for and a later collect claims its result.
2356
+ deadlineMs: Math.max(0, deadline - Date.now()),
2088
2357
  onStatus: (status, counts) => opts.onStatus?.(pass.kind, status, counts),
2089
2358
  fetcher: opts.fetcher,
2090
2359
  });
@@ -2107,10 +2376,20 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2107
2376
  }
2108
2377
  }
2109
2378
  }
2110
- else if (batch.error) {
2379
+ else if (BROADSIDE_DEAD_BATCH_STATUSES.includes(String(batch.status))) {
2380
+ // The batch will never produce a result, so retire the pass.
2381
+ // This used to require `batch.error`, leaving an expired or
2382
+ // cancelled batch parked at "submitted" forever — and since a
2383
+ // resumed collect re-polls anything still "submitted", it
2384
+ // would re-poll a dead batch on every future run.
2111
2385
  pass.entry.status = "failed";
2386
+ if (batch.error)
2387
+ pass.entry.error = batch.error instanceof Error ? batch.error.message : String(batch.error);
2112
2388
  }
2113
- await saveBroadsideState(broadsideDir, state);
2389
+ // A "timeout" is deliberately left at "submitted": the batch is
2390
+ // still running server-side and has already been paid for, so a
2391
+ // later collect should claim its result rather than discard it.
2392
+ await persistBroadsideRun(broadsideDir, run);
2114
2393
  }
2115
2394
  }
2116
2395
  }
@@ -2120,7 +2399,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2120
2399
  });
2121
2400
  run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
2122
2401
  run.totalCost = totalCost;
2123
- await saveBroadsideState(broadsideDir, state);
2402
+ await persistBroadsideRun(broadsideDir, run);
2124
2403
  await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
2125
2404
  experimental: true,
2126
2405
  method: "Broad-Side (OpenRouter Batch API)",
@@ -2223,9 +2502,31 @@ function parseSynthesisTopFindings(content) {
2223
2502
  }
2224
2503
  }
2225
2504
  // ---------- formatting helpers for tool output ----------
2505
+ function describeIncrementalFallback(reason) {
2506
+ switch (reason) {
2507
+ case "dirty-worktree":
2508
+ return "the working tree has uncommitted changes, so there is no committed state to diff against";
2509
+ case "no-baseline":
2510
+ return "no earlier run recorded a commit to diff against";
2511
+ case "diff-failed":
2512
+ return "the diff against the previous run's commit could not be read";
2513
+ default:
2514
+ return "no baseline was available";
2515
+ }
2516
+ }
2226
2517
  export function estimateSubmitText(result, lenses) {
2518
+ // Count the lenses that actually got a batch, not every lens considered. A
2519
+ // lens with nothing to scan is reported as `skipped (0 request(s))` two
2520
+ // lines below, so counting it here made the header contradict its own body:
2521
+ // a Rust CLI with no server surface reported "submitted 6 batch(es)" over a
2522
+ // list showing four batches and two skips.
2523
+ const entries = Object.values(result.batches ?? {});
2524
+ const submittedCount = entries.filter((entry) => entry.batchId).length;
2525
+ const withoutBatch = entries.length - submittedCount;
2227
2526
  const lines = [
2228
- `Broad-Side submitted ${result.batches ? Object.keys(result.batches).length : 0} batch(es).`,
2527
+ withoutBatch > 0
2528
+ ? `Broad-Side submitted ${submittedCount} batch(es); ${withoutBatch} lens(es) produced none (see below).`
2529
+ : `Broad-Side submitted ${submittedCount} batch(es).`,
2229
2530
  ];
2230
2531
  for (const lens of lenses) {
2231
2532
  const entry = result.batches[lens.id];
@@ -2235,6 +2536,12 @@ export function estimateSubmitText(result, lenses) {
2235
2536
  const override = entry.model ? ` on ${entry.model}` : "";
2236
2537
  lines.push(` ${lens.name}: ${status} (${entry.requests} request(s), ~$${entry.estimatedCost.toFixed(4)})${override}`);
2237
2538
  }
2539
+ const incremental = result.incremental;
2540
+ if (incremental?.requested) {
2541
+ lines.push(incremental.applied
2542
+ ? `Incremental: scanning only what changed since ${(incremental.baseHead ?? "").slice(0, 8)}.`
2543
+ : `Incremental: requested but NOT applied — ${describeIncrementalFallback(incremental.reason)}. Every module was scanned, at full cost.`);
2544
+ }
2238
2545
  lines.push(`Estimated total: ~$${result.estimatedTotalCost.toFixed(4)}`, `Pricing: $${result.pricing.inputPerM.toFixed(4)}/M in, $${result.pricing.outputPerM.toFixed(4)}/M out (${result.pricing.source})`);
2239
2546
  if (result.modelInfo.contextLength) {
2240
2547
  lines.push(`Model: ${result.modelInfo.contextLength.toLocaleString()} context, ${result.modelInfo.maxCompletionTokens?.toLocaleString() ?? "?"} max output`);
@@ -39,4 +39,17 @@ export interface DashboardInputs {
39
39
  narration?: DashboardNarration;
40
40
  }
41
41
  export declare function renderDashboard(inputs: DashboardInputs): string;
42
+ /**
43
+ * A dashboard link target, or undefined when the path is not a safe relative
44
+ * reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
45
+ * refused outright.
46
+ *
47
+ * Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
48
+ * `dir\..\secret` survives the segment scan, because that scan splits on `/`
49
+ * — but `encodeURIComponent` turns the separators into `%5C`, which no URL
50
+ * parser treats as a path separator, so the traversal cannot resolve. Removing
51
+ * the encoding, or "simplifying" it to a raw join, reopens it. Exported so
52
+ * that property has a test.
53
+ */
54
+ export declare function safeRelativeHref(path: string): string | undefined;
42
55
  export declare function escapeHtml(input: string): string;
@@ -660,7 +660,19 @@ function renderSafeLink(path, label, className) {
660
660
  const classAttr = className ? ` class="${escapeAttr(className)}"` : "";
661
661
  return `<a${classAttr} href="${escapeAttr(href)}">${escapeHtml(label)}</a>`;
662
662
  }
663
- function safeRelativeHref(path) {
663
+ /**
664
+ * A dashboard link target, or undefined when the path is not a safe relative
665
+ * reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
666
+ * refused outright.
667
+ *
668
+ * Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
669
+ * `dir\..\secret` survives the segment scan, because that scan splits on `/`
670
+ * — but `encodeURIComponent` turns the separators into `%5C`, which no URL
671
+ * parser treats as a path separator, so the traversal cannot resolve. Removing
672
+ * the encoding, or "simplifying" it to a raw join, reopens it. Exported so
673
+ * that property has a test.
674
+ */
675
+ export function safeRelativeHref(path) {
664
676
  if (!path || path.startsWith("/") || path.startsWith("\\") || /^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(path))
665
677
  return undefined;
666
678
  const parts = path.split("/");