codecartographer-pi 0.19.0 → 0.19.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -125,6 +125,13 @@ from OpenRouter's runtime cost tracking: it predicts from file sizes before
125
125
  spend, it does not stop a batch mid-flight. Actual spend appears in
126
126
  `run-meta.json` after collect.
127
127
 
128
+ Budget for the **ceiling, not the estimate**. OpenRouter charges the worst
129
+ case when it accepts a batch — every request's full `max_tokens` — and
130
+ refunds the unused output as each request settles. A six-lens scan of this
131
+ repository held $0.6972 and settled at $0.3806, so the balance a run needs to
132
+ *start* is roughly double what it ends up costing. The submit estimate sits
133
+ between the two: above what the run should settle at, below what it can hold.
134
+
128
135
  ## Resilience notes
129
136
 
130
137
  - **Truncation is spoken.** A lens output whose JSON does not parse — even
@@ -68,8 +68,8 @@
68
68
  # lookup fails (offline, private model) or you want to assert a ceiling.
69
69
  #
70
70
  # pricing:
71
- # input_per_m: 0.1875
72
- # output_per_m: 0.9375
71
+ # input_per_m: 0.375
72
+ # output_per_m: 1.875
73
73
  # ---------------------------------------------------------------------------
74
74
  # Run defaults. Each key below mirrors a codecarto_broadside parameter of the
75
75
  # same name and sets this repository's default for it; an explicit parameter on
@@ -3,4 +3,4 @@
3
3
  # workspace's framework-owned files (GUIDE.md, templates/, workflow/ pipelines
4
4
  # and VALIDATE.md) predate the running release. Written at release time and
5
5
  # copied verbatim by init — never edit by hand.
6
- scaffold_version: 0.19.0
6
+ scaffold_version: 0.19.1
@@ -6,8 +6,8 @@ export declare const BROADSIDE_SKILL_NAME = "broadside";
6
6
  export declare const BROADSIDE_STATE_FILE = "state.json";
7
7
  export declare const BROADSIDE_CONFIG_FILE = "config.yaml";
8
8
  export declare const BROADSIDE_STATE_SCHEMA_VERSION = 1;
9
- export declare const BROADSIDE_INPUT_PRICE_PER_M = 0.1875;
10
- export declare const BROADSIDE_OUTPUT_PRICE_PER_M = 0.9375;
9
+ export declare const BROADSIDE_INPUT_PRICE_PER_M = 0.375;
10
+ export declare const BROADSIDE_OUTPUT_PRICE_PER_M = 1.875;
11
11
  export declare const BROADSIDE_MODELS_URL = "https://openrouter.ai/api/v1/models";
12
12
  export declare const BROADSIDE_BENCHMARKS_URL = "https://openrouter.ai/api/v1/benchmarks";
13
13
  export declare const BROADSIDE_CATALOG_CACHE_FILE = "model-catalog.json";
@@ -49,7 +49,13 @@ export type CodingBenchmarks = {
49
49
  export type BroadsideCatalogResult = {
50
50
  model: string;
51
51
  source: "built-in" | "config" | "live" | "cache";
52
- entry: CatalogEntry | null;
52
+ /**
53
+ * Always resolved. `resolveCatalogEntry` either returns an entry — from
54
+ * config, cache, the live catalog, or the compile-time fallback — or throws
55
+ * naming the model it could not price. This was declared nullable, which is
56
+ * the only reason the single consumer needed a non-null assertion to read it.
57
+ */
58
+ entry: CatalogEntry;
53
59
  benchmarks?: CodingBenchmarks;
54
60
  };
55
61
  export type JsonSchemaDef = {
@@ -115,6 +121,8 @@ export type BroadsideSynthesisEntry = {
115
121
  batchId?: string;
116
122
  status: "pending" | "submitted" | "completed" | "failed";
117
123
  cost?: number;
124
+ /** Why the pass was retired, when the batch reported one. */
125
+ error?: string;
118
126
  };
119
127
  /** One triage item — a scouting lead turned into a work-order entry. */
120
128
  export type TriageItem = {
@@ -220,6 +228,8 @@ export type BroadsideEstimate = {
220
228
  /** Set when incremental scouting found a baseline to diff against. */
221
229
  baseHead: string | null;
222
230
  sourceDirty: boolean;
231
+ /** Whether a requested incremental run actually narrowed this estimate. */
232
+ incremental: BroadsideIncrementalOutcome;
223
233
  /** The provider's completion ceiling, when the catalog advertises one. */
224
234
  outputCap?: number;
225
235
  };
@@ -227,6 +237,22 @@ export type BroadsideEstimate = {
227
237
  export declare class BroadsideCancelledError extends Error {
228
238
  constructor(message?: string);
229
239
  }
240
+ /**
241
+ * Whether incremental scouting actually narrowed the run.
242
+ *
243
+ * A request for incremental falls back to a full scan whenever there is nothing
244
+ * to diff against, and that fallback costs real money — the caller asked for the
245
+ * cheap mode and gets the expensive one. It must be reported, not inferred from
246
+ * the request counts.
247
+ */
248
+ export type BroadsideIncrementalOutcome = {
249
+ requested: boolean;
250
+ applied: boolean;
251
+ /** The commit the run diffed against, when one was found. */
252
+ baseHead: string | null;
253
+ /** Why a requested incremental run did not apply. */
254
+ reason?: "dirty-worktree" | "no-baseline" | "diff-failed";
255
+ };
230
256
  export type BroadsideSubmitResult = {
231
257
  runId: string;
232
258
  outputDir: string;
@@ -242,6 +268,7 @@ export type BroadsideSubmitResult = {
242
268
  supportsStructuredOutputs?: boolean;
243
269
  expirationDate?: string | null;
244
270
  };
271
+ incremental: BroadsideIncrementalOutcome;
245
272
  };
246
273
  export type BroadsideCollectResult = {
247
274
  runId: string;
@@ -287,7 +314,22 @@ export declare function listLenses(): LensDefinition[];
287
314
  export declare function collectRepoInfo(targetDir: string): Promise<RepoInfo>;
288
315
  export declare function gatherSlices(targetDir: string, lens: LensDefinition, info: RepoInfo): Promise<FileSlice[]>;
289
316
  export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number): BatchRequest;
290
- export declare function estimateCost(lens: LensDefinition, slices: FileSlice[], pricing: ModelPricing, maxTokensOverride?: number): {
317
+ /**
318
+ * Pre-flight cost estimate for one lens.
319
+ *
320
+ * Every slice is its own batch request, so both halves scale with the slice
321
+ * count. The output half used to be a single `maxTokens * 0.75` for the whole
322
+ * lens no matter how many requests it sent — on a repository that sliced into
323
+ * 13 modules that budgeted one request's output and shipped thirteen, and a
324
+ * live run came in at roughly 3x its estimate. Since this number is what
325
+ * `max_cost` binds against, under-counting it lets a run outspend the cap the
326
+ * user set.
327
+ *
328
+ * @param info - Repo info, when the caller has it: lets the estimate include
329
+ * the system prompt and JSON schema each request carries. Omitted, the
330
+ * estimate covers slice content only, which is what the old signature did.
331
+ */
332
+ export declare function estimateCost(lens: LensDefinition, slices: FileSlice[], pricing: ModelPricing, maxTokensOverride?: number, info?: RepoInfo): {
291
333
  inputTokens: number;
292
334
  outputTokens: number;
293
335
  cost: number;
@@ -313,7 +355,40 @@ export declare function readBroadsideSkill(cwd: string): Promise<{
313
355
  }>;
314
356
  export declare function defaultBroadsideState(): BroadsideStateFile;
315
357
  export declare function loadBroadsideState(broadsideDir: string): Promise<BroadsideStateFile>;
358
+ /**
359
+ * Overwrite `state.json` wholesale with `state`.
360
+ *
361
+ * Prefer {@link persistBroadsideRun} anywhere a live operation is recording its
362
+ * own progress — this entry point replaces the file, so any run a concurrent
363
+ * process recorded in the meantime is erased. It remains the right call for
364
+ * seeding a fresh workspace and for test fixtures, where "make the file exactly
365
+ * this" is the intent.
366
+ */
316
367
  export declare function saveBroadsideState(broadsideDir: string, state: BroadsideStateFile): Promise<void>;
368
+ /**
369
+ * Read-modify-write `state.json` under a lock.
370
+ *
371
+ * The lock is held only for the read-modify-write, never for the surrounding
372
+ * operation: a `collect` can poll for the better part of an hour, and holding
373
+ * the lock across that would push every concurrent caller past the 5s lock
374
+ * timeout.
375
+ */
376
+ export declare function updateBroadsideStateAtomically(broadsideDir: string, mutate: (state: BroadsideStateFile) => void | Promise<void>): Promise<BroadsideStateFile>;
377
+ /**
378
+ * Record one run's current shape, merged into whatever is on disk *now*.
379
+ *
380
+ * Broad-Side operations are long-lived and hold their state in memory while
381
+ * they poll. Writing that snapshot back wholesale silently erased any run a
382
+ * concurrent operation had recorded since it was loaded, orphaning that run's
383
+ * paid results on disk — present as files, invisible to `list`, and unreachable
384
+ * by `collect`, which finds its run by position in `state.runs`. Observed live:
385
+ * a submit at 23:35 was erased by a collect that had loaded state before it and
386
+ * wrote back at 00:08.
387
+ *
388
+ * Merging by run id also self-heals: a run erased by an older writer is
389
+ * restored the next time its own operation checkpoints.
390
+ */
391
+ export declare function persistBroadsideRun(broadsideDir: string, run: BroadsideRun): Promise<BroadsideStateFile>;
317
392
  export declare function loadBroadsideConfig(broadsideDir: string): Promise<BroadsideConfig>;
318
393
  export declare function builtInCatalogEntry(model: string): CatalogEntry | null;
319
394
  export declare function builtInPricing(model: string): ModelPricing | null;
@@ -336,6 +411,14 @@ export declare function submitBatch(batchRequests: BatchRequest[], apiKey: strin
336
411
  error?: unknown;
337
412
  }>;
338
413
  export declare function fetchBatch(batchId: string, apiKey: string, fetcher?: FetchLike): Promise<Record<string, unknown>>;
414
+ /**
415
+ * Batch statuses that will never produce a result.
416
+ *
417
+ * Deliberately excludes the synthetic `timeout` this module returns when a poll
418
+ * budget expires: that batch is still running server-side and has already been
419
+ * charged, so callers must come back for it rather than retire it.
420
+ */
421
+ export declare const BROADSIDE_DEAD_BATCH_STATUSES: string[];
339
422
  export declare function pollBatchUntilTerminal(batchId: string, apiKey: string, opts?: {
340
423
  deadlineMs?: number;
341
424
  onStatus?: (status: string, counts: Record<string, unknown>) => void;
@@ -398,6 +481,18 @@ export type StoredLensResult = {
398
481
  */
399
482
  export declare function parseLensJson(content: string): unknown | null;
400
483
  export declare function saveLensResults(runDir: string, lensId: BroadsideLensId, batch: Record<string, unknown>): Promise<StoredLensResult[]>;
484
+ /**
485
+ * Rebuild lens results from what a previous collect already wrote to disk.
486
+ *
487
+ * The post-passes are gated on having lens findings in hand, and a collect
488
+ * only holds the ones *it* polled. When an earlier collect saved every lens
489
+ * and then died before synthesis and triage ran — the batch window is long
490
+ * and a poll can easily be interrupted — the next collect finds every lens
491
+ * already terminal, skips them all, and would otherwise reach the post-pass
492
+ * gate with nothing to hand it. Reading the saved results back is what makes
493
+ * "a resumed collect can finish whichever is still pending" true.
494
+ */
495
+ export declare function loadSavedLensResults(runDir: string, lenses: BroadsideLensId[]): Promise<StoredLensResult[]>;
401
496
  export declare function runBroadsideCollect(cwd: string, apiKey: string, opts?: {
402
497
  waitMs?: number;
403
498
  includeSynthesis?: boolean;
@@ -31,11 +31,12 @@
31
31
  // executable surfaces (Pi and MCP), not the pure template. What the template does
32
32
  // carry is the reading guide for its output — `.codecarto/broadside/SKILL.md`,
33
33
  // served by codecarto_skill under the name `broadside` (see readBroadsideSkill).
34
- import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
34
+ import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
35
35
  import { execFile } from "node:child_process";
36
36
  import { promisify } from "node:util";
37
- import { join } from "node:path";
37
+ import { join, relative } from "node:path";
38
38
  import { pathExists, sleep } from "./utils.js";
39
+ import { acquireLock } from "./status.js";
39
40
  import { loadYamlFile } from "./yaml.js";
40
41
  import { packagedWorkspaceDir } from "./workspace.js";
41
42
  const execFileAsync = promisify(execFile);
@@ -49,8 +50,14 @@ export const BROADSIDE_STATE_FILE = "state.json";
49
50
  export const BROADSIDE_CONFIG_FILE = "config.yaml";
50
51
  export const BROADSIDE_STATE_SCHEMA_VERSION = 1;
51
52
  // Per-token pricing in USD (OpenRouter, google/gemini-3.7-flash:batch).
52
- export const BROADSIDE_INPUT_PRICE_PER_M = 0.1875;
53
- export const BROADSIDE_OUTPUT_PRICE_PER_M = 0.9375;
53
+ // OpenRouter's listed rates for the `:batch` variant, which already carry the
54
+ // batch discount — the sync model is $0.75/$3.75. These were half these values
55
+ // until a live run compared them against the catalog: the batch discount had
56
+ // been applied a second time by hand, so every estimate for the default model
57
+ // came out at half its true cost and `max_cost` bound at twice what the user
58
+ // asked for. They are the offline fallback only; the live catalog wins.
59
+ export const BROADSIDE_INPUT_PRICE_PER_M = 0.375;
60
+ export const BROADSIDE_OUTPUT_PRICE_PER_M = 1.875;
54
61
  // OpenRouter's public model catalog; pricing, context, and capabilities live
55
62
  // per model id. The benchmarks endpoint adds coding/intelligence indices.
56
63
  export const BROADSIDE_MODELS_URL = "https://openrouter.ai/api/v1/models";
@@ -848,7 +855,10 @@ async function walkFiles(rootDir, dir, depth, remaining) {
848
855
  out = out.concat(children);
849
856
  }
850
857
  else if (entry.isFile()) {
851
- const rel = join(dir, entry.name).slice(rootDir.length + 1).split("\\").join("/");
858
+ // relative() rather than slice(rootDir.length + 1): the hand-rolled
859
+ // slice cut one character too many whenever rootDir carried a trailing
860
+ // separator, and mangled every path outright when rootDir was "/".
861
+ const rel = relative(rootDir, join(dir, entry.name)).split("\\").join("/");
852
862
  out.push(rel);
853
863
  }
854
864
  }
@@ -1142,9 +1152,30 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
1142
1152
  },
1143
1153
  };
1144
1154
  }
1145
- export function estimateCost(lens, slices, pricing, maxTokensOverride) {
1146
- const inputTokens = Math.ceil(slices.reduce((sum, s) => sum + (lens.maxChars === 0 ? 6000 : s.chars), 0) / 4);
1147
- const outputTokens = Math.ceil((maxTokensOverride ?? lens.maxTokens) * 0.75);
1155
+ /**
1156
+ * Pre-flight cost estimate for one lens.
1157
+ *
1158
+ * Every slice is its own batch request, so both halves scale with the slice
1159
+ * count. The output half used to be a single `maxTokens * 0.75` for the whole
1160
+ * lens no matter how many requests it sent — on a repository that sliced into
1161
+ * 13 modules that budgeted one request's output and shipped thirteen, and a
1162
+ * live run came in at roughly 3x its estimate. Since this number is what
1163
+ * `max_cost` binds against, under-counting it lets a run outspend the cap the
1164
+ * user set.
1165
+ *
1166
+ * @param info - Repo info, when the caller has it: lets the estimate include
1167
+ * the system prompt and JSON schema each request carries. Omitted, the
1168
+ * estimate covers slice content only, which is what the old signature did.
1169
+ */
1170
+ export function estimateCost(lens, slices, pricing, maxTokensOverride, info) {
1171
+ const sliceChars = slices.reduce((sum, s) => sum + (lens.maxChars === 0 ? 6000 : s.chars), 0);
1172
+ // The system prompt and the response schema ride on every request, so they
1173
+ // are paid once per slice rather than once per lens.
1174
+ const perRequestOverhead = info
1175
+ ? (lens.systemPrompt(info)?.length ?? 0) + JSON.stringify(SCHEMAS[lens.schemaName] ?? {}).length
1176
+ : 0;
1177
+ const inputTokens = Math.ceil((sliceChars + perRequestOverhead * slices.length) / 4);
1178
+ const outputTokens = slices.length * Math.ceil((maxTokensOverride ?? lens.maxTokens) * 0.75);
1148
1179
  const cost = (inputTokens / 1_000_000) * pricing.inputPerM +
1149
1180
  (outputTokens / 1_000_000) * pricing.outputPerM;
1150
1181
  return { inputTokens, outputTokens, cost };
@@ -1195,9 +1226,76 @@ export async function loadBroadsideState(broadsideDir) {
1195
1226
  return defaultBroadsideState();
1196
1227
  }
1197
1228
  }
1229
+ /**
1230
+ * Overwrite `state.json` wholesale with `state`.
1231
+ *
1232
+ * Prefer {@link persistBroadsideRun} anywhere a live operation is recording its
1233
+ * own progress — this entry point replaces the file, so any run a concurrent
1234
+ * process recorded in the meantime is erased. It remains the right call for
1235
+ * seeding a fresh workspace and for test fixtures, where "make the file exactly
1236
+ * this" is the intent.
1237
+ */
1198
1238
  export async function saveBroadsideState(broadsideDir, state) {
1199
1239
  await mkdir(broadsideDir, { recursive: true });
1200
- await writeFile(join(broadsideDir, BROADSIDE_STATE_FILE), `${JSON.stringify(state, null, "\t")}\n`, "utf8");
1240
+ const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
1241
+ const lock = await acquireLock(`${statePath}.lock`);
1242
+ try {
1243
+ await writeBroadsideStateFile(statePath, state);
1244
+ }
1245
+ finally {
1246
+ await lock.release();
1247
+ }
1248
+ }
1249
+ /** Serialize through a temp file so a crash mid-write cannot truncate state.json. */
1250
+ async function writeBroadsideStateFile(statePath, state) {
1251
+ const tempPath = `${statePath}.${process.pid}.${Date.now()}.tmp`;
1252
+ await writeFile(tempPath, `${JSON.stringify(state, null, "\t")}\n`, "utf8");
1253
+ await rename(tempPath, statePath);
1254
+ }
1255
+ /**
1256
+ * Read-modify-write `state.json` under a lock.
1257
+ *
1258
+ * The lock is held only for the read-modify-write, never for the surrounding
1259
+ * operation: a `collect` can poll for the better part of an hour, and holding
1260
+ * the lock across that would push every concurrent caller past the 5s lock
1261
+ * timeout.
1262
+ */
1263
+ export async function updateBroadsideStateAtomically(broadsideDir, mutate) {
1264
+ await mkdir(broadsideDir, { recursive: true });
1265
+ const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
1266
+ const lock = await acquireLock(`${statePath}.lock`);
1267
+ try {
1268
+ const state = await loadBroadsideState(broadsideDir);
1269
+ await mutate(state);
1270
+ await writeBroadsideStateFile(statePath, state);
1271
+ return state;
1272
+ }
1273
+ finally {
1274
+ await lock.release();
1275
+ }
1276
+ }
1277
+ /**
1278
+ * Record one run's current shape, merged into whatever is on disk *now*.
1279
+ *
1280
+ * Broad-Side operations are long-lived and hold their state in memory while
1281
+ * they poll. Writing that snapshot back wholesale silently erased any run a
1282
+ * concurrent operation had recorded since it was loaded, orphaning that run's
1283
+ * paid results on disk — present as files, invisible to `list`, and unreachable
1284
+ * by `collect`, which finds its run by position in `state.runs`. Observed live:
1285
+ * a submit at 23:35 was erased by a collect that had loaded state before it and
1286
+ * wrote back at 00:08.
1287
+ *
1288
+ * Merging by run id also self-heals: a run erased by an older writer is
1289
+ * restored the next time its own operation checkpoints.
1290
+ */
1291
+ export async function persistBroadsideRun(broadsideDir, run) {
1292
+ return updateBroadsideStateAtomically(broadsideDir, (state) => {
1293
+ const index = state.runs.findIndex((candidate) => candidate.id === run.id);
1294
+ if (index === -1)
1295
+ state.runs.push(run);
1296
+ else
1297
+ state.runs[index] = run;
1298
+ });
1201
1299
  }
1202
1300
  export async function loadBroadsideConfig(broadsideDir) {
1203
1301
  const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
@@ -1332,10 +1430,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
1332
1430
  },
1333
1431
  };
1334
1432
  }
1335
- const builtIn = builtInCatalogEntry(model);
1336
- if (builtIn)
1337
- return { model, source: "built-in", entry: builtIn };
1338
- // Unknown model: on-disk cache first, then the live catalog.
1433
+ // On-disk cache first, then the live catalog — for the default model too.
1434
+ // Hardcoded rates used to short-circuit here, which meant a stale constant
1435
+ // could never self-correct even though the catalog was already being
1436
+ // fetched for every other model. The authoritative source wins; the
1437
+ // constants below are what we fall back to when the network is unavailable.
1339
1438
  const cache = await readCatalogCache(broadsideDir);
1340
1439
  const cached = cache?.models[model];
1341
1440
  if (cached && Date.now() - new Date(cache.fetched_at).getTime() < BROADSIDE_CATALOG_CACHE_TTL_MS) {
@@ -1366,6 +1465,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
1366
1465
  await writeCatalogCache(broadsideDir, updated);
1367
1466
  return { model, source: "live", entry: live };
1368
1467
  }
1468
+ // Offline fallback: the default model's rates and capabilities are known at
1469
+ // compile time, so a network failure does not have to stop a run.
1470
+ const builtIn = builtInCatalogEntry(model);
1471
+ if (builtIn)
1472
+ return { model, source: "built-in", entry: builtIn };
1369
1473
  throw new Error(`Could not resolve per-token pricing for batch model "${model}". ` +
1370
1474
  "Set pricing.input_per_m and pricing.output_per_m in .codecarto/broadside/config.yaml " +
1371
1475
  "(USD per million tokens), or check the model id against https://openrouter.ai/models?variant=batch.");
@@ -1469,6 +1573,14 @@ export async function fetchBatch(batchId, apiKey, fetcher = fetch) {
1469
1573
  data.http_status = resp.status;
1470
1574
  return data;
1471
1575
  }
1576
+ /**
1577
+ * Batch statuses that will never produce a result.
1578
+ *
1579
+ * Deliberately excludes the synthetic `timeout` this module returns when a poll
1580
+ * budget expires: that batch is still running server-side and has already been
1581
+ * charged, so callers must come back for it rather than retire it.
1582
+ */
1583
+ export const BROADSIDE_DEAD_BATCH_STATUSES = ["failed", "expired", "cancelled", "auth-failed"];
1472
1584
  export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
1473
1585
  const deadline = Date.now() + (opts.deadlineMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
1474
1586
  const intervalMs = opts.pollIntervalMs ?? BROADSIDE_POLL_INTERVAL_MS;
@@ -1491,7 +1603,7 @@ export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
1491
1603
  const status = String(batch.status ?? "unknown");
1492
1604
  const counts = (batch.request_counts ?? {});
1493
1605
  opts.onStatus?.(status, counts);
1494
- if (["completed", "failed", "expired", "cancelled", "auth-failed"].includes(status))
1606
+ if (status === "completed" || BROADSIDE_DEAD_BATCH_STATUSES.includes(status))
1495
1607
  return batch;
1496
1608
  if (Date.now() >= deadline)
1497
1609
  return { id: batchId, status: "timeout" };
@@ -1563,14 +1675,30 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1563
1675
  const sourceDirty = await gitDirty(cwd);
1564
1676
  let baseHead = null;
1565
1677
  let changed = null;
1678
+ const incrementalOutcome = {
1679
+ requested: opts.incremental === true,
1680
+ applied: false,
1681
+ baseHead: null,
1682
+ };
1566
1683
  if (opts.incremental) {
1567
1684
  const state = await loadBroadsideState(broadsideDir);
1568
1685
  // The baseline is the most recent run that recorded a HEAD — a
1569
1686
  // submit-only run (never collected) is still a valid committed base.
1570
1687
  const previous = [...state.runs].reverse().find((r) => r.sourceHead);
1571
- if (previous?.sourceHead && !sourceDirty) {
1688
+ if (sourceDirty) {
1689
+ incrementalOutcome.reason = "dirty-worktree";
1690
+ }
1691
+ else if (!previous?.sourceHead) {
1692
+ incrementalOutcome.reason = "no-baseline";
1693
+ }
1694
+ else {
1572
1695
  baseHead = previous.sourceHead;
1573
1696
  changed = await changedFilesSince(cwd, baseHead);
1697
+ incrementalOutcome.baseHead = baseHead;
1698
+ if (changed)
1699
+ incrementalOutcome.applied = true;
1700
+ else
1701
+ incrementalOutcome.reason = "diff-failed";
1574
1702
  }
1575
1703
  }
1576
1704
  // Slice offline first so the estimate covers every request we would send.
@@ -1591,7 +1719,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1591
1719
  const lensModel = modelForLens(lensId);
1592
1720
  const { pricing: lensPricing, outputCap: lensOutputCap } = resolved.get(lensModel);
1593
1721
  const maxTokens = lensOutputCap ? Math.min(lens.maxTokens, lensOutputCap) : lens.maxTokens;
1594
- const estimate = estimateCost(lens, slices, lensPricing, maxTokens);
1722
+ const estimate = estimateCost(lens, slices, lensPricing, maxTokens, info);
1595
1723
  estimatedInputTokens += estimate.inputTokens;
1596
1724
  estimatedOutputTokens += estimate.outputTokens;
1597
1725
  estimatedTotalCost += estimate.cost;
@@ -1626,6 +1754,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1626
1754
  exceedsLimit,
1627
1755
  baseHead,
1628
1756
  sourceDirty,
1757
+ incremental: incrementalOutcome,
1629
1758
  ...(outputCap !== undefined && { outputCap }),
1630
1759
  });
1631
1760
  if (!approved)
@@ -1659,7 +1788,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1659
1788
  baseHead,
1660
1789
  };
1661
1790
  state.runs.push(run);
1662
- await saveBroadsideState(broadsideDir, state);
1791
+ await persistBroadsideRun(broadsideDir, run);
1663
1792
  const requestsByCustomId = {};
1664
1793
  const submissions = [];
1665
1794
  // Submit from the estimate rather than recomputing: the user approved that
@@ -1707,7 +1836,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1707
1836
  })());
1708
1837
  }
1709
1838
  await Promise.allSettled(submissions);
1710
- await saveBroadsideState(broadsideDir, state);
1839
+ await persistBroadsideRun(broadsideDir, run);
1711
1840
  // Persist the exact request bodies so collect can re-submit a truncated
1712
1841
  // slice (bumped output cap) without re-walking the repo (#133). The run
1713
1842
  // dir is created here rather than waiting for collect so a crash between
@@ -1734,6 +1863,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1734
1863
  : resolved.get(model).supportsStructuredOutputs,
1735
1864
  expirationDate: defaultEntry.expirationDate ?? null,
1736
1865
  },
1866
+ incremental: incrementalOutcome,
1737
1867
  };
1738
1868
  }
1739
1869
  function extractContent(result) {
@@ -1799,6 +1929,46 @@ export async function saveLensResults(runDir, lensId, batch) {
1799
1929
  }
1800
1930
  return out;
1801
1931
  }
1932
+ /**
1933
+ * Rebuild lens results from what a previous collect already wrote to disk.
1934
+ *
1935
+ * The post-passes are gated on having lens findings in hand, and a collect
1936
+ * only holds the ones *it* polled. When an earlier collect saved every lens
1937
+ * and then died before synthesis and triage ran — the batch window is long
1938
+ * and a poll can easily be interrupted — the next collect finds every lens
1939
+ * already terminal, skips them all, and would otherwise reach the post-pass
1940
+ * gate with nothing to hand it. Reading the saved results back is what makes
1941
+ * "a resumed collect can finish whichever is still pending" true.
1942
+ */
1943
+ export async function loadSavedLensResults(runDir, lenses) {
1944
+ if (!(await pathExists(runDir)))
1945
+ return [];
1946
+ const reserved = new Set(["requests.json", "run-meta.json", "synthesis.json", "triage.json"]);
1947
+ const out = [];
1948
+ // Longest lens id first: no id is a prefix of another today, but ordering
1949
+ // keeps that from becoming a silent misattribution if one ever is.
1950
+ const ordered = [...lenses].sort((a, b) => b.length - a.length);
1951
+ for (const name of (await readdir(runDir)).sort()) {
1952
+ if (!name.endsWith(".json") || name.endsWith(".error.json") || reserved.has(name) || name.startsWith("raw-"))
1953
+ continue;
1954
+ const customId = name.slice(0, -".json".length);
1955
+ const lensId = ordered.find((id) => customId === id || customId.startsWith(`${id}-`));
1956
+ if (!lensId)
1957
+ continue;
1958
+ const content = await readFile(join(runDir, name), "utf8").catch(() => null);
1959
+ if (content === null)
1960
+ continue;
1961
+ out.push({
1962
+ lensId,
1963
+ customId,
1964
+ moduleName: customId.replace(/^[a-z]+-/, ""),
1965
+ content,
1966
+ raw: {},
1967
+ truncated: parseLensJson(content) === null,
1968
+ });
1969
+ }
1970
+ return out;
1971
+ }
1802
1972
  async function loadStoredRequests(runDir) {
1803
1973
  const path = join(runDir, "requests.json");
1804
1974
  if (!(await pathExists(path)))
@@ -1956,11 +2126,19 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
1956
2126
  await writeFile(join(runDir, `raw-${lensId}.json`), `${JSON.stringify(batch, null, "\t")}\n`, "utf8");
1957
2127
  lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, truncated };
1958
2128
  }
1959
- else if (batch.error) {
1960
- entry.error = batch.error;
2129
+ else {
2130
+ // Every non-completed outcome still has to reach the report.
2131
+ // `lensOutcomes` is what the caller renders, and this branch used to
2132
+ // require `batch.error` — but the commonest failure here is the
2133
+ // synthetic `{ status: "timeout" }` the poll returns when its budget
2134
+ // expires with the batch still in flight, and that carries no error.
2135
+ // A lens that never came back was therefore omitted entirely,
2136
+ // indistinguishable in the output from one that was never requested.
2137
+ if (batch.error)
2138
+ entry.error = batch.error;
1961
2139
  lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount };
1962
2140
  }
1963
- await saveBroadsideState(broadsideDir, state);
2141
+ await persistBroadsideRun(broadsideDir, run);
1964
2142
  }
1965
2143
  // #133: re-submit truncated slices once with a bumped output cap. Batch
1966
2144
  // requests are pure, so re-running is always safe; the aim is to recover
@@ -1993,7 +2171,11 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
1993
2171
  if (error)
1994
2172
  continue;
1995
2173
  const batch = await pollBatchUntilTerminal(batchId, apiKey, {
1996
- deadlineMs: BROADSIDE_DEFAULT_POLL_BUDGET_MS,
2174
+ // Share the caller's deadline. Each of these polls used to
2175
+ // start a fresh 25-minute budget, so `wait_seconds` bounded
2176
+ // only the lens poll and a collect could run for the caller's
2177
+ // budget plus fifty minutes.
2178
+ deadlineMs: Math.max(0, deadline - Date.now()),
1997
2179
  onStatus: (status, counts) => opts.onStatus?.(`${stored.lensId}:retry`, status, counts),
1998
2180
  fetcher: opts.fetcher,
1999
2181
  });
@@ -2023,7 +2205,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2023
2205
  outcome.truncated = allLensResults.filter((s) => s.lensId === lensId && s.truncated).length;
2024
2206
  }
2025
2207
  }
2026
- await saveBroadsideState(broadsideDir, state);
2208
+ await persistBroadsideRun(broadsideDir, run);
2027
2209
  }
2028
2210
  // Synthesis + triage: cross-lens post-passes, only after every lens batch
2029
2211
  // is terminal. Triage turns the leads into a prioritized work order.
@@ -2032,12 +2214,25 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2032
2214
  let topTriageItems = [];
2033
2215
  const wantSynthesis = opts.includeSynthesis !== false;
2034
2216
  const wantTriage = opts.includeTriage !== false;
2217
+ // A resumed collect polls nothing — every lens is already terminal — so the
2218
+ // findings the post-passes need have to come back off disk, or a run whose
2219
+ // first collect was interrupted could never produce its executive report
2220
+ // and work order, however many times it was re-run.
2221
+ const postPassUnfinished = (entry) => entry.status === "pending" || entry.status === "submitted";
2222
+ if ((wantSynthesis || wantTriage) && allLensResults.length === 0
2223
+ && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
2224
+ const restored = await loadSavedLensResults(runDir, run.lenses);
2225
+ if (restored.length > 0) {
2226
+ allLensResults.push(...restored);
2227
+ truncatedCount = restored.filter((s) => s.truncated).length;
2228
+ }
2229
+ }
2035
2230
  if ((wantSynthesis || wantTriage) && allLensResults.length > 0) {
2036
2231
  const allTerminal = run.lenses.every((lensId) => {
2037
2232
  const entry = run.batches[lensId];
2038
2233
  return entry && ["completed", "failed", "expired", "cancelled", "auth-failed", "skipped", "rejected"].includes(entry.status);
2039
2234
  });
2040
- if (allTerminal && (run.synthesis.status === "pending" || run.triage.status === "pending")) {
2235
+ if (allTerminal && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
2041
2236
  const findingsText = allLensResults
2042
2237
  .map((r) => `## ${r.lensId} — ${r.customId}\n\n${r.content}\n`)
2043
2238
  .join("\n");
@@ -2066,6 +2261,24 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2066
2261
  : []),
2067
2262
  ];
2068
2263
  const submitted = new Map();
2264
+ // A pass can be left at "submitted" when an earlier collect returned
2265
+ // before its batch reached a terminal status — the batch still runs
2266
+ // and is still charged, so the result exists and is simply unclaimed.
2267
+ // Nothing above would ever look at it again: the pass list is built
2268
+ // from "pending" entries only. Poll those regardless of the want
2269
+ // flags, because the spend already happened and discarding a
2270
+ // finished result is worse than saving one the caller opted out of.
2271
+ for (const kind of ["synthesis", "triage"]) {
2272
+ const entry = kind === "synthesis" ? run.synthesis : run.triage;
2273
+ if (entry.status !== "submitted" || !entry.batchId)
2274
+ continue;
2275
+ if (submitted.has(entry.batchId))
2276
+ continue;
2277
+ submitted.set(entry.batchId, {
2278
+ batchId: entry.batchId,
2279
+ pass: { kind, request: undefined, entry },
2280
+ });
2281
+ }
2069
2282
  await Promise.allSettled(passes.map(async (pass) => {
2070
2283
  pass.entry.status = "submitted";
2071
2284
  try {
@@ -2081,10 +2294,13 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2081
2294
  pass.entry.status = "failed";
2082
2295
  }
2083
2296
  }));
2084
- await saveBroadsideState(broadsideDir, state);
2297
+ await persistBroadsideRun(broadsideDir, run);
2085
2298
  for (const { batchId, pass } of submitted.values()) {
2086
2299
  const batch = await pollBatchUntilTerminal(batchId, apiKey, {
2087
- deadlineMs: BROADSIDE_DEFAULT_POLL_BUDGET_MS,
2300
+ // Shares the caller's deadline, as the retry poll above does.
2301
+ // A pass whose poll runs out stays `submitted`, so the batch
2302
+ // is already paid for and a later collect claims its result.
2303
+ deadlineMs: Math.max(0, deadline - Date.now()),
2088
2304
  onStatus: (status, counts) => opts.onStatus?.(pass.kind, status, counts),
2089
2305
  fetcher: opts.fetcher,
2090
2306
  });
@@ -2107,10 +2323,20 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2107
2323
  }
2108
2324
  }
2109
2325
  }
2110
- else if (batch.error) {
2326
+ else if (BROADSIDE_DEAD_BATCH_STATUSES.includes(String(batch.status))) {
2327
+ // The batch will never produce a result, so retire the pass.
2328
+ // This used to require `batch.error`, leaving an expired or
2329
+ // cancelled batch parked at "submitted" forever — and since a
2330
+ // resumed collect re-polls anything still "submitted", it
2331
+ // would re-poll a dead batch on every future run.
2111
2332
  pass.entry.status = "failed";
2333
+ if (batch.error)
2334
+ pass.entry.error = batch.error instanceof Error ? batch.error.message : String(batch.error);
2112
2335
  }
2113
- await saveBroadsideState(broadsideDir, state);
2336
+ // A "timeout" is deliberately left at "submitted": the batch is
2337
+ // still running server-side and has already been paid for, so a
2338
+ // later collect should claim its result rather than discard it.
2339
+ await persistBroadsideRun(broadsideDir, run);
2114
2340
  }
2115
2341
  }
2116
2342
  }
@@ -2120,7 +2346,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
2120
2346
  });
2121
2347
  run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
2122
2348
  run.totalCost = totalCost;
2123
- await saveBroadsideState(broadsideDir, state);
2349
+ await persistBroadsideRun(broadsideDir, run);
2124
2350
  await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
2125
2351
  experimental: true,
2126
2352
  method: "Broad-Side (OpenRouter Batch API)",
@@ -2223,9 +2449,31 @@ function parseSynthesisTopFindings(content) {
2223
2449
  }
2224
2450
  }
2225
2451
  // ---------- formatting helpers for tool output ----------
2452
+ function describeIncrementalFallback(reason) {
2453
+ switch (reason) {
2454
+ case "dirty-worktree":
2455
+ return "the working tree has uncommitted changes, so there is no committed state to diff against";
2456
+ case "no-baseline":
2457
+ return "no earlier run recorded a commit to diff against";
2458
+ case "diff-failed":
2459
+ return "the diff against the previous run's commit could not be read";
2460
+ default:
2461
+ return "no baseline was available";
2462
+ }
2463
+ }
2226
2464
  export function estimateSubmitText(result, lenses) {
2465
+ // Count the lenses that actually got a batch, not every lens considered. A
2466
+ // lens with nothing to scan is reported as `skipped (0 request(s))` two
2467
+ // lines below, so counting it here made the header contradict its own body:
2468
+ // a Rust CLI with no server surface reported "submitted 6 batch(es)" over a
2469
+ // list showing four batches and two skips.
2470
+ const entries = Object.values(result.batches ?? {});
2471
+ const submittedCount = entries.filter((entry) => entry.batchId).length;
2472
+ const withoutBatch = entries.length - submittedCount;
2227
2473
  const lines = [
2228
- `Broad-Side submitted ${result.batches ? Object.keys(result.batches).length : 0} batch(es).`,
2474
+ withoutBatch > 0
2475
+ ? `Broad-Side submitted ${submittedCount} batch(es); ${withoutBatch} lens(es) produced none (see below).`
2476
+ : `Broad-Side submitted ${submittedCount} batch(es).`,
2229
2477
  ];
2230
2478
  for (const lens of lenses) {
2231
2479
  const entry = result.batches[lens.id];
@@ -2235,6 +2483,12 @@ export function estimateSubmitText(result, lenses) {
2235
2483
  const override = entry.model ? ` on ${entry.model}` : "";
2236
2484
  lines.push(` ${lens.name}: ${status} (${entry.requests} request(s), ~$${entry.estimatedCost.toFixed(4)})${override}`);
2237
2485
  }
2486
+ const incremental = result.incremental;
2487
+ if (incremental?.requested) {
2488
+ lines.push(incremental.applied
2489
+ ? `Incremental: scanning only what changed since ${(incremental.baseHead ?? "").slice(0, 8)}.`
2490
+ : `Incremental: requested but NOT applied — ${describeIncrementalFallback(incremental.reason)}. Every module was scanned, at full cost.`);
2491
+ }
2238
2492
  lines.push(`Estimated total: ~$${result.estimatedTotalCost.toFixed(4)}`, `Pricing: $${result.pricing.inputPerM.toFixed(4)}/M in, $${result.pricing.outputPerM.toFixed(4)}/M out (${result.pricing.source})`);
2239
2493
  if (result.modelInfo.contextLength) {
2240
2494
  lines.push(`Model: ${result.modelInfo.contextLength.toLocaleString()} context, ${result.modelInfo.maxCompletionTokens?.toLocaleString() ?? "?"} max output`);
@@ -39,4 +39,17 @@ export interface DashboardInputs {
39
39
  narration?: DashboardNarration;
40
40
  }
41
41
  export declare function renderDashboard(inputs: DashboardInputs): string;
42
+ /**
43
+ * A dashboard link target, or undefined when the path is not a safe relative
44
+ * reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
45
+ * refused outright.
46
+ *
47
+ * Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
48
+ * `dir\..\secret` survives the segment scan, because that scan splits on `/`
49
+ * — but `encodeURIComponent` turns the separators into `%5C`, which no URL
50
+ * parser treats as a path separator, so the traversal cannot resolve. Removing
51
+ * the encoding, or "simplifying" it to a raw join, reopens it. Exported so
52
+ * that property has a test.
53
+ */
54
+ export declare function safeRelativeHref(path: string): string | undefined;
42
55
  export declare function escapeHtml(input: string): string;
@@ -660,7 +660,19 @@ function renderSafeLink(path, label, className) {
660
660
  const classAttr = className ? ` class="${escapeAttr(className)}"` : "";
661
661
  return `<a${classAttr} href="${escapeAttr(href)}">${escapeHtml(label)}</a>`;
662
662
  }
663
- function safeRelativeHref(path) {
663
+ /**
664
+ * A dashboard link target, or undefined when the path is not a safe relative
665
+ * reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
666
+ * refused outright.
667
+ *
668
+ * Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
669
+ * `dir\..\secret` survives the segment scan, because that scan splits on `/`
670
+ * — but `encodeURIComponent` turns the separators into `%5C`, which no URL
671
+ * parser treats as a path separator, so the traversal cannot resolve. Removing
672
+ * the encoding, or "simplifying" it to a raw join, reopens it. Exported so
673
+ * that property has a test.
674
+ */
675
+ export function safeRelativeHref(path) {
664
676
  if (!path || path.startsWith("/") || path.startsWith("\\") || /^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(path))
665
677
  return undefined;
666
678
  const parts = path.split("/");
package/dist/core/yaml.js CHANGED
@@ -126,6 +126,19 @@ export function parseSimpleYaml(raw) {
126
126
  };
127
127
  const parseMapping = (indent) => {
128
128
  const result = {};
129
+ // Duplicate detection tracks keys explicitly rather than testing
130
+ // `key in result`. A bare object inherits from Object.prototype, so
131
+ // `"constructor" in result` is already true before anything is parsed —
132
+ // a document whose key is `constructor`, `toString`, `valueOf` or any
133
+ // other prototype member was rejected as a duplicate on first sight.
134
+ const seen = new Set();
135
+ // Assignment goes through defineProperty for the same reason: plain
136
+ // `result[key] = value` with the key `__proto__` invokes the prototype
137
+ // setter instead of creating an entry, so a hand-edited YAML file could
138
+ // change the shape of every object in the process rather than parse.
139
+ const assign = (key, value) => {
140
+ Object.defineProperty(result, key, { value, writable: true, enumerable: true, configurable: true });
141
+ };
129
142
  while (index < lines.length) {
130
143
  skipBlank();
131
144
  if (index >= lines.length)
@@ -147,9 +160,10 @@ export function parseSimpleYaml(raw) {
147
160
  const key = trimmed.slice(0, separator).trim();
148
161
  const rawValue = trimmed.slice(separator + 1).trim();
149
162
  index++;
150
- if (key in result) {
163
+ if (seen.has(key)) {
151
164
  throw new Error(`Duplicate YAML key: ${key} near line: ${line.trim()}`);
152
165
  }
166
+ seen.add(key);
153
167
  if (rawValue === "|" || rawValue === "|-") {
154
168
  const blockLines = [];
155
169
  let contentIndent = null;
@@ -168,19 +182,19 @@ export function parseSimpleYaml(raw) {
168
182
  index++;
169
183
  }
170
184
  const content = blockLines.join("\n").replace(/\n+$/, "");
171
- result[key] = rawValue === "|" ? `${content}\n` : content;
185
+ assign(key, rawValue === "|" ? `${content}\n` : content);
172
186
  continue;
173
187
  }
174
188
  if (rawValue !== "") {
175
- result[key] = parseYamlScalar(rawValue);
189
+ assign(key, parseYamlScalar(rawValue));
176
190
  continue;
177
191
  }
178
192
  skipBlank();
179
193
  if (index < lines.length && countIndent(lines[index] ?? "") > indent) {
180
- result[key] = parseBlock(countIndent(lines[index] ?? ""));
194
+ assign(key, parseBlock(countIndent(lines[index] ?? "")));
181
195
  }
182
196
  else {
183
- result[key] = null;
197
+ assign(key, null);
184
198
  }
185
199
  }
186
200
  return result;
@@ -198,7 +212,13 @@ export function parseSimpleYaml(raw) {
198
212
  const trimmed = line.slice(lineIndent);
199
213
  if (lineIndent !== indent || (!trimmed.startsWith("- ") && trimmed !== "-"))
200
214
  break;
201
- const rawItem = trimmed === "-" ? "" : trimmed.slice(2).trim();
215
+ const afterDash = trimmed === "-" ? "" : trimmed.slice(2);
216
+ const rawItem = afterDash.trim();
217
+ // A sequence item's mapping continues at the column where its own
218
+ // content starts, which is not always two past the dash: `- id: x`
219
+ // aligns its siblings under the `i`, four columns in. Hardcoding two
220
+ // rejected that valid layout as bad indentation.
221
+ const itemIndent = indent + 2 + (afterDash.length - afterDash.trimStart().length);
202
222
  index++;
203
223
  if (rawItem === "") {
204
224
  skipBlank();
@@ -221,7 +241,7 @@ export function parseSimpleYaml(raw) {
221
241
  item[key] = parseBlock(countIndent(lines[index] ?? ""));
222
242
  }
223
243
  if (index < lines.length && countIndent(lines[index] ?? "") > indent) {
224
- const nested = parseMapping(indent + 2);
244
+ const nested = parseMapping(itemIndent);
225
245
  for (const [nestedKey, nestedValue] of Object.entries(nested))
226
246
  item[nestedKey] = nestedValue;
227
247
  }
@@ -44,7 +44,7 @@ export interface AutoCompleteResult {
44
44
  /** Non-gating closure-integrity notes from completion (#122). */
45
45
  warnings: string[];
46
46
  }
47
- export declare function autoCompletePhase(ctx: ExtensionContext, validation: ValidationResult): Promise<AutoCompleteResult>;
47
+ export declare function autoCompletePhase(cwd: string, validation: ValidationResult): Promise<AutoCompleteResult>;
48
48
  export type AutoOutcome = "complete" | "stopped" | "aborted";
49
49
  export interface AutoRunOptions {
50
50
  strict: boolean;
@@ -151,9 +151,13 @@ export function isPhaseRunning(phaseId) {
151
151
  const existing = getPhaseActivity(phaseId);
152
152
  return existing?.status === "running";
153
153
  }
154
- export async function autoCompletePhase(ctx, validation) {
155
- const result = await completeValidatedPhase(ctx.cwd, validation, "/codecarto-complete");
156
- void writeDashboard(ctx.cwd, PACKAGE_VERSION);
154
+ export async function autoCompletePhase(
155
+ // A directory, not a ctx: this only ever needed `ctx.cwd`, and its callers
156
+ // run after a sub-agent has invalidated the ctx they captured, where every
157
+ // property access throws.
158
+ cwd, validation) {
159
+ const result = await completeValidatedPhase(cwd, validation, "/codecarto-complete");
160
+ void writeDashboard(cwd, PACKAGE_VERSION);
157
161
  return result;
158
162
  }
159
163
  export function decideAfterPhase(phaseStatus, phaseError, validation, strict) {
@@ -185,6 +189,9 @@ export function decideAfterPhase(phaseStatus, phaseError, validation, strict) {
185
189
  }
186
190
  export async function runAuto(ctx, pi, initialState, options) {
187
191
  const startedAt = Date.now();
192
+ // Captured once: the loop below spawns a sub-agent per phase, and each one
193
+ // invalidates this ctx, after which reading `ctx.cwd` throws.
194
+ const autoCwd = ctx.cwd;
188
195
  const phasesRun = [];
189
196
  const totalTokens = { input: 0, output: 0, cacheWrite: 0 };
190
197
  const totalPhases = initialState.pipeline.phase_order.length;
@@ -262,7 +269,7 @@ export async function runAuto(ctx, pi, initialState, options) {
262
269
  // decision.action === "continue" → auto-complete and loop.
263
270
  // validation is guaranteed non-null on the continue branch.
264
271
  try {
265
- const { updatedState } = await autoCompletePhase(ctx, validation);
272
+ const { updatedState } = await autoCompletePhase(autoCwd, validation);
266
273
  state = updatedState;
267
274
  phasesRun.push(phase.id);
268
275
  options.onPhaseAdvanced?.(state);
@@ -83,8 +83,42 @@ function buildStatusLines(state, extraLines = []) {
83
83
  }
84
84
  return lines;
85
85
  }
86
+ /**
87
+ * Whether `ctx` still belongs to the live session.
88
+ *
89
+ * Pi invalidates an extension ctx when the session is replaced, and from then
90
+ * on *every* property access on it throws — `ctx.cwd` and `ctx.hasUI` included.
91
+ * A phase runs as a sub-agent, so by the time post-phase work fires, the ctx
92
+ * captured when the command started may already be dead. That is an ordinary
93
+ * outcome rather than an error: the UI it would have refreshed is gone with the
94
+ * session. Callers skip their UI work instead of throwing into a `void` call
95
+ * that nothing is waiting on.
96
+ */
97
+ function isCtxLive(ctx) {
98
+ try {
99
+ return typeof ctx.cwd === "string";
100
+ }
101
+ catch {
102
+ return false;
103
+ }
104
+ }
105
+ /**
106
+ * Notify through `ctx`, dropping the message if the session it belonged to is
107
+ * gone.
108
+ *
109
+ * `ctx.hasUI` throws on a stale ctx rather than returning false, so the usual
110
+ * `if (ctx.hasUI) ctx.ui.notify(...)` guard was itself a throw site. Inside a
111
+ * promise chain that was worse than a lost message: the `.catch` handler threw
112
+ * while reporting the original failure, and that second rejection had nothing
113
+ * left to catch it.
114
+ */
115
+ function notifyCtx(ctx, message, level) {
116
+ if (!isCtxLive(ctx) || !ctx.hasUI)
117
+ return;
118
+ ctx.ui.notify(message, level);
119
+ }
86
120
  function setUiState(ctx, state, extraLines = []) {
87
- if (!ctx.hasUI)
121
+ if (!isCtxLive(ctx) || !ctx.hasUI)
88
122
  return;
89
123
  if (!state) {
90
124
  ctx.ui.setStatus(STATUS_LINE_ID, undefined);
@@ -260,6 +294,11 @@ export default function codeCartographerExtension(pi) {
260
294
  // remembered here for the completers that list files under .codecarto/.
261
295
  let sessionCwd;
262
296
  const readWorkspaceState = async (ctx, notifyOnError = true) => {
297
+ // `ctx.cwd` was read before the try, so a stale ctx made this reject
298
+ // rather than return null as its signature promises — and the callers
299
+ // that fire it without awaiting turned that into an unhandled rejection.
300
+ if (!isCtxLive(ctx))
301
+ return null;
263
302
  sessionCwd = ctx.cwd;
264
303
  try {
265
304
  return await getWorkspaceState(ctx.cwd);
@@ -268,12 +307,16 @@ export default function codeCartographerExtension(pi) {
268
307
  const message = error instanceof Error ? error.message : String(error);
269
308
  lastFeedbackLines = [message];
270
309
  setUiState(ctx, null);
271
- if (notifyOnError && ctx.hasUI)
310
+ // The ctx can die between the read above and here, so the error
311
+ // path must not assume it is still usable either.
312
+ if (notifyOnError && isCtxLive(ctx) && ctx.hasUI)
272
313
  ctx.ui.notify(message, "error");
273
314
  return null;
274
315
  }
275
316
  };
276
317
  const refreshWorkspaceUi = async (ctx, extraLines) => {
318
+ if (!isCtxLive(ctx))
319
+ return null;
277
320
  if (!codecartoModeActive) {
278
321
  setUiState(ctx, null);
279
322
  return null;
@@ -290,15 +333,19 @@ export default function codeCartographerExtension(pi) {
290
333
  const ensureWorkspaceState = async (ctx) => {
291
334
  if (!codecartoModeActive) {
292
335
  setUiState(ctx, null);
293
- ctx.ui.notify("CodeCartographer is not active in this session. Run /codecarto-init first.", "warning");
336
+ notifyCtx(ctx, "CodeCartographer is not active in this session. Run /codecarto-init first.", "warning");
294
337
  return null;
295
338
  }
296
339
  const state = await readWorkspaceState(ctx);
297
340
  if (state)
298
341
  return state;
342
+ // Reached when the ctx is stale as well as when there is no workspace,
343
+ // so neither `ctx.cwd` nor the notify below may assume a live ctx.
344
+ if (!isCtxLive(ctx))
345
+ return null;
299
346
  const hasWorkspace = await pathExists(join(ctx.cwd, ".codecarto", "workflow", "status.yaml"));
300
347
  if (!hasWorkspace)
301
- ctx.ui.notify("No .codecarto/ workspace found. Run /codecarto-init first.", "warning");
348
+ notifyCtx(ctx, "No .codecarto/ workspace found. Run /codecarto-init first.", "warning");
302
349
  return null;
303
350
  };
304
351
  pi.on("session_start", async (_event, ctx) => {
@@ -607,34 +654,36 @@ export default function codeCartographerExtension(pi) {
607
654
  // phase so status.yaml advances without requiring the user to manually
608
655
  // run /codecarto-validate then /codecarto-complete. This mirrors what
609
656
  // the auto loop (runAuto) does after each phase.
657
+ // The sub-agent replaces the session, which invalidates this ctx —
658
+ // every later property access on it throws. Capture the directory
659
+ // now so the post-phase work below does not depend on the ctx
660
+ // surviving, and route UI updates through notifyCtx, which drops
661
+ // them if it has not.
662
+ const phaseCwd = ctx.cwd;
610
663
  void runSinglePhase(ctx, pi, state, phase, { llmSteerEnabled, signal: ctx.signal, preflight })
611
664
  .then(async (result) => {
612
665
  if (result.status !== "completed")
613
666
  return;
614
667
  // Refresh state from disk — the sub-agent may have written
615
668
  // findings that the validator needs to read.
616
- const stateForValidation = (await getWorkspaceState(ctx.cwd)) ?? state;
669
+ const stateForValidation = (await getWorkspaceState(phaseCwd)) ?? state;
617
670
  const validation = await validatePhaseOutput(stateForValidation, phase.id).catch((error) => (error instanceof Error ? error : new Error(String(error))));
618
671
  if (validation instanceof Error) {
619
- if (ctx.hasUI)
620
- ctx.ui.notify(`Auto-validation error for ${phase.id}: ${validation.message}`, "warning");
672
+ notifyCtx(ctx, `Auto-validation error for ${phase.id}: ${validation.message}`, "warning");
621
673
  lastFeedbackLines = [`Validation error: ${validation.message}`, "Run `/codecarto-validate` then `/codecarto-complete` manually."];
622
674
  return;
623
675
  }
624
676
  if (validation.overall === "FAIL" || validation.overall === "MISSING") {
625
- if (ctx.hasUI)
626
- ctx.ui.notify(`Phase ${phase.id} validation: ${validation.overall}. Fix the output, then re-run /codecarto-next.`, "warning");
677
+ notifyCtx(ctx, `Phase ${phase.id} validation: ${validation.overall}. Fix the output, then re-run /codecarto-next.`, "warning");
627
678
  lastFeedbackLines = buildValidationSummary(validation);
628
679
  return;
629
680
  }
630
681
  // PASS or PASS WITH GAPS — auto-complete the phase.
631
682
  try {
632
- const { updatedState, closeoutNotice } = await autoCompletePhase(ctx, validation);
633
- if (ctx.hasUI) {
634
- ctx.ui.notify(`Phase ${phase.id} auto-completed (validation: ${validation.overall}).`, validation.overall === "PASS WITH GAPS" ? "warning" : "info");
635
- if (closeoutNotice)
636
- ctx.ui.notify(closeoutNotice, "info");
637
- }
683
+ const { updatedState, closeoutNotice } = await autoCompletePhase(phaseCwd, validation);
684
+ notifyCtx(ctx, `Phase ${phase.id} auto-completed (validation: ${validation.overall}).`, validation.overall === "PASS WITH GAPS" ? "warning" : "info");
685
+ if (closeoutNotice)
686
+ notifyCtx(ctx, closeoutNotice, "info");
638
687
  lastFeedbackLines = [
639
688
  `Completed phase: ${validation.phaseId}`,
640
689
  `Validation: ${validation.overall}`,
@@ -645,22 +694,20 @@ export default function codeCartographerExtension(pi) {
645
694
  }
646
695
  catch (error) {
647
696
  const message = error instanceof Error ? error.message : String(error);
648
- if (ctx.hasUI)
649
- ctx.ui.notify(`Auto-completion failed for ${phase.id}: ${message}. Run /codecarto-complete manually.`, "warning");
697
+ notifyCtx(ctx, `Auto-completion failed for ${phase.id}: ${message}. Run /codecarto-complete manually.`, "warning");
650
698
  lastFeedbackLines = [`Auto-completion failed: ${message}`, "Run `/codecarto-complete` manually."];
651
699
  }
652
700
  })
653
701
  .catch((error) => {
654
702
  const message = error instanceof Error ? error.message : String(error);
655
- if (ctx.hasUI)
656
- ctx.ui.notify(`Post-phase processing error for ${phase.id}: ${message}`, "warning");
703
+ notifyCtx(ctx, `Post-phase processing error for ${phase.id}: ${message}`, "warning");
657
704
  lastFeedbackLines = [`Post-phase error: ${message}`];
658
705
  })
659
706
  .finally(() => {
660
707
  // Refresh the status widget after the phase resolves so the
661
708
  // "Open questions / Carry-forward / Next" lines reflect any
662
709
  // owner_notes the sub-agent wrote to status.yaml.
663
- void refreshWorkspaceUi(ctx);
710
+ void refreshWorkspaceUi(ctx).catch(() => undefined);
664
711
  });
665
712
  },
666
713
  });
@@ -738,7 +785,7 @@ export default function codeCartographerExtension(pi) {
738
785
  ctx.ui.notify(`Cannot complete ${validation.phaseId}: ${validation.overall}`, "error");
739
786
  return;
740
787
  }
741
- const { updatedState, closeoutNotice, warnings } = await autoCompletePhase(ctx, validation);
788
+ const { updatedState, closeoutNotice, warnings } = await autoCompletePhase(ctx.cwd, validation);
742
789
  lastFeedbackLines = [
743
790
  `Completed phase: ${validation.phaseId}`,
744
791
  `Validation: ${validation.overall}`,
@@ -884,7 +931,7 @@ export default function codeCartographerExtension(pi) {
884
931
  }
885
932
  lastFeedbackLines = [`Queued the CodeCartographer guide: ${document.topic}`];
886
933
  if (codecartoModeActive)
887
- void refreshWorkspaceUi(ctx, lastFeedbackLines);
934
+ void refreshWorkspaceUi(ctx, lastFeedbackLines).catch(() => undefined);
888
935
  ctx.ui.notify(`Queued the CodeCartographer guide (${document.topic})`, "info");
889
936
  },
890
937
  });
@@ -917,7 +964,7 @@ export default function codeCartographerExtension(pi) {
917
964
  // A workspace session already has a widget; fold the result into it.
918
965
  if (ctx.hasUI)
919
966
  ctx.ui.setWidget(BROADSIDE_WIDGET_ID, undefined);
920
- void refreshWorkspaceUi(ctx, lines);
967
+ void refreshWorkspaceUi(ctx, lines).catch(() => undefined);
921
968
  }
922
969
  else if (ctx.hasUI) {
923
970
  // Scout-only repository: the Broad-Side widget is the only place
@@ -24,6 +24,20 @@ import { initLibrary } from "../core/library.js";
24
24
  import { loadUserConfig, resolveUserConfigPath } from "../core/orchestrator-config.js";
25
25
  import { writeDashboard } from "../extensions/codecarto/dashboard-writer.js";
26
26
  // ---------- input helpers ----------
27
+ /**
28
+ * Normalize an optional `phase` argument. A client can send any JSON, and
29
+ * `args.phase?.trim()` throws a bare TypeError on a number or an object —
30
+ * surfacing as an opaque InternalError rather than telling the caller which
31
+ * argument was wrong. Sibling handlers already guard the required case.
32
+ */
33
+ function requireOptionalPhase(phase) {
34
+ if (phase === undefined || phase === null)
35
+ return undefined;
36
+ if (typeof phase !== "string") {
37
+ throw new McpError(ErrorCode.InvalidParams, `phase must be a string when provided, got ${typeof phase}`);
38
+ }
39
+ return phase.trim() || undefined;
40
+ }
27
41
  async function validateCwd(cwd) {
28
42
  if (typeof cwd !== "string" || !cwd.trim()) {
29
43
  throw new McpError(ErrorCode.InvalidParams, "cwd is required");
@@ -148,7 +162,7 @@ export async function handleStatus(args) {
148
162
  const currentPhase = nextPhase?.id ?? state.status.current_phase ?? "complete";
149
163
  const completed = state.pipeline.phase_order.filter((id) => state.status.phases[id]?.status === "complete").length;
150
164
  const totalCarryForward = Object.values(state.status.phases).reduce((sum, phase) => sum + (phase.carry_forward?.length ?? 0), 0);
151
- const currentOpenQuestions = currentPhase === "complete" ? 0 : state.status.phases[currentPhase]?.open_questions.length ?? 0;
165
+ const currentOpenQuestions = currentPhase === "complete" ? 0 : state.status.phases[currentPhase]?.open_questions?.length ?? 0;
152
166
  const terminalOpenQuestions = Object.values(state.status.phases).reduce((sum, phase) => sum + (phase.open_questions?.length ?? 0), 0);
153
167
  const postPipelinePending = state.status.post_pipeline.filter((entry) => entry.status !== "resolved").length;
154
168
  const scaffoldNotice = describeScaffoldStaleness(state);
@@ -237,7 +251,7 @@ export async function handlePhase(args) {
237
251
  export async function handleValidate(args) {
238
252
  const cwd = await validateCwd(args.cwd);
239
253
  const state = await requireWorkspace(cwd);
240
- const validation = await validatePhaseOutput(state, args.phase?.trim() || undefined).catch((error) => {
254
+ const validation = await validatePhaseOutput(state, requireOptionalPhase(args.phase)).catch((error) => {
241
255
  throw new McpError(ErrorCode.InvalidParams, error instanceof Error ? error.message : String(error));
242
256
  });
243
257
  const summary = buildValidationSummary(validation).join("\n");
@@ -256,7 +270,7 @@ export async function handleValidate(args) {
256
270
  export async function handleComplete(args) {
257
271
  const cwd = await validateCwd(args.cwd);
258
272
  const initialState = await requireWorkspace(cwd);
259
- const validation = await validatePhaseOutput(initialState, args.phase?.trim() || undefined).catch((error) => {
273
+ const validation = await validatePhaseOutput(initialState, requireOptionalPhase(args.phase)).catch((error) => {
260
274
  throw new McpError(ErrorCode.InvalidParams, error instanceof Error ? error.message : String(error));
261
275
  });
262
276
  if (validation.overall === "FAIL" || validation.overall === "MISSING") {
@@ -377,9 +391,17 @@ async function resolveLibraryPath(args) {
377
391
  * the publish tool enforces.
378
392
  */
379
393
  async function loadEffectiveConfig(cwd) {
380
- return typeof cwd === "string" && cwd.trim() !== ""
381
- ? loadCodecartoConfig(join(cwd.trim(), ".codecarto"))
382
- : loadUserConfig();
394
+ if (typeof cwd !== "string" || cwd.trim() === "")
395
+ return loadUserConfig();
396
+ // A relative path here resolves against the server process's working
397
+ // directory, not the caller's, so it would quietly read some other
398
+ // workspace's config — and this config decides whether publish_confirm
399
+ // gates the write. Refuse rather than answer from the wrong file.
400
+ const trimmed = cwd.trim();
401
+ if (!isAbsolute(trimmed)) {
402
+ throw new McpError(ErrorCode.InvalidParams, `cwd must be an absolute path, got: ${trimmed}`);
403
+ }
404
+ return loadCodecartoConfig(join(trimmed, ".codecarto"));
383
405
  }
384
406
  function asStringArray(value, fieldName) {
385
407
  if (!Array.isArray(value)) {
@@ -731,6 +753,13 @@ export async function handleLibraryInit(args) {
731
753
  }
732
754
  export async function handleVision(args) {
733
755
  const cwd = await validateCwd(args.cwd);
756
+ // raw_text is interpolated straight into the returned prompt, so an absent
757
+ // value silently becomes the literal string "undefined" for the agent to
758
+ // synthesize a vision brief from. Every sibling handler validates its
759
+ // required string argument; this one did not.
760
+ if (typeof args.raw_text !== "string" || !args.raw_text.trim()) {
761
+ throw new McpError(ErrorCode.InvalidParams, "raw_text is required (the user's raw product description)");
762
+ }
734
763
  const workspaceDir = join(cwd, ".codecarto");
735
764
  const interviewPath = join(workspaceDir, "findings", "vision-capture", "INTERVIEW.md");
736
765
  const visionPath = join(workspaceDir, "inputs", "vision.md");
@@ -757,8 +786,8 @@ export async function handleVision(args) {
757
786
  });
758
787
  }
759
788
  export async function handleConfig(args) {
760
- const config = args.cwd
761
- ? await loadCodecartoConfig(join(args.cwd, ".codecarto"))
789
+ const config = args.cwd !== undefined && args.cwd !== null
790
+ ? await loadCodecartoConfig(join(await validateCwd(args.cwd), ".codecarto"))
762
791
  : await loadUserConfig();
763
792
  const userConfigPath = resolveUserConfigPath();
764
793
  const workspaceConfigPath = args.cwd ? join(args.cwd, ".codecarto", "workflow", "config.yaml") : null;
@@ -976,6 +1005,11 @@ export async function handleBroadside(args) {
976
1005
  includeTriage,
977
1006
  retryTruncated,
978
1007
  onStatus: (lensId, status, counts) => lines.push(` ${lensId}: ${status} (${counts.completed ?? 0}/${counts.total ?? "?"})`),
1008
+ }).catch((error) => {
1009
+ // The `collect` action normalizes this same call; without it here,
1010
+ // a failure during submit-with-wait reached the client as an
1011
+ // opaque InternalError instead of naming its cause.
1012
+ throw new McpError(ErrorCode.InvalidRequest, error instanceof Error ? error.message : String(error));
979
1013
  });
980
1014
  lines.push("", collectResultText(collect));
981
1015
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "codecartographer-pi",
3
- "version": "0.19.0",
3
+ "version": "0.19.1",
4
4
  "mcpName": "io.github.HuginnIndustries/codecartographer",
5
5
  "description": "Turn an unfamiliar codebase into a validated reimplementation spec, then synthesize confirmed specs and a product vision into a traceable plan.",
6
6
  "type": "module",