codecartographer-pi 0.19.0 → 0.19.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +18 -0
- package/.codecarto/broadside/config.yaml +43 -4
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/dist/core/broadside.d.ts +144 -5
- package/dist/core/broadside.js +339 -32
- package/dist/core/dashboard.d.ts +13 -0
- package/dist/core/dashboard.js +13 -1
- package/dist/core/yaml.js +27 -7
- package/dist/extensions/codecarto/agent-rewriter.js +2 -0
- package/dist/extensions/codecarto/agent-runner.js +2 -0
- package/dist/extensions/codecarto/auto-runner.d.ts +1 -1
- package/dist/extensions/codecarto/auto-runner.js +11 -4
- package/dist/extensions/codecarto/child-model-runtime.d.ts +2 -0
- package/dist/extensions/codecarto/child-model-runtime.js +44 -0
- package/dist/extensions/codecarto/dashboard-narrator.js +2 -0
- package/dist/extensions/codecarto/index.js +88 -23
- package/dist/mcp-server/server.js +42 -8
- package/package.json +1 -1
package/dist/core/broadside.js
CHANGED
|
@@ -31,11 +31,12 @@
|
|
|
31
31
|
// executable surfaces (Pi and MCP), not the pure template. What the template does
|
|
32
32
|
// carry is the reading guide for its output — `.codecarto/broadside/SKILL.md`,
|
|
33
33
|
// served by codecarto_skill under the name `broadside` (see readBroadsideSkill).
|
|
34
|
-
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
34
|
+
import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
|
|
35
35
|
import { execFile } from "node:child_process";
|
|
36
36
|
import { promisify } from "node:util";
|
|
37
|
-
import { join } from "node:path";
|
|
37
|
+
import { join, relative } from "node:path";
|
|
38
38
|
import { pathExists, sleep } from "./utils.js";
|
|
39
|
+
import { acquireLock } from "./status.js";
|
|
39
40
|
import { loadYamlFile } from "./yaml.js";
|
|
40
41
|
import { packagedWorkspaceDir } from "./workspace.js";
|
|
41
42
|
const execFileAsync = promisify(execFile);
|
|
@@ -49,8 +50,14 @@ export const BROADSIDE_STATE_FILE = "state.json";
|
|
|
49
50
|
export const BROADSIDE_CONFIG_FILE = "config.yaml";
|
|
50
51
|
export const BROADSIDE_STATE_SCHEMA_VERSION = 1;
|
|
51
52
|
// Per-token pricing in USD (OpenRouter, google/gemini-3.7-flash:batch).
|
|
52
|
-
|
|
53
|
-
|
|
53
|
+
// OpenRouter's listed rates for the `:batch` variant, which already carry the
|
|
54
|
+
// batch discount — the sync model is $0.75/$3.75. These were half these values
|
|
55
|
+
// until a live run compared them against the catalog: the batch discount had
|
|
56
|
+
// been applied a second time by hand, so every estimate for the default model
|
|
57
|
+
// came out at half its true cost and `max_cost` bound at twice what the user
|
|
58
|
+
// asked for. They are the offline fallback only; the live catalog wins.
|
|
59
|
+
export const BROADSIDE_INPUT_PRICE_PER_M = 0.375;
|
|
60
|
+
export const BROADSIDE_OUTPUT_PRICE_PER_M = 1.875;
|
|
54
61
|
// OpenRouter's public model catalog; pricing, context, and capabilities live
|
|
55
62
|
// per model id. The benchmarks endpoint adds coding/intelligence indices.
|
|
56
63
|
export const BROADSIDE_MODELS_URL = "https://openrouter.ai/api/v1/models";
|
|
@@ -67,6 +74,34 @@ export const BROADSIDE_LENS_IDS = [
|
|
|
67
74
|
];
|
|
68
75
|
export const BROADSIDE_POLL_INTERVAL_MS = 15_000;
|
|
69
76
|
export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
|
|
77
|
+
/**
|
|
78
|
+
* The share of a lens's output budget reasoning may spend.
|
|
79
|
+
*
|
|
80
|
+
* `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
|
|
81
|
+
* at the remaining quarter makes that assumption true by construction and
|
|
82
|
+
* guarantees the answer has room. A floor keeps the cap sane for a small lens.
|
|
83
|
+
*/
|
|
84
|
+
export const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
|
|
85
|
+
export const BROADSIDE_MIN_REASONING_TOKENS = 512;
|
|
86
|
+
/**
|
|
87
|
+
* Cap reasoning for a lens request — deliberately a cap, not an off switch.
|
|
88
|
+
*
|
|
89
|
+
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
90
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
91
|
+
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
92
|
+
* or not a provider allows reasoning to be switched off.
|
|
93
|
+
*
|
|
94
|
+
* The failure this prevents is the budget being spent thinking rather than
|
|
95
|
+
* answering. Measured on one run: 5,758 of a 6,000-token budget went to
|
|
96
|
+
* reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
|
|
97
|
+
* those tokens bill at the full output rate. The shipped default model does the
|
|
98
|
+
* same thing less consistently (reasoning tokens from 0 to 5,757 across 13
|
|
99
|
+
* slices, three of them cut off at `finish_reason: length`), so this is not a
|
|
100
|
+
* multi-model concern.
|
|
101
|
+
*/
|
|
102
|
+
export function defaultReasoningFor(maxTokens) {
|
|
103
|
+
return { max_tokens: Math.max(BROADSIDE_MIN_REASONING_TOKENS, Math.floor(maxTokens * BROADSIDE_REASONING_BUDGET_FRACTION)) };
|
|
104
|
+
}
|
|
70
105
|
/** Thrown when a confirm hook declines a run. Nothing was submitted. */
|
|
71
106
|
export class BroadsideCancelledError extends Error {
|
|
72
107
|
constructor(message = "Broad-Side submission cancelled. Nothing was submitted.") {
|
|
@@ -848,7 +883,10 @@ async function walkFiles(rootDir, dir, depth, remaining) {
|
|
|
848
883
|
out = out.concat(children);
|
|
849
884
|
}
|
|
850
885
|
else if (entry.isFile()) {
|
|
851
|
-
|
|
886
|
+
// relative() rather than slice(rootDir.length + 1): the hand-rolled
|
|
887
|
+
// slice cut one character too many whenever rootDir carried a trailing
|
|
888
|
+
// separator, and mangled every path outright when rootDir was "/".
|
|
889
|
+
const rel = relative(rootDir, join(dir, entry.name)).split("\\").join("/");
|
|
852
890
|
out.push(rel);
|
|
853
891
|
}
|
|
854
892
|
}
|
|
@@ -1126,7 +1164,7 @@ async function sumFileSizes(targetDir, files) {
|
|
|
1126
1164
|
return total;
|
|
1127
1165
|
}
|
|
1128
1166
|
// ---------- request building ----------
|
|
1129
|
-
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride) {
|
|
1167
|
+
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride, reasoningOverride) {
|
|
1130
1168
|
const moduleTag = sanitizeId(slice.moduleName);
|
|
1131
1169
|
const customId = sliceCount > 1 ? `${lens.id}-${moduleTag}-${index + 1}` : `${lens.id}-${moduleTag}`;
|
|
1132
1170
|
return {
|
|
@@ -1139,12 +1177,36 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1139
1177
|
],
|
|
1140
1178
|
response_format: { type: "json_schema", json_schema: SCHEMAS[lens.schemaName] },
|
|
1141
1179
|
max_tokens: maxTokensOverride ?? lens.maxTokens,
|
|
1180
|
+
// Always sent, never inherited: an absent field means the model's
|
|
1181
|
+
// own default, and that default is what truncated the JSON.
|
|
1182
|
+
reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(maxTokensOverride ?? lens.maxTokens),
|
|
1142
1183
|
},
|
|
1143
1184
|
};
|
|
1144
1185
|
}
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1186
|
+
/**
|
|
1187
|
+
* Pre-flight cost estimate for one lens.
|
|
1188
|
+
*
|
|
1189
|
+
* Every slice is its own batch request, so both halves scale with the slice
|
|
1190
|
+
* count. The output half used to be a single `maxTokens * 0.75` for the whole
|
|
1191
|
+
* lens no matter how many requests it sent — on a repository that sliced into
|
|
1192
|
+
* 13 modules that budgeted one request's output and shipped thirteen, and a
|
|
1193
|
+
* live run came in at roughly 3x its estimate. Since this number is what
|
|
1194
|
+
* `max_cost` binds against, under-counting it lets a run outspend the cap the
|
|
1195
|
+
* user set.
|
|
1196
|
+
*
|
|
1197
|
+
* @param info - Repo info, when the caller has it: lets the estimate include
|
|
1198
|
+
* the system prompt and JSON schema each request carries. Omitted, the
|
|
1199
|
+
* estimate covers slice content only, which is what the old signature did.
|
|
1200
|
+
*/
|
|
1201
|
+
export function estimateCost(lens, slices, pricing, maxTokensOverride, info) {
|
|
1202
|
+
const sliceChars = slices.reduce((sum, s) => sum + (lens.maxChars === 0 ? 6000 : s.chars), 0);
|
|
1203
|
+
// The system prompt and the response schema ride on every request, so they
|
|
1204
|
+
// are paid once per slice rather than once per lens.
|
|
1205
|
+
const perRequestOverhead = info
|
|
1206
|
+
? (lens.systemPrompt(info)?.length ?? 0) + JSON.stringify(SCHEMAS[lens.schemaName] ?? {}).length
|
|
1207
|
+
: 0;
|
|
1208
|
+
const inputTokens = Math.ceil((sliceChars + perRequestOverhead * slices.length) / 4);
|
|
1209
|
+
const outputTokens = slices.length * Math.ceil((maxTokensOverride ?? lens.maxTokens) * 0.75);
|
|
1148
1210
|
const cost = (inputTokens / 1_000_000) * pricing.inputPerM +
|
|
1149
1211
|
(outputTokens / 1_000_000) * pricing.outputPerM;
|
|
1150
1212
|
return { inputTokens, outputTokens, cost };
|
|
@@ -1195,9 +1257,94 @@ export async function loadBroadsideState(broadsideDir) {
|
|
|
1195
1257
|
return defaultBroadsideState();
|
|
1196
1258
|
}
|
|
1197
1259
|
}
|
|
1260
|
+
/**
|
|
1261
|
+
* Overwrite `state.json` wholesale with `state`.
|
|
1262
|
+
*
|
|
1263
|
+
* Prefer {@link persistBroadsideRun} anywhere a live operation is recording its
|
|
1264
|
+
* own progress — this entry point replaces the file, so any run a concurrent
|
|
1265
|
+
* process recorded in the meantime is erased. It remains the right call for
|
|
1266
|
+
* seeding a fresh workspace and for test fixtures, where "make the file exactly
|
|
1267
|
+
* this" is the intent.
|
|
1268
|
+
*/
|
|
1198
1269
|
export async function saveBroadsideState(broadsideDir, state) {
|
|
1199
1270
|
await mkdir(broadsideDir, { recursive: true });
|
|
1200
|
-
|
|
1271
|
+
const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
|
|
1272
|
+
const lock = await acquireLock(`${statePath}.lock`);
|
|
1273
|
+
try {
|
|
1274
|
+
await writeBroadsideStateFile(statePath, state);
|
|
1275
|
+
}
|
|
1276
|
+
finally {
|
|
1277
|
+
await lock.release();
|
|
1278
|
+
}
|
|
1279
|
+
}
|
|
1280
|
+
/** Serialize through a temp file so a crash mid-write cannot truncate state.json. */
|
|
1281
|
+
async function writeBroadsideStateFile(statePath, state) {
|
|
1282
|
+
const tempPath = `${statePath}.${process.pid}.${Date.now()}.tmp`;
|
|
1283
|
+
await writeFile(tempPath, `${JSON.stringify(state, null, "\t")}\n`, "utf8");
|
|
1284
|
+
await rename(tempPath, statePath);
|
|
1285
|
+
}
|
|
1286
|
+
/**
|
|
1287
|
+
* Read-modify-write `state.json` under a lock.
|
|
1288
|
+
*
|
|
1289
|
+
* The lock is held only for the read-modify-write, never for the surrounding
|
|
1290
|
+
* operation: a `collect` can poll for the better part of an hour, and holding
|
|
1291
|
+
* the lock across that would push every concurrent caller past the 5s lock
|
|
1292
|
+
* timeout.
|
|
1293
|
+
*/
|
|
1294
|
+
export async function updateBroadsideStateAtomically(broadsideDir, mutate) {
|
|
1295
|
+
await mkdir(broadsideDir, { recursive: true });
|
|
1296
|
+
const statePath = join(broadsideDir, BROADSIDE_STATE_FILE);
|
|
1297
|
+
const lock = await acquireLock(`${statePath}.lock`);
|
|
1298
|
+
try {
|
|
1299
|
+
const state = await loadBroadsideState(broadsideDir);
|
|
1300
|
+
await mutate(state);
|
|
1301
|
+
await writeBroadsideStateFile(statePath, state);
|
|
1302
|
+
return state;
|
|
1303
|
+
}
|
|
1304
|
+
finally {
|
|
1305
|
+
await lock.release();
|
|
1306
|
+
}
|
|
1307
|
+
}
|
|
1308
|
+
/**
|
|
1309
|
+
* Record one run's current shape, merged into whatever is on disk *now*.
|
|
1310
|
+
*
|
|
1311
|
+
* Broad-Side operations are long-lived and hold their state in memory while
|
|
1312
|
+
* they poll. Writing that snapshot back wholesale silently erased any run a
|
|
1313
|
+
* concurrent operation had recorded since it was loaded, orphaning that run's
|
|
1314
|
+
* paid results on disk — present as files, invisible to `list`, and unreachable
|
|
1315
|
+
* by `collect`, which finds its run by position in `state.runs`. Observed live:
|
|
1316
|
+
* a submit at 23:35 was erased by a collect that had loaded state before it and
|
|
1317
|
+
* wrote back at 00:08.
|
|
1318
|
+
*
|
|
1319
|
+
* Merging by run id also self-heals: a run erased by an older writer is
|
|
1320
|
+
* restored the next time its own operation checkpoints.
|
|
1321
|
+
*/
|
|
1322
|
+
export async function persistBroadsideRun(broadsideDir, run) {
|
|
1323
|
+
return updateBroadsideStateAtomically(broadsideDir, (state) => {
|
|
1324
|
+
const index = state.runs.findIndex((candidate) => candidate.id === run.id);
|
|
1325
|
+
if (index === -1)
|
|
1326
|
+
state.runs.push(run);
|
|
1327
|
+
else
|
|
1328
|
+
state.runs[index] = run;
|
|
1329
|
+
});
|
|
1330
|
+
}
|
|
1331
|
+
/** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
|
|
1332
|
+
function parseReasoningConfig(raw) {
|
|
1333
|
+
if (raw === false)
|
|
1334
|
+
return { enabled: false };
|
|
1335
|
+
if (raw === true)
|
|
1336
|
+
return { enabled: true };
|
|
1337
|
+
if (!raw || typeof raw !== "object")
|
|
1338
|
+
return null;
|
|
1339
|
+
const value = raw;
|
|
1340
|
+
const out = {};
|
|
1341
|
+
if (typeof value.enabled === "boolean")
|
|
1342
|
+
out.enabled = value.enabled;
|
|
1343
|
+
if (value.effort === "minimal" || value.effort === "low" || value.effort === "medium" || value.effort === "high")
|
|
1344
|
+
out.effort = value.effort;
|
|
1345
|
+
if (typeof value.max_tokens === "number" && value.max_tokens > 0)
|
|
1346
|
+
out.max_tokens = value.max_tokens;
|
|
1347
|
+
return Object.keys(out).length > 0 ? out : null;
|
|
1201
1348
|
}
|
|
1202
1349
|
export async function loadBroadsideConfig(broadsideDir) {
|
|
1203
1350
|
const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
|
|
@@ -1239,6 +1386,10 @@ export async function loadBroadsideConfig(broadsideDir) {
|
|
|
1239
1386
|
? { inputPerM: inputOverride, outputPerM: outputOverride }
|
|
1240
1387
|
: null,
|
|
1241
1388
|
lensModels,
|
|
1389
|
+
// An escape hatch, not a knob to reach for: a model whose reasoning is
|
|
1390
|
+
// worth paying for needs its lens maxTokens raised to cover both the
|
|
1391
|
+
// thinking and the answer, or the JSON truncates exactly as before.
|
|
1392
|
+
reasoning: parseReasoningConfig(raw.reasoning),
|
|
1242
1393
|
incremental: flag("incremental", false),
|
|
1243
1394
|
retryTruncated: flag("retry_truncated", true),
|
|
1244
1395
|
includeSynthesis: flag("include_synthesis", true),
|
|
@@ -1332,10 +1483,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
|
|
|
1332
1483
|
},
|
|
1333
1484
|
};
|
|
1334
1485
|
}
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
//
|
|
1486
|
+
// On-disk cache first, then the live catalog — for the default model too.
|
|
1487
|
+
// Hardcoded rates used to short-circuit here, which meant a stale constant
|
|
1488
|
+
// could never self-correct even though the catalog was already being
|
|
1489
|
+
// fetched for every other model. The authoritative source wins; the
|
|
1490
|
+
// constants below are what we fall back to when the network is unavailable.
|
|
1339
1491
|
const cache = await readCatalogCache(broadsideDir);
|
|
1340
1492
|
const cached = cache?.models[model];
|
|
1341
1493
|
if (cached && Date.now() - new Date(cache.fetched_at).getTime() < BROADSIDE_CATALOG_CACHE_TTL_MS) {
|
|
@@ -1366,6 +1518,11 @@ export async function resolveCatalogEntry(broadsideDir, config, model, apiKey, f
|
|
|
1366
1518
|
await writeCatalogCache(broadsideDir, updated);
|
|
1367
1519
|
return { model, source: "live", entry: live };
|
|
1368
1520
|
}
|
|
1521
|
+
// Offline fallback: the default model's rates and capabilities are known at
|
|
1522
|
+
// compile time, so a network failure does not have to stop a run.
|
|
1523
|
+
const builtIn = builtInCatalogEntry(model);
|
|
1524
|
+
if (builtIn)
|
|
1525
|
+
return { model, source: "built-in", entry: builtIn };
|
|
1369
1526
|
throw new Error(`Could not resolve per-token pricing for batch model "${model}". ` +
|
|
1370
1527
|
"Set pricing.input_per_m and pricing.output_per_m in .codecarto/broadside/config.yaml " +
|
|
1371
1528
|
"(USD per million tokens), or check the model id against https://openrouter.ai/models?variant=batch.");
|
|
@@ -1469,6 +1626,14 @@ export async function fetchBatch(batchId, apiKey, fetcher = fetch) {
|
|
|
1469
1626
|
data.http_status = resp.status;
|
|
1470
1627
|
return data;
|
|
1471
1628
|
}
|
|
1629
|
+
/**
|
|
1630
|
+
* Batch statuses that will never produce a result.
|
|
1631
|
+
*
|
|
1632
|
+
* Deliberately excludes the synthetic `timeout` this module returns when a poll
|
|
1633
|
+
* budget expires: that batch is still running server-side and has already been
|
|
1634
|
+
* charged, so callers must come back for it rather than retire it.
|
|
1635
|
+
*/
|
|
1636
|
+
export const BROADSIDE_DEAD_BATCH_STATUSES = ["failed", "expired", "cancelled", "auth-failed"];
|
|
1472
1637
|
export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
|
|
1473
1638
|
const deadline = Date.now() + (opts.deadlineMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
|
|
1474
1639
|
const intervalMs = opts.pollIntervalMs ?? BROADSIDE_POLL_INTERVAL_MS;
|
|
@@ -1491,7 +1656,7 @@ export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
|
|
|
1491
1656
|
const status = String(batch.status ?? "unknown");
|
|
1492
1657
|
const counts = (batch.request_counts ?? {});
|
|
1493
1658
|
opts.onStatus?.(status, counts);
|
|
1494
|
-
if (
|
|
1659
|
+
if (status === "completed" || BROADSIDE_DEAD_BATCH_STATUSES.includes(status))
|
|
1495
1660
|
return batch;
|
|
1496
1661
|
if (Date.now() >= deadline)
|
|
1497
1662
|
return { id: batchId, status: "timeout" };
|
|
@@ -1563,14 +1728,30 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1563
1728
|
const sourceDirty = await gitDirty(cwd);
|
|
1564
1729
|
let baseHead = null;
|
|
1565
1730
|
let changed = null;
|
|
1731
|
+
const incrementalOutcome = {
|
|
1732
|
+
requested: opts.incremental === true,
|
|
1733
|
+
applied: false,
|
|
1734
|
+
baseHead: null,
|
|
1735
|
+
};
|
|
1566
1736
|
if (opts.incremental) {
|
|
1567
1737
|
const state = await loadBroadsideState(broadsideDir);
|
|
1568
1738
|
// The baseline is the most recent run that recorded a HEAD — a
|
|
1569
1739
|
// submit-only run (never collected) is still a valid committed base.
|
|
1570
1740
|
const previous = [...state.runs].reverse().find((r) => r.sourceHead);
|
|
1571
|
-
if (
|
|
1741
|
+
if (sourceDirty) {
|
|
1742
|
+
incrementalOutcome.reason = "dirty-worktree";
|
|
1743
|
+
}
|
|
1744
|
+
else if (!previous?.sourceHead) {
|
|
1745
|
+
incrementalOutcome.reason = "no-baseline";
|
|
1746
|
+
}
|
|
1747
|
+
else {
|
|
1572
1748
|
baseHead = previous.sourceHead;
|
|
1573
1749
|
changed = await changedFilesSince(cwd, baseHead);
|
|
1750
|
+
incrementalOutcome.baseHead = baseHead;
|
|
1751
|
+
if (changed)
|
|
1752
|
+
incrementalOutcome.applied = true;
|
|
1753
|
+
else
|
|
1754
|
+
incrementalOutcome.reason = "diff-failed";
|
|
1574
1755
|
}
|
|
1575
1756
|
}
|
|
1576
1757
|
// Slice offline first so the estimate covers every request we would send.
|
|
@@ -1591,7 +1772,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1591
1772
|
const lensModel = modelForLens(lensId);
|
|
1592
1773
|
const { pricing: lensPricing, outputCap: lensOutputCap } = resolved.get(lensModel);
|
|
1593
1774
|
const maxTokens = lensOutputCap ? Math.min(lens.maxTokens, lensOutputCap) : lens.maxTokens;
|
|
1594
|
-
const estimate = estimateCost(lens, slices, lensPricing, maxTokens);
|
|
1775
|
+
const estimate = estimateCost(lens, slices, lensPricing, maxTokens, info);
|
|
1595
1776
|
estimatedInputTokens += estimate.inputTokens;
|
|
1596
1777
|
estimatedOutputTokens += estimate.outputTokens;
|
|
1597
1778
|
estimatedTotalCost += estimate.cost;
|
|
@@ -1626,6 +1807,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1626
1807
|
exceedsLimit,
|
|
1627
1808
|
baseHead,
|
|
1628
1809
|
sourceDirty,
|
|
1810
|
+
incremental: incrementalOutcome,
|
|
1629
1811
|
...(outputCap !== undefined && { outputCap }),
|
|
1630
1812
|
});
|
|
1631
1813
|
if (!approved)
|
|
@@ -1659,7 +1841,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1659
1841
|
baseHead,
|
|
1660
1842
|
};
|
|
1661
1843
|
state.runs.push(run);
|
|
1662
|
-
await
|
|
1844
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
1663
1845
|
const requestsByCustomId = {};
|
|
1664
1846
|
const submissions = [];
|
|
1665
1847
|
// Submit from the estimate rather than recomputing: the user approved that
|
|
@@ -1668,7 +1850,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1668
1850
|
const { lens, maxTokens, lensModel, lensOutputCap } = priced;
|
|
1669
1851
|
const lensId = lens.id;
|
|
1670
1852
|
const slices = slicesByLens.get(lensId) ?? [];
|
|
1671
|
-
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens));
|
|
1853
|
+
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens, config.reasoning ?? undefined));
|
|
1672
1854
|
for (const request of requests)
|
|
1673
1855
|
requestsByCustomId[request.custom_id] = request;
|
|
1674
1856
|
const entry = {
|
|
@@ -1707,7 +1889,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1707
1889
|
})());
|
|
1708
1890
|
}
|
|
1709
1891
|
await Promise.allSettled(submissions);
|
|
1710
|
-
await
|
|
1892
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
1711
1893
|
// Persist the exact request bodies so collect can re-submit a truncated
|
|
1712
1894
|
// slice (bumped output cap) without re-walking the repo (#133). The run
|
|
1713
1895
|
// dir is created here rather than waiting for collect so a crash between
|
|
@@ -1734,6 +1916,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1734
1916
|
: resolved.get(model).supportsStructuredOutputs,
|
|
1735
1917
|
expirationDate: defaultEntry.expirationDate ?? null,
|
|
1736
1918
|
},
|
|
1919
|
+
incremental: incrementalOutcome,
|
|
1737
1920
|
};
|
|
1738
1921
|
}
|
|
1739
1922
|
function extractContent(result) {
|
|
@@ -1799,6 +1982,46 @@ export async function saveLensResults(runDir, lensId, batch) {
|
|
|
1799
1982
|
}
|
|
1800
1983
|
return out;
|
|
1801
1984
|
}
|
|
1985
|
+
/**
|
|
1986
|
+
* Rebuild lens results from what a previous collect already wrote to disk.
|
|
1987
|
+
*
|
|
1988
|
+
* The post-passes are gated on having lens findings in hand, and a collect
|
|
1989
|
+
* only holds the ones *it* polled. When an earlier collect saved every lens
|
|
1990
|
+
* and then died before synthesis and triage ran — the batch window is long
|
|
1991
|
+
* and a poll can easily be interrupted — the next collect finds every lens
|
|
1992
|
+
* already terminal, skips them all, and would otherwise reach the post-pass
|
|
1993
|
+
* gate with nothing to hand it. Reading the saved results back is what makes
|
|
1994
|
+
* "a resumed collect can finish whichever is still pending" true.
|
|
1995
|
+
*/
|
|
1996
|
+
export async function loadSavedLensResults(runDir, lenses) {
|
|
1997
|
+
if (!(await pathExists(runDir)))
|
|
1998
|
+
return [];
|
|
1999
|
+
const reserved = new Set(["requests.json", "run-meta.json", "synthesis.json", "triage.json"]);
|
|
2000
|
+
const out = [];
|
|
2001
|
+
// Longest lens id first: no id is a prefix of another today, but ordering
|
|
2002
|
+
// keeps that from becoming a silent misattribution if one ever is.
|
|
2003
|
+
const ordered = [...lenses].sort((a, b) => b.length - a.length);
|
|
2004
|
+
for (const name of (await readdir(runDir)).sort()) {
|
|
2005
|
+
if (!name.endsWith(".json") || name.endsWith(".error.json") || reserved.has(name) || name.startsWith("raw-"))
|
|
2006
|
+
continue;
|
|
2007
|
+
const customId = name.slice(0, -".json".length);
|
|
2008
|
+
const lensId = ordered.find((id) => customId === id || customId.startsWith(`${id}-`));
|
|
2009
|
+
if (!lensId)
|
|
2010
|
+
continue;
|
|
2011
|
+
const content = await readFile(join(runDir, name), "utf8").catch(() => null);
|
|
2012
|
+
if (content === null)
|
|
2013
|
+
continue;
|
|
2014
|
+
out.push({
|
|
2015
|
+
lensId,
|
|
2016
|
+
customId,
|
|
2017
|
+
moduleName: customId.replace(/^[a-z]+-/, ""),
|
|
2018
|
+
content,
|
|
2019
|
+
raw: {},
|
|
2020
|
+
truncated: parseLensJson(content) === null,
|
|
2021
|
+
});
|
|
2022
|
+
}
|
|
2023
|
+
return out;
|
|
2024
|
+
}
|
|
1802
2025
|
async function loadStoredRequests(runDir) {
|
|
1803
2026
|
const path = join(runDir, "requests.json");
|
|
1804
2027
|
if (!(await pathExists(path)))
|
|
@@ -1956,11 +2179,19 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
1956
2179
|
await writeFile(join(runDir, `raw-${lensId}.json`), `${JSON.stringify(batch, null, "\t")}\n`, "utf8");
|
|
1957
2180
|
lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, truncated };
|
|
1958
2181
|
}
|
|
1959
|
-
else
|
|
1960
|
-
|
|
2182
|
+
else {
|
|
2183
|
+
// Every non-completed outcome still has to reach the report.
|
|
2184
|
+
// `lensOutcomes` is what the caller renders, and this branch used to
|
|
2185
|
+
// require `batch.error` — but the commonest failure here is the
|
|
2186
|
+
// synthetic `{ status: "timeout" }` the poll returns when its budget
|
|
2187
|
+
// expires with the batch still in flight, and that carries no error.
|
|
2188
|
+
// A lens that never came back was therefore omitted entirely,
|
|
2189
|
+
// indistinguishable in the output from one that was never requested.
|
|
2190
|
+
if (batch.error)
|
|
2191
|
+
entry.error = batch.error;
|
|
1961
2192
|
lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount };
|
|
1962
2193
|
}
|
|
1963
|
-
await
|
|
2194
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
1964
2195
|
}
|
|
1965
2196
|
// #133: re-submit truncated slices once with a bumped output cap. Batch
|
|
1966
2197
|
// requests are pure, so re-running is always safe; the aim is to recover
|
|
@@ -1993,7 +2224,11 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
1993
2224
|
if (error)
|
|
1994
2225
|
continue;
|
|
1995
2226
|
const batch = await pollBatchUntilTerminal(batchId, apiKey, {
|
|
1996
|
-
|
|
2227
|
+
// Share the caller's deadline. Each of these polls used to
|
|
2228
|
+
// start a fresh 25-minute budget, so `wait_seconds` bounded
|
|
2229
|
+
// only the lens poll and a collect could run for the caller's
|
|
2230
|
+
// budget plus fifty minutes.
|
|
2231
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
1997
2232
|
onStatus: (status, counts) => opts.onStatus?.(`${stored.lensId}:retry`, status, counts),
|
|
1998
2233
|
fetcher: opts.fetcher,
|
|
1999
2234
|
});
|
|
@@ -2023,7 +2258,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2023
2258
|
outcome.truncated = allLensResults.filter((s) => s.lensId === lensId && s.truncated).length;
|
|
2024
2259
|
}
|
|
2025
2260
|
}
|
|
2026
|
-
await
|
|
2261
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
2027
2262
|
}
|
|
2028
2263
|
// Synthesis + triage: cross-lens post-passes, only after every lens batch
|
|
2029
2264
|
// is terminal. Triage turns the leads into a prioritized work order.
|
|
@@ -2032,12 +2267,25 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2032
2267
|
let topTriageItems = [];
|
|
2033
2268
|
const wantSynthesis = opts.includeSynthesis !== false;
|
|
2034
2269
|
const wantTriage = opts.includeTriage !== false;
|
|
2270
|
+
// A resumed collect polls nothing — every lens is already terminal — so the
|
|
2271
|
+
// findings the post-passes need have to come back off disk, or a run whose
|
|
2272
|
+
// first collect was interrupted could never produce its executive report
|
|
2273
|
+
// and work order, however many times it was re-run.
|
|
2274
|
+
const postPassUnfinished = (entry) => entry.status === "pending" || entry.status === "submitted";
|
|
2275
|
+
if ((wantSynthesis || wantTriage) && allLensResults.length === 0
|
|
2276
|
+
&& (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
|
|
2277
|
+
const restored = await loadSavedLensResults(runDir, run.lenses);
|
|
2278
|
+
if (restored.length > 0) {
|
|
2279
|
+
allLensResults.push(...restored);
|
|
2280
|
+
truncatedCount = restored.filter((s) => s.truncated).length;
|
|
2281
|
+
}
|
|
2282
|
+
}
|
|
2035
2283
|
if ((wantSynthesis || wantTriage) && allLensResults.length > 0) {
|
|
2036
2284
|
const allTerminal = run.lenses.every((lensId) => {
|
|
2037
2285
|
const entry = run.batches[lensId];
|
|
2038
2286
|
return entry && ["completed", "failed", "expired", "cancelled", "auth-failed", "skipped", "rejected"].includes(entry.status);
|
|
2039
2287
|
});
|
|
2040
|
-
if (allTerminal && (run.synthesis
|
|
2288
|
+
if (allTerminal && (postPassUnfinished(run.synthesis) || postPassUnfinished(run.triage))) {
|
|
2041
2289
|
const findingsText = allLensResults
|
|
2042
2290
|
.map((r) => `## ${r.lensId} — ${r.customId}\n\n${r.content}\n`)
|
|
2043
2291
|
.join("\n");
|
|
@@ -2066,6 +2314,24 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2066
2314
|
: []),
|
|
2067
2315
|
];
|
|
2068
2316
|
const submitted = new Map();
|
|
2317
|
+
// A pass can be left at "submitted" when an earlier collect returned
|
|
2318
|
+
// before its batch reached a terminal status — the batch still runs
|
|
2319
|
+
// and is still charged, so the result exists and is simply unclaimed.
|
|
2320
|
+
// Nothing above would ever look at it again: the pass list is built
|
|
2321
|
+
// from "pending" entries only. Poll those regardless of the want
|
|
2322
|
+
// flags, because the spend already happened and discarding a
|
|
2323
|
+
// finished result is worse than saving one the caller opted out of.
|
|
2324
|
+
for (const kind of ["synthesis", "triage"]) {
|
|
2325
|
+
const entry = kind === "synthesis" ? run.synthesis : run.triage;
|
|
2326
|
+
if (entry.status !== "submitted" || !entry.batchId)
|
|
2327
|
+
continue;
|
|
2328
|
+
if (submitted.has(entry.batchId))
|
|
2329
|
+
continue;
|
|
2330
|
+
submitted.set(entry.batchId, {
|
|
2331
|
+
batchId: entry.batchId,
|
|
2332
|
+
pass: { kind, request: undefined, entry },
|
|
2333
|
+
});
|
|
2334
|
+
}
|
|
2069
2335
|
await Promise.allSettled(passes.map(async (pass) => {
|
|
2070
2336
|
pass.entry.status = "submitted";
|
|
2071
2337
|
try {
|
|
@@ -2081,10 +2347,13 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2081
2347
|
pass.entry.status = "failed";
|
|
2082
2348
|
}
|
|
2083
2349
|
}));
|
|
2084
|
-
await
|
|
2350
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
2085
2351
|
for (const { batchId, pass } of submitted.values()) {
|
|
2086
2352
|
const batch = await pollBatchUntilTerminal(batchId, apiKey, {
|
|
2087
|
-
|
|
2353
|
+
// Shares the caller's deadline, as the retry poll above does.
|
|
2354
|
+
// A pass whose poll runs out stays `submitted`, so the batch
|
|
2355
|
+
// is already paid for and a later collect claims its result.
|
|
2356
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
2088
2357
|
onStatus: (status, counts) => opts.onStatus?.(pass.kind, status, counts),
|
|
2089
2358
|
fetcher: opts.fetcher,
|
|
2090
2359
|
});
|
|
@@ -2107,10 +2376,20 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2107
2376
|
}
|
|
2108
2377
|
}
|
|
2109
2378
|
}
|
|
2110
|
-
else if (batch.
|
|
2379
|
+
else if (BROADSIDE_DEAD_BATCH_STATUSES.includes(String(batch.status))) {
|
|
2380
|
+
// The batch will never produce a result, so retire the pass.
|
|
2381
|
+
// This used to require `batch.error`, leaving an expired or
|
|
2382
|
+
// cancelled batch parked at "submitted" forever — and since a
|
|
2383
|
+
// resumed collect re-polls anything still "submitted", it
|
|
2384
|
+
// would re-poll a dead batch on every future run.
|
|
2111
2385
|
pass.entry.status = "failed";
|
|
2386
|
+
if (batch.error)
|
|
2387
|
+
pass.entry.error = batch.error instanceof Error ? batch.error.message : String(batch.error);
|
|
2112
2388
|
}
|
|
2113
|
-
|
|
2389
|
+
// A "timeout" is deliberately left at "submitted": the batch is
|
|
2390
|
+
// still running server-side and has already been paid for, so a
|
|
2391
|
+
// later collect should claim its result rather than discard it.
|
|
2392
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
2114
2393
|
}
|
|
2115
2394
|
}
|
|
2116
2395
|
}
|
|
@@ -2120,7 +2399,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2120
2399
|
});
|
|
2121
2400
|
run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
|
|
2122
2401
|
run.totalCost = totalCost;
|
|
2123
|
-
await
|
|
2402
|
+
await persistBroadsideRun(broadsideDir, run);
|
|
2124
2403
|
await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
|
|
2125
2404
|
experimental: true,
|
|
2126
2405
|
method: "Broad-Side (OpenRouter Batch API)",
|
|
@@ -2223,9 +2502,31 @@ function parseSynthesisTopFindings(content) {
|
|
|
2223
2502
|
}
|
|
2224
2503
|
}
|
|
2225
2504
|
// ---------- formatting helpers for tool output ----------
|
|
2505
|
+
function describeIncrementalFallback(reason) {
|
|
2506
|
+
switch (reason) {
|
|
2507
|
+
case "dirty-worktree":
|
|
2508
|
+
return "the working tree has uncommitted changes, so there is no committed state to diff against";
|
|
2509
|
+
case "no-baseline":
|
|
2510
|
+
return "no earlier run recorded a commit to diff against";
|
|
2511
|
+
case "diff-failed":
|
|
2512
|
+
return "the diff against the previous run's commit could not be read";
|
|
2513
|
+
default:
|
|
2514
|
+
return "no baseline was available";
|
|
2515
|
+
}
|
|
2516
|
+
}
|
|
2226
2517
|
export function estimateSubmitText(result, lenses) {
|
|
2518
|
+
// Count the lenses that actually got a batch, not every lens considered. A
|
|
2519
|
+
// lens with nothing to scan is reported as `skipped (0 request(s))` two
|
|
2520
|
+
// lines below, so counting it here made the header contradict its own body:
|
|
2521
|
+
// a Rust CLI with no server surface reported "submitted 6 batch(es)" over a
|
|
2522
|
+
// list showing four batches and two skips.
|
|
2523
|
+
const entries = Object.values(result.batches ?? {});
|
|
2524
|
+
const submittedCount = entries.filter((entry) => entry.batchId).length;
|
|
2525
|
+
const withoutBatch = entries.length - submittedCount;
|
|
2227
2526
|
const lines = [
|
|
2228
|
-
|
|
2527
|
+
withoutBatch > 0
|
|
2528
|
+
? `Broad-Side submitted ${submittedCount} batch(es); ${withoutBatch} lens(es) produced none (see below).`
|
|
2529
|
+
: `Broad-Side submitted ${submittedCount} batch(es).`,
|
|
2229
2530
|
];
|
|
2230
2531
|
for (const lens of lenses) {
|
|
2231
2532
|
const entry = result.batches[lens.id];
|
|
@@ -2235,6 +2536,12 @@ export function estimateSubmitText(result, lenses) {
|
|
|
2235
2536
|
const override = entry.model ? ` on ${entry.model}` : "";
|
|
2236
2537
|
lines.push(` ${lens.name}: ${status} (${entry.requests} request(s), ~$${entry.estimatedCost.toFixed(4)})${override}`);
|
|
2237
2538
|
}
|
|
2539
|
+
const incremental = result.incremental;
|
|
2540
|
+
if (incremental?.requested) {
|
|
2541
|
+
lines.push(incremental.applied
|
|
2542
|
+
? `Incremental: scanning only what changed since ${(incremental.baseHead ?? "").slice(0, 8)}.`
|
|
2543
|
+
: `Incremental: requested but NOT applied — ${describeIncrementalFallback(incremental.reason)}. Every module was scanned, at full cost.`);
|
|
2544
|
+
}
|
|
2238
2545
|
lines.push(`Estimated total: ~$${result.estimatedTotalCost.toFixed(4)}`, `Pricing: $${result.pricing.inputPerM.toFixed(4)}/M in, $${result.pricing.outputPerM.toFixed(4)}/M out (${result.pricing.source})`);
|
|
2239
2546
|
if (result.modelInfo.contextLength) {
|
|
2240
2547
|
lines.push(`Model: ${result.modelInfo.contextLength.toLocaleString()} context, ${result.modelInfo.maxCompletionTokens?.toLocaleString() ?? "?"} max output`);
|
package/dist/core/dashboard.d.ts
CHANGED
|
@@ -39,4 +39,17 @@ export interface DashboardInputs {
|
|
|
39
39
|
narration?: DashboardNarration;
|
|
40
40
|
}
|
|
41
41
|
export declare function renderDashboard(inputs: DashboardInputs): string;
|
|
42
|
+
/**
|
|
43
|
+
* A dashboard link target, or undefined when the path is not a safe relative
|
|
44
|
+
* reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
|
|
45
|
+
* refused outright.
|
|
46
|
+
*
|
|
47
|
+
* Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
|
|
48
|
+
* `dir\..\secret` survives the segment scan, because that scan splits on `/`
|
|
49
|
+
* — but `encodeURIComponent` turns the separators into `%5C`, which no URL
|
|
50
|
+
* parser treats as a path separator, so the traversal cannot resolve. Removing
|
|
51
|
+
* the encoding, or "simplifying" it to a raw join, reopens it. Exported so
|
|
52
|
+
* that property has a test.
|
|
53
|
+
*/
|
|
54
|
+
export declare function safeRelativeHref(path: string): string | undefined;
|
|
42
55
|
export declare function escapeHtml(input: string): string;
|
package/dist/core/dashboard.js
CHANGED
|
@@ -660,7 +660,19 @@ function renderSafeLink(path, label, className) {
|
|
|
660
660
|
const classAttr = className ? ` class="${escapeAttr(className)}"` : "";
|
|
661
661
|
return `<a${classAttr} href="${escapeAttr(href)}">${escapeHtml(label)}</a>`;
|
|
662
662
|
}
|
|
663
|
-
|
|
663
|
+
/**
|
|
664
|
+
* A dashboard link target, or undefined when the path is not a safe relative
|
|
665
|
+
* reference. Absolute paths, scheme-bearing URLs and `.`/`..` segments are
|
|
666
|
+
* refused outright.
|
|
667
|
+
*
|
|
668
|
+
* Percent-encoding each segment is load-bearing, not cosmetic: a Windows-style
|
|
669
|
+
* `dir\..\secret` survives the segment scan, because that scan splits on `/`
|
|
670
|
+
* — but `encodeURIComponent` turns the separators into `%5C`, which no URL
|
|
671
|
+
* parser treats as a path separator, so the traversal cannot resolve. Removing
|
|
672
|
+
* the encoding, or "simplifying" it to a raw join, reopens it. Exported so
|
|
673
|
+
* that property has a test.
|
|
674
|
+
*/
|
|
675
|
+
export function safeRelativeHref(path) {
|
|
664
676
|
if (!path || path.startsWith("/") || path.startsWith("\\") || /^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(path))
|
|
665
677
|
return undefined;
|
|
666
678
|
const parts = path.split("/");
|