codecartographer-pi 0.24.1 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +20 -1
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +5 -4
- package/agent-skill/codecartographer/references/broadside.md +5 -1
- package/dist/core/broadside/client.d.ts +56 -0
- package/dist/core/broadside/client.js +200 -0
- package/dist/core/broadside/collect.d.ts +68 -0
- package/dist/core/broadside/collect.js +676 -0
- package/dist/core/broadside/constants.d.ts +51 -0
- package/dist/core/broadside/constants.js +74 -0
- package/dist/core/broadside/lenses.d.ts +31 -0
- package/dist/core/broadside/lenses.js +312 -0
- package/dist/core/broadside/models.d.ts +46 -0
- package/dist/core/broadside/models.js +321 -0
- package/dist/core/broadside/render.d.ts +20 -0
- package/dist/core/broadside/render.js +285 -0
- package/dist/core/broadside/repo.d.ts +58 -0
- package/dist/core/broadside/repo.js +592 -0
- package/dist/core/broadside/requests.d.ts +23 -0
- package/dist/core/broadside/requests.js +71 -0
- package/dist/core/broadside/results.d.ts +36 -0
- package/dist/core/broadside/results.js +163 -0
- package/dist/core/broadside/schemas.d.ts +2 -0
- package/dist/core/broadside/schemas.js +342 -0
- package/dist/core/broadside/state.d.ts +99 -0
- package/dist/core/broadside/state.js +384 -0
- package/dist/core/broadside/submit.d.ts +30 -0
- package/dist/core/broadside/submit.js +350 -0
- package/dist/core/broadside/types.d.ts +491 -0
- package/dist/core/broadside/types.js +107 -0
- package/dist/core/{broadside-verify.d.ts → broadside/verify.d.ts} +23 -2
- package/dist/core/{broadside-verify.js → broadside/verify.js} +43 -5
- package/dist/core/broadside.d.ts +14 -890
- package/dist/core/broadside.js +25 -3564
- package/dist/core/completion.js +91 -72
- package/dist/core/dashboard-writer.js +9 -1
- package/dist/core/index.d.ts +0 -1
- package/dist/core/index.js +0 -1
- package/dist/core/library.d.ts +24 -1
- package/dist/core/library.js +46 -15
- package/dist/core/orchestrator-config.js +22 -8
- package/dist/core/status.d.ts +42 -23
- package/dist/core/status.js +163 -137
- package/dist/core/workspace.d.ts +2 -0
- package/dist/core/workspace.js +49 -25
- package/dist/core/yaml.js +9 -3
- package/dist/extensions/codecarto/auto-runner.d.ts +7 -0
- package/dist/extensions/codecarto/auto-runner.js +54 -23
- package/dist/extensions/codecarto/broadside-flags.d.ts +3 -1
- package/dist/extensions/codecarto/broadside-flags.js +13 -0
- package/dist/extensions/codecarto/index.js +13 -7
- package/dist/extensions/codecarto/phase-compaction.js +6 -2
- package/dist/mcp-server/server.d.ts +1 -0
- package/dist/mcp-server/server.js +28 -5
- package/package.json +1 -1
|
@@ -35,6 +35,10 @@ not replace any phase; it tells phases where to look.
|
|
|
35
35
|
it describes), **discarded** (the claim is wrong about the code, with the
|
|
36
36
|
guard or line that shows it), or **unclear**. Start from the confirmed
|
|
37
37
|
ones; treat a discarded one as answered unless the reasoning is thin.
|
|
38
|
+
Then make sure the work order was built from it: `status` shows
|
|
39
|
+
`triage: completed (built from N verdicts)`; if it says `(no verdicts)`,
|
|
40
|
+
run `collect --regenerate` (`regenerate_post_passes: true`) before
|
|
41
|
+
reading `triage.md`.
|
|
38
42
|
Measured on this repository, the top twelve findings by severity were two
|
|
39
43
|
real defects and ten that a look at the guard, the caller, or the tsconfig
|
|
40
44
|
dismissed — the pass agreed with a reviewer on all twelve for about a cent
|
|
@@ -85,6 +89,7 @@ Broad-Side is an executable-surface feature. On the Pi extension:
|
|
|
85
89
|
/codecarto-broadside status # show recorded runs
|
|
86
90
|
/codecarto-broadside models # compare batch models
|
|
87
91
|
/codecarto-broadside verify --top=10 # read the top findings against the source
|
|
92
|
+
/codecarto-broadside collect --regenerate # rebuild synthesis and triage from the verdicts
|
|
88
93
|
```
|
|
89
94
|
|
|
90
95
|
On the MCP server:
|
|
@@ -95,6 +100,7 @@ codecarto_broadside {cwd, action: "collect"} # poll, save, synt
|
|
|
95
100
|
codecarto_broadside {cwd, action: "status"} # show recorded runs
|
|
96
101
|
codecarto_broadside {cwd, action: "models"} # compare batch models
|
|
97
102
|
codecarto_broadside {cwd, action: "verify", top: 10} # read the top findings against the source
|
|
103
|
+
codecarto_broadside {cwd, action: "collect", regenerate_post_passes: true} # rebuild synthesis and triage from the verdicts
|
|
98
104
|
```
|
|
99
105
|
|
|
100
106
|
The `models` action lists every `:batch` variant on OpenRouter — pricing per
|
|
@@ -139,6 +145,16 @@ cap here, since a sync call's cost is known only when it returns: the pass
|
|
|
139
145
|
stops before the next finding once the calls so far have reached it and
|
|
140
146
|
reports `partial`. About a cent a finding on the default model.
|
|
141
147
|
|
|
148
|
+
The post-passes read the verdicts when they exist: a synthesis or triage
|
|
149
|
+
built after a `verify` ranks the confirmed findings first, drops the
|
|
150
|
+
discarded ones, lists the not-a-defect ones apart in `omitted`, and says in
|
|
151
|
+
its summary how many verdicts it was built from; `status` and the collect
|
|
152
|
+
report show `(built from N verdicts)` or `(no verdicts)` on each pass. A run
|
|
153
|
+
collected before it was verified has a work order built from batch
|
|
154
|
+
severities alone — `collect --regenerate` (`regenerate_post_passes: true`)
|
|
155
|
+
resets the settled passes and runs them again with the verdicts, for another
|
|
156
|
+
post-pass pair's cost; a pass still in flight is left to finish.
|
|
157
|
+
|
|
142
158
|
Two caveats apply to any model you pick. The `models` action lists every id
|
|
143
159
|
OpenRouter advertises a `:batch` variant for, and many of those variants do not
|
|
144
160
|
exist — submitting one returns `does not have a :batch endpoint`, with nothing in
|
|
@@ -206,7 +222,10 @@ is about the *presence* of a hardcoded credential at that location; the
|
|
|
206
222
|
value was never sent. This is a safety net against an accidental upload
|
|
207
223
|
with deliberately low-false-positive patterns, not a secret scanner:
|
|
208
224
|
anything it does not recognise goes as written. `redact_secrets: false` in
|
|
209
|
-
`config.yaml` turns the content pass off (the by-name skip stays).
|
|
225
|
+
`config.yaml` turns the content pass off (the by-name skip stays). The same
|
|
226
|
+
pass runs over every line `verify`'s read-only tools hand the model — a key in
|
|
227
|
+
an ordinary source file is exactly what a finding points a verifier at — and
|
|
228
|
+
the verify report says how many values it redacted.
|
|
210
229
|
|
|
211
230
|
The `max_cost` guardrail is an **estimate-based pre-flight limit**, distinct
|
|
212
231
|
from OpenRouter's runtime cost tracking: it predicts from file sizes before
|
package/README.md
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
```
|
|
21
21
|
|
|
22
22
|
<p align="center">
|
|
23
|
-
<img src="docs/demo-dashboard-hero.png" alt="CodeCartographer dashboard
|
|
23
|
+
<img src="docs/demo-dashboard-hero.png" alt="CodeCartographer's dashboard after running the full deep-audit pipeline on its own repository — 7/7 phases complete, 50M tokens and 618 tool uses across 77 minutes, seven open questions and twelve post-pipeline items routed, no blocking artifact gaps.">
|
|
24
24
|
</p>
|
|
25
25
|
|
|
26
26
|
---
|
|
@@ -305,7 +305,7 @@ Beyond the slash commands, the Pi extension layers on:
|
|
|
305
305
|
|
|
306
306
|
**Per-phase usage tracking.** Each phase run is appended to `.codecarto/workflow/.usage.local.yaml`. `/codecarto-usage` reports cumulative + per-phase token, runtime, tool-use, and compaction totals, including threshold/overflow/manual triggers and successful/failed/aborted outcomes.
|
|
307
307
|
|
|
308
|
-
**Tool interception.** `bash` is blocked outright; `edit` and `write` are confined to `.codecarto/`, plus the configured, marker-validated CodeCartographer library when one is configured.
|
|
308
|
+
**Tool interception.** `bash` is blocked outright; `edit` and `write` are confined to `.codecarto/`, plus the configured, marker-validated CodeCartographer library when one is configured. Phase sub-agents get the same hook over a narrower root — `.codecarto/` alone, since a phase writes findings and handoffs and never publishes.
|
|
309
309
|
|
|
310
310
|
### Slash commands
|
|
311
311
|
|
|
@@ -395,15 +395,16 @@ codecarto_broadside {cwd, action: "submit", lenses: [...]} # fire the batches
|
|
|
395
395
|
codecarto_broadside {cwd, action: "status"} # what is in flight
|
|
396
396
|
codecarto_broadside {cwd, action: "collect"} # poll, save, synthesize, triage
|
|
397
397
|
codecarto_broadside {cwd, action: "verify", top: 10} # read the top findings against the source
|
|
398
|
+
codecarto_broadside {cwd, action: "collect", regenerate_post_passes: true} # rebuild the report and work order from the verdicts
|
|
398
399
|
```
|
|
399
400
|
|
|
400
|
-
A third action, `verify`, reads the top defect and security findings of a collected run against the real source — one sync-priced call each with read-only tools (`read_file`, `grep`, `list_dir`) confined to the repository — and writes `verified.md` with a verdict per finding: **confirmed** (a reachable failure, with the trigger), **not-a-defect**, **discarded**, or **unclear**. Measured on this repository, the top twelve findings by severity were two real defects and ten claims a look at the guard or the caller dismissed; the pass agreed with a reviewer on all twelve for a cent a finding. Read `verified.md` before `triage.md
|
|
401
|
+
A third action, `verify`, reads the top defect and security findings of a collected run against the real source — one sync-priced call each with read-only tools (`read_file`, `grep`, `list_dir`) confined to the repository — and writes `verified.md` with a verdict per finding: **confirmed** (a reachable failure, with the trigger), **not-a-defect**, **discarded**, or **unclear**. Measured on this repository, the top twelve findings by severity were two real defects and ten claims a look at the guard or the caller dismissed; the pass agreed with a reviewer on all twelve for a cent a finding. Read `verified.md` before `triage.md` — and have `triage.md` built from it: the synthesis and triage passes rank on the verdicts when they exist (confirmed first, discarded dropped, not-a-defect listed apart), and `collect` with `regenerate_post_passes: true` (`--regenerate` on Pi) rebuilds both for a run that was collected before it was verified.
|
|
401
402
|
|
|
402
403
|
Submit and collect are separate because batch jobs routinely take tens of minutes; collect is resumable and picks up whatever is still in flight (`wait_seconds: 0`, the default, polls once and returns), and two collects on one run — a retried tool call, a second session — never pay for the synthesis, triage, or truncation retry twice: each is claimed in the run's state before it is submitted, and a collect whose client has gone away stops polling and submits nothing further. Submit prices the run from the collected file sizes against the model's live per-token pricing (cached 24h) and refuses when the estimate exceeds `max_cost` — $1.00 unless the config or the call sets another value, `0` for no limit — unless `force: true` is passed — a pre-flight estimate, not a runtime stop. Actual spend lands in each run's `run-meta.json`.
|
|
403
404
|
|
|
404
405
|
Repository defaults live in `.codecarto/broadside/config.yaml` (`model`, `api_key`, `default_lenses`, `max_cost`, `pricing` overrides, `lens_models`, `incremental`, `retry_truncated`, `include_synthesis`, `include_triage`, `wait_seconds`); an explicit tool parameter always wins. `lens_models` routes individual lenses to their own batch model — a stronger model changes security and defect findings far more than it changes an architecture map — and each override is priced, capability-checked, and clamped exactly like the default, with the estimate broken out per lens so a mixed-model run cannot be approved without seeing which lens costs what. CodeCartographer ships no stronger default: which model earns its price depends on your repository and budget, so compare candidates with the `models` action and choose — for one run with the `model` and `lens_models` parameters (Pi: `--model=ID`, `--lens-model=LENS:ID`), or for the repository in `config.yaml`. The `models` listing is advisory: OpenRouter's catalog returns a `:batch` id for some models its Batch API then refuses (`does not have a :batch endpoint`), at no cost, and nothing in the catalog tells them apart — so the listing tags the ids this repository's own submits have seen accepted or refused, and a refused lens says why in the submit report. `codecarto_skill {cwd, name: "broadside"}` returns the reading guide for a completed run, and unlike post-pipeline skills it is not gated on a finished pipeline.
|
|
405
406
|
|
|
406
|
-
On the Pi extension the same run is `/codecarto-broadside [submit|collect|status|models|verify] [lenses…] [--model=ID] [--lens-model=LENS:ID] [--top=N]`, with tab-completion for actions, lens names, and flags and live per-lens progress while batches poll. The two surfaces differ in one deliberate place: MCP cannot ask a human, so it refuses a run over `max_cost` until you pass `force`; Pi shows the per-lens breakdown and asks, and your approval *is* the force flag. Neither surface takes an API key as a command argument — a key typed into a slash command lands in the session transcript.
|
|
407
|
+
On the Pi extension the same run is `/codecarto-broadside [submit|collect|status|models|verify] [lenses…] [--model=ID] [--lens-model=LENS:ID] [--top=N] [--regenerate]`, with tab-completion for actions, lens names, and flags and live per-lens progress while batches poll. The two surfaces differ in one deliberate place: MCP cannot ask a human, so it refuses a run over `max_cost` until you pass `force`; Pi shows the per-lens breakdown and asks, and your approval *is* the force flag. Neither surface takes an API key as a command argument — a key typed into a slash command lands in the session transcript.
|
|
407
408
|
|
|
408
409
|
Broad-Side needs runtime code, so firing a run is an executable-surface feature: Pi and MCP have it, the pure drop-in template does not (it carries only the reading guide).
|
|
409
410
|
|
|
@@ -99,7 +99,11 @@ Results land in `.codecarto/broadside/<run>/`. If `verified.md` is there, read
|
|
|
99
99
|
it before anything else: `action: "verify"` has read the top defect and
|
|
100
100
|
security findings against the source with read-only tools and given each a
|
|
101
101
|
verdict (confirmed with its trigger, not-a-defect, discarded with the guard
|
|
102
|
-
that shows it, unclear).
|
|
102
|
+
that shows it, unclear). The post-passes rank on those verdicts when they
|
|
103
|
+
exist; a run collected before it was verified has a work order built from
|
|
104
|
+
batch severities alone (`status` says `(no verdicts)`), and
|
|
105
|
+
`action: "collect", regenerate_post_passes: true` rebuilds both passes from
|
|
106
|
+
the verdicts for another post-pass pair's cost. Then, in this order:
|
|
103
107
|
|
|
104
108
|
1. `synthesis.md` — executive summary, severity counts, top cross-lens
|
|
105
109
|
findings, per-module risk.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { type BroadsideLensId } from "./constants.ts";
|
|
2
|
+
import { type BatchRequest } from "./types.ts";
|
|
3
|
+
export type FetchLike = (url: string, init: Record<string, unknown>) => Promise<Response>;
|
|
4
|
+
export declare function submitBatch(batchRequests: BatchRequest[], apiKey: string, fetcher?: FetchLike, model?: string): Promise<{
|
|
5
|
+
batchId: string;
|
|
6
|
+
status: string;
|
|
7
|
+
error?: unknown;
|
|
8
|
+
}>;
|
|
9
|
+
export declare function fetchBatch(batchId: string, apiKey: string, fetcher?: FetchLike): Promise<Record<string, unknown>>;
|
|
10
|
+
export declare function pollBatchUntilTerminal(batchId: string, apiKey: string, opts?: {
|
|
11
|
+
deadlineMs?: number;
|
|
12
|
+
onStatus?: (status: string, counts: Record<string, unknown>) => void;
|
|
13
|
+
fetcher?: FetchLike;
|
|
14
|
+
pollIntervalMs?: number;
|
|
15
|
+
/**
|
|
16
|
+
* Stops polling early with the same synthetic `timeout` a spent budget
|
|
17
|
+
* returns: the batch keeps running server-side and a later collect
|
|
18
|
+
* claims it. The MCP server aborts when its client disconnects (#322).
|
|
19
|
+
*/
|
|
20
|
+
signal?: AbortSignal;
|
|
21
|
+
}): Promise<Record<string, unknown>>;
|
|
22
|
+
/**
|
|
23
|
+
* Poll several batch ids in parallel against one shared deadline. Collect
|
|
24
|
+
* previously polled one lens at a time, so a slow first lens serialized the
|
|
25
|
+
* wall clock for lenses that had already finished server-side (#136). The
|
|
26
|
+
* onStatus callback identifies the lens so progress output stays readable
|
|
27
|
+
* even while the polls interleave.
|
|
28
|
+
*/
|
|
29
|
+
export declare function pollBatchesConcurrently(entries: Array<{
|
|
30
|
+
lensId: BroadsideLensId;
|
|
31
|
+
batchId: string;
|
|
32
|
+
}>, apiKey: string, opts?: {
|
|
33
|
+
deadlineMs?: number;
|
|
34
|
+
fetcher?: FetchLike;
|
|
35
|
+
pollIntervalMs?: number;
|
|
36
|
+
signal?: AbortSignal;
|
|
37
|
+
onStatus?: (lensId: string, status: string, counts: Record<string, unknown>) => void;
|
|
38
|
+
}): Promise<Map<string, Record<string, unknown>>>;
|
|
39
|
+
export declare const NO_BATCH_ENDPOINT_RE: RegExp;
|
|
40
|
+
/** The refusal for a full per-account concurrent batch-job quota. */
|
|
41
|
+
export declare const BATCH_QUOTA_RE: RegExp;
|
|
42
|
+
/** One line of a batch's error field, whatever shape the provider gave it. */
|
|
43
|
+
export declare function describeBatchError(error: unknown): string | null;
|
|
44
|
+
/**
|
|
45
|
+
* A provider refusal plus what to do about it, for the two refusals a batch
|
|
46
|
+
* run meets in practice and cannot fix by itself (#141):
|
|
47
|
+
*
|
|
48
|
+
* - `Model '<id>' does not have a :batch endpoint.` — the catalog advertises a
|
|
49
|
+
* `:batch` id that OpenRouter runs no batch endpoint for. Nothing in the
|
|
50
|
+
* catalog distinguishes these; the `models` action marks ids this
|
|
51
|
+
* repository has seen refused.
|
|
52
|
+
* - `job-submission-count … in use: 16, quota: 16` — the per-account limit
|
|
53
|
+
* on concurrent batch jobs. Broad-Side submits one job per lens, so a few
|
|
54
|
+
* runs in flight on the same key fill it; the refusal costs nothing.
|
|
55
|
+
*/
|
|
56
|
+
export declare function explainBatchError(error: unknown): string | null;
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
// The OpenRouter Batch API client: submit, fetch, poll one or many batches to a terminal status.
|
|
2
|
+
//
|
|
3
|
+
// Split out of core/broadside.ts (#339); the barrel there re-exports every
|
|
4
|
+
// name, so `core/index.ts` and the tests see one module as before.
|
|
5
|
+
import { sleep } from "../utils.js";
|
|
6
|
+
import { BROADSIDE_BATCH_URL, BROADSIDE_DEAD_BATCH_STATUSES, BROADSIDE_DEFAULT_POLL_BUDGET_MS, BROADSIDE_MODEL, BROADSIDE_POLL_INTERVAL_MS } from "./constants.js";
|
|
7
|
+
export async function submitBatch(batchRequests, apiKey, fetcher = fetch, model = BROADSIDE_MODEL) {
|
|
8
|
+
// The OpenRouter batch endpoint stream-parses the body and requires
|
|
9
|
+
// `endpoint` and `model` to serialize before `requests` — key order matters.
|
|
10
|
+
const payload = {
|
|
11
|
+
endpoint: "/v1/chat/completions",
|
|
12
|
+
model,
|
|
13
|
+
requests: batchRequests,
|
|
14
|
+
};
|
|
15
|
+
const resp = await fetcher(BROADSIDE_BATCH_URL, {
|
|
16
|
+
method: "POST",
|
|
17
|
+
headers: {
|
|
18
|
+
Authorization: `Bearer ${apiKey}`,
|
|
19
|
+
"Content-Type": "application/json",
|
|
20
|
+
},
|
|
21
|
+
body: JSON.stringify(payload),
|
|
22
|
+
signal: AbortSignal.timeout(30_000),
|
|
23
|
+
});
|
|
24
|
+
const data = (await resp.json());
|
|
25
|
+
if (resp.status !== 202) {
|
|
26
|
+
return { batchId: "", status: "rejected", error: data };
|
|
27
|
+
}
|
|
28
|
+
return { batchId: String(data.id), status: String(data.status) };
|
|
29
|
+
}
|
|
30
|
+
export async function fetchBatch(batchId, apiKey, fetcher = fetch) {
|
|
31
|
+
const resp = await fetcher(`${BROADSIDE_BATCH_URL}/${batchId}`, {
|
|
32
|
+
method: "GET",
|
|
33
|
+
headers: { Authorization: `Bearer ${apiKey}` },
|
|
34
|
+
signal: AbortSignal.timeout(30_000),
|
|
35
|
+
});
|
|
36
|
+
let data;
|
|
37
|
+
try {
|
|
38
|
+
data = (await resp.json());
|
|
39
|
+
}
|
|
40
|
+
catch (error) {
|
|
41
|
+
// A gateway error page is not JSON. It used to throw out of here and
|
|
42
|
+
// be retried as if the network were down; keep the status instead.
|
|
43
|
+
data = { error: `non-JSON response (${error instanceof Error ? error.message : String(error)})` };
|
|
44
|
+
}
|
|
45
|
+
if (!data || typeof data !== "object")
|
|
46
|
+
data = { error: "empty response" };
|
|
47
|
+
// Surface the HTTP status so the poller can bail fast on auth expiry
|
|
48
|
+
// instead of retrying a dead key for the whole budget.
|
|
49
|
+
data.http_status = resp.status;
|
|
50
|
+
return data;
|
|
51
|
+
}
|
|
52
|
+
export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
|
|
53
|
+
const deadline = Date.now() + (opts.deadlineMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
|
|
54
|
+
const intervalMs = opts.pollIntervalMs ?? BROADSIDE_POLL_INTERVAL_MS;
|
|
55
|
+
const fetcher = opts.fetcher ?? fetch;
|
|
56
|
+
// A poll that runs out of budget without one good response is not a slow
|
|
57
|
+
// batch. The last thing that went wrong rides on the timeout so the report
|
|
58
|
+
// can tell a dead network or a failing gateway from a batch still running.
|
|
59
|
+
let lastError = null;
|
|
60
|
+
let sawBatch = false;
|
|
61
|
+
const timedOut = () => ({
|
|
62
|
+
id: batchId,
|
|
63
|
+
status: "timeout",
|
|
64
|
+
...(lastError && !sawBatch && { error: `no successful poll response; last error: ${lastError}` }),
|
|
65
|
+
...(lastError && sawBatch && { last_error: lastError }),
|
|
66
|
+
});
|
|
67
|
+
for (;;) {
|
|
68
|
+
if (opts.signal?.aborted)
|
|
69
|
+
return { ...timedOut(), aborted: true };
|
|
70
|
+
let batch;
|
|
71
|
+
try {
|
|
72
|
+
batch = await fetchBatch(batchId, apiKey, fetcher);
|
|
73
|
+
}
|
|
74
|
+
catch (error) {
|
|
75
|
+
lastError = `fetch failed (${error instanceof Error ? error.message : String(error)})`;
|
|
76
|
+
if (Date.now() >= deadline)
|
|
77
|
+
return timedOut();
|
|
78
|
+
await sleep(intervalMs);
|
|
79
|
+
continue;
|
|
80
|
+
}
|
|
81
|
+
const httpStatus = Number(batch.http_status ?? 200);
|
|
82
|
+
if (httpStatus === 401 || httpStatus === 403) {
|
|
83
|
+
return { id: batchId, status: "auth-failed", error: batch.error ?? batch };
|
|
84
|
+
}
|
|
85
|
+
if (httpStatus >= 400) {
|
|
86
|
+
// A gateway or server error: retry within the budget, remembered.
|
|
87
|
+
const detail = typeof batch.error === "string" ? batch.error : JSON.stringify(batch.error ?? "");
|
|
88
|
+
lastError = `HTTP ${httpStatus}${detail ? ` (${detail.slice(0, 200)})` : ""}`;
|
|
89
|
+
if (Date.now() >= deadline)
|
|
90
|
+
return timedOut();
|
|
91
|
+
await sleep(intervalMs);
|
|
92
|
+
continue;
|
|
93
|
+
}
|
|
94
|
+
sawBatch = true;
|
|
95
|
+
const status = String(batch.status ?? "unknown");
|
|
96
|
+
const counts = (batch.request_counts ?? {});
|
|
97
|
+
opts.onStatus?.(status, counts);
|
|
98
|
+
if (status === "completed" || BROADSIDE_DEAD_BATCH_STATUSES.includes(status))
|
|
99
|
+
return batch;
|
|
100
|
+
if (Date.now() >= deadline)
|
|
101
|
+
return timedOut();
|
|
102
|
+
await sleepUnlessAborted(intervalMs, opts.signal);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
/** Sleep, but wake at once when the signal fires so an abort is not a poll interval late. */
|
|
106
|
+
function sleepUnlessAborted(ms, signal) {
|
|
107
|
+
if (!signal)
|
|
108
|
+
return sleep(ms);
|
|
109
|
+
if (signal.aborted)
|
|
110
|
+
return Promise.resolve();
|
|
111
|
+
return new Promise((resolve) => {
|
|
112
|
+
const timer = setTimeout(() => {
|
|
113
|
+
signal.removeEventListener("abort", onAbort);
|
|
114
|
+
resolve();
|
|
115
|
+
}, ms);
|
|
116
|
+
const onAbort = () => {
|
|
117
|
+
clearTimeout(timer);
|
|
118
|
+
resolve();
|
|
119
|
+
};
|
|
120
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* Poll several batch ids in parallel against one shared deadline. Collect
|
|
125
|
+
* previously polled one lens at a time, so a slow first lens serialized the
|
|
126
|
+
* wall clock for lenses that had already finished server-side (#136). The
|
|
127
|
+
* onStatus callback identifies the lens so progress output stays readable
|
|
128
|
+
* even while the polls interleave.
|
|
129
|
+
*/
|
|
130
|
+
export async function pollBatchesConcurrently(entries, apiKey, opts = {}) {
|
|
131
|
+
const results = new Map();
|
|
132
|
+
const deadlineMs = opts.deadlineMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS;
|
|
133
|
+
await Promise.all(entries.map(async ({ lensId, batchId }) => {
|
|
134
|
+
const batch = await pollBatchUntilTerminal(batchId, apiKey, {
|
|
135
|
+
deadlineMs,
|
|
136
|
+
fetcher: opts.fetcher,
|
|
137
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
138
|
+
signal: opts.signal,
|
|
139
|
+
onStatus: (status, counts) => opts.onStatus?.(lensId, status, counts),
|
|
140
|
+
});
|
|
141
|
+
results.set(batchId, batch);
|
|
142
|
+
}));
|
|
143
|
+
return results;
|
|
144
|
+
}
|
|
145
|
+
// ---------- provider errors ----------
|
|
146
|
+
export const NO_BATCH_ENDPOINT_RE = /does not have a :batch endpoint/i;
|
|
147
|
+
/** The refusal for a full per-account concurrent batch-job quota. */
|
|
148
|
+
export const BATCH_QUOTA_RE = /job-submission-count/i;
|
|
149
|
+
/** One line of a batch's error field, whatever shape the provider gave it. */
|
|
150
|
+
export function describeBatchError(error) {
|
|
151
|
+
if (error === undefined || error === null || error === "")
|
|
152
|
+
return null;
|
|
153
|
+
if (typeof error === "string")
|
|
154
|
+
return error.slice(0, 300);
|
|
155
|
+
if (typeof error === "object") {
|
|
156
|
+
const message = error.message;
|
|
157
|
+
if (typeof message === "string" && message)
|
|
158
|
+
return message.slice(0, 300);
|
|
159
|
+
// OpenRouter wraps a submit refusal as `{ error: { message } }`.
|
|
160
|
+
const nested = error.error;
|
|
161
|
+
if (nested && typeof nested === "object") {
|
|
162
|
+
const inner = nested.message;
|
|
163
|
+
if (typeof inner === "string" && inner)
|
|
164
|
+
return inner.slice(0, 300);
|
|
165
|
+
}
|
|
166
|
+
if (typeof nested === "string" && nested)
|
|
167
|
+
return nested.slice(0, 300);
|
|
168
|
+
try {
|
|
169
|
+
return JSON.stringify(error).slice(0, 300);
|
|
170
|
+
}
|
|
171
|
+
catch {
|
|
172
|
+
return String(error);
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
return String(error);
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* A provider refusal plus what to do about it, for the two refusals a batch
|
|
179
|
+
* run meets in practice and cannot fix by itself (#141):
|
|
180
|
+
*
|
|
181
|
+
* - `Model '<id>' does not have a :batch endpoint.` — the catalog advertises a
|
|
182
|
+
* `:batch` id that OpenRouter runs no batch endpoint for. Nothing in the
|
|
183
|
+
* catalog distinguishes these; the `models` action marks ids this
|
|
184
|
+
* repository has seen refused.
|
|
185
|
+
* - `job-submission-count … in use: 16, quota: 16` — the per-account limit
|
|
186
|
+
* on concurrent batch jobs. Broad-Side submits one job per lens, so a few
|
|
187
|
+
* runs in flight on the same key fill it; the refusal costs nothing.
|
|
188
|
+
*/
|
|
189
|
+
export function explainBatchError(error) {
|
|
190
|
+
const message = describeBatchError(error);
|
|
191
|
+
if (!message)
|
|
192
|
+
return null;
|
|
193
|
+
if (NO_BATCH_ENDPOINT_RE.test(message)) {
|
|
194
|
+
return `${message} — the catalog lists this id, but OpenRouter runs no batch endpoint for it. Nothing was charged; pick another model (the models action marks ids this repository has seen refused).`;
|
|
195
|
+
}
|
|
196
|
+
if (BATCH_QUOTA_RE.test(message)) {
|
|
197
|
+
return `${message} — OpenRouter's per-account limit on concurrent batch jobs is full. Broad-Side submits one job per lens, so a few runs in flight on this key (in any repository) fill it. Nothing was charged; collect or wait out the runs in flight, then re-submit.`;
|
|
198
|
+
}
|
|
199
|
+
return message;
|
|
200
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { type BroadsideCollectResult, type BroadsideStateFile } from "./types.ts";
|
|
2
|
+
import { type FetchLike } from "./client.ts";
|
|
3
|
+
/**
|
|
4
|
+
* One verdict from a run's `verified.json` (written by the verify pass in
|
|
5
|
+
* `verify.ts`), reduced to what the post-passes are told.
|
|
6
|
+
*/
|
|
7
|
+
export type PostPassVerdict = {
|
|
8
|
+
lensId: string;
|
|
9
|
+
customId: string;
|
|
10
|
+
severity: string;
|
|
11
|
+
title: string;
|
|
12
|
+
location: string;
|
|
13
|
+
verdict: string;
|
|
14
|
+
confidence: string;
|
|
15
|
+
evidence: Array<{
|
|
16
|
+
file: string;
|
|
17
|
+
lines: string;
|
|
18
|
+
note: string;
|
|
19
|
+
}>;
|
|
20
|
+
reasoning: string;
|
|
21
|
+
};
|
|
22
|
+
/**
|
|
23
|
+
* The verdicts a verify pass left in the run directory, or null when none
|
|
24
|
+
* has run (#338). A file that does not parse is treated as absent: the
|
|
25
|
+
* post-passes then run from the findings alone, which is what they did
|
|
26
|
+
* before verdicts existed, and `status` shows the pass carried no verdicts.
|
|
27
|
+
*/
|
|
28
|
+
export declare function loadPostPassVerdicts(runDir: string): Promise<PostPassVerdict[] | null>;
|
|
29
|
+
/**
|
|
30
|
+
* The verdicts as a section of the post-pass user message: one line per
|
|
31
|
+
* finding with the verdict, the evidence the verifier cited, and its
|
|
32
|
+
* reasoning, so the pass can rank on them rather than on the batch model's
|
|
33
|
+
* own severities (#338).
|
|
34
|
+
*/
|
|
35
|
+
export declare function renderPostPassVerdicts(verdicts: PostPassVerdict[]): string;
|
|
36
|
+
export declare function runBroadsideCollect(cwd: string, apiKey: string, opts?: {
|
|
37
|
+
waitMs?: number;
|
|
38
|
+
includeSynthesis?: boolean;
|
|
39
|
+
includeTriage?: boolean;
|
|
40
|
+
/** Re-submit truncated slices once with a doubled output cap (#133). */
|
|
41
|
+
retryTruncated?: boolean;
|
|
42
|
+
onStatus?: (lensId: string, status: string, counts: Record<string, unknown>) => void;
|
|
43
|
+
fetcher?: FetchLike;
|
|
44
|
+
/**
|
|
45
|
+
* Which run to collect. Absent, the most recent — which used to be the
|
|
46
|
+
* only choice, so an older run still in flight could not be collected
|
|
47
|
+
* once a newer submit existed (#268). `status` lists the ids.
|
|
48
|
+
*/
|
|
49
|
+
runId?: string;
|
|
50
|
+
/**
|
|
51
|
+
* Stops polling and submits nothing further once fired; what was
|
|
52
|
+
* already submitted keeps running server-side for a later collect to
|
|
53
|
+
* claim. The MCP server fires it when its client disconnects (#322).
|
|
54
|
+
*/
|
|
55
|
+
signal?: AbortSignal;
|
|
56
|
+
/** Poll cadence override; tests drive the loop faster than 15 s. */
|
|
57
|
+
pollIntervalMs?: number;
|
|
58
|
+
/**
|
|
59
|
+
* Reset the wanted post-passes of a collected run and run them again
|
|
60
|
+
* (#338) — after a `verify`, so the executive report and the work order
|
|
61
|
+
* are built from the verdicts. A pass still in flight is left to finish;
|
|
62
|
+
* a run whose lens batches are still running is refused.
|
|
63
|
+
*/
|
|
64
|
+
regeneratePostPasses?: boolean;
|
|
65
|
+
}): Promise<BroadsideCollectResult>;
|
|
66
|
+
export declare function runBroadsideStatus(cwd: string): Promise<{
|
|
67
|
+
state: BroadsideStateFile;
|
|
68
|
+
}>;
|