codecartographer-pi 0.22.0 → 0.22.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +8 -5
- package/.codecarto/broadside/config.yaml +41 -20
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +1 -1
- package/dist/core/broadside.d.ts +110 -20
- package/dist/core/broadside.js +362 -65
- package/dist/extensions/codecarto/index.js +32 -8
- package/dist/mcp-server/server.js +27 -3
- package/package.json +1 -1
|
@@ -95,11 +95,14 @@ OpenRouter advertises a `:batch` variant for, and many of those variants do not
|
|
|
95
95
|
exist — submitting one returns `does not have a :batch endpoint`, with nothing in
|
|
96
96
|
the catalog to distinguish it beforehand. A rejected batch costs nothing, so
|
|
97
97
|
probe a candidate on a single lens first. And reasoning competes with the answer for
|
|
98
|
-
`max_tokens`: Broad-Side
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
98
|
+
`max_tokens`: Broad-Side asks every model for low reasoning effort so the output
|
|
99
|
+
budget stays with the JSON, which is the split the cost estimate already
|
|
100
|
+
assumes. An effort level rather than a token cap, because Gemini 3.x ignores a
|
|
101
|
+
cap (measured: 11,518 thinking tokens under a 5,800 cap) and honours the level;
|
|
102
|
+
a level rather than off, because some endpoints refuse to be switched off. A
|
|
103
|
+
slice that still truncates is retried once with a doubled budget at low effort.
|
|
104
|
+
Override with `reasoning:` in `config.yaml` only alongside a raised output
|
|
105
|
+
budget, and set `effort` or `max_tokens`, never both.
|
|
103
106
|
|
|
104
107
|
Lenses do not all have to run on the same model. `lens_models` in `config.yaml`
|
|
105
108
|
routes individual lenses to their own batch model — the usual reason being that
|
|
@@ -72,29 +72,50 @@
|
|
|
72
72
|
|
|
73
73
|
# Reasoning control, sent on every lens request.
|
|
74
74
|
#
|
|
75
|
-
# By default Broad-Side
|
|
76
|
-
# leaving the
|
|
77
|
-
# the cost estimate already assumes.
|
|
78
|
-
#
|
|
79
|
-
# The
|
|
80
|
-
# One measured run spent 5,758 of a 6,000-token budget thinking
|
|
81
|
-
# tokens for the JSON, which truncated mid-structure on 11 of 13
|
|
82
|
-
# tokens bill at the full output rate, so it paid for ~6,000
|
|
83
|
-
# slice to receive ~230 usable ones. The shipped default
|
|
84
|
-
# thing less consistently — reasoning from 0 to 5,757
|
|
85
|
-
# three of them cut off — so this is not something
|
|
86
|
-
#
|
|
87
|
-
#
|
|
88
|
-
#
|
|
89
|
-
#
|
|
90
|
-
#
|
|
91
|
-
#
|
|
92
|
-
#
|
|
93
|
-
#
|
|
75
|
+
# By default Broad-Side asks every model for low reasoning effort
|
|
76
|
+
# (`effort: low`), leaving the output budget for the answer — which is the
|
|
77
|
+
# split the cost estimate already assumes.
|
|
78
|
+
#
|
|
79
|
+
# The control exists because reasoning competes with the answer for
|
|
80
|
+
# `max_tokens`. One measured run spent 5,758 of a 6,000-token budget thinking
|
|
81
|
+
# and left ~230 tokens for the JSON, which truncated mid-structure on 11 of 13
|
|
82
|
+
# slices; those tokens bill at the full output rate, so it paid for ~6,000
|
|
83
|
+
# output tokens per slice to receive ~230 usable ones. The shipped default
|
|
84
|
+
# model does the same thing less consistently — reasoning from 0 to 5,757
|
|
85
|
+
# tokens across 13 slices, three of them cut off — so this is not something
|
|
86
|
+
# only exotic models do.
|
|
87
|
+
#
|
|
88
|
+
# It is an effort level rather than a token cap because a cap is not honoured
|
|
89
|
+
# everywhere. Gemini 3.x models take a thinking *level*, not a budget: under
|
|
90
|
+
# `max_tokens: 5800`, google/gemini-3.8-flash:batch reasoned 5,218 tokens on
|
|
91
|
+
# the first pass and 11,518 on the doubled-budget retry — the cap changed
|
|
92
|
+
# nothing, both results truncated, and the retry cost twice the original for
|
|
93
|
+
# no JSON. The same lens at `effort: low` reasoned 0 tokens, returned valid
|
|
94
|
+
# JSON, and cost a twelfth as much. `effort` is what OpenRouter can translate
|
|
95
|
+
# for every provider; a token cap reaches only the providers that take one.
|
|
96
|
+
#
|
|
97
|
+
# And it is a level rather than an off switch on purpose. Some endpoints refuse
|
|
98
|
+
# to be switched off: `google/gemini-3.8-flash:batch` rejects the entire batch
|
|
99
|
+
# with "Reasoning is mandatory for this endpoint and cannot be disabled", which
|
|
100
|
+
# turns a partial result into none at all. Low effort works either way.
|
|
101
|
+
#
|
|
102
|
+
# A slice that still truncates is re-submitted once with a doubled `max_tokens`
|
|
103
|
+
# and, whatever this block says, low effort — the cutoff is usually thinking.
|
|
104
|
+
#
|
|
105
|
+
# Set `effort` OR `max_tokens`, not both: OpenRouter refuses a request that
|
|
106
|
+
# carries both ("Only one of reasoning.effort and reasoning.max_tokens can be
|
|
107
|
+
# specified") — per request, after the batch is accepted, so every lens fails
|
|
108
|
+
# at $0. Submit refuses a config.yaml that sets both. Raise the effort only
|
|
109
|
+
# together with a raised lens output budget, or the JSON truncates exactly as
|
|
110
|
+
# above.
|
|
94
111
|
#
|
|
95
112
|
# reasoning:
|
|
96
113
|
# effort: low # minimal | low | medium | high
|
|
97
|
-
#
|
|
114
|
+
#
|
|
115
|
+
# reasoning:
|
|
116
|
+
# max_tokens: 2000 # a thinking budget, where the provider honours one
|
|
117
|
+
#
|
|
118
|
+
# reasoning:
|
|
98
119
|
# enabled: false # only where the provider allows it
|
|
99
120
|
|
|
100
121
|
# Approximate run expense limit in USD. The default is 1.00, and it applies
|
package/README.md
CHANGED
|
@@ -396,7 +396,7 @@ codecarto_broadside {cwd, action: "status"} # what is in fligh
|
|
|
396
396
|
codecarto_broadside {cwd, action: "collect"} # poll, save, synthesize, triage
|
|
397
397
|
```
|
|
398
398
|
|
|
399
|
-
Submit and collect are separate because batch jobs routinely take tens of minutes; collect is resumable and picks up whatever is still in flight (`wait_seconds: 0`, the default, polls once and returns). Submit prices the run from the collected file sizes against the model's live per-token pricing (cached 24h) and refuses when the estimate exceeds `max_cost` — $1.00 unless the config or the call sets another value, `0` for no limit — unless `force: true` is passed — a pre-flight estimate, not a runtime stop. Actual spend lands in each run's `run-meta.json`.
|
|
399
|
+
Submit and collect are separate because batch jobs routinely take tens of minutes; collect is resumable and picks up whatever is still in flight (`wait_seconds: 0`, the default, polls once and returns), and two collects on one run — a retried tool call, a second session — never pay for the synthesis, triage, or truncation retry twice: each is claimed in the run's state before it is submitted, and a collect whose client has gone away stops polling and submits nothing further. Submit prices the run from the collected file sizes against the model's live per-token pricing (cached 24h) and refuses when the estimate exceeds `max_cost` — $1.00 unless the config or the call sets another value, `0` for no limit — unless `force: true` is passed — a pre-flight estimate, not a runtime stop. Actual spend lands in each run's `run-meta.json`.
|
|
400
400
|
|
|
401
401
|
Repository defaults live in `.codecarto/broadside/config.yaml` (`model`, `api_key`, `default_lenses`, `max_cost`, `pricing` overrides, `lens_models`, `incremental`, `retry_truncated`, `include_synthesis`, `include_triage`, `wait_seconds`); an explicit tool parameter always wins. `lens_models` routes individual lenses to their own batch model — a stronger model changes security and defect findings far more than it changes an architecture map — and each override is priced, capability-checked, and clamped exactly like the default, with the estimate broken out per lens so a mixed-model run cannot be approved without seeing which lens costs what. CodeCartographer ships no stronger default: which model earns its price depends on your repository and budget, so compare candidates with the `models` action and choose — for one run with the `model` and `lens_models` parameters (Pi: `--model=ID`, `--lens-model=LENS:ID`), or for the repository in `config.yaml`. The `models` listing is advisory: OpenRouter's catalog returns a `:batch` id for some models its Batch API then refuses (`does not have a :batch endpoint`), at no cost, and nothing in the catalog tells them apart — so the listing tags the ids this repository's own submits have seen accepted or refused, and a refused lens says why in the submit report. `codecarto_skill {cwd, name: "broadside"}` returns the reading guide for a completed run, and unlike post-pipeline skills it is not gated on a finished pipeline.
|
|
402
402
|
|
package/dist/core/broadside.d.ts
CHANGED
|
@@ -137,31 +137,43 @@ export type BroadsideReasoning = {
|
|
|
137
137
|
max_tokens?: number;
|
|
138
138
|
};
|
|
139
139
|
/**
|
|
140
|
-
* The
|
|
140
|
+
* The reasoning control every lens request carries: low effort.
|
|
141
141
|
*
|
|
142
|
-
*
|
|
143
|
-
*
|
|
144
|
-
*
|
|
142
|
+
* It used to be a token cap — `max_tokens` at a quarter of the lens's output
|
|
143
|
+
* budget, so three quarters stayed for the answer. Measured live on
|
|
144
|
+
* `google/gemini-3.8-flash:batch` (0.22.0 verification, defect lens, cap
|
|
145
|
+
* 5,800 of a 6,000 budget): the model reasoned 5,218 tokens on the first
|
|
146
|
+
* pass and **11,518 under the same cap** on the doubled-budget retry —
|
|
147
|
+
* thinking scaled with `max_tokens` and the cap changed nothing, both
|
|
148
|
+
* results truncated, and the retry cost twice the original for no JSON.
|
|
149
|
+
* The same lens with `effort: "low"` reasoned 0 tokens, finished with
|
|
150
|
+
* `stop`, returned valid JSON, and cost a twelfth as much. Gemini 3.x
|
|
151
|
+
* models take a thinking *level*, not a budget, and OpenRouter forwards a
|
|
152
|
+
* `max_tokens` cap to them as nothing at all; `effort` is what it can
|
|
153
|
+
* translate for every provider (a level where the provider has levels, a
|
|
154
|
+
* fraction of the budget where it takes a budget). So the default asks for
|
|
155
|
+
* little thinking in the one vocabulary that reaches everyone.
|
|
156
|
+
*
|
|
157
|
+
* Deliberately not `enabled: false`: `google/gemini-3.8-flash:batch` refuses
|
|
158
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot
|
|
159
|
+
* be disabled"*, turning a partial result into none at all. Low effort works
|
|
160
|
+
* whether or not a provider allows reasoning to be switched off.
|
|
145
161
|
*/
|
|
146
|
-
export declare const
|
|
147
|
-
|
|
162
|
+
export declare const BROADSIDE_DEFAULT_REASONING: Readonly<BroadsideReasoning>;
|
|
163
|
+
/** The reasoning control a lens request carries when config.yaml sets none. */
|
|
164
|
+
export declare function defaultReasoningFor(): BroadsideReasoning;
|
|
148
165
|
/**
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
152
|
-
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
153
|
-
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
154
|
-
* or not a provider allows reasoning to be switched off.
|
|
166
|
+
* The reasoning control a truncated slice is re-submitted with.
|
|
155
167
|
*
|
|
156
|
-
*
|
|
157
|
-
*
|
|
158
|
-
*
|
|
159
|
-
*
|
|
160
|
-
*
|
|
161
|
-
*
|
|
162
|
-
*
|
|
168
|
+
* A truncation on a reasoning-capable model is usually thinking that ate the
|
|
169
|
+
* answer's budget, and doubling `max_tokens` doubles the thinking where the
|
|
170
|
+
* provider ignores a token cap (see {@link BROADSIDE_DEFAULT_REASONING}). The
|
|
171
|
+
* retry therefore asks for low effort as well, replacing a `max_tokens` cap
|
|
172
|
+
* (OpenRouter refuses a request carrying both) and lowering a higher effort.
|
|
173
|
+
* An explicit `enabled: false` and an effort already at or below low are left
|
|
174
|
+
* as they are.
|
|
163
175
|
*/
|
|
164
|
-
export declare function
|
|
176
|
+
export declare function retryReasoningFor(original: BroadsideReasoning | undefined): BroadsideReasoning;
|
|
165
177
|
export type BatchRequest = {
|
|
166
178
|
custom_id: string;
|
|
167
179
|
body: {
|
|
@@ -189,6 +201,8 @@ export type BroadsideBatchEntry = {
|
|
|
189
201
|
cost?: number;
|
|
190
202
|
resultCount?: number;
|
|
191
203
|
error?: unknown;
|
|
204
|
+
/** Why a `skipped` lens had nothing to submit: the globs that matched no file. */
|
|
205
|
+
reason?: string;
|
|
192
206
|
/** Set when this lens used a model other than the run default. */
|
|
193
207
|
model?: string;
|
|
194
208
|
/** The completion ceiling of this lens's model; bounds the truncation retry. */
|
|
@@ -216,7 +230,24 @@ export type BroadsideTriageEntry = {
|
|
|
216
230
|
batchId?: string;
|
|
217
231
|
status: "pending" | "submitted" | "completed" | "failed";
|
|
218
232
|
cost?: number;
|
|
233
|
+
error?: string;
|
|
234
|
+
};
|
|
235
|
+
/** The truncation retry pass of one run: one batch per model (#206). */
|
|
236
|
+
export type BroadsideRetryEntry = {
|
|
237
|
+
status: "submitted" | "completed" | "failed";
|
|
238
|
+
batches: Array<{
|
|
239
|
+
model: string;
|
|
240
|
+
batchId: string;
|
|
241
|
+
}>;
|
|
242
|
+
/** When the owning collect claimed the pass (#322). */
|
|
243
|
+
claimedAt: string;
|
|
219
244
|
};
|
|
245
|
+
/**
|
|
246
|
+
* The parts of a run that cost money to submit and that exactly one collect
|
|
247
|
+
* may own: the two post-passes and the truncation retry (#322).
|
|
248
|
+
*/
|
|
249
|
+
export type BroadsideRunSlot = "synthesis" | "triage" | "retry";
|
|
250
|
+
export declare const BROADSIDE_RUN_SLOTS: readonly BroadsideRunSlot[];
|
|
220
251
|
export type BroadsideRun = {
|
|
221
252
|
id: string;
|
|
222
253
|
createdAt: string;
|
|
@@ -227,6 +258,11 @@ export type BroadsideRun = {
|
|
|
227
258
|
batches: Partial<Record<BroadsideLensId, BroadsideBatchEntry>>;
|
|
228
259
|
synthesis: BroadsideSynthesisEntry;
|
|
229
260
|
triage: BroadsideTriageEntry;
|
|
261
|
+
/**
|
|
262
|
+
* The truncation retry pass (#133), recorded so that two collects on one
|
|
263
|
+
* run cannot both submit it (#322). Absent until a collect claims it.
|
|
264
|
+
*/
|
|
265
|
+
retry?: BroadsideRetryEntry;
|
|
230
266
|
totalCost?: number;
|
|
231
267
|
pricing?: ModelPricing;
|
|
232
268
|
maxCost?: number;
|
|
@@ -424,6 +460,8 @@ export type BroadsideCollectResult = {
|
|
|
424
460
|
truncatedCount: number;
|
|
425
461
|
/** Truncated slices recovered by the automatic re-submit pass (#133). */
|
|
426
462
|
retriedCount: number;
|
|
463
|
+
/** Another collect on this run owns the retry pass; its result lands on a later collect (#322). */
|
|
464
|
+
retryElsewhere?: boolean;
|
|
427
465
|
lensOutcomes: Partial<Record<BroadsideLensId, {
|
|
428
466
|
status: string;
|
|
429
467
|
cost?: number;
|
|
@@ -541,6 +579,35 @@ export declare function updateBroadsideStateAtomically(broadsideDir: string, mut
|
|
|
541
579
|
* restored the next time its own operation checkpoints.
|
|
542
580
|
*/
|
|
543
581
|
export declare function persistBroadsideRun(broadsideDir: string, run: BroadsideRun): Promise<BroadsideStateFile>;
|
|
582
|
+
/**
|
|
583
|
+
* Record a collect's view of its run, keeping whatever is further along on
|
|
584
|
+
* disk (#322).
|
|
585
|
+
*
|
|
586
|
+
* Two collects on one run each hold the run in memory and each used to write
|
|
587
|
+
* the whole thing back, so the last writer replaced the other's post-pass
|
|
588
|
+
* entries with its own — and both had submitted their own post-passes, since
|
|
589
|
+
* each decided from the copy it loaded at entry. This writer merges slot by
|
|
590
|
+
* slot: a post-pass or retry entry that is further along on disk (claimed
|
|
591
|
+
* over pending, submitted over claimed, settled over submitted) wins and is
|
|
592
|
+
* copied into `run`, so the caller reports what is true; a lens entry never
|
|
593
|
+
* goes backwards from terminal to polling. A tie keeps this collect's copy,
|
|
594
|
+
* so the collect that settled a pass records its cost. Submitting is guarded
|
|
595
|
+
* separately by {@link claimRunSlot}.
|
|
596
|
+
*/
|
|
597
|
+
export declare function persistBroadsideRunMerging(broadsideDir: string, run: BroadsideRun): Promise<BroadsideStateFile>;
|
|
598
|
+
/**
|
|
599
|
+
* Claim one spending slot of a run for this collect (#322).
|
|
600
|
+
*
|
|
601
|
+
* Read-modify-write under the state lock: if the slot on disk is still
|
|
602
|
+
* unclaimed (`pending`, or absent for the retry), it is marked `submitted`
|
|
603
|
+
* with no batch id *before* any network call and `true` comes back — this
|
|
604
|
+
* collect owns it and may submit. Otherwise another collect got there first:
|
|
605
|
+
* its entry is copied into `run` and `false` comes back. An adopted entry
|
|
606
|
+
* with a batch id can be polled (polling is idempotent); one without an id
|
|
607
|
+
* is a claim whose owner has not recorded the id yet, and is reported as in
|
|
608
|
+
* flight elsewhere.
|
|
609
|
+
*/
|
|
610
|
+
export declare function claimRunSlot(broadsideDir: string, run: BroadsideRun, slot: BroadsideRunSlot): Promise<boolean>;
|
|
544
611
|
export declare function loadBroadsideConfig(broadsideDir: string): Promise<BroadsideConfig>;
|
|
545
612
|
/** The shipped defaults: what an absent config.yaml means. */
|
|
546
613
|
export declare function defaultBroadsideConfig(): BroadsideConfig;
|
|
@@ -610,6 +677,12 @@ export declare function pollBatchUntilTerminal(batchId: string, apiKey: string,
|
|
|
610
677
|
onStatus?: (status: string, counts: Record<string, unknown>) => void;
|
|
611
678
|
fetcher?: FetchLike;
|
|
612
679
|
pollIntervalMs?: number;
|
|
680
|
+
/**
|
|
681
|
+
* Stops polling early with the same synthetic `timeout` a spent budget
|
|
682
|
+
* returns: the batch keeps running server-side and a later collect
|
|
683
|
+
* claims it. The MCP server aborts when its client disconnects (#322).
|
|
684
|
+
*/
|
|
685
|
+
signal?: AbortSignal;
|
|
613
686
|
}): Promise<Record<string, unknown>>;
|
|
614
687
|
/**
|
|
615
688
|
* Poll several batch ids in parallel against one shared deadline. Collect
|
|
@@ -625,6 +698,7 @@ export declare function pollBatchesConcurrently(entries: Array<{
|
|
|
625
698
|
deadlineMs?: number;
|
|
626
699
|
fetcher?: FetchLike;
|
|
627
700
|
pollIntervalMs?: number;
|
|
701
|
+
signal?: AbortSignal;
|
|
628
702
|
onStatus?: (lensId: string, status: string, counts: Record<string, unknown>) => void;
|
|
629
703
|
}): Promise<Map<string, Record<string, unknown>>>;
|
|
630
704
|
export declare function runBroadsideSubmit(cwd: string, apiKey: string, opts?: {
|
|
@@ -699,11 +773,20 @@ export declare function runBroadsideCollect(cwd: string, apiKey: string, opts?:
|
|
|
699
773
|
* once a newer submit existed (#268). `status` lists the ids.
|
|
700
774
|
*/
|
|
701
775
|
runId?: string;
|
|
776
|
+
/**
|
|
777
|
+
* Stops polling and submits nothing further once fired; what was
|
|
778
|
+
* already submitted keeps running server-side for a later collect to
|
|
779
|
+
* claim. The MCP server fires it when its client disconnects (#322).
|
|
780
|
+
*/
|
|
781
|
+
signal?: AbortSignal;
|
|
782
|
+
/** Poll cadence override; tests drive the loop faster than 15 s. */
|
|
783
|
+
pollIntervalMs?: number;
|
|
702
784
|
}): Promise<BroadsideCollectResult>;
|
|
703
785
|
export declare function runBroadsideStatus(cwd: string): Promise<{
|
|
704
786
|
state: BroadsideStateFile;
|
|
705
787
|
}>;
|
|
706
788
|
export declare function renderFindingsMarkdown(content: string): string;
|
|
789
|
+
export declare function describeIncrementalFallback(reason: BroadsideIncrementalOutcome["reason"]): string;
|
|
707
790
|
export declare function estimateSubmitText(result: BroadsideSubmitResult, lenses: LensDefinition[]): string;
|
|
708
791
|
export declare function modelsText(entries: CatalogEntry[], opts: {
|
|
709
792
|
benchmarks: CodingBenchmarks | null;
|
|
@@ -724,5 +807,12 @@ export declare function modelsText(entries: CatalogEntry[], opts: {
|
|
|
724
807
|
*/
|
|
725
808
|
export declare function explainBatchError(error: unknown): string | null;
|
|
726
809
|
export declare function collectResultText(result: BroadsideCollectResult): string;
|
|
810
|
+
/**
|
|
811
|
+
* An `onStatus` callback that appends one line to `lines` per *change* of a
|
|
812
|
+
* lens's polled status. Every poll used to append a line, so a four-minute
|
|
813
|
+
* wait returned twenty-six identical "in_progress (0/1)" lines per lens
|
|
814
|
+
* before the result (0.22.0 live run).
|
|
815
|
+
*/
|
|
816
|
+
export declare function statusLineWriter(lines: string[]): (lensId: string, status: string, counts: Record<string, unknown>) => void;
|
|
727
817
|
export declare function statusText(state: BroadsideStateFile): string;
|
|
728
818
|
export {};
|
package/dist/core/broadside.js
CHANGED
|
@@ -94,33 +94,53 @@ export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
|
|
|
94
94
|
*/
|
|
95
95
|
export const BROADSIDE_DEFAULT_MAX_COST = 1;
|
|
96
96
|
/**
|
|
97
|
-
* The
|
|
97
|
+
* The reasoning control every lens request carries: low effort.
|
|
98
98
|
*
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
99
|
+
* It used to be a token cap — `max_tokens` at a quarter of the lens's output
|
|
100
|
+
* budget, so three quarters stayed for the answer. Measured live on
|
|
101
|
+
* `google/gemini-3.8-flash:batch` (0.22.0 verification, defect lens, cap
|
|
102
|
+
* 5,800 of a 6,000 budget): the model reasoned 5,218 tokens on the first
|
|
103
|
+
* pass and **11,518 under the same cap** on the doubled-budget retry —
|
|
104
|
+
* thinking scaled with `max_tokens` and the cap changed nothing, both
|
|
105
|
+
* results truncated, and the retry cost twice the original for no JSON.
|
|
106
|
+
* The same lens with `effort: "low"` reasoned 0 tokens, finished with
|
|
107
|
+
* `stop`, returned valid JSON, and cost a twelfth as much. Gemini 3.x
|
|
108
|
+
* models take a thinking *level*, not a budget, and OpenRouter forwards a
|
|
109
|
+
* `max_tokens` cap to them as nothing at all; `effort` is what it can
|
|
110
|
+
* translate for every provider (a level where the provider has levels, a
|
|
111
|
+
* fraction of the budget where it takes a budget). So the default asks for
|
|
112
|
+
* little thinking in the one vocabulary that reaches everyone.
|
|
113
|
+
*
|
|
114
|
+
* Deliberately not `enabled: false`: `google/gemini-3.8-flash:batch` refuses
|
|
115
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot
|
|
116
|
+
* be disabled"*, turning a partial result into none at all. Low effort works
|
|
117
|
+
* whether or not a provider allows reasoning to be switched off.
|
|
102
118
|
*/
|
|
103
|
-
export const
|
|
104
|
-
|
|
119
|
+
export const BROADSIDE_DEFAULT_REASONING = Object.freeze({ effort: "low" });
|
|
120
|
+
/** The reasoning control a lens request carries when config.yaml sets none. */
|
|
121
|
+
export function defaultReasoningFor() {
|
|
122
|
+
return { ...BROADSIDE_DEFAULT_REASONING };
|
|
123
|
+
}
|
|
105
124
|
/**
|
|
106
|
-
*
|
|
107
|
-
*
|
|
108
|
-
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
109
|
-
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
110
|
-
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
111
|
-
* or not a provider allows reasoning to be switched off.
|
|
125
|
+
* The reasoning control a truncated slice is re-submitted with.
|
|
112
126
|
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
*
|
|
118
|
-
*
|
|
119
|
-
*
|
|
127
|
+
* A truncation on a reasoning-capable model is usually thinking that ate the
|
|
128
|
+
* answer's budget, and doubling `max_tokens` doubles the thinking where the
|
|
129
|
+
* provider ignores a token cap (see {@link BROADSIDE_DEFAULT_REASONING}). The
|
|
130
|
+
* retry therefore asks for low effort as well, replacing a `max_tokens` cap
|
|
131
|
+
* (OpenRouter refuses a request carrying both) and lowering a higher effort.
|
|
132
|
+
* An explicit `enabled: false` and an effort already at or below low are left
|
|
133
|
+
* as they are.
|
|
120
134
|
*/
|
|
121
|
-
export function
|
|
122
|
-
|
|
123
|
-
}
|
|
135
|
+
export function retryReasoningFor(original) {
|
|
136
|
+
if (original?.enabled === false)
|
|
137
|
+
return { ...original };
|
|
138
|
+
if (original?.effort === "minimal" || original?.effort === "low")
|
|
139
|
+
return { ...original };
|
|
140
|
+
const { max_tokens: _cap, effort: _effort, ...rest } = original ?? {};
|
|
141
|
+
return { ...rest, effort: "low" };
|
|
142
|
+
}
|
|
143
|
+
export const BROADSIDE_RUN_SLOTS = ["synthesis", "triage", "retry"];
|
|
124
144
|
/**
|
|
125
145
|
* OpenRouter rejected the API key (HTTP 401/403). Thrown from the catalog
|
|
126
146
|
* lookup rather than swallowed into "could not price" or a silent built-in
|
|
@@ -1348,7 +1368,7 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1348
1368
|
max_tokens: maxTokensOverride ?? lens.maxTokens,
|
|
1349
1369
|
// Always sent, never inherited: an absent field means the model's
|
|
1350
1370
|
// own default, and that default is what truncated the JSON.
|
|
1351
|
-
reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(
|
|
1371
|
+
reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(),
|
|
1352
1372
|
},
|
|
1353
1373
|
};
|
|
1354
1374
|
}
|
|
@@ -1515,6 +1535,110 @@ export async function persistBroadsideRun(broadsideDir, run) {
|
|
|
1515
1535
|
state.runs[index] = run;
|
|
1516
1536
|
});
|
|
1517
1537
|
}
|
|
1538
|
+
/** Where a lens batch entry stands, for keeping the more advanced of two. */
|
|
1539
|
+
function batchEntryRank(entry) {
|
|
1540
|
+
if (!entry)
|
|
1541
|
+
return -1;
|
|
1542
|
+
if (BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status))
|
|
1543
|
+
return 2;
|
|
1544
|
+
if (entry.batchId)
|
|
1545
|
+
return 1;
|
|
1546
|
+
return 0;
|
|
1547
|
+
}
|
|
1548
|
+
/** Where a post-pass entry stands: unclaimed, claimed, submitted, settled. */
|
|
1549
|
+
function passEntryRank(entry) {
|
|
1550
|
+
if (!entry || entry.status === "pending")
|
|
1551
|
+
return 0;
|
|
1552
|
+
if (entry.status === "submitted")
|
|
1553
|
+
return entry.batchId ? 2 : 1;
|
|
1554
|
+
return 3;
|
|
1555
|
+
}
|
|
1556
|
+
/** Where the retry pass stands: absent, claimed, submitted, settled. */
|
|
1557
|
+
function retryEntryRank(entry) {
|
|
1558
|
+
if (!entry)
|
|
1559
|
+
return 0;
|
|
1560
|
+
if (entry.status === "submitted")
|
|
1561
|
+
return entry.batches.length > 0 ? 2 : 1;
|
|
1562
|
+
return 3;
|
|
1563
|
+
}
|
|
1564
|
+
/**
|
|
1565
|
+
* Record a collect's view of its run, keeping whatever is further along on
|
|
1566
|
+
* disk (#322).
|
|
1567
|
+
*
|
|
1568
|
+
* Two collects on one run each hold the run in memory and each used to write
|
|
1569
|
+
* the whole thing back, so the last writer replaced the other's post-pass
|
|
1570
|
+
* entries with its own — and both had submitted their own post-passes, since
|
|
1571
|
+
* each decided from the copy it loaded at entry. This writer merges slot by
|
|
1572
|
+
* slot: a post-pass or retry entry that is further along on disk (claimed
|
|
1573
|
+
* over pending, submitted over claimed, settled over submitted) wins and is
|
|
1574
|
+
* copied into `run`, so the caller reports what is true; a lens entry never
|
|
1575
|
+
* goes backwards from terminal to polling. A tie keeps this collect's copy,
|
|
1576
|
+
* so the collect that settled a pass records its cost. Submitting is guarded
|
|
1577
|
+
* separately by {@link claimRunSlot}.
|
|
1578
|
+
*/
|
|
1579
|
+
export async function persistBroadsideRunMerging(broadsideDir, run) {
|
|
1580
|
+
return updateBroadsideStateAtomically(broadsideDir, (state) => {
|
|
1581
|
+
const index = state.runs.findIndex((candidate) => candidate.id === run.id);
|
|
1582
|
+
const onDisk = index === -1 ? undefined : state.runs[index];
|
|
1583
|
+
if (onDisk) {
|
|
1584
|
+
if (passEntryRank(onDisk.synthesis) > passEntryRank(run.synthesis))
|
|
1585
|
+
run.synthesis = onDisk.synthesis;
|
|
1586
|
+
if (passEntryRank(onDisk.triage) > passEntryRank(run.triage))
|
|
1587
|
+
run.triage = onDisk.triage;
|
|
1588
|
+
if (retryEntryRank(onDisk.retry) > retryEntryRank(run.retry))
|
|
1589
|
+
run.retry = onDisk.retry;
|
|
1590
|
+
for (const [lensId, theirs] of Object.entries(onDisk.batches)) {
|
|
1591
|
+
if (theirs && batchEntryRank(theirs) > batchEntryRank(run.batches[lensId]))
|
|
1592
|
+
run.batches[lensId] = theirs;
|
|
1593
|
+
}
|
|
1594
|
+
}
|
|
1595
|
+
if (index === -1)
|
|
1596
|
+
state.runs.push(run);
|
|
1597
|
+
else
|
|
1598
|
+
state.runs[index] = run;
|
|
1599
|
+
});
|
|
1600
|
+
}
|
|
1601
|
+
/**
|
|
1602
|
+
* Claim one spending slot of a run for this collect (#322).
|
|
1603
|
+
*
|
|
1604
|
+
* Read-modify-write under the state lock: if the slot on disk is still
|
|
1605
|
+
* unclaimed (`pending`, or absent for the retry), it is marked `submitted`
|
|
1606
|
+
* with no batch id *before* any network call and `true` comes back — this
|
|
1607
|
+
* collect owns it and may submit. Otherwise another collect got there first:
|
|
1608
|
+
* its entry is copied into `run` and `false` comes back. An adopted entry
|
|
1609
|
+
* with a batch id can be polled (polling is idempotent); one without an id
|
|
1610
|
+
* is a claim whose owner has not recorded the id yet, and is reported as in
|
|
1611
|
+
* flight elsewhere.
|
|
1612
|
+
*/
|
|
1613
|
+
export async function claimRunSlot(broadsideDir, run, slot) {
|
|
1614
|
+
let owned = false;
|
|
1615
|
+
const claimedAt = new Date().toISOString();
|
|
1616
|
+
await updateBroadsideStateAtomically(broadsideDir, (state) => {
|
|
1617
|
+
const index = state.runs.findIndex((candidate) => candidate.id === run.id);
|
|
1618
|
+
const onDisk = index === -1 ? undefined : state.runs[index];
|
|
1619
|
+
const theirs = onDisk?.[slot];
|
|
1620
|
+
const unclaimed = slot === "retry" ? theirs === undefined : theirs?.status === "pending";
|
|
1621
|
+
if (onDisk && !unclaimed) {
|
|
1622
|
+
run[slot] = theirs;
|
|
1623
|
+
owned = false;
|
|
1624
|
+
return;
|
|
1625
|
+
}
|
|
1626
|
+
owned = true;
|
|
1627
|
+
if (slot === "retry") {
|
|
1628
|
+
run.retry = { status: "submitted", batches: [], claimedAt };
|
|
1629
|
+
}
|
|
1630
|
+
else {
|
|
1631
|
+
run[slot] = { ...run[slot], status: "submitted", batchId: undefined };
|
|
1632
|
+
}
|
|
1633
|
+
if (!onDisk) {
|
|
1634
|
+
state.runs.push(run);
|
|
1635
|
+
}
|
|
1636
|
+
else {
|
|
1637
|
+
onDisk[slot] = run[slot];
|
|
1638
|
+
}
|
|
1639
|
+
});
|
|
1640
|
+
return owned;
|
|
1641
|
+
}
|
|
1518
1642
|
/** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
|
|
1519
1643
|
function parseReasoningConfig(raw) {
|
|
1520
1644
|
if (raw === false)
|
|
@@ -1549,6 +1673,21 @@ export async function loadBroadsideConfig(broadsideDir) {
|
|
|
1549
1673
|
throw new BroadsideConfigError(configPath, "is not a YAML mapping");
|
|
1550
1674
|
raw = parsed;
|
|
1551
1675
|
}
|
|
1676
|
+
// OpenRouter accepts `reasoning.effort` or `reasoning.max_tokens`, not
|
|
1677
|
+
// both: a request carrying both is refused per request *after* the batch
|
|
1678
|
+
// is accepted, so every lens fails at $0 with the reason in each
|
|
1679
|
+
// result's error. Seen live on 0.22.0 with the two keys set together.
|
|
1680
|
+
// Refuse here, where the file can be fixed, rather than submit a run
|
|
1681
|
+
// that cannot produce a result.
|
|
1682
|
+
const reasoning = raw.reasoning;
|
|
1683
|
+
if (reasoning && typeof reasoning === "object" && !Array.isArray(reasoning)) {
|
|
1684
|
+
const value = reasoning;
|
|
1685
|
+
const hasEffort = typeof value.effort === "string";
|
|
1686
|
+
const hasBudget = typeof value.max_tokens === "number" && value.max_tokens > 0;
|
|
1687
|
+
if (hasEffort && hasBudget) {
|
|
1688
|
+
throw new BroadsideConfigError(configPath, 'sets both reasoning.effort and reasoning.max_tokens; OpenRouter accepts one or the other ("Only one of reasoning.effort and reasoning.max_tokens can be specified"), and every lens request would fail after the batch is accepted. Keep one');
|
|
1689
|
+
}
|
|
1690
|
+
}
|
|
1552
1691
|
}
|
|
1553
1692
|
return buildBroadsideConfig(raw);
|
|
1554
1693
|
}
|
|
@@ -1991,6 +2130,8 @@ export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
|
|
|
1991
2130
|
...(lastError && sawBatch && { last_error: lastError }),
|
|
1992
2131
|
});
|
|
1993
2132
|
for (;;) {
|
|
2133
|
+
if (opts.signal?.aborted)
|
|
2134
|
+
return { ...timedOut(), aborted: true };
|
|
1994
2135
|
let batch;
|
|
1995
2136
|
try {
|
|
1996
2137
|
batch = await fetchBatch(batchId, apiKey, fetcher);
|
|
@@ -2023,8 +2164,26 @@ export async function pollBatchUntilTerminal(batchId, apiKey, opts = {}) {
|
|
|
2023
2164
|
return batch;
|
|
2024
2165
|
if (Date.now() >= deadline)
|
|
2025
2166
|
return timedOut();
|
|
2026
|
-
await
|
|
2027
|
-
}
|
|
2167
|
+
await sleepUnlessAborted(intervalMs, opts.signal);
|
|
2168
|
+
}
|
|
2169
|
+
}
|
|
2170
|
+
/** Sleep, but wake at once when the signal fires so an abort is not a poll interval late. */
|
|
2171
|
+
function sleepUnlessAborted(ms, signal) {
|
|
2172
|
+
if (!signal)
|
|
2173
|
+
return sleep(ms);
|
|
2174
|
+
if (signal.aborted)
|
|
2175
|
+
return Promise.resolve();
|
|
2176
|
+
return new Promise((resolve) => {
|
|
2177
|
+
const timer = setTimeout(() => {
|
|
2178
|
+
signal.removeEventListener("abort", onAbort);
|
|
2179
|
+
resolve();
|
|
2180
|
+
}, ms);
|
|
2181
|
+
const onAbort = () => {
|
|
2182
|
+
clearTimeout(timer);
|
|
2183
|
+
resolve();
|
|
2184
|
+
};
|
|
2185
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
2186
|
+
});
|
|
2028
2187
|
}
|
|
2029
2188
|
/**
|
|
2030
2189
|
* Poll several batch ids in parallel against one shared deadline. Collect
|
|
@@ -2041,6 +2200,7 @@ export async function pollBatchesConcurrently(entries, apiKey, opts = {}) {
|
|
|
2041
2200
|
deadlineMs,
|
|
2042
2201
|
fetcher: opts.fetcher,
|
|
2043
2202
|
pollIntervalMs: opts.pollIntervalMs,
|
|
2203
|
+
signal: opts.signal,
|
|
2044
2204
|
onStatus: (status, counts) => opts.onStatus?.(lensId, status, counts),
|
|
2045
2205
|
});
|
|
2046
2206
|
results.set(batchId, batch);
|
|
@@ -2132,6 +2292,8 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
2132
2292
|
}
|
|
2133
2293
|
// Slice offline first so the estimate covers every request we would send.
|
|
2134
2294
|
const slicesByLens = new Map();
|
|
2295
|
+
// Why a lens ended up with nothing to submit, for the report (see below).
|
|
2296
|
+
const skipReasons = new Map();
|
|
2135
2297
|
let estimatedInputTokens = 0;
|
|
2136
2298
|
let estimatedOutputTokens = 0;
|
|
2137
2299
|
let estimatedTotalCost = 0;
|
|
@@ -2149,12 +2311,21 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
2149
2311
|
for (const file of slice.redactedFiles ?? [])
|
|
2150
2312
|
redactedFiles.add(file);
|
|
2151
2313
|
}
|
|
2314
|
+
const matchedBeforeIncremental = slices.length;
|
|
2152
2315
|
if (changed) {
|
|
2153
2316
|
// Repo-info slices (empty files, e.g. architecture) always run;
|
|
2154
2317
|
// file-backed slices run only when one of their files changed.
|
|
2155
2318
|
slices = slices.filter((s) => s.files.length === 0 || s.files.some((f) => changed.has(f)));
|
|
2156
2319
|
}
|
|
2157
2320
|
slicesByLens.set(lensId, slices);
|
|
2321
|
+
if (slices.length === 0) {
|
|
2322
|
+
const globs = lens.globsFor(info).filter(Boolean);
|
|
2323
|
+
skipReasons.set(lensId, globs.length === 0
|
|
2324
|
+
? "the lens has no file patterns for this language"
|
|
2325
|
+
: matchedBeforeIncremental > 0
|
|
2326
|
+
? "incremental: none of this lens's files changed since the previous run"
|
|
2327
|
+
: `no files matched ${globs.join(", ")}${lens.skipTestFiles ? " (test files excluded)" : ""}`);
|
|
2328
|
+
}
|
|
2158
2329
|
const lensModel = modelForLens(lensId);
|
|
2159
2330
|
const { pricing: lensPricing, outputCap: lensOutputCap } = resolved.get(lensModel);
|
|
2160
2331
|
const maxTokens = lensOutputCap ? Math.min(lens.maxTokens, lensOutputCap) : lens.maxTokens;
|
|
@@ -2263,7 +2434,15 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
2263
2434
|
if (requests.length === 0) {
|
|
2264
2435
|
// No files matched the lens's globs. That is a coverage gap to
|
|
2265
2436
|
// report, not a batch to submit — the API rejects empty batches.
|
|
2437
|
+
// Name the globs: a JavaScript service whose server lives at
|
|
2438
|
+
// src/server.js gets no security review (that lens reads server/**,
|
|
2439
|
+
// **/auth*, **/middleware/**), and "skipped (0 request(s))" alone
|
|
2440
|
+
// read as an empty repository rather than a lens that looked in
|
|
2441
|
+
// the wrong place.
|
|
2266
2442
|
entry.status = "skipped";
|
|
2443
|
+
const reason = skipReasons.get(lensId);
|
|
2444
|
+
if (reason)
|
|
2445
|
+
entry.reason = reason;
|
|
2267
2446
|
continue;
|
|
2268
2447
|
}
|
|
2269
2448
|
submissions.push((async () => {
|
|
@@ -2284,6 +2463,12 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
2284
2463
|
})());
|
|
2285
2464
|
}
|
|
2286
2465
|
await Promise.allSettled(submissions);
|
|
2466
|
+
// A run with no batch behind it has nothing in flight. Every lens was
|
|
2467
|
+
// skipped or refused, so no poll will ever complete it; leaving it
|
|
2468
|
+
// "in-flight" had status listing a refused run above the completed ones
|
|
2469
|
+
// with synthesis and triage "pending" forever.
|
|
2470
|
+
if (!Object.values(run.batches).some((entry) => entry.batchId))
|
|
2471
|
+
run.status = "failed";
|
|
2287
2472
|
await persistBroadsideRun(broadsideDir, run);
|
|
2288
2473
|
// What the provider just said about each model's batch endpoint outlives
|
|
2289
2474
|
// the run: the `models` action reads it back (#141).
|
|
@@ -2547,6 +2732,12 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2547
2732
|
}
|
|
2548
2733
|
const runDir = join(broadsideDir, run.outputDir);
|
|
2549
2734
|
await mkdir(runDir, { recursive: true });
|
|
2735
|
+
// The spending slots this collect has claimed (#322); only a claimed slot
|
|
2736
|
+
// is ever submitted from here. Every write-back merges with the file, so a
|
|
2737
|
+
// slot another collect has moved further along is never overwritten.
|
|
2738
|
+
const owned = new Set();
|
|
2739
|
+
const persist = () => persistBroadsideRunMerging(broadsideDir, run);
|
|
2740
|
+
const aborted = () => opts.signal?.aborted === true;
|
|
2550
2741
|
const deadline = Date.now() + (opts.waitMs ?? BROADSIDE_DEFAULT_POLL_BUDGET_MS);
|
|
2551
2742
|
let totalCost = 0;
|
|
2552
2743
|
let resultCount = 0;
|
|
@@ -2574,6 +2765,8 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2574
2765
|
const polled = await pollBatchesConcurrently(inFlight, apiKey, {
|
|
2575
2766
|
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
2576
2767
|
fetcher: opts.fetcher,
|
|
2768
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
2769
|
+
signal: opts.signal,
|
|
2577
2770
|
onStatus: opts.onStatus,
|
|
2578
2771
|
});
|
|
2579
2772
|
for (const { lensId } of inFlight) {
|
|
@@ -2623,11 +2816,14 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2623
2816
|
const error = explainBatchError(batch.error);
|
|
2624
2817
|
lensOutcomes[lensId] = { status, cost: entry.cost, resultCount: entry.resultCount, ...(error && { error }) };
|
|
2625
2818
|
}
|
|
2626
|
-
await
|
|
2819
|
+
await persist();
|
|
2627
2820
|
}
|
|
2628
|
-
// #133: re-submit truncated slices once with a bumped output cap
|
|
2629
|
-
// requests are pure, so re-running is always safe;
|
|
2630
|
-
// coverage the first pass lost to a max_tokens
|
|
2821
|
+
// #133: re-submit truncated slices once with a bumped output cap and low
|
|
2822
|
+
// reasoning effort. Batch requests are pure, so re-running is always safe;
|
|
2823
|
+
// the aim is to recover coverage the first pass lost to a max_tokens
|
|
2824
|
+
// cutoff, not to loop forever. Low effort because the cutoff is usually
|
|
2825
|
+
// thinking, and a doubled budget doubled the thinking where a token cap
|
|
2826
|
+
// was ignored (see retryReasoningFor).
|
|
2631
2827
|
//
|
|
2632
2828
|
// All bumped requests for one model go out as ONE batch, and the batches
|
|
2633
2829
|
// (one per model, since a batch carries a single model) are polled
|
|
@@ -2638,7 +2834,36 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2638
2834
|
// lens pass, still present here (#206). Grouping also keeps the retry to
|
|
2639
2835
|
// one job per model against OpenRouter's 16-concurrent-job quota.
|
|
2640
2836
|
let retriedCount = 0;
|
|
2641
|
-
|
|
2837
|
+
let retryElsewhere = false;
|
|
2838
|
+
// A collect that polled nothing — every lens already terminal — still owes
|
|
2839
|
+
// the retry if the collect that saved the results never got to it (it
|
|
2840
|
+
// died, or its client did: #322). Read the saved results back and let the
|
|
2841
|
+
// claim decide; a recovered slice re-parses clean, so this costs nothing
|
|
2842
|
+
// once the retry has run.
|
|
2843
|
+
if (opts.retryTruncated !== false && allLensResults.length === 0 && !aborted()) {
|
|
2844
|
+
const everyLensTerminal = run.lenses.every((lensId) => {
|
|
2845
|
+
const entry = run.batches[lensId];
|
|
2846
|
+
return entry && BROADSIDE_TERMINAL_ENTRY_STATUSES.includes(entry.status);
|
|
2847
|
+
});
|
|
2848
|
+
if (everyLensTerminal) {
|
|
2849
|
+
const restored = await loadSavedLensResults(runDir, run.lenses);
|
|
2850
|
+
if (restored.some((s) => s.truncated)) {
|
|
2851
|
+
allLensResults.push(...restored);
|
|
2852
|
+
truncatedCount = restored.filter((s) => s.truncated).length;
|
|
2853
|
+
}
|
|
2854
|
+
}
|
|
2855
|
+
}
|
|
2856
|
+
if (opts.retryTruncated !== false && truncatedCount > 0 && !aborted()) {
|
|
2857
|
+
// Claim the pass before spending: a second collect on this run finds the
|
|
2858
|
+
// claim and leaves the retry to the first (#322). A retry another
|
|
2859
|
+
// collect has already settled is not run again — its truncation is
|
|
2860
|
+
// what it is.
|
|
2861
|
+
if (await claimRunSlot(broadsideDir, run, "retry"))
|
|
2862
|
+
owned.add("retry");
|
|
2863
|
+
else if (run.retry?.status === "submitted")
|
|
2864
|
+
retryElsewhere = true;
|
|
2865
|
+
}
|
|
2866
|
+
if (opts.retryTruncated !== false && truncatedCount > 0 && owned.has("retry")) {
|
|
2642
2867
|
const requestsByCustomId = await loadStoredRequests(runDir);
|
|
2643
2868
|
const byModel = new Map();
|
|
2644
2869
|
for (const stored of allLensResults) {
|
|
@@ -2658,13 +2883,18 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2658
2883
|
if (bumpedMax <= previousMax)
|
|
2659
2884
|
continue; // already at the ceiling
|
|
2660
2885
|
const group = byModel.get(lensModel) ?? { requests: [], slices: new Map() };
|
|
2661
|
-
group.requests.push({
|
|
2886
|
+
group.requests.push({
|
|
2887
|
+
...original,
|
|
2888
|
+
body: { ...original.body, max_tokens: bumpedMax, reasoning: retryReasoningFor(original.body.reasoning) },
|
|
2889
|
+
});
|
|
2662
2890
|
group.slices.set(stored.customId, stored);
|
|
2663
2891
|
byModel.set(lensModel, group);
|
|
2664
2892
|
}
|
|
2665
2893
|
// Submit every group, then poll whatever was accepted, together.
|
|
2666
2894
|
const submitted = [];
|
|
2667
2895
|
for (const [model, group] of byModel) {
|
|
2896
|
+
if (aborted())
|
|
2897
|
+
break;
|
|
2668
2898
|
try {
|
|
2669
2899
|
const { batchId, error } = await submitBatch(group.requests, apiKey, opts.fetcher, model);
|
|
2670
2900
|
if (!error && batchId)
|
|
@@ -2675,6 +2905,15 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2675
2905
|
// truncated results in place — nothing is lost.
|
|
2676
2906
|
}
|
|
2677
2907
|
}
|
|
2908
|
+
// Record the ids under the claim so a later collect can see what was
|
|
2909
|
+
// paid for, even if this one never returns. No group at all means every
|
|
2910
|
+
// truncated slice was already at its model's ceiling: nothing to retry.
|
|
2911
|
+
run.retry = {
|
|
2912
|
+
...run.retry,
|
|
2913
|
+
batches: submitted,
|
|
2914
|
+
status: submitted.length > 0 ? "submitted" : byModel.size === 0 ? "completed" : "failed",
|
|
2915
|
+
};
|
|
2916
|
+
await persist();
|
|
2678
2917
|
const polled = await pollBatchesConcurrently(submitted.map(({ model, batchId }) => ({ lensId: `retry:${model}`, batchId })), apiKey, {
|
|
2679
2918
|
// Share the caller's deadline. Each of these polls used to start a
|
|
2680
2919
|
// fresh 25-minute budget, so `wait_seconds` bounded only the lens
|
|
@@ -2682,6 +2921,8 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2682
2921
|
// minutes.
|
|
2683
2922
|
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
2684
2923
|
fetcher: opts.fetcher,
|
|
2924
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
2925
|
+
signal: opts.signal,
|
|
2685
2926
|
onStatus: opts.onStatus,
|
|
2686
2927
|
});
|
|
2687
2928
|
for (const { model, batchId } of submitted) {
|
|
@@ -2709,13 +2950,17 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2709
2950
|
retriedCount += 1;
|
|
2710
2951
|
}
|
|
2711
2952
|
}
|
|
2953
|
+
// Every retry batch reached a terminal status, or the poll ran out.
|
|
2954
|
+
if (submitted.length > 0 && submitted.every(({ batchId }) => polled.get(batchId)?.status === "completed")) {
|
|
2955
|
+
run.retry = { ...run.retry, status: "completed" };
|
|
2956
|
+
}
|
|
2712
2957
|
truncatedCount = allLensResults.filter((s) => s.truncated).length;
|
|
2713
2958
|
for (const [lensId, outcome] of Object.entries(lensOutcomes)) {
|
|
2714
2959
|
if (outcome.truncated !== undefined) {
|
|
2715
2960
|
outcome.truncated = allLensResults.filter((s) => s.lensId === lensId && s.truncated).length;
|
|
2716
2961
|
}
|
|
2717
2962
|
}
|
|
2718
|
-
await
|
|
2963
|
+
await persist();
|
|
2719
2964
|
}
|
|
2720
2965
|
// Synthesis + triage: cross-lens post-passes, only after every lens batch
|
|
2721
2966
|
// is terminal. Triage turns the leads into a prioritized work order.
|
|
@@ -2754,22 +2999,27 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2754
2999
|
// Both post-passes consume the same findings; they run as two
|
|
2755
3000
|
// batches (different response_format schemas cannot share one)
|
|
2756
3001
|
// submitted together and polled in turn.
|
|
2757
|
-
|
|
2758
|
-
|
|
2759
|
-
|
|
2760
|
-
|
|
2761
|
-
|
|
2762
|
-
|
|
2763
|
-
|
|
2764
|
-
|
|
2765
|
-
|
|
2766
|
-
|
|
2767
|
-
|
|
2768
|
-
|
|
2769
|
-
|
|
2770
|
-
|
|
2771
|
-
|
|
2772
|
-
|
|
3002
|
+
// Claim each wanted, still-pending pass before building its request:
|
|
3003
|
+
// a second collect on this run adopts the first one's entry instead
|
|
3004
|
+
// of submitting its own (#322). An abort submits nothing further.
|
|
3005
|
+
const passes = [];
|
|
3006
|
+
for (const kind of ["synthesis", "triage"]) {
|
|
3007
|
+
const want = kind === "synthesis" ? wantSynthesis : wantTriage;
|
|
3008
|
+
if (!want || aborted())
|
|
3009
|
+
continue;
|
|
3010
|
+
if ((kind === "synthesis" ? run.synthesis : run.triage).status !== "pending")
|
|
3011
|
+
continue;
|
|
3012
|
+
if (!(await claimRunSlot(broadsideDir, run, kind)))
|
|
3013
|
+
continue;
|
|
3014
|
+
owned.add(kind);
|
|
3015
|
+
passes.push({
|
|
3016
|
+
kind,
|
|
3017
|
+
request: kind === "synthesis"
|
|
3018
|
+
? buildSynthesisRequest(findingsText, truncatedNote, run.model)
|
|
3019
|
+
: buildTriageRequest(findingsText, truncatedNote, run.model),
|
|
3020
|
+
entry: kind === "synthesis" ? run.synthesis : run.triage,
|
|
3021
|
+
});
|
|
3022
|
+
}
|
|
2773
3023
|
const submitted = new Map();
|
|
2774
3024
|
// A pass can be left at "submitted" when an earlier collect returned
|
|
2775
3025
|
// before its batch reached a terminal status — the batch still runs
|
|
@@ -2804,16 +3054,22 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2804
3054
|
pass.entry.status = "failed";
|
|
2805
3055
|
}
|
|
2806
3056
|
}));
|
|
2807
|
-
await
|
|
3057
|
+
await persist();
|
|
3058
|
+
// Poll both passes together against the shared deadline. Polled in
|
|
3059
|
+
// turn, the first pass could spend the whole budget and leave the
|
|
3060
|
+
// second a single poll (0.22.1 live run: triage settled, synthesis
|
|
3061
|
+
// left running though it had been submitted at the same moment).
|
|
3062
|
+
// A pass whose poll runs out stays `submitted`, so the batch is
|
|
3063
|
+
// already paid for and a later collect claims its result.
|
|
3064
|
+
const polledPasses = await pollBatchesConcurrently([...submitted.values()].map(({ batchId, pass }) => ({ lensId: pass.kind, batchId })), apiKey, {
|
|
3065
|
+
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
3066
|
+
fetcher: opts.fetcher,
|
|
3067
|
+
pollIntervalMs: opts.pollIntervalMs,
|
|
3068
|
+
signal: opts.signal,
|
|
3069
|
+
onStatus: opts.onStatus,
|
|
3070
|
+
});
|
|
2808
3071
|
for (const { batchId, pass } of submitted.values()) {
|
|
2809
|
-
const batch =
|
|
2810
|
-
// Shares the caller's deadline, as the retry poll above does.
|
|
2811
|
-
// A pass whose poll runs out stays `submitted`, so the batch
|
|
2812
|
-
// is already paid for and a later collect claims its result.
|
|
2813
|
-
deadlineMs: Math.max(0, deadline - Date.now()),
|
|
2814
|
-
onStatus: (status, counts) => opts.onStatus?.(pass.kind, status, counts),
|
|
2815
|
-
fetcher: opts.fetcher,
|
|
2816
|
-
});
|
|
3072
|
+
const batch = polledPasses.get(batchId) ?? { id: batchId, status: "timeout" };
|
|
2817
3073
|
if (batch.status === "completed") {
|
|
2818
3074
|
const usage = (batch.usage ?? {});
|
|
2819
3075
|
const cost = typeof usage.cost === "number" ? usage.cost : undefined;
|
|
@@ -2846,7 +3102,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2846
3102
|
// A "timeout" is deliberately left at "submitted": the batch is
|
|
2847
3103
|
// still running server-side and has already been paid for, so a
|
|
2848
3104
|
// later collect should claim its result rather than discard it.
|
|
2849
|
-
await
|
|
3105
|
+
await persist();
|
|
2850
3106
|
}
|
|
2851
3107
|
}
|
|
2852
3108
|
}
|
|
@@ -2856,7 +3112,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2856
3112
|
});
|
|
2857
3113
|
run.status = terminal ? (resultCount > 0 ? "completed" : "failed") : "partial";
|
|
2858
3114
|
run.totalCost = totalCost;
|
|
2859
|
-
await
|
|
3115
|
+
await persist();
|
|
2860
3116
|
await writeFile(join(runDir, "run-meta.json"), `${JSON.stringify({
|
|
2861
3117
|
experimental: true,
|
|
2862
3118
|
method: "Broad-Side (OpenRouter Batch API)",
|
|
@@ -2888,6 +3144,7 @@ export async function runBroadsideCollect(cwd, apiKey, opts = {}) {
|
|
|
2888
3144
|
resultCount,
|
|
2889
3145
|
truncatedCount,
|
|
2890
3146
|
retriedCount,
|
|
3147
|
+
...(retryElsewhere && { retryElsewhere: true }),
|
|
2891
3148
|
lensOutcomes,
|
|
2892
3149
|
synthesis: run.synthesis,
|
|
2893
3150
|
triage: run.triage,
|
|
@@ -2959,7 +3216,7 @@ function parseSynthesisTopFindings(content) {
|
|
|
2959
3216
|
}
|
|
2960
3217
|
}
|
|
2961
3218
|
// ---------- formatting helpers for tool output ----------
|
|
2962
|
-
function describeIncrementalFallback(reason) {
|
|
3219
|
+
export function describeIncrementalFallback(reason) {
|
|
2963
3220
|
switch (reason) {
|
|
2964
3221
|
case "dirty-worktree":
|
|
2965
3222
|
return "the working tree has uncommitted changes, so there is no committed state to diff against";
|
|
@@ -2993,8 +3250,13 @@ export function estimateSubmitText(result, lenses) {
|
|
|
2993
3250
|
const override = entry.model ? ` on ${entry.model}` : "";
|
|
2994
3251
|
// A rejected lens says why: the message is the only way to tell a
|
|
2995
3252
|
// catalog id with no batch endpoint from a full job quota, and both
|
|
2996
|
-
// used to read as a bare "rejected".
|
|
2997
|
-
|
|
3253
|
+
// used to read as a bare "rejected". A skipped lens names the globs
|
|
3254
|
+
// that matched nothing.
|
|
3255
|
+
const reason = !entry.batchId && entry.error
|
|
3256
|
+
? ` — ${explainBatchError(entry.error)}`
|
|
3257
|
+
: !entry.batchId && entry.reason
|
|
3258
|
+
? ` — ${entry.reason}`
|
|
3259
|
+
: "";
|
|
2998
3260
|
lines.push(` ${lens.name}: ${status} (${entry.requests} request(s), ~$${entry.estimatedCost.toFixed(4)})${override}${reason}`);
|
|
2999
3261
|
}
|
|
3000
3262
|
if (result.repo) {
|
|
@@ -3150,9 +3412,24 @@ export function collectResultText(result) {
|
|
|
3150
3412
|
if (result.retriedCount > 0) {
|
|
3151
3413
|
lines.push(` ↻ ${result.retriedCount} truncated result(s) recovered by re-submission with a doubled output cap.`);
|
|
3152
3414
|
}
|
|
3415
|
+
if (result.retryElsewhere) {
|
|
3416
|
+
lines.push(" ↻ The truncation retry is in flight in another collect on this run; collect again for its result.");
|
|
3417
|
+
}
|
|
3153
3418
|
if (result.truncatedCount > 0) {
|
|
3154
3419
|
lines.push(` ⚠ ${result.truncatedCount} result(s) still truncated after retry — their modules are unscouted, not clean.`);
|
|
3155
3420
|
}
|
|
3421
|
+
// A pass still in flight or retired must appear: a run reported
|
|
3422
|
+
// "completed" with no synthesis line read as "no synthesis was run",
|
|
3423
|
+
// when the batch was running and a later collect would have claimed it
|
|
3424
|
+
// (0.22.1 live run — the collect's wait ran out during the pass).
|
|
3425
|
+
const passInFlight = (kind, entry) => {
|
|
3426
|
+
if (entry.status === "submitted") {
|
|
3427
|
+
lines.push(` ${kind}: ${entry.batchId ? "still running" : "in flight in another collect"} — collect again for its result.`);
|
|
3428
|
+
}
|
|
3429
|
+
else if (entry.status === "failed") {
|
|
3430
|
+
lines.push(` ${kind}: failed${entry.error ? ` — ${explainBatchError(entry.error)}` : ""}`);
|
|
3431
|
+
}
|
|
3432
|
+
};
|
|
3156
3433
|
if (result.synthesis.status === "completed") {
|
|
3157
3434
|
lines.push(` synthesis: completed, $${(result.synthesis.cost ?? 0).toFixed(6)}`);
|
|
3158
3435
|
if (result.topFindings.length > 0) {
|
|
@@ -3162,6 +3439,9 @@ export function collectResultText(result) {
|
|
|
3162
3439
|
}
|
|
3163
3440
|
}
|
|
3164
3441
|
}
|
|
3442
|
+
else {
|
|
3443
|
+
passInFlight("synthesis", result.synthesis);
|
|
3444
|
+
}
|
|
3165
3445
|
if (result.triage.status === "completed") {
|
|
3166
3446
|
lines.push(` triage: completed, $${(result.triage.cost ?? 0).toFixed(6)}`);
|
|
3167
3447
|
if (result.topTriageItems.length > 0) {
|
|
@@ -3172,12 +3452,28 @@ export function collectResultText(result) {
|
|
|
3172
3452
|
}
|
|
3173
3453
|
}
|
|
3174
3454
|
}
|
|
3175
|
-
else
|
|
3176
|
-
|
|
3455
|
+
else {
|
|
3456
|
+
passInFlight("triage", result.triage);
|
|
3177
3457
|
}
|
|
3178
3458
|
lines.push("", "Disclaimer: Broad-Side findings are unverified scouting signals from a batch model, not validated claims.");
|
|
3179
3459
|
return lines.join("\n");
|
|
3180
3460
|
}
|
|
3461
|
+
/**
|
|
3462
|
+
* An `onStatus` callback that appends one line to `lines` per *change* of a
|
|
3463
|
+
* lens's polled status. Every poll used to append a line, so a four-minute
|
|
3464
|
+
* wait returned twenty-six identical "in_progress (0/1)" lines per lens
|
|
3465
|
+
* before the result (0.22.0 live run).
|
|
3466
|
+
*/
|
|
3467
|
+
export function statusLineWriter(lines) {
|
|
3468
|
+
const last = new Map();
|
|
3469
|
+
return (lensId, status, counts) => {
|
|
3470
|
+
const line = ` ${lensId}: ${status} (${counts.completed ?? 0}/${counts.total ?? "?"})`;
|
|
3471
|
+
if (last.get(lensId) === line)
|
|
3472
|
+
return;
|
|
3473
|
+
last.set(lensId, line);
|
|
3474
|
+
lines.push(line);
|
|
3475
|
+
};
|
|
3476
|
+
}
|
|
3181
3477
|
export function statusText(state) {
|
|
3182
3478
|
if (state.runs.length === 0) {
|
|
3183
3479
|
return "No Broad-Side runs recorded. Call codecarto_broadside with action 'submit' first.";
|
|
@@ -3195,7 +3491,8 @@ export function statusText(state) {
|
|
|
3195
3491
|
const entry = run.batches[lensId];
|
|
3196
3492
|
if (!entry)
|
|
3197
3493
|
continue;
|
|
3198
|
-
lines.push(` ${lensId}: ${entry.status}${entry.batchId ? ` (${entry.batchId})` : ""}${entry.cost !== undefined ? `, $${entry.cost.toFixed(6)}` : ""}`
|
|
3494
|
+
lines.push(` ${lensId}: ${entry.status}${entry.batchId ? ` (${entry.batchId})` : ""}${entry.cost !== undefined ? `, $${entry.cost.toFixed(6)}` : ""}` +
|
|
3495
|
+
(entry.status === "skipped" && entry.reason ? ` — ${entry.reason}` : ""));
|
|
3199
3496
|
}
|
|
3200
3497
|
lines.push(` synthesis: ${run.synthesis.status}`);
|
|
3201
3498
|
lines.push(` triage: ${run.triage?.status ?? "pending"}`);
|
|
@@ -9,7 +9,7 @@ import { parseNextFlags } from "./next-flags.js";
|
|
|
9
9
|
import { buildPiGuideMessage } from "./guide-framing.js";
|
|
10
10
|
import { isCtxLive, notifyCtx } from "./notify.js";
|
|
11
11
|
import { phaseCompactionExtension } from "./phase-compaction.js";
|
|
12
|
-
import { applyAmendment, buildPhasePrompt, buildSkillPrompt, buildValidationSummary, backupWorkspaceState, canonicalPath, copyPackagedWorkspace, computePerPhaseTotals, computeTotals, ConfidentialityMismatchError, createEmptyStatus, DEFAULT_PIPELINE_PATH, describeScaffoldStaleness, deriveSlug, discoverLibrary, describeDanglingCarryForward, describeMissingCompletedOutputs, describeStuckPipeline, expandTilde, getPipelineLabel, getWorkspaceState, isWithinPath, resolveExistingPrefix, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideConfigError, defaultBroadsideConfig, BroadsideCancelledError, broadsideDirFor, collectResultText, estimateSubmitText, getLens, listAmendmentNames, listBatchModels, listGuideTopics, listScaffoldRefreshFiles, listMissingCompletedOutputs, listSkillNames, resolvePipelineOutcome, resolveSkillName, loadAmendmentFile, loadBroadsideConfig, modelsText, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, statusText, describeConfigProblems, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, packagedWorkspaceDir, pathExists, PACKAGE_VERSION, readBroadsideSkill, readGuide, refreshScaffold, PhasePreflightError, PIPELINE_ALIASES, publishEntry, resolvePhase, resolvePipelineChoice, resolvePublishSourceRepo, SourceRepoMismatchError, runPhasePreflight, SCAFFOLD_REFRESH_PROTECTED, seedOrchestratorFiles, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, writeDashboard, } from "../../core/index.js";
|
|
12
|
+
import { applyAmendment, buildPhasePrompt, buildSkillPrompt, buildValidationSummary, backupWorkspaceState, canonicalPath, copyPackagedWorkspace, computePerPhaseTotals, computeTotals, ConfidentialityMismatchError, createEmptyStatus, DEFAULT_PIPELINE_PATH, describeScaffoldStaleness, deriveSlug, discoverLibrary, describeDanglingCarryForward, describeMissingCompletedOutputs, describeStuckPipeline, expandTilde, getPipelineLabel, getWorkspaceState, isWithinPath, resolveExistingPrefix, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideConfigError, defaultBroadsideConfig, BroadsideCancelledError, broadsideDirFor, collectResultText, estimateSubmitText, getLens, listAmendmentNames, listBatchModels, listGuideTopics, listScaffoldRefreshFiles, listMissingCompletedOutputs, listSkillNames, resolvePipelineOutcome, resolveSkillName, loadAmendmentFile, loadBroadsideConfig, modelsText, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, statusText, describeConfigProblems, describeIncrementalFallback, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, packagedWorkspaceDir, pathExists, PACKAGE_VERSION, readBroadsideSkill, readGuide, refreshScaffold, PhasePreflightError, PIPELINE_ALIASES, publishEntry, resolvePhase, resolvePipelineChoice, resolvePublishSourceRepo, SourceRepoMismatchError, runPhasePreflight, SCAFFOLD_REFRESH_PROTECTED, seedOrchestratorFiles, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, writeDashboard, } from "../../core/index.js";
|
|
13
13
|
import { initLibrary } from "../../core/library.js";
|
|
14
14
|
import { resolveUserConfigPath } from "../../core/orchestrator-config.js";
|
|
15
15
|
const STATUS_WIDGET_ID = "codecarto-widget";
|
|
@@ -144,11 +144,14 @@ function describeBroadsideEstimate(estimate) {
|
|
|
144
144
|
? `This EXCEEDS the configured max_cost of $${estimate.maxCost.toFixed(2)}. Approving here overrides it for this run.`
|
|
145
145
|
: `Within the configured max_cost of $${estimate.maxCost.toFixed(2)}.`);
|
|
146
146
|
}
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
147
|
+
// Only a requested incremental run has anything to say here. The dialog
|
|
148
|
+
// used to print "Incremental was requested but the tree is dirty" on every
|
|
149
|
+
// dirty tree, requested or not — a full scan that nobody asked to shrink
|
|
150
|
+
// read as a fallback.
|
|
151
|
+
if (estimate.incremental?.requested) {
|
|
152
|
+
lines.push(estimate.incremental.applied
|
|
153
|
+
? `Incremental: only modules changed since ${(estimate.baseHead ?? "").slice(0, 8)} are included.`
|
|
154
|
+
: `Incremental was requested but NOT applied — ${describeIncrementalFallback(estimate.incremental.reason)}. This is a full scan.`);
|
|
152
155
|
}
|
|
153
156
|
lines.push("", "The estimate is a pre-flight prediction from file sizes; OpenRouter bills actual usage.");
|
|
154
157
|
return lines.join("\n");
|
|
@@ -1114,6 +1117,9 @@ export default function codeCartographerExtension(pi) {
|
|
|
1114
1117
|
const lenses = flags.lenses.length > 0 ? flags.lenses : config.defaultLenses;
|
|
1115
1118
|
renderProgress("Slicing the repository and pricing the run…");
|
|
1116
1119
|
let submit;
|
|
1120
|
+
// Set when a headless run was refused over max_cost, so the cancel
|
|
1121
|
+
// message says so instead of reading as a user's "no".
|
|
1122
|
+
let headlessRefusal = null;
|
|
1117
1123
|
try {
|
|
1118
1124
|
submit = await runBroadsideSubmit(ctx.cwd, apiKey, {
|
|
1119
1125
|
lenses,
|
|
@@ -1128,14 +1134,32 @@ export default function codeCartographerExtension(pi) {
|
|
|
1128
1134
|
incremental: flags.incremental ?? config.incremental,
|
|
1129
1135
|
// Pi can ask, so it asks instead of refusing over max_cost the
|
|
1130
1136
|
// way MCP has to. An approval here IS the force flag.
|
|
1131
|
-
confirm: (estimate) =>
|
|
1137
|
+
confirm: (estimate) => {
|
|
1138
|
+
if (ctx.hasUI) {
|
|
1139
|
+
return ctx.ui.confirm(`Broad-Side will spend about $${estimate.totalCost.toFixed(4)}`, describeBroadsideEstimate(estimate));
|
|
1140
|
+
}
|
|
1141
|
+
// No dialog under `pi -p`: the stub answered "no" to every
|
|
1142
|
+
// estimate, so a headless submit could never fire. Behave as
|
|
1143
|
+
// the MCP surface does — an estimate within max_cost is
|
|
1144
|
+
// approved by the cap itself; one over it is refused, since
|
|
1145
|
+
// nobody is here to say yes — and print the breakdown either
|
|
1146
|
+
// way, because the dialog was the only place it showed.
|
|
1147
|
+
notifyCtx(ctx, describeBroadsideEstimate(estimate), "info");
|
|
1148
|
+
if (estimate.exceedsLimit) {
|
|
1149
|
+
headlessRefusal =
|
|
1150
|
+
`Broad-Side refused: the estimate ~$${estimate.totalCost.toFixed(4)} exceeds max_cost $${estimate.maxCost.toFixed(2)} ` +
|
|
1151
|
+
"and there is no dialog to approve it in a headless run. Raise --max-cost (0 for no limit) or run interactively.";
|
|
1152
|
+
return false;
|
|
1153
|
+
}
|
|
1154
|
+
return true;
|
|
1155
|
+
},
|
|
1132
1156
|
});
|
|
1133
1157
|
}
|
|
1134
1158
|
catch (error) {
|
|
1135
1159
|
if (ctx.hasUI)
|
|
1136
1160
|
ctx.ui.setWidget(BROADSIDE_WIDGET_ID, undefined);
|
|
1137
1161
|
if (error instanceof BroadsideCancelledError) {
|
|
1138
|
-
notifyCtx(ctx, "Broad-Side cancelled. Nothing was submitted.", "info");
|
|
1162
|
+
notifyCtx(ctx, headlessRefusal ?? "Broad-Side cancelled. Nothing was submitted.", headlessRefusal ? "error" : "info");
|
|
1139
1163
|
return;
|
|
1140
1164
|
}
|
|
1141
1165
|
notifyCtx(ctx, `Broad-Side submit failed: ${error instanceof Error ? error.message : String(error)}`, "error");
|
|
@@ -17,7 +17,7 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
|
|
|
17
17
|
import { CallToolRequestSchema, ErrorCode, ListToolsRequestSchema, McpError, } from "@modelcontextprotocol/sdk/types.js";
|
|
18
18
|
import { mkdir, readFile, readdir, rename, writeFile } from "node:fs/promises";
|
|
19
19
|
import { basename, isAbsolute, join } from "node:path";
|
|
20
|
-
import { buildPhasePrompt, buildSkillPrompt, buildValidationSummary, BROADSIDE_DIR, broadsideDirFor, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideConfigError, defaultBroadsideConfig, backupWorkspaceState, canonicalPath, copyPackagedWorkspace, collectResultText, completeValidatedPhase, computePerPhaseTotals, computeTotals, createEmptyStatus, DEFAULT_PIPELINE_PATH, deriveSlug, discoverLibrary, describeScaffoldStaleness, detectProvenanceConflicts, estimateSubmitText, getLens, describeDanglingCarryForward, describeMissingCompletedOutputs, describeStuckPipeline, getPipelineLabel, getWorkspaceState, isValidSlug, isWithinPathResolved, listBatchModels, listEntries, listGuideTopics, loadBroadsideConfig, modelsText, readGuide, listMissingCompletedOutputs, listSkillNames, resolvePipelineOutcome, resolveSkillName, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, PACKAGE_VERSION, packagedWorkspaceDir, pathExists, PhasePreflightError, previewPublishVersion, publishEntry, reindex as libraryReindex, refreshScaffold, resolvePhase, resolvePipelineChoice, readBroadsideSkill, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, seedOrchestratorFiles, statusText, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, writeDashboard, } from "../core/index.js";
|
|
20
|
+
import { buildPhasePrompt, buildSkillPrompt, buildValidationSummary, BROADSIDE_DIR, broadsideDirFor, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideConfigError, defaultBroadsideConfig, backupWorkspaceState, canonicalPath, copyPackagedWorkspace, collectResultText, completeValidatedPhase, computePerPhaseTotals, computeTotals, createEmptyStatus, DEFAULT_PIPELINE_PATH, deriveSlug, discoverLibrary, describeScaffoldStaleness, detectProvenanceConflicts, estimateSubmitText, getLens, describeDanglingCarryForward, describeMissingCompletedOutputs, describeStuckPipeline, getPipelineLabel, getWorkspaceState, isValidSlug, isWithinPathResolved, listBatchModels, listEntries, listGuideTopics, loadBroadsideConfig, modelsText, readGuide, listMissingCompletedOutputs, listSkillNames, resolvePipelineOutcome, resolveSkillName, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, PACKAGE_VERSION, packagedWorkspaceDir, pathExists, PhasePreflightError, previewPublishVersion, publishEntry, reindex as libraryReindex, refreshScaffold, resolvePhase, resolvePipelineChoice, readBroadsideSkill, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, seedOrchestratorFiles, statusLineWriter, statusText, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, writeDashboard, } from "../core/index.js";
|
|
21
21
|
import { applyAmendment } from "../core/amendment.js";
|
|
22
22
|
import { appendUsageRun } from "../core/usage.js";
|
|
23
23
|
import { initLibrary } from "../core/library.js";
|
|
@@ -1151,7 +1151,11 @@ export async function handleBroadside(args) {
|
|
|
1151
1151
|
includeSynthesis,
|
|
1152
1152
|
includeTriage,
|
|
1153
1153
|
retryTruncated,
|
|
1154
|
-
|
|
1154
|
+
signal: serverLifetime?.signal,
|
|
1155
|
+
// One line per *change* of a lens's status. Every poll used to
|
|
1156
|
+
// append a line, so a four-minute wait returned twenty-six
|
|
1157
|
+
// "in_progress (0/1)" lines before the result (0.22.0 live run).
|
|
1158
|
+
onStatus: statusLineWriter(lines),
|
|
1155
1159
|
}).catch((error) => {
|
|
1156
1160
|
// The `collect` action normalizes this same call; without it here,
|
|
1157
1161
|
// a failure during submit-with-wait reached the client as an
|
|
@@ -1176,6 +1180,7 @@ export async function handleBroadside(args) {
|
|
|
1176
1180
|
includeSynthesis,
|
|
1177
1181
|
includeTriage,
|
|
1178
1182
|
retryTruncated,
|
|
1183
|
+
signal: serverLifetime?.signal,
|
|
1179
1184
|
...(runId && { runId }),
|
|
1180
1185
|
}).catch((error) => {
|
|
1181
1186
|
throw new McpError(ErrorCode.InvalidRequest, error instanceof Error ? error.message : String(error));
|
|
@@ -1515,7 +1520,7 @@ const TOOLS = [
|
|
|
1515
1520
|
},
|
|
1516
1521
|
wait_seconds: {
|
|
1517
1522
|
type: "number",
|
|
1518
|
-
description: "For submit: after submitting, poll up to this many seconds before returning. For collect: poll up to this many seconds before returning with partial state. 0 polls each in-flight batch once and returns without waiting. Falls back to wait_seconds in .codecarto/broadside/config.yaml (default 0).",
|
|
1523
|
+
description: "For submit: after submitting, poll up to this many seconds before returning. For collect: poll up to this many seconds before returning with partial state. 0 polls each in-flight batch once and returns without waiting. Falls back to wait_seconds in .codecarto/broadside/config.yaml (default 0). The wait is also bounded by the host's own tool-call timeout: if the host gives up first, the server stops polling and submits nothing further, the batches keep running server-side, and the next collect claims them — so prefer submit, then collect later, over a wait longer than the host allows.",
|
|
1519
1524
|
},
|
|
1520
1525
|
include_synthesis: {
|
|
1521
1526
|
type: "boolean",
|
|
@@ -1614,8 +1619,27 @@ export function buildServer() {
|
|
|
1614
1619
|
});
|
|
1615
1620
|
return server;
|
|
1616
1621
|
}
|
|
1622
|
+
/**
|
|
1623
|
+
* Fires when the stdio client goes away, so a Broad-Side wait that outlived
|
|
1624
|
+
* the request that asked for it stops polling and submits nothing further
|
|
1625
|
+
* (#322). Batches already accepted keep running server-side; the next collect
|
|
1626
|
+
* claims them. Set only by {@link startStdioServer}; handlers driven directly
|
|
1627
|
+
* (tests, in-process callers) see no signal.
|
|
1628
|
+
*/
|
|
1629
|
+
let serverLifetime = null;
|
|
1617
1630
|
export async function startStdioServer() {
|
|
1618
1631
|
const server = buildServer();
|
|
1619
1632
|
const transport = new StdioServerTransport();
|
|
1633
|
+
serverLifetime = new AbortController();
|
|
1634
|
+
const lifetime = serverLifetime;
|
|
1635
|
+
server.onclose = () => lifetime.abort();
|
|
1636
|
+
// The SDK's stdio transport listens for stdin `data` and `error` only — it
|
|
1637
|
+
// never sees the end of the stream — so a client that exits mid-request
|
|
1638
|
+
// leaves the server polling with nobody to answer to (#322, observed: a
|
|
1639
|
+
// server outlived its client by fifteen minutes and submitted two paid
|
|
1640
|
+
// post-passes on its own). End of stdin is the client going away.
|
|
1641
|
+
const gone = () => lifetime.abort();
|
|
1642
|
+
process.stdin.once("end", gone);
|
|
1643
|
+
process.stdin.once("close", gone);
|
|
1620
1644
|
await server.connect(transport);
|
|
1621
1645
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "codecartographer-pi",
|
|
3
|
-
"version": "0.22.
|
|
3
|
+
"version": "0.22.2",
|
|
4
4
|
"mcpName": "io.github.HuginnIndustries/codecartographer",
|
|
5
5
|
"description": "Turn an unfamiliar codebase into a validated reimplementation spec, then synthesize confirmed specs and a product vision into a traceable plan.",
|
|
6
6
|
"type": "module",
|