codecartographer-pi 0.19.1 → 0.19.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +11 -0
- package/.codecarto/broadside/config.yaml +41 -2
- package/.codecarto/templates/phase-handoff.yaml +2 -1
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/dist/core/broadside.d.ts +45 -1
- package/dist/core/broadside.js +55 -2
- package/dist/core/yaml.js +58 -3
- package/dist/extensions/codecarto/agent-rewriter.js +2 -0
- package/dist/extensions/codecarto/agent-runner.js +2 -0
- package/dist/extensions/codecarto/child-model-runtime.d.ts +2 -0
- package/dist/extensions/codecarto/child-model-runtime.js +44 -0
- package/dist/extensions/codecarto/dashboard-narrator.js +2 -0
- package/dist/extensions/codecarto/guide-framing.d.ts +11 -0
- package/dist/extensions/codecarto/guide-framing.js +55 -0
- package/dist/extensions/codecarto/index.js +24 -5
- package/package.json +1 -1
|
@@ -90,6 +90,17 @@ Collect runs two cross-lens post-passes by default: **synthesis** (the
|
|
|
90
90
|
executive report) and **triage** (the prioritized work order). Pass
|
|
91
91
|
`include_synthesis: false` or `include_triage: false` on collect to skip one.
|
|
92
92
|
|
|
93
|
+
Two caveats apply to any model you pick. The `models` action lists every id
|
|
94
|
+
OpenRouter advertises a `:batch` variant for, and many of those variants do not
|
|
95
|
+
exist — submitting one returns `does not have a :batch endpoint`, with nothing in
|
|
96
|
+
the catalog to distinguish it beforehand. A rejected batch costs nothing, so
|
|
97
|
+
probe a candidate on a single lens first. And reasoning competes with the answer for
|
|
98
|
+
`max_tokens`: Broad-Side caps thinking at a quarter of each lens's output budget
|
|
99
|
+
so three quarters remain for the JSON, which is the split the cost estimate
|
|
100
|
+
already assumes. It caps rather than disables because some endpoints refuse to
|
|
101
|
+
be switched off entirely. Override with `reasoning:` in `config.yaml` only
|
|
102
|
+
alongside a raised output budget.
|
|
103
|
+
|
|
93
104
|
Lenses do not all have to run on the same model. `lens_models` in `config.yaml`
|
|
94
105
|
routes individual lenses to their own batch model — the usual reason being that
|
|
95
106
|
a stronger model changes security and defect findings more than it changes an
|
|
@@ -46,9 +46,48 @@
|
|
|
46
46
|
# and picking one for you would spend your money on our guess. Compare
|
|
47
47
|
# candidates with the `models` action first.
|
|
48
48
|
#
|
|
49
|
+
# Two things to check before committing to a model, both learned the hard way:
|
|
50
|
+
#
|
|
51
|
+
# 1. The `models` action lists whatever OpenRouter advertises a `:batch`
|
|
52
|
+
# variant for, and a good number of those variants do not actually exist —
|
|
53
|
+
# submitting one comes back `Model '<id>' does not have a :batch endpoint.`
|
|
54
|
+
# Nothing in the catalog distinguishes them. Every Anthropic and OpenAI
|
|
55
|
+
# batch id tried so far is rejected this way; Google and DeepSeek work.
|
|
56
|
+
# A rejected batch costs nothing, so probe a candidate on one lens before
|
|
57
|
+
# relying on it.
|
|
58
|
+
# 2. A reasoning-capable model spends its output budget thinking, and the
|
|
59
|
+
# thinking is billed at the full output rate. See `reasoning:` below.
|
|
60
|
+
#
|
|
49
61
|
# lens_models:
|
|
50
|
-
# security:
|
|
51
|
-
# defect:
|
|
62
|
+
# security: deepseek/deepseek-v4-pro-0813:batch
|
|
63
|
+
# defect: deepseek/deepseek-v4-pro-0813:batch
|
|
64
|
+
|
|
65
|
+
# Reasoning control, sent on every lens request.
|
|
66
|
+
#
|
|
67
|
+
# By default Broad-Side caps thinking at a quarter of the lens's output budget,
|
|
68
|
+
# leaving the other three quarters for the answer — which is exactly the split
|
|
69
|
+
# the cost estimate already assumes.
|
|
70
|
+
#
|
|
71
|
+
# The cap exists because reasoning competes with the answer for `max_tokens`.
|
|
72
|
+
# One measured run spent 5,758 of a 6,000-token budget thinking and left ~230
|
|
73
|
+
# tokens for the JSON, which truncated mid-structure on 11 of 13 slices; those
|
|
74
|
+
# tokens bill at the full output rate, so it paid for ~6,000 output tokens per
|
|
75
|
+
# slice to receive ~230 usable ones. The shipped default model does the same
|
|
76
|
+
# thing less consistently — reasoning from 0 to 5,757 tokens across 13 slices,
|
|
77
|
+
# three of them cut off — so this is not something only exotic models do.
|
|
78
|
+
#
|
|
79
|
+
# It is a cap rather than an off switch on purpose. Some endpoints refuse to be
|
|
80
|
+
# switched off: `google/gemini-3.8-flash:batch` rejects the entire batch with
|
|
81
|
+
# "Reasoning is mandatory for this endpoint and cannot be disabled", which turns
|
|
82
|
+
# a partial result into none at all. Capping works either way.
|
|
83
|
+
#
|
|
84
|
+
# Override only with a raised lens output budget, or the JSON truncates exactly
|
|
85
|
+
# as above.
|
|
86
|
+
#
|
|
87
|
+
# reasoning:
|
|
88
|
+
# effort: low # minimal | low | medium | high
|
|
89
|
+
# max_tokens: 2000 # or set the thinking budget directly
|
|
90
|
+
# enabled: false # only where the provider allows it
|
|
52
91
|
|
|
53
92
|
# Approximate run expense limit in USD (0 = no limit). Before submitting,
|
|
54
93
|
# Broad-Side estimates the run cost from the collected file sizes and the
|
|
@@ -39,7 +39,8 @@ decisions: []
|
|
|
39
39
|
# evidence: <optional: where the pattern showed up>
|
|
40
40
|
proposed_conventions: []
|
|
41
41
|
closeout_summary: ""
|
|
42
|
-
# Optional full closeout Markdown.
|
|
42
|
+
# Optional full closeout Markdown. Block scalars are supported — literal (|, |-,
|
|
43
|
+
# |+) to keep line breaks, folded (>, >-, >+) to join wrapped prose into one line.
|
|
43
44
|
closeout_content: |-
|
|
44
45
|
# Closeout — <phase-id>
|
|
45
46
|
|
package/dist/core/broadside.d.ts
CHANGED
|
@@ -86,6 +86,46 @@ export type FileSlice = {
|
|
|
86
86
|
/** Repo-relative paths of the files folded into this slice. */
|
|
87
87
|
files: string[];
|
|
88
88
|
};
|
|
89
|
+
/**
|
|
90
|
+
* OpenRouter's unified `reasoning` control, as sent on a lens request.
|
|
91
|
+
*
|
|
92
|
+
* Left unsent, each model applies its own default — which is how a
|
|
93
|
+
* reasoning-capable model came to spend 5,758 of a 6,000-token output budget
|
|
94
|
+
* thinking, leaving ~230 tokens for JSON that then truncated mid-structure. The
|
|
95
|
+
* thinking is billed at the full *output* rate, so the run paid for roughly
|
|
96
|
+
* 6,000 output tokens per slice to receive 230 usable ones.
|
|
97
|
+
*/
|
|
98
|
+
export type BroadsideReasoning = {
|
|
99
|
+
enabled?: boolean;
|
|
100
|
+
effort?: "minimal" | "low" | "medium" | "high";
|
|
101
|
+
max_tokens?: number;
|
|
102
|
+
};
|
|
103
|
+
/**
|
|
104
|
+
* The share of a lens's output budget reasoning may spend.
|
|
105
|
+
*
|
|
106
|
+
* `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
|
|
107
|
+
* at the remaining quarter makes that assumption true by construction and
|
|
108
|
+
* guarantees the answer has room. A floor keeps the cap sane for a small lens.
|
|
109
|
+
*/
|
|
110
|
+
export declare const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
|
|
111
|
+
export declare const BROADSIDE_MIN_REASONING_TOKENS = 512;
|
|
112
|
+
/**
|
|
113
|
+
* Cap reasoning for a lens request — deliberately a cap, not an off switch.
|
|
114
|
+
*
|
|
115
|
+
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
116
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
117
|
+
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
118
|
+
* or not a provider allows reasoning to be switched off.
|
|
119
|
+
*
|
|
120
|
+
* The failure this prevents is the budget being spent thinking rather than
|
|
121
|
+
* answering. Measured on one run: 5,758 of a 6,000-token budget went to
|
|
122
|
+
* reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
|
|
123
|
+
* those tokens bill at the full output rate. The shipped default model does the
|
|
124
|
+
* same thing less consistently (reasoning tokens from 0 to 5,757 across 13
|
|
125
|
+
* slices, three of them cut off at `finish_reason: length`), so this is not a
|
|
126
|
+
* multi-model concern.
|
|
127
|
+
*/
|
|
128
|
+
export declare function defaultReasoningFor(maxTokens: number): BroadsideReasoning;
|
|
89
129
|
export type BatchRequest = {
|
|
90
130
|
custom_id: string;
|
|
91
131
|
body: {
|
|
@@ -99,6 +139,7 @@ export type BatchRequest = {
|
|
|
99
139
|
json_schema: JsonSchemaDef;
|
|
100
140
|
};
|
|
101
141
|
max_tokens: number;
|
|
142
|
+
reasoning?: BroadsideReasoning;
|
|
102
143
|
};
|
|
103
144
|
};
|
|
104
145
|
export type BatchTerminalStatus = "completed" | "failed" | "expired" | "cancelled";
|
|
@@ -185,6 +226,8 @@ export type BroadsideConfig = {
|
|
|
185
226
|
* trade-off is not the same for every lens.
|
|
186
227
|
*/
|
|
187
228
|
lensModels: Partial<Record<BroadsideLensId, string>>;
|
|
229
|
+
/** Overrides every lens's reasoning setting when present. */
|
|
230
|
+
reasoning: BroadsideReasoning | null;
|
|
188
231
|
/**
|
|
189
232
|
* Repo defaults for the per-call run knobs. Each mirrors a tool parameter
|
|
190
233
|
* of the same name; an explicit parameter always wins. They live here so a
|
|
@@ -304,6 +347,7 @@ type LensDefinition = {
|
|
|
304
347
|
sliceBy: "none" | "directory" | "auto";
|
|
305
348
|
maxChars: number;
|
|
306
349
|
maxTokens: number;
|
|
350
|
+
reasoning?: BroadsideReasoning;
|
|
307
351
|
skipTestFiles?: boolean;
|
|
308
352
|
globsFor: (info: RepoInfo) => string[];
|
|
309
353
|
systemPrompt: (info: RepoInfo) => string;
|
|
@@ -313,7 +357,7 @@ export declare function getLens(lensId: BroadsideLensId): LensDefinition;
|
|
|
313
357
|
export declare function listLenses(): LensDefinition[];
|
|
314
358
|
export declare function collectRepoInfo(targetDir: string): Promise<RepoInfo>;
|
|
315
359
|
export declare function gatherSlices(targetDir: string, lens: LensDefinition, info: RepoInfo): Promise<FileSlice[]>;
|
|
316
|
-
export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number): BatchRequest;
|
|
360
|
+
export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number, reasoningOverride?: BroadsideReasoning): BatchRequest;
|
|
317
361
|
/**
|
|
318
362
|
* Pre-flight cost estimate for one lens.
|
|
319
363
|
*
|
package/dist/core/broadside.js
CHANGED
|
@@ -74,6 +74,34 @@ export const BROADSIDE_LENS_IDS = [
|
|
|
74
74
|
];
|
|
75
75
|
export const BROADSIDE_POLL_INTERVAL_MS = 15_000;
|
|
76
76
|
export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
|
|
77
|
+
/**
|
|
78
|
+
* The share of a lens's output budget reasoning may spend.
|
|
79
|
+
*
|
|
80
|
+
* `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
|
|
81
|
+
* at the remaining quarter makes that assumption true by construction and
|
|
82
|
+
* guarantees the answer has room. A floor keeps the cap sane for a small lens.
|
|
83
|
+
*/
|
|
84
|
+
export const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
|
|
85
|
+
export const BROADSIDE_MIN_REASONING_TOKENS = 512;
|
|
86
|
+
/**
|
|
87
|
+
* Cap reasoning for a lens request — deliberately a cap, not an off switch.
|
|
88
|
+
*
|
|
89
|
+
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
90
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
91
|
+
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
92
|
+
* or not a provider allows reasoning to be switched off.
|
|
93
|
+
*
|
|
94
|
+
* The failure this prevents is the budget being spent thinking rather than
|
|
95
|
+
* answering. Measured on one run: 5,758 of a 6,000-token budget went to
|
|
96
|
+
* reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
|
|
97
|
+
* those tokens bill at the full output rate. The shipped default model does the
|
|
98
|
+
* same thing less consistently (reasoning tokens from 0 to 5,757 across 13
|
|
99
|
+
* slices, three of them cut off at `finish_reason: length`), so this is not a
|
|
100
|
+
* multi-model concern.
|
|
101
|
+
*/
|
|
102
|
+
export function defaultReasoningFor(maxTokens) {
|
|
103
|
+
return { max_tokens: Math.max(BROADSIDE_MIN_REASONING_TOKENS, Math.floor(maxTokens * BROADSIDE_REASONING_BUDGET_FRACTION)) };
|
|
104
|
+
}
|
|
77
105
|
/** Thrown when a confirm hook declines a run. Nothing was submitted. */
|
|
78
106
|
export class BroadsideCancelledError extends Error {
|
|
79
107
|
constructor(message = "Broad-Side submission cancelled. Nothing was submitted.") {
|
|
@@ -1136,7 +1164,7 @@ async function sumFileSizes(targetDir, files) {
|
|
|
1136
1164
|
return total;
|
|
1137
1165
|
}
|
|
1138
1166
|
// ---------- request building ----------
|
|
1139
|
-
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride) {
|
|
1167
|
+
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride, reasoningOverride) {
|
|
1140
1168
|
const moduleTag = sanitizeId(slice.moduleName);
|
|
1141
1169
|
const customId = sliceCount > 1 ? `${lens.id}-${moduleTag}-${index + 1}` : `${lens.id}-${moduleTag}`;
|
|
1142
1170
|
return {
|
|
@@ -1149,6 +1177,9 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1149
1177
|
],
|
|
1150
1178
|
response_format: { type: "json_schema", json_schema: SCHEMAS[lens.schemaName] },
|
|
1151
1179
|
max_tokens: maxTokensOverride ?? lens.maxTokens,
|
|
1180
|
+
// Always sent, never inherited: an absent field means the model's
|
|
1181
|
+
// own default, and that default is what truncated the JSON.
|
|
1182
|
+
reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(maxTokensOverride ?? lens.maxTokens),
|
|
1152
1183
|
},
|
|
1153
1184
|
};
|
|
1154
1185
|
}
|
|
@@ -1297,6 +1328,24 @@ export async function persistBroadsideRun(broadsideDir, run) {
|
|
|
1297
1328
|
state.runs[index] = run;
|
|
1298
1329
|
});
|
|
1299
1330
|
}
|
|
1331
|
+
/** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
|
|
1332
|
+
function parseReasoningConfig(raw) {
|
|
1333
|
+
if (raw === false)
|
|
1334
|
+
return { enabled: false };
|
|
1335
|
+
if (raw === true)
|
|
1336
|
+
return { enabled: true };
|
|
1337
|
+
if (!raw || typeof raw !== "object")
|
|
1338
|
+
return null;
|
|
1339
|
+
const value = raw;
|
|
1340
|
+
const out = {};
|
|
1341
|
+
if (typeof value.enabled === "boolean")
|
|
1342
|
+
out.enabled = value.enabled;
|
|
1343
|
+
if (value.effort === "minimal" || value.effort === "low" || value.effort === "medium" || value.effort === "high")
|
|
1344
|
+
out.effort = value.effort;
|
|
1345
|
+
if (typeof value.max_tokens === "number" && value.max_tokens > 0)
|
|
1346
|
+
out.max_tokens = value.max_tokens;
|
|
1347
|
+
return Object.keys(out).length > 0 ? out : null;
|
|
1348
|
+
}
|
|
1300
1349
|
export async function loadBroadsideConfig(broadsideDir) {
|
|
1301
1350
|
const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
|
|
1302
1351
|
let raw = {};
|
|
@@ -1337,6 +1386,10 @@ export async function loadBroadsideConfig(broadsideDir) {
|
|
|
1337
1386
|
? { inputPerM: inputOverride, outputPerM: outputOverride }
|
|
1338
1387
|
: null,
|
|
1339
1388
|
lensModels,
|
|
1389
|
+
// An escape hatch, not a knob to reach for: a model whose reasoning is
|
|
1390
|
+
// worth paying for needs its lens maxTokens raised to cover both the
|
|
1391
|
+
// thinking and the answer, or the JSON truncates exactly as before.
|
|
1392
|
+
reasoning: parseReasoningConfig(raw.reasoning),
|
|
1340
1393
|
incremental: flag("incremental", false),
|
|
1341
1394
|
retryTruncated: flag("retry_truncated", true),
|
|
1342
1395
|
includeSynthesis: flag("include_synthesis", true),
|
|
@@ -1797,7 +1850,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1797
1850
|
const { lens, maxTokens, lensModel, lensOutputCap } = priced;
|
|
1798
1851
|
const lensId = lens.id;
|
|
1799
1852
|
const slices = slicesByLens.get(lensId) ?? [];
|
|
1800
|
-
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens));
|
|
1853
|
+
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens, config.reasoning ?? undefined));
|
|
1801
1854
|
for (const request of requests)
|
|
1802
1855
|
requestsByCustomId[request.custom_id] = request;
|
|
1803
1856
|
const entry = {
|
package/dist/core/yaml.js
CHANGED
|
@@ -105,6 +105,61 @@ export function parseYamlScalar(rawValue) {
|
|
|
105
105
|
}
|
|
106
106
|
return trimmed;
|
|
107
107
|
}
|
|
108
|
+
function parseBlockScalarHeader(rawValue) {
|
|
109
|
+
const match = /^([|>])([-+]?)$/.exec(rawValue);
|
|
110
|
+
if (!match)
|
|
111
|
+
return null;
|
|
112
|
+
return {
|
|
113
|
+
literal: match[1] === "|",
|
|
114
|
+
chomp: match[2] === "-" ? "strip" : match[2] === "+" ? "keep" : "clip",
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Fold a block scalar's lines per YAML's folding rules: a single line break
|
|
119
|
+
* between two content lines becomes a space, and a run of k blank lines becomes
|
|
120
|
+
* k newlines. Lines indented deeper than the block's own content indent are
|
|
121
|
+
* "more indented" and keep their breaks literally, which is what lets a folded
|
|
122
|
+
* block hold an indented snippet without it being flattened onto one line.
|
|
123
|
+
*/
|
|
124
|
+
function foldBlockLines(blockLines) {
|
|
125
|
+
let result = "";
|
|
126
|
+
let pendingBreaks = 0;
|
|
127
|
+
let started = false;
|
|
128
|
+
let previousMoreIndented = false;
|
|
129
|
+
for (const line of blockLines) {
|
|
130
|
+
if (line.trim() === "") {
|
|
131
|
+
pendingBreaks++;
|
|
132
|
+
continue;
|
|
133
|
+
}
|
|
134
|
+
const moreIndented = /^[ \t]/.test(line);
|
|
135
|
+
if (!started) {
|
|
136
|
+
result = line;
|
|
137
|
+
started = true;
|
|
138
|
+
previousMoreIndented = moreIndented;
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
if (pendingBreaks > 0) {
|
|
142
|
+
result += "\n".repeat(pendingBreaks);
|
|
143
|
+
pendingBreaks = 0;
|
|
144
|
+
}
|
|
145
|
+
else if (moreIndented || previousMoreIndented) {
|
|
146
|
+
result += "\n";
|
|
147
|
+
}
|
|
148
|
+
else {
|
|
149
|
+
result += " ";
|
|
150
|
+
}
|
|
151
|
+
result += line;
|
|
152
|
+
previousMoreIndented = moreIndented;
|
|
153
|
+
}
|
|
154
|
+
return started ? result + "\n".repeat(pendingBreaks) : "";
|
|
155
|
+
}
|
|
156
|
+
function applyBlockScalar(blockLines, header) {
|
|
157
|
+
const content = header.literal ? blockLines.join("\n") : foldBlockLines(blockLines);
|
|
158
|
+
if (header.chomp === "keep")
|
|
159
|
+
return content;
|
|
160
|
+
const stripped = content.replace(/\n+$/, "");
|
|
161
|
+
return header.chomp === "strip" ? stripped : `${stripped}\n`;
|
|
162
|
+
}
|
|
108
163
|
export function parseSimpleYaml(raw) {
|
|
109
164
|
const lines = raw.split(/\r?\n/);
|
|
110
165
|
let index = 0;
|
|
@@ -164,7 +219,8 @@ export function parseSimpleYaml(raw) {
|
|
|
164
219
|
throw new Error(`Duplicate YAML key: ${key} near line: ${line.trim()}`);
|
|
165
220
|
}
|
|
166
221
|
seen.add(key);
|
|
167
|
-
|
|
222
|
+
const blockHeader = parseBlockScalarHeader(rawValue);
|
|
223
|
+
if (blockHeader) {
|
|
168
224
|
const blockLines = [];
|
|
169
225
|
let contentIndent = null;
|
|
170
226
|
while (index < lines.length) {
|
|
@@ -181,8 +237,7 @@ export function parseSimpleYaml(raw) {
|
|
|
181
237
|
blockLines.push(blockLine.slice(Math.min(contentIndent, blockIndent)));
|
|
182
238
|
index++;
|
|
183
239
|
}
|
|
184
|
-
|
|
185
|
-
assign(key, rawValue === "|" ? `${content}\n` : content);
|
|
240
|
+
assign(key, applyBlockScalar(blockLines, blockHeader));
|
|
186
241
|
continue;
|
|
187
242
|
}
|
|
188
243
|
if (rawValue !== "") {
|
|
@@ -11,6 +11,7 @@ import { readFile, readdir } from "node:fs/promises";
|
|
|
11
11
|
import { join } from "node:path";
|
|
12
12
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
13
13
|
import { closeoutFileName, pathExists } from "../../core/index.js";
|
|
14
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
14
15
|
/** Closeout content over this many bytes is truncated before being passed to
|
|
15
16
|
* the rewriter. Keeps the orchestrator-side cost predictable. */
|
|
16
17
|
const CLOSEOUT_BYTE_BUDGET = 8000;
|
|
@@ -143,6 +144,7 @@ async function runRewriterOnce(ctx, prompt) {
|
|
|
143
144
|
const { session } = await createAgentSession({
|
|
144
145
|
cwd,
|
|
145
146
|
agentDir,
|
|
147
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
146
148
|
sessionManager: SessionManager.inMemory(cwd),
|
|
147
149
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
148
150
|
model: ctx.model,
|
|
@@ -13,6 +13,7 @@ import { access } from "node:fs/promises";
|
|
|
13
13
|
import { join, resolve } from "node:path";
|
|
14
14
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
15
15
|
import { canonicalPath, isWithinPath } from "../../core/index.js";
|
|
16
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
16
17
|
import { phaseCompactionExtension } from "./phase-compaction.js";
|
|
17
18
|
// Tools available to the phase sub-agent. Matches the codecarto interception
|
|
18
19
|
// allowlist (SAFE_TOOL_NAMES in extensions/codecarto/index.ts), minus bash.
|
|
@@ -104,6 +105,7 @@ export async function runPhase(ctx, prompt, callbacks = {}, options = {}, signal
|
|
|
104
105
|
const { session } = await createAgentSession({
|
|
105
106
|
cwd,
|
|
106
107
|
agentDir,
|
|
108
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
107
109
|
sessionManager,
|
|
108
110
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
109
111
|
model: ctx.model,
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// Model runtime for codecarto's child sessions (phase sub-agents, the
|
|
2
|
+
// next-phase rewriter, the dashboard narrator).
|
|
3
|
+
//
|
|
4
|
+
// All three child sessions load with `noExtensions: true` so a globally
|
|
5
|
+
// installed codecarto doesn't register its commands and tool guards twice
|
|
6
|
+
// inside its own sub-agent. That isolation has a side effect: providers
|
|
7
|
+
// registered by *other* global extensions via `pi.registerProvider()` — an
|
|
8
|
+
// Ollama Cloud bridge, a company gateway, any custom `streamSimple` provider —
|
|
9
|
+
// are registered onto the parent's ModelRuntime by the resource loader that
|
|
10
|
+
// loaded them. A child that builds a fresh ModelRuntime never sees them, so
|
|
11
|
+
// the parent's selected model resolves to a provider the child does not know,
|
|
12
|
+
// and the session throws `No API key found for <provider>` before its first
|
|
13
|
+
// turn.
|
|
14
|
+
//
|
|
15
|
+
// Carrying the parent's registered provider configs across keeps the child on
|
|
16
|
+
// the same model the user picked without reloading (and re-registering) the
|
|
17
|
+
// extensions themselves.
|
|
18
|
+
import { join } from "node:path";
|
|
19
|
+
import { ModelRuntime } from "@earendil-works/pi-coding-agent";
|
|
20
|
+
export async function createChildModelRuntime(ctx, agentDir) {
|
|
21
|
+
const runtime = await ModelRuntime.create({
|
|
22
|
+
authPath: join(agentDir, "auth.json"),
|
|
23
|
+
modelsPath: join(agentDir, "models.json"),
|
|
24
|
+
});
|
|
25
|
+
for (const providerId of ctx.modelRegistry.getRegisteredProviderIds()) {
|
|
26
|
+
const config = ctx.modelRegistry.getRegisteredProviderConfig(providerId);
|
|
27
|
+
if (!config)
|
|
28
|
+
continue;
|
|
29
|
+
try {
|
|
30
|
+
runtime.registerProvider(providerId, config);
|
|
31
|
+
}
|
|
32
|
+
catch {
|
|
33
|
+
// A provider the child can't accept is not worth failing the phase
|
|
34
|
+
// over: the child either doesn't need it (the user's model comes
|
|
35
|
+
// from a different provider) or fails later with the provider-
|
|
36
|
+
// specific error, which is more useful than one thrown here.
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
// Recompose the provider table so the newly registered providers land in
|
|
40
|
+
// the availability snapshot that `hasConfiguredAuth` reads. Offline: the
|
|
41
|
+
// parent already paid for any network catalog refresh.
|
|
42
|
+
await runtime.refresh({ allowNetwork: false });
|
|
43
|
+
return runtime;
|
|
44
|
+
}
|
|
@@ -14,6 +14,7 @@ import { readFile, readdir, rename, writeFile } from "node:fs/promises";
|
|
|
14
14
|
import { join } from "node:path";
|
|
15
15
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
16
16
|
import { computeTotals, NARRATION_CACHE_RELATIVE_PATH, loadUsage, pathExists, stringifySimpleYaml, } from "../../core/index.js";
|
|
17
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
17
18
|
// Per-closeout byte budget when stuffing the narrator's input. Three
|
|
18
19
|
// closeouts × 4 KB each ≈ 12 KB of prompt context, which is well under any
|
|
19
20
|
// reasonable model's input window.
|
|
@@ -153,6 +154,7 @@ async function runNarratorOnce(ctx, prompt) {
|
|
|
153
154
|
const { session } = await createAgentSession({
|
|
154
155
|
cwd,
|
|
155
156
|
agentDir,
|
|
157
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
156
158
|
sessionManager: SessionManager.inMemory(cwd),
|
|
157
159
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
158
160
|
model: ctx.model,
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/** Tool names in the guide that have no Pi slash command. */
|
|
2
|
+
export declare const MCP_ONLY_TOOLS: readonly ["codecarto_library_list", "codecarto_library_reindex"];
|
|
3
|
+
export declare const GUIDE_PREAMBLE: string;
|
|
4
|
+
export declare const PI_SURFACE_ADDENDUM: string;
|
|
5
|
+
/**
|
|
6
|
+
* Assemble the message /codecarto-guide queues. The guide document is embedded
|
|
7
|
+
* whole and unmodified between the framing and the addendum — `guide.test.mjs`
|
|
8
|
+
* pins that documents are served entire rather than summarized, and that holds
|
|
9
|
+
* on this surface too.
|
|
10
|
+
*/
|
|
11
|
+
export declare function buildPiGuideMessage(documentContent: string, otherTopics: readonly string[]): string;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
// Surface framing for the packaged agent guide when it is read into a Pi
|
|
2
|
+
// session.
|
|
3
|
+
//
|
|
4
|
+
// Two problems this solves, both invisible on the MCP surface:
|
|
5
|
+
//
|
|
6
|
+
// 1. The guide arrives as a user message. /codecarto-guide queues the document
|
|
7
|
+
// through pi.sendUserMessage, so ~200 lines of imperative instructions land
|
|
8
|
+
// as if the user had typed them, with no task attached. A model handed
|
|
9
|
+
// instructions and no task either starts driving immediately or stalls
|
|
10
|
+
// asking what to do; both are wrong for what is a reference lookup.
|
|
11
|
+
//
|
|
12
|
+
// 2. The guide is written for the MCP surface. It tells the agent to call
|
|
13
|
+
// codecarto_* tools — but the Pi extension registers no tools at all, only
|
|
14
|
+
// slash commands the *user* invokes. A model that tries to follow it finds
|
|
15
|
+
// nothing to call and reasonably concludes the server is missing. The drive
|
|
16
|
+
// loop differs too: on Pi, /codecarto-next runs the phase as an isolated
|
|
17
|
+
// sub-agent and then auto-validates and auto-completes it, so the guide's
|
|
18
|
+
// hand-written execute → handoff → validate → complete loop does not
|
|
19
|
+
// describe a Pi session.
|
|
20
|
+
//
|
|
21
|
+
// The guide text itself stays untouched — agent-skill/ is the single source and
|
|
22
|
+
// core/guide.ts serves it verbatim to every surface. Per-surface adaptation
|
|
23
|
+
// belongs in the wrapper, which is here.
|
|
24
|
+
/** Tool names in the guide that have no Pi slash command. */
|
|
25
|
+
export const MCP_ONLY_TOOLS = ["codecarto_library_list", "codecarto_library_reindex"];
|
|
26
|
+
export const GUIDE_PREAMBLE = [
|
|
27
|
+
"**CodeCartographer guide — reference material, not a task.**",
|
|
28
|
+
"",
|
|
29
|
+
"The user ran `/codecarto-guide`, which queues the guide below into this session so it is available when needed. Nothing is being asked of you yet.",
|
|
30
|
+
"",
|
|
31
|
+
"Do not start a workflow, do not begin a phase, and do not ask which repository or pipeline to use. Acknowledge in a sentence that you have read it, then wait. Answer from it when the user asks.",
|
|
32
|
+
].join("\n");
|
|
33
|
+
export const PI_SURFACE_ADDENDUM = [
|
|
34
|
+
"---",
|
|
35
|
+
"",
|
|
36
|
+
"## Reading this guide in a Pi session",
|
|
37
|
+
"",
|
|
38
|
+
"The guide above is written for the MCP surface, where an agent drives the workflow by calling `codecarto_*` tools. **This session is the Pi extension, which registers no tools.** There is nothing named `codecarto_*` for you to call, and their absence does not mean a server is missing or misconfigured.",
|
|
39
|
+
"",
|
|
40
|
+
`- **Every tool name maps to a slash command the user runs**, mechanically: \`codecarto_status\` → \`/codecarto-status\`, \`codecarto_next\` → \`/codecarto-next\`, and so on. Two have no Pi equivalent: ${MCP_ONLY_TOOLS.map((name) => `\`${name}\``).join(" and ")}.`,
|
|
41
|
+
"- **Ignore \"every tool takes an absolute `cwd`\".** Slash commands act on the session's own directory; there is no `cwd` argument to pass.",
|
|
42
|
+
"- **The drive loop is different.** `/codecarto-next` executes the phase itself, as an isolated sub-agent, and then auto-validates and auto-completes it. The guide's hand-written loop — take the prompt, execute it, write the handoff, then validate and complete yourself — describes the MCP surface. On Pi the user drives and the extension executes; your job is to explain what the framework is doing and answer questions about it, not to reproduce that loop by hand.",
|
|
43
|
+
].join("\n");
|
|
44
|
+
/**
|
|
45
|
+
* Assemble the message /codecarto-guide queues. The guide document is embedded
|
|
46
|
+
* whole and unmodified between the framing and the addendum — `guide.test.mjs`
|
|
47
|
+
* pins that documents are served entire rather than summarized, and that holds
|
|
48
|
+
* on this surface too.
|
|
49
|
+
*/
|
|
50
|
+
export function buildPiGuideMessage(documentContent, otherTopics) {
|
|
51
|
+
const footer = otherTopics.length > 0
|
|
52
|
+
? `\n\n---\nOther guide topics: ${otherTopics.join(", ")} (run /codecarto-guide <topic>).`
|
|
53
|
+
: "";
|
|
54
|
+
return `${GUIDE_PREAMBLE}\n\n---\n\n${documentContent}\n\n${PI_SURFACE_ADDENDUM}${footer}`;
|
|
55
|
+
}
|
|
@@ -8,6 +8,7 @@ import { narrateDashboard } from "./dashboard-narrator.js";
|
|
|
8
8
|
import { writeDashboard } from "./dashboard-writer.js";
|
|
9
9
|
import { parseBroadsideFlags, KNOWN_BROADSIDE_TOKENS } from "./broadside-flags.js";
|
|
10
10
|
import { parseNextFlags } from "./next-flags.js";
|
|
11
|
+
import { buildPiGuideMessage } from "./guide-framing.js";
|
|
11
12
|
import { phaseCompactionExtension } from "./phase-compaction.js";
|
|
12
13
|
import { applyAmendment, buildPhasePrompt, buildSkillPrompt, buildValidationSummary, canonicalPath, copyPackagedWorkspace, computePerPhaseTotals, computeTotals, ConfidentialityMismatchError, createEmptyStatus, DEFAULT_PIPELINE_PATH, describeScaffoldStaleness, deriveSlug, discoverLibrary, getNextEligiblePhase, getPipelineLabel, getWorkspaceState, isWithinPathResolved, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideCancelledError, broadsideDirFor, collectResultText, estimateSubmitText, getLens, listAmendmentNames, listBatchModels, listGuideTopics, listScaffoldRefreshFiles, listSkillNames, loadAmendmentFile, loadBroadsideConfig, modelsText, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, statusText, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, packagedWorkspaceDir, pathExists, PACKAGE_VERSION, readBroadsideSkill, readGuide, refreshScaffold, PhasePreflightError, PIPELINE_ALIASES, publishEntry, resolvePhase, resolvePipelineChoice, resolvePublishSourceRepo, SourceRepoMismatchError, runPhasePreflight, SCAFFOLD_REFRESH_PROTECTED, seedOrchestratorFiles, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, } from "../../core/index.js";
|
|
13
14
|
import { initLibrary } from "../../core/library.js";
|
|
@@ -785,7 +786,25 @@ export default function codeCartographerExtension(pi) {
|
|
|
785
786
|
ctx.ui.notify(`Cannot complete ${validation.phaseId}: ${validation.overall}`, "error");
|
|
786
787
|
return;
|
|
787
788
|
}
|
|
788
|
-
|
|
789
|
+
// Completion refuses for reasons the framework words carefully — a
|
|
790
|
+
// missing phase handoff, a carry-forward without `derives_from`, a
|
|
791
|
+
// closure lacking runtime evidence. Those messages are the whole
|
|
792
|
+
// point of the refusal, and this was the one call in this file that
|
|
793
|
+
// let them escape as a rejection instead of showing them. The
|
|
794
|
+
// irony was sharp: /codecarto-next catches this same throw and tells
|
|
795
|
+
// the user to run /codecarto-complete manually, which then threw.
|
|
796
|
+
let completion;
|
|
797
|
+
try {
|
|
798
|
+
completion = await autoCompletePhase(ctx.cwd, validation);
|
|
799
|
+
}
|
|
800
|
+
catch (error) {
|
|
801
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
802
|
+
lastFeedbackLines = [`Completion refused: ${message}`];
|
|
803
|
+
setUiState(ctx, currentState, lastFeedbackLines);
|
|
804
|
+
ctx.ui.notify(message, "error");
|
|
805
|
+
return;
|
|
806
|
+
}
|
|
807
|
+
const { updatedState, closeoutNotice, warnings } = completion;
|
|
789
808
|
lastFeedbackLines = [
|
|
790
809
|
`Completed phase: ${validation.phaseId}`,
|
|
791
810
|
`Validation: ${validation.overall}`,
|
|
@@ -919,10 +938,10 @@ export default function codeCartographerExtension(pi) {
|
|
|
919
938
|
return;
|
|
920
939
|
}
|
|
921
940
|
const other = topics.filter((name) => name !== document.topic);
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
const message =
|
|
941
|
+
// Framed, not bare: the guide is MCP-centric text arriving as a user
|
|
942
|
+
// message, so it needs both a "this is reference, not a task" header
|
|
943
|
+
// and a Pi-surface addendum. See guide-framing.ts.
|
|
944
|
+
const message = buildPiGuideMessage(document.content, other);
|
|
926
945
|
if (ctx.isIdle()) {
|
|
927
946
|
pi.sendUserMessage(message);
|
|
928
947
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "codecartographer-pi",
|
|
3
|
-
"version": "0.19.
|
|
3
|
+
"version": "0.19.3",
|
|
4
4
|
"mcpName": "io.github.HuginnIndustries/codecartographer",
|
|
5
5
|
"description": "Turn an unfamiliar codebase into a validated reimplementation spec, then synthesize confirmed specs and a product vision into a traceable plan.",
|
|
6
6
|
"type": "module",
|