codecartographer-pi 0.19.1 → 0.19.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -90,6 +90,17 @@ Collect runs two cross-lens post-passes by default: **synthesis** (the
90
90
  executive report) and **triage** (the prioritized work order). Pass
91
91
  `include_synthesis: false` or `include_triage: false` on collect to skip one.
92
92
 
93
+ Two caveats apply to any model you pick. The `models` action lists every id
94
+ OpenRouter advertises a `:batch` variant for, and many of those variants do not
95
+ exist — submitting one returns `does not have a :batch endpoint`, with nothing in
96
+ the catalog to distinguish it beforehand. A rejected batch costs nothing, so
97
+ probe a candidate on a single lens first. And reasoning competes with the answer for
98
+ `max_tokens`: Broad-Side caps thinking at a quarter of each lens's output budget
99
+ so three quarters remain for the JSON, which is the split the cost estimate
100
+ already assumes. It caps rather than disables because some endpoints refuse to
101
+ be switched off entirely. Override with `reasoning:` in `config.yaml` only
102
+ alongside a raised output budget.
103
+
93
104
  Lenses do not all have to run on the same model. `lens_models` in `config.yaml`
94
105
  routes individual lenses to their own batch model — the usual reason being that
95
106
  a stronger model changes security and defect findings more than it changes an
@@ -46,9 +46,48 @@
46
46
  # and picking one for you would spend your money on our guess. Compare
47
47
  # candidates with the `models` action first.
48
48
  #
49
+ # Two things to check before committing to a model, both learned the hard way:
50
+ #
51
+ # 1. The `models` action lists whatever OpenRouter advertises a `:batch`
52
+ # variant for, and a good number of those variants do not actually exist —
53
+ # submitting one comes back `Model '<id>' does not have a :batch endpoint.`
54
+ # Nothing in the catalog distinguishes them. Every Anthropic and OpenAI
55
+ # batch id tried so far is rejected this way; Google and DeepSeek work.
56
+ # A rejected batch costs nothing, so probe a candidate on one lens before
57
+ # relying on it.
58
+ # 2. A reasoning-capable model spends its output budget thinking, and the
59
+ # thinking is billed at the full output rate. See `reasoning:` below.
60
+ #
49
61
  # lens_models:
50
- # security: anthropic/claude-opus-4.5:batch
51
- # defect: anthropic/claude-opus-4.5:batch
62
+ # security: deepseek/deepseek-v4-pro-0813:batch
63
+ # defect: deepseek/deepseek-v4-pro-0813:batch
64
+
65
+ # Reasoning control, sent on every lens request.
66
+ #
67
+ # By default Broad-Side caps thinking at a quarter of the lens's output budget,
68
+ # leaving the other three quarters for the answer — which is exactly the split
69
+ # the cost estimate already assumes.
70
+ #
71
+ # The cap exists because reasoning competes with the answer for `max_tokens`.
72
+ # One measured run spent 5,758 of a 6,000-token budget thinking and left ~230
73
+ # tokens for the JSON, which truncated mid-structure on 11 of 13 slices; those
74
+ # tokens bill at the full output rate, so it paid for ~6,000 output tokens per
75
+ # slice to receive ~230 usable ones. The shipped default model does the same
76
+ # thing less consistently — reasoning from 0 to 5,757 tokens across 13 slices,
77
+ # three of them cut off — so this is not something only exotic models do.
78
+ #
79
+ # It is a cap rather than an off switch on purpose. Some endpoints refuse to be
80
+ # switched off: `google/gemini-3.8-flash:batch` rejects the entire batch with
81
+ # "Reasoning is mandatory for this endpoint and cannot be disabled", which turns
82
+ # a partial result into none at all. Capping works either way.
83
+ #
84
+ # Override only with a raised lens output budget, or the JSON truncates exactly
85
+ # as above.
86
+ #
87
+ # reasoning:
88
+ # effort: low # minimal | low | medium | high
89
+ # max_tokens: 2000 # or set the thinking budget directly
90
+ # enabled: false # only where the provider allows it
52
91
 
53
92
  # Approximate run expense limit in USD (0 = no limit). Before submitting,
54
93
  # Broad-Side estimates the run cost from the collected file sizes and the
@@ -39,7 +39,8 @@ decisions: []
39
39
  # evidence: <optional: where the pattern showed up>
40
40
  proposed_conventions: []
41
41
  closeout_summary: ""
42
- # Optional full closeout Markdown. YAML literal blocks are supported.
42
+ # Optional full closeout Markdown. Block scalars are supported — literal (|, |-,
43
+ # |+) to keep line breaks, folded (>, >-, >+) to join wrapped prose into one line.
43
44
  closeout_content: |-
44
45
  # Closeout — <phase-id>
45
46
 
@@ -3,4 +3,4 @@
3
3
  # workspace's framework-owned files (GUIDE.md, templates/, workflow/ pipelines
4
4
  # and VALIDATE.md) predate the running release. Written at release time and
5
5
  # copied verbatim by init — never edit by hand.
6
- scaffold_version: 0.19.1
6
+ scaffold_version: 0.19.3
@@ -86,6 +86,46 @@ export type FileSlice = {
86
86
  /** Repo-relative paths of the files folded into this slice. */
87
87
  files: string[];
88
88
  };
89
+ /**
90
+ * OpenRouter's unified `reasoning` control, as sent on a lens request.
91
+ *
92
+ * Left unsent, each model applies its own default — which is how a
93
+ * reasoning-capable model came to spend 5,758 of a 6,000-token output budget
94
+ * thinking, leaving ~230 tokens for JSON that then truncated mid-structure. The
95
+ * thinking is billed at the full *output* rate, so the run paid for roughly
96
+ * 6,000 output tokens per slice to receive 230 usable ones.
97
+ */
98
+ export type BroadsideReasoning = {
99
+ enabled?: boolean;
100
+ effort?: "minimal" | "low" | "medium" | "high";
101
+ max_tokens?: number;
102
+ };
103
+ /**
104
+ * The share of a lens's output budget reasoning may spend.
105
+ *
106
+ * `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
107
+ * at the remaining quarter makes that assumption true by construction and
108
+ * guarantees the answer has room. A floor keeps the cap sane for a small lens.
109
+ */
110
+ export declare const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
111
+ export declare const BROADSIDE_MIN_REASONING_TOKENS = 512;
112
+ /**
113
+ * Cap reasoning for a lens request — deliberately a cap, not an off switch.
114
+ *
115
+ * Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
116
+ * the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
117
+ * disabled"*, turning a partial result into none at all. Capping works whether
118
+ * or not a provider allows reasoning to be switched off.
119
+ *
120
+ * The failure this prevents is the budget being spent thinking rather than
121
+ * answering. Measured on one run: 5,758 of a 6,000-token budget went to
122
+ * reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
123
+ * those tokens bill at the full output rate. The shipped default model does the
124
+ * same thing less consistently (reasoning tokens from 0 to 5,757 across 13
125
+ * slices, three of them cut off at `finish_reason: length`), so this is not a
126
+ * multi-model concern.
127
+ */
128
+ export declare function defaultReasoningFor(maxTokens: number): BroadsideReasoning;
89
129
  export type BatchRequest = {
90
130
  custom_id: string;
91
131
  body: {
@@ -99,6 +139,7 @@ export type BatchRequest = {
99
139
  json_schema: JsonSchemaDef;
100
140
  };
101
141
  max_tokens: number;
142
+ reasoning?: BroadsideReasoning;
102
143
  };
103
144
  };
104
145
  export type BatchTerminalStatus = "completed" | "failed" | "expired" | "cancelled";
@@ -185,6 +226,8 @@ export type BroadsideConfig = {
185
226
  * trade-off is not the same for every lens.
186
227
  */
187
228
  lensModels: Partial<Record<BroadsideLensId, string>>;
229
+ /** Overrides every lens's reasoning setting when present. */
230
+ reasoning: BroadsideReasoning | null;
188
231
  /**
189
232
  * Repo defaults for the per-call run knobs. Each mirrors a tool parameter
190
233
  * of the same name; an explicit parameter always wins. They live here so a
@@ -304,6 +347,7 @@ type LensDefinition = {
304
347
  sliceBy: "none" | "directory" | "auto";
305
348
  maxChars: number;
306
349
  maxTokens: number;
350
+ reasoning?: BroadsideReasoning;
307
351
  skipTestFiles?: boolean;
308
352
  globsFor: (info: RepoInfo) => string[];
309
353
  systemPrompt: (info: RepoInfo) => string;
@@ -313,7 +357,7 @@ export declare function getLens(lensId: BroadsideLensId): LensDefinition;
313
357
  export declare function listLenses(): LensDefinition[];
314
358
  export declare function collectRepoInfo(targetDir: string): Promise<RepoInfo>;
315
359
  export declare function gatherSlices(targetDir: string, lens: LensDefinition, info: RepoInfo): Promise<FileSlice[]>;
316
- export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number): BatchRequest;
360
+ export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number, reasoningOverride?: BroadsideReasoning): BatchRequest;
317
361
  /**
318
362
  * Pre-flight cost estimate for one lens.
319
363
  *
@@ -74,6 +74,34 @@ export const BROADSIDE_LENS_IDS = [
74
74
  ];
75
75
  export const BROADSIDE_POLL_INTERVAL_MS = 15_000;
76
76
  export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
77
+ /**
78
+ * The share of a lens's output budget reasoning may spend.
79
+ *
80
+ * `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
81
+ * at the remaining quarter makes that assumption true by construction and
82
+ * guarantees the answer has room. A floor keeps the cap sane for a small lens.
83
+ */
84
+ export const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
85
+ export const BROADSIDE_MIN_REASONING_TOKENS = 512;
86
+ /**
87
+ * Cap reasoning for a lens request — deliberately a cap, not an off switch.
88
+ *
89
+ * Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
90
+ * the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
91
+ * disabled"*, turning a partial result into none at all. Capping works whether
92
+ * or not a provider allows reasoning to be switched off.
93
+ *
94
+ * The failure this prevents is the budget being spent thinking rather than
95
+ * answering. Measured on one run: 5,758 of a 6,000-token budget went to
96
+ * reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
97
+ * those tokens bill at the full output rate. The shipped default model does the
98
+ * same thing less consistently (reasoning tokens from 0 to 5,757 across 13
99
+ * slices, three of them cut off at `finish_reason: length`), so this is not a
100
+ * multi-model concern.
101
+ */
102
+ export function defaultReasoningFor(maxTokens) {
103
+ return { max_tokens: Math.max(BROADSIDE_MIN_REASONING_TOKENS, Math.floor(maxTokens * BROADSIDE_REASONING_BUDGET_FRACTION)) };
104
+ }
77
105
  /** Thrown when a confirm hook declines a run. Nothing was submitted. */
78
106
  export class BroadsideCancelledError extends Error {
79
107
  constructor(message = "Broad-Side submission cancelled. Nothing was submitted.") {
@@ -1136,7 +1164,7 @@ async function sumFileSizes(targetDir, files) {
1136
1164
  return total;
1137
1165
  }
1138
1166
  // ---------- request building ----------
1139
- export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride) {
1167
+ export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride, reasoningOverride) {
1140
1168
  const moduleTag = sanitizeId(slice.moduleName);
1141
1169
  const customId = sliceCount > 1 ? `${lens.id}-${moduleTag}-${index + 1}` : `${lens.id}-${moduleTag}`;
1142
1170
  return {
@@ -1149,6 +1177,9 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
1149
1177
  ],
1150
1178
  response_format: { type: "json_schema", json_schema: SCHEMAS[lens.schemaName] },
1151
1179
  max_tokens: maxTokensOverride ?? lens.maxTokens,
1180
+ // Always sent, never inherited: an absent field means the model's
1181
+ // own default, and that default is what truncated the JSON.
1182
+ reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(maxTokensOverride ?? lens.maxTokens),
1152
1183
  },
1153
1184
  };
1154
1185
  }
@@ -1297,6 +1328,24 @@ export async function persistBroadsideRun(broadsideDir, run) {
1297
1328
  state.runs[index] = run;
1298
1329
  });
1299
1330
  }
1331
+ /** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
1332
+ function parseReasoningConfig(raw) {
1333
+ if (raw === false)
1334
+ return { enabled: false };
1335
+ if (raw === true)
1336
+ return { enabled: true };
1337
+ if (!raw || typeof raw !== "object")
1338
+ return null;
1339
+ const value = raw;
1340
+ const out = {};
1341
+ if (typeof value.enabled === "boolean")
1342
+ out.enabled = value.enabled;
1343
+ if (value.effort === "minimal" || value.effort === "low" || value.effort === "medium" || value.effort === "high")
1344
+ out.effort = value.effort;
1345
+ if (typeof value.max_tokens === "number" && value.max_tokens > 0)
1346
+ out.max_tokens = value.max_tokens;
1347
+ return Object.keys(out).length > 0 ? out : null;
1348
+ }
1300
1349
  export async function loadBroadsideConfig(broadsideDir) {
1301
1350
  const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
1302
1351
  let raw = {};
@@ -1337,6 +1386,10 @@ export async function loadBroadsideConfig(broadsideDir) {
1337
1386
  ? { inputPerM: inputOverride, outputPerM: outputOverride }
1338
1387
  : null,
1339
1388
  lensModels,
1389
+ // An escape hatch, not a knob to reach for: a model whose reasoning is
1390
+ // worth paying for needs its lens maxTokens raised to cover both the
1391
+ // thinking and the answer, or the JSON truncates exactly as before.
1392
+ reasoning: parseReasoningConfig(raw.reasoning),
1340
1393
  incremental: flag("incremental", false),
1341
1394
  retryTruncated: flag("retry_truncated", true),
1342
1395
  includeSynthesis: flag("include_synthesis", true),
@@ -1797,7 +1850,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
1797
1850
  const { lens, maxTokens, lensModel, lensOutputCap } = priced;
1798
1851
  const lensId = lens.id;
1799
1852
  const slices = slicesByLens.get(lensId) ?? [];
1800
- const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens));
1853
+ const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens, config.reasoning ?? undefined));
1801
1854
  for (const request of requests)
1802
1855
  requestsByCustomId[request.custom_id] = request;
1803
1856
  const entry = {
package/dist/core/yaml.js CHANGED
@@ -105,6 +105,61 @@ export function parseYamlScalar(rawValue) {
105
105
  }
106
106
  return trimmed;
107
107
  }
108
+ function parseBlockScalarHeader(rawValue) {
109
+ const match = /^([|>])([-+]?)$/.exec(rawValue);
110
+ if (!match)
111
+ return null;
112
+ return {
113
+ literal: match[1] === "|",
114
+ chomp: match[2] === "-" ? "strip" : match[2] === "+" ? "keep" : "clip",
115
+ };
116
+ }
117
+ /**
118
+ * Fold a block scalar's lines per YAML's folding rules: a single line break
119
+ * between two content lines becomes a space, and a run of k blank lines becomes
120
+ * k newlines. Lines indented deeper than the block's own content indent are
121
+ * "more indented" and keep their breaks literally, which is what lets a folded
122
+ * block hold an indented snippet without it being flattened onto one line.
123
+ */
124
+ function foldBlockLines(blockLines) {
125
+ let result = "";
126
+ let pendingBreaks = 0;
127
+ let started = false;
128
+ let previousMoreIndented = false;
129
+ for (const line of blockLines) {
130
+ if (line.trim() === "") {
131
+ pendingBreaks++;
132
+ continue;
133
+ }
134
+ const moreIndented = /^[ \t]/.test(line);
135
+ if (!started) {
136
+ result = line;
137
+ started = true;
138
+ previousMoreIndented = moreIndented;
139
+ continue;
140
+ }
141
+ if (pendingBreaks > 0) {
142
+ result += "\n".repeat(pendingBreaks);
143
+ pendingBreaks = 0;
144
+ }
145
+ else if (moreIndented || previousMoreIndented) {
146
+ result += "\n";
147
+ }
148
+ else {
149
+ result += " ";
150
+ }
151
+ result += line;
152
+ previousMoreIndented = moreIndented;
153
+ }
154
+ return started ? result + "\n".repeat(pendingBreaks) : "";
155
+ }
156
+ function applyBlockScalar(blockLines, header) {
157
+ const content = header.literal ? blockLines.join("\n") : foldBlockLines(blockLines);
158
+ if (header.chomp === "keep")
159
+ return content;
160
+ const stripped = content.replace(/\n+$/, "");
161
+ return header.chomp === "strip" ? stripped : `${stripped}\n`;
162
+ }
108
163
  export function parseSimpleYaml(raw) {
109
164
  const lines = raw.split(/\r?\n/);
110
165
  let index = 0;
@@ -164,7 +219,8 @@ export function parseSimpleYaml(raw) {
164
219
  throw new Error(`Duplicate YAML key: ${key} near line: ${line.trim()}`);
165
220
  }
166
221
  seen.add(key);
167
- if (rawValue === "|" || rawValue === "|-") {
222
+ const blockHeader = parseBlockScalarHeader(rawValue);
223
+ if (blockHeader) {
168
224
  const blockLines = [];
169
225
  let contentIndent = null;
170
226
  while (index < lines.length) {
@@ -181,8 +237,7 @@ export function parseSimpleYaml(raw) {
181
237
  blockLines.push(blockLine.slice(Math.min(contentIndent, blockIndent)));
182
238
  index++;
183
239
  }
184
- const content = blockLines.join("\n").replace(/\n+$/, "");
185
- assign(key, rawValue === "|" ? `${content}\n` : content);
240
+ assign(key, applyBlockScalar(blockLines, blockHeader));
186
241
  continue;
187
242
  }
188
243
  if (rawValue !== "") {
@@ -11,6 +11,7 @@ import { readFile, readdir } from "node:fs/promises";
11
11
  import { join } from "node:path";
12
12
  import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
13
13
  import { closeoutFileName, pathExists } from "../../core/index.js";
14
+ import { createChildModelRuntime } from "./child-model-runtime.js";
14
15
  /** Closeout content over this many bytes is truncated before being passed to
15
16
  * the rewriter. Keeps the orchestrator-side cost predictable. */
16
17
  const CLOSEOUT_BYTE_BUDGET = 8000;
@@ -143,6 +144,7 @@ async function runRewriterOnce(ctx, prompt) {
143
144
  const { session } = await createAgentSession({
144
145
  cwd,
145
146
  agentDir,
147
+ modelRuntime: await createChildModelRuntime(ctx, agentDir),
146
148
  sessionManager: SessionManager.inMemory(cwd),
147
149
  settingsManager: SettingsManager.create(cwd, agentDir),
148
150
  model: ctx.model,
@@ -13,6 +13,7 @@ import { access } from "node:fs/promises";
13
13
  import { join, resolve } from "node:path";
14
14
  import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
15
15
  import { canonicalPath, isWithinPath } from "../../core/index.js";
16
+ import { createChildModelRuntime } from "./child-model-runtime.js";
16
17
  import { phaseCompactionExtension } from "./phase-compaction.js";
17
18
  // Tools available to the phase sub-agent. Matches the codecarto interception
18
19
  // allowlist (SAFE_TOOL_NAMES in extensions/codecarto/index.ts), minus bash.
@@ -104,6 +105,7 @@ export async function runPhase(ctx, prompt, callbacks = {}, options = {}, signal
104
105
  const { session } = await createAgentSession({
105
106
  cwd,
106
107
  agentDir,
108
+ modelRuntime: await createChildModelRuntime(ctx, agentDir),
107
109
  sessionManager,
108
110
  settingsManager: SettingsManager.create(cwd, agentDir),
109
111
  model: ctx.model,
@@ -0,0 +1,2 @@
1
+ import { type ExtensionContext, ModelRuntime } from "@earendil-works/pi-coding-agent";
2
+ export declare function createChildModelRuntime(ctx: ExtensionContext, agentDir: string): Promise<ModelRuntime>;
@@ -0,0 +1,44 @@
1
+ // Model runtime for codecarto's child sessions (phase sub-agents, the
2
+ // next-phase rewriter, the dashboard narrator).
3
+ //
4
+ // All three child sessions load with `noExtensions: true` so a globally
5
+ // installed codecarto doesn't register its commands and tool guards twice
6
+ // inside its own sub-agent. That isolation has a side effect: providers
7
+ // registered by *other* global extensions via `pi.registerProvider()` — an
8
+ // Ollama Cloud bridge, a company gateway, any custom `streamSimple` provider —
9
+ // are registered onto the parent's ModelRuntime by the resource loader that
10
+ // loaded them. A child that builds a fresh ModelRuntime never sees them, so
11
+ // the parent's selected model resolves to a provider the child does not know,
12
+ // and the session throws `No API key found for <provider>` before its first
13
+ // turn.
14
+ //
15
+ // Carrying the parent's registered provider configs across keeps the child on
16
+ // the same model the user picked without reloading (and re-registering) the
17
+ // extensions themselves.
18
+ import { join } from "node:path";
19
+ import { ModelRuntime } from "@earendil-works/pi-coding-agent";
20
+ export async function createChildModelRuntime(ctx, agentDir) {
21
+ const runtime = await ModelRuntime.create({
22
+ authPath: join(agentDir, "auth.json"),
23
+ modelsPath: join(agentDir, "models.json"),
24
+ });
25
+ for (const providerId of ctx.modelRegistry.getRegisteredProviderIds()) {
26
+ const config = ctx.modelRegistry.getRegisteredProviderConfig(providerId);
27
+ if (!config)
28
+ continue;
29
+ try {
30
+ runtime.registerProvider(providerId, config);
31
+ }
32
+ catch {
33
+ // A provider the child can't accept is not worth failing the phase
34
+ // over: the child either doesn't need it (the user's model comes
35
+ // from a different provider) or fails later with the provider-
36
+ // specific error, which is more useful than one thrown here.
37
+ }
38
+ }
39
+ // Recompose the provider table so the newly registered providers land in
40
+ // the availability snapshot that `hasConfiguredAuth` reads. Offline: the
41
+ // parent already paid for any network catalog refresh.
42
+ await runtime.refresh({ allowNetwork: false });
43
+ return runtime;
44
+ }
@@ -14,6 +14,7 @@ import { readFile, readdir, rename, writeFile } from "node:fs/promises";
14
14
  import { join } from "node:path";
15
15
  import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
16
16
  import { computeTotals, NARRATION_CACHE_RELATIVE_PATH, loadUsage, pathExists, stringifySimpleYaml, } from "../../core/index.js";
17
+ import { createChildModelRuntime } from "./child-model-runtime.js";
17
18
  // Per-closeout byte budget when stuffing the narrator's input. Three
18
19
  // closeouts × 4 KB each ≈ 12 KB of prompt context, which is well under any
19
20
  // reasonable model's input window.
@@ -153,6 +154,7 @@ async function runNarratorOnce(ctx, prompt) {
153
154
  const { session } = await createAgentSession({
154
155
  cwd,
155
156
  agentDir,
157
+ modelRuntime: await createChildModelRuntime(ctx, agentDir),
156
158
  sessionManager: SessionManager.inMemory(cwd),
157
159
  settingsManager: SettingsManager.create(cwd, agentDir),
158
160
  model: ctx.model,
@@ -0,0 +1,11 @@
1
+ /** Tool names in the guide that have no Pi slash command. */
2
+ export declare const MCP_ONLY_TOOLS: readonly ["codecarto_library_list", "codecarto_library_reindex"];
3
+ export declare const GUIDE_PREAMBLE: string;
4
+ export declare const PI_SURFACE_ADDENDUM: string;
5
+ /**
6
+ * Assemble the message /codecarto-guide queues. The guide document is embedded
7
+ * whole and unmodified between the framing and the addendum — `guide.test.mjs`
8
+ * pins that documents are served entire rather than summarized, and that holds
9
+ * on this surface too.
10
+ */
11
+ export declare function buildPiGuideMessage(documentContent: string, otherTopics: readonly string[]): string;
@@ -0,0 +1,55 @@
1
+ // Surface framing for the packaged agent guide when it is read into a Pi
2
+ // session.
3
+ //
4
+ // Two problems this solves, both invisible on the MCP surface:
5
+ //
6
+ // 1. The guide arrives as a user message. /codecarto-guide queues the document
7
+ // through pi.sendUserMessage, so ~200 lines of imperative instructions land
8
+ // as if the user had typed them, with no task attached. A model handed
9
+ // instructions and no task either starts driving immediately or stalls
10
+ // asking what to do; both are wrong for what is a reference lookup.
11
+ //
12
+ // 2. The guide is written for the MCP surface. It tells the agent to call
13
+ // codecarto_* tools — but the Pi extension registers no tools at all, only
14
+ // slash commands the *user* invokes. A model that tries to follow it finds
15
+ // nothing to call and reasonably concludes the server is missing. The drive
16
+ // loop differs too: on Pi, /codecarto-next runs the phase as an isolated
17
+ // sub-agent and then auto-validates and auto-completes it, so the guide's
18
+ // hand-written execute → handoff → validate → complete loop does not
19
+ // describe a Pi session.
20
+ //
21
+ // The guide text itself stays untouched — agent-skill/ is the single source and
22
+ // core/guide.ts serves it verbatim to every surface. Per-surface adaptation
23
+ // belongs in the wrapper, which is here.
24
+ /** Tool names in the guide that have no Pi slash command. */
25
+ export const MCP_ONLY_TOOLS = ["codecarto_library_list", "codecarto_library_reindex"];
26
+ export const GUIDE_PREAMBLE = [
27
+ "**CodeCartographer guide — reference material, not a task.**",
28
+ "",
29
+ "The user ran `/codecarto-guide`, which queues the guide below into this session so it is available when needed. Nothing is being asked of you yet.",
30
+ "",
31
+ "Do not start a workflow, do not begin a phase, and do not ask which repository or pipeline to use. Acknowledge in a sentence that you have read it, then wait. Answer from it when the user asks.",
32
+ ].join("\n");
33
+ export const PI_SURFACE_ADDENDUM = [
34
+ "---",
35
+ "",
36
+ "## Reading this guide in a Pi session",
37
+ "",
38
+ "The guide above is written for the MCP surface, where an agent drives the workflow by calling `codecarto_*` tools. **This session is the Pi extension, which registers no tools.** There is nothing named `codecarto_*` for you to call, and their absence does not mean a server is missing or misconfigured.",
39
+ "",
40
+ `- **Every tool name maps to a slash command the user runs**, mechanically: \`codecarto_status\` → \`/codecarto-status\`, \`codecarto_next\` → \`/codecarto-next\`, and so on. Two have no Pi equivalent: ${MCP_ONLY_TOOLS.map((name) => `\`${name}\``).join(" and ")}.`,
41
+ "- **Ignore \"every tool takes an absolute `cwd`\".** Slash commands act on the session's own directory; there is no `cwd` argument to pass.",
42
+ "- **The drive loop is different.** `/codecarto-next` executes the phase itself, as an isolated sub-agent, and then auto-validates and auto-completes it. The guide's hand-written loop — take the prompt, execute it, write the handoff, then validate and complete yourself — describes the MCP surface. On Pi the user drives and the extension executes; your job is to explain what the framework is doing and answer questions about it, not to reproduce that loop by hand.",
43
+ ].join("\n");
44
+ /**
45
+ * Assemble the message /codecarto-guide queues. The guide document is embedded
46
+ * whole and unmodified between the framing and the addendum — `guide.test.mjs`
47
+ * pins that documents are served entire rather than summarized, and that holds
48
+ * on this surface too.
49
+ */
50
+ export function buildPiGuideMessage(documentContent, otherTopics) {
51
+ const footer = otherTopics.length > 0
52
+ ? `\n\n---\nOther guide topics: ${otherTopics.join(", ")} (run /codecarto-guide <topic>).`
53
+ : "";
54
+ return `${GUIDE_PREAMBLE}\n\n---\n\n${documentContent}\n\n${PI_SURFACE_ADDENDUM}${footer}`;
55
+ }
@@ -8,6 +8,7 @@ import { narrateDashboard } from "./dashboard-narrator.js";
8
8
  import { writeDashboard } from "./dashboard-writer.js";
9
9
  import { parseBroadsideFlags, KNOWN_BROADSIDE_TOKENS } from "./broadside-flags.js";
10
10
  import { parseNextFlags } from "./next-flags.js";
11
+ import { buildPiGuideMessage } from "./guide-framing.js";
11
12
  import { phaseCompactionExtension } from "./phase-compaction.js";
12
13
  import { applyAmendment, buildPhasePrompt, buildSkillPrompt, buildValidationSummary, canonicalPath, copyPackagedWorkspace, computePerPhaseTotals, computeTotals, ConfidentialityMismatchError, createEmptyStatus, DEFAULT_PIPELINE_PATH, describeScaffoldStaleness, deriveSlug, discoverLibrary, getNextEligiblePhase, getPipelineLabel, getWorkspaceState, isWithinPathResolved, BROADSIDE_LENS_IDS, BROADSIDE_SKILL_NAME, BroadsideCancelledError, broadsideDirFor, collectResultText, estimateSubmitText, getLens, listAmendmentNames, listBatchModels, listGuideTopics, listScaffoldRefreshFiles, listSkillNames, loadAmendmentFile, loadBroadsideConfig, modelsText, runBroadsideCollect, runBroadsideStatus, runBroadsideSubmit, statusText, loadCodecartoConfig, loadUsage, loadYamlFile, normalizeForComparison, packagedWorkspaceDir, pathExists, PACKAGE_VERSION, readBroadsideSkill, readGuide, refreshScaffold, PhasePreflightError, PIPELINE_ALIASES, publishEntry, resolvePhase, resolvePipelineChoice, resolvePublishSourceRepo, SourceRepoMismatchError, runPhasePreflight, SCAFFOLD_REFRESH_PROTECTED, seedOrchestratorFiles, stringifySimpleYaml, switchPipeline, validatePhaseOutput, writeLibraryConfig, } from "../../core/index.js";
13
14
  import { initLibrary } from "../../core/library.js";
@@ -785,7 +786,25 @@ export default function codeCartographerExtension(pi) {
785
786
  ctx.ui.notify(`Cannot complete ${validation.phaseId}: ${validation.overall}`, "error");
786
787
  return;
787
788
  }
788
- const { updatedState, closeoutNotice, warnings } = await autoCompletePhase(ctx.cwd, validation);
789
+ // Completion refuses for reasons the framework words carefully — a
790
+ // missing phase handoff, a carry-forward without `derives_from`, a
791
+ // closure lacking runtime evidence. Those messages are the whole
792
+ // point of the refusal, and this was the one call in this file that
793
+ // let them escape as a rejection instead of showing them. The
794
+ // irony was sharp: /codecarto-next catches this same throw and tells
795
+ // the user to run /codecarto-complete manually, which then threw.
796
+ let completion;
797
+ try {
798
+ completion = await autoCompletePhase(ctx.cwd, validation);
799
+ }
800
+ catch (error) {
801
+ const message = error instanceof Error ? error.message : String(error);
802
+ lastFeedbackLines = [`Completion refused: ${message}`];
803
+ setUiState(ctx, currentState, lastFeedbackLines);
804
+ ctx.ui.notify(message, "error");
805
+ return;
806
+ }
807
+ const { updatedState, closeoutNotice, warnings } = completion;
789
808
  lastFeedbackLines = [
790
809
  `Completed phase: ${validation.phaseId}`,
791
810
  `Validation: ${validation.overall}`,
@@ -919,10 +938,10 @@ export default function codeCartographerExtension(pi) {
919
938
  return;
920
939
  }
921
940
  const other = topics.filter((name) => name !== document.topic);
922
- const footer = other.length > 0
923
- ? `\n\n---\nOther guide topics: ${other.join(", ")} (run /codecarto-guide <topic>).`
924
- : "";
925
- const message = `${document.content}${footer}`;
941
+ // Framed, not bare: the guide is MCP-centric text arriving as a user
942
+ // message, so it needs both a "this is reference, not a task" header
943
+ // and a Pi-surface addendum. See guide-framing.ts.
944
+ const message = buildPiGuideMessage(document.content, other);
926
945
  if (ctx.isIdle()) {
927
946
  pi.sendUserMessage(message);
928
947
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "codecartographer-pi",
3
- "version": "0.19.1",
3
+ "version": "0.19.3",
4
4
  "mcpName": "io.github.HuginnIndustries/codecartographer",
5
5
  "description": "Turn an unfamiliar codebase into a validated reimplementation spec, then synthesize confirmed specs and a product vision into a traceable plan.",
6
6
  "type": "module",