codecartographer-pi 0.19.1 → 0.19.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/broadside/SKILL.md +11 -0
- package/.codecarto/broadside/config.yaml +41 -2
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/dist/core/broadside.d.ts +45 -1
- package/dist/core/broadside.js +55 -2
- package/dist/extensions/codecarto/agent-rewriter.js +2 -0
- package/dist/extensions/codecarto/agent-runner.js +2 -0
- package/dist/extensions/codecarto/child-model-runtime.d.ts +2 -0
- package/dist/extensions/codecarto/child-model-runtime.js +44 -0
- package/dist/extensions/codecarto/dashboard-narrator.js +2 -0
- package/dist/extensions/codecarto/index.js +19 -1
- package/package.json +1 -1
|
@@ -90,6 +90,17 @@ Collect runs two cross-lens post-passes by default: **synthesis** (the
|
|
|
90
90
|
executive report) and **triage** (the prioritized work order). Pass
|
|
91
91
|
`include_synthesis: false` or `include_triage: false` on collect to skip one.
|
|
92
92
|
|
|
93
|
+
Two caveats apply to any model you pick. The `models` action lists every id
|
|
94
|
+
OpenRouter advertises a `:batch` variant for, and many of those variants do not
|
|
95
|
+
exist — submitting one returns `does not have a :batch endpoint`, with nothing in
|
|
96
|
+
the catalog to distinguish it beforehand. A rejected batch costs nothing, so
|
|
97
|
+
probe a candidate on a single lens first. And reasoning competes with the answer for
|
|
98
|
+
`max_tokens`: Broad-Side caps thinking at a quarter of each lens's output budget
|
|
99
|
+
so three quarters remain for the JSON, which is the split the cost estimate
|
|
100
|
+
already assumes. It caps rather than disables because some endpoints refuse to
|
|
101
|
+
be switched off entirely. Override with `reasoning:` in `config.yaml` only
|
|
102
|
+
alongside a raised output budget.
|
|
103
|
+
|
|
93
104
|
Lenses do not all have to run on the same model. `lens_models` in `config.yaml`
|
|
94
105
|
routes individual lenses to their own batch model — the usual reason being that
|
|
95
106
|
a stronger model changes security and defect findings more than it changes an
|
|
@@ -46,9 +46,48 @@
|
|
|
46
46
|
# and picking one for you would spend your money on our guess. Compare
|
|
47
47
|
# candidates with the `models` action first.
|
|
48
48
|
#
|
|
49
|
+
# Two things to check before committing to a model, both learned the hard way:
|
|
50
|
+
#
|
|
51
|
+
# 1. The `models` action lists whatever OpenRouter advertises a `:batch`
|
|
52
|
+
# variant for, and a good number of those variants do not actually exist —
|
|
53
|
+
# submitting one comes back `Model '<id>' does not have a :batch endpoint.`
|
|
54
|
+
# Nothing in the catalog distinguishes them. Every Anthropic and OpenAI
|
|
55
|
+
# batch id tried so far is rejected this way; Google and DeepSeek work.
|
|
56
|
+
# A rejected batch costs nothing, so probe a candidate on one lens before
|
|
57
|
+
# relying on it.
|
|
58
|
+
# 2. A reasoning-capable model spends its output budget thinking, and the
|
|
59
|
+
# thinking is billed at the full output rate. See `reasoning:` below.
|
|
60
|
+
#
|
|
49
61
|
# lens_models:
|
|
50
|
-
# security:
|
|
51
|
-
# defect:
|
|
62
|
+
# security: deepseek/deepseek-v4-pro-0813:batch
|
|
63
|
+
# defect: deepseek/deepseek-v4-pro-0813:batch
|
|
64
|
+
|
|
65
|
+
# Reasoning control, sent on every lens request.
|
|
66
|
+
#
|
|
67
|
+
# By default Broad-Side caps thinking at a quarter of the lens's output budget,
|
|
68
|
+
# leaving the other three quarters for the answer — which is exactly the split
|
|
69
|
+
# the cost estimate already assumes.
|
|
70
|
+
#
|
|
71
|
+
# The cap exists because reasoning competes with the answer for `max_tokens`.
|
|
72
|
+
# One measured run spent 5,758 of a 6,000-token budget thinking and left ~230
|
|
73
|
+
# tokens for the JSON, which truncated mid-structure on 11 of 13 slices; those
|
|
74
|
+
# tokens bill at the full output rate, so it paid for ~6,000 output tokens per
|
|
75
|
+
# slice to receive ~230 usable ones. The shipped default model does the same
|
|
76
|
+
# thing less consistently — reasoning from 0 to 5,757 tokens across 13 slices,
|
|
77
|
+
# three of them cut off — so this is not something only exotic models do.
|
|
78
|
+
#
|
|
79
|
+
# It is a cap rather than an off switch on purpose. Some endpoints refuse to be
|
|
80
|
+
# switched off: `google/gemini-3.8-flash:batch` rejects the entire batch with
|
|
81
|
+
# "Reasoning is mandatory for this endpoint and cannot be disabled", which turns
|
|
82
|
+
# a partial result into none at all. Capping works either way.
|
|
83
|
+
#
|
|
84
|
+
# Override only with a raised lens output budget, or the JSON truncates exactly
|
|
85
|
+
# as above.
|
|
86
|
+
#
|
|
87
|
+
# reasoning:
|
|
88
|
+
# effort: low # minimal | low | medium | high
|
|
89
|
+
# max_tokens: 2000 # or set the thinking budget directly
|
|
90
|
+
# enabled: false # only where the provider allows it
|
|
52
91
|
|
|
53
92
|
# Approximate run expense limit in USD (0 = no limit). Before submitting,
|
|
54
93
|
# Broad-Side estimates the run cost from the collected file sizes and the
|
package/dist/core/broadside.d.ts
CHANGED
|
@@ -86,6 +86,46 @@ export type FileSlice = {
|
|
|
86
86
|
/** Repo-relative paths of the files folded into this slice. */
|
|
87
87
|
files: string[];
|
|
88
88
|
};
|
|
89
|
+
/**
|
|
90
|
+
* OpenRouter's unified `reasoning` control, as sent on a lens request.
|
|
91
|
+
*
|
|
92
|
+
* Left unsent, each model applies its own default — which is how a
|
|
93
|
+
* reasoning-capable model came to spend 5,758 of a 6,000-token output budget
|
|
94
|
+
* thinking, leaving ~230 tokens for JSON that then truncated mid-structure. The
|
|
95
|
+
* thinking is billed at the full *output* rate, so the run paid for roughly
|
|
96
|
+
* 6,000 output tokens per slice to receive 230 usable ones.
|
|
97
|
+
*/
|
|
98
|
+
export type BroadsideReasoning = {
|
|
99
|
+
enabled?: boolean;
|
|
100
|
+
effort?: "minimal" | "low" | "medium" | "high";
|
|
101
|
+
max_tokens?: number;
|
|
102
|
+
};
|
|
103
|
+
/**
|
|
104
|
+
* The share of a lens's output budget reasoning may spend.
|
|
105
|
+
*
|
|
106
|
+
* `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
|
|
107
|
+
* at the remaining quarter makes that assumption true by construction and
|
|
108
|
+
* guarantees the answer has room. A floor keeps the cap sane for a small lens.
|
|
109
|
+
*/
|
|
110
|
+
export declare const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
|
|
111
|
+
export declare const BROADSIDE_MIN_REASONING_TOKENS = 512;
|
|
112
|
+
/**
|
|
113
|
+
* Cap reasoning for a lens request — deliberately a cap, not an off switch.
|
|
114
|
+
*
|
|
115
|
+
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
116
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
117
|
+
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
118
|
+
* or not a provider allows reasoning to be switched off.
|
|
119
|
+
*
|
|
120
|
+
* The failure this prevents is the budget being spent thinking rather than
|
|
121
|
+
* answering. Measured on one run: 5,758 of a 6,000-token budget went to
|
|
122
|
+
* reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
|
|
123
|
+
* those tokens bill at the full output rate. The shipped default model does the
|
|
124
|
+
* same thing less consistently (reasoning tokens from 0 to 5,757 across 13
|
|
125
|
+
* slices, three of them cut off at `finish_reason: length`), so this is not a
|
|
126
|
+
* multi-model concern.
|
|
127
|
+
*/
|
|
128
|
+
export declare function defaultReasoningFor(maxTokens: number): BroadsideReasoning;
|
|
89
129
|
export type BatchRequest = {
|
|
90
130
|
custom_id: string;
|
|
91
131
|
body: {
|
|
@@ -99,6 +139,7 @@ export type BatchRequest = {
|
|
|
99
139
|
json_schema: JsonSchemaDef;
|
|
100
140
|
};
|
|
101
141
|
max_tokens: number;
|
|
142
|
+
reasoning?: BroadsideReasoning;
|
|
102
143
|
};
|
|
103
144
|
};
|
|
104
145
|
export type BatchTerminalStatus = "completed" | "failed" | "expired" | "cancelled";
|
|
@@ -185,6 +226,8 @@ export type BroadsideConfig = {
|
|
|
185
226
|
* trade-off is not the same for every lens.
|
|
186
227
|
*/
|
|
187
228
|
lensModels: Partial<Record<BroadsideLensId, string>>;
|
|
229
|
+
/** Overrides every lens's reasoning setting when present. */
|
|
230
|
+
reasoning: BroadsideReasoning | null;
|
|
188
231
|
/**
|
|
189
232
|
* Repo defaults for the per-call run knobs. Each mirrors a tool parameter
|
|
190
233
|
* of the same name; an explicit parameter always wins. They live here so a
|
|
@@ -304,6 +347,7 @@ type LensDefinition = {
|
|
|
304
347
|
sliceBy: "none" | "directory" | "auto";
|
|
305
348
|
maxChars: number;
|
|
306
349
|
maxTokens: number;
|
|
350
|
+
reasoning?: BroadsideReasoning;
|
|
307
351
|
skipTestFiles?: boolean;
|
|
308
352
|
globsFor: (info: RepoInfo) => string[];
|
|
309
353
|
systemPrompt: (info: RepoInfo) => string;
|
|
@@ -313,7 +357,7 @@ export declare function getLens(lensId: BroadsideLensId): LensDefinition;
|
|
|
313
357
|
export declare function listLenses(): LensDefinition[];
|
|
314
358
|
export declare function collectRepoInfo(targetDir: string): Promise<RepoInfo>;
|
|
315
359
|
export declare function gatherSlices(targetDir: string, lens: LensDefinition, info: RepoInfo): Promise<FileSlice[]>;
|
|
316
|
-
export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number): BatchRequest;
|
|
360
|
+
export declare function buildBatchRequest(lens: LensDefinition, info: RepoInfo, slice: FileSlice, index: number, sliceCount: number, model?: string, maxTokensOverride?: number, reasoningOverride?: BroadsideReasoning): BatchRequest;
|
|
317
361
|
/**
|
|
318
362
|
* Pre-flight cost estimate for one lens.
|
|
319
363
|
*
|
package/dist/core/broadside.js
CHANGED
|
@@ -74,6 +74,34 @@ export const BROADSIDE_LENS_IDS = [
|
|
|
74
74
|
];
|
|
75
75
|
export const BROADSIDE_POLL_INTERVAL_MS = 15_000;
|
|
76
76
|
export const BROADSIDE_DEFAULT_POLL_BUDGET_MS = 25 * 60 * 1000;
|
|
77
|
+
/**
|
|
78
|
+
* The share of a lens's output budget reasoning may spend.
|
|
79
|
+
*
|
|
80
|
+
* `estimateCost` already budgets output at 75% of `maxTokens`; capping thinking
|
|
81
|
+
* at the remaining quarter makes that assumption true by construction and
|
|
82
|
+
* guarantees the answer has room. A floor keeps the cap sane for a small lens.
|
|
83
|
+
*/
|
|
84
|
+
export const BROADSIDE_REASONING_BUDGET_FRACTION = 0.25;
|
|
85
|
+
export const BROADSIDE_MIN_REASONING_TOKENS = 512;
|
|
86
|
+
/**
|
|
87
|
+
* Cap reasoning for a lens request — deliberately a cap, not an off switch.
|
|
88
|
+
*
|
|
89
|
+
* Disabling outright is not portable: `google/gemini-3.8-flash:batch` refuses
|
|
90
|
+
* the whole batch with *"Reasoning is mandatory for this endpoint and cannot be
|
|
91
|
+
* disabled"*, turning a partial result into none at all. Capping works whether
|
|
92
|
+
* or not a provider allows reasoning to be switched off.
|
|
93
|
+
*
|
|
94
|
+
* The failure this prevents is the budget being spent thinking rather than
|
|
95
|
+
* answering. Measured on one run: 5,758 of a 6,000-token budget went to
|
|
96
|
+
* reasoning, leaving ~230 tokens for JSON that truncated mid-structure — and
|
|
97
|
+
* those tokens bill at the full output rate. The shipped default model does the
|
|
98
|
+
* same thing less consistently (reasoning tokens from 0 to 5,757 across 13
|
|
99
|
+
* slices, three of them cut off at `finish_reason: length`), so this is not a
|
|
100
|
+
* multi-model concern.
|
|
101
|
+
*/
|
|
102
|
+
export function defaultReasoningFor(maxTokens) {
|
|
103
|
+
return { max_tokens: Math.max(BROADSIDE_MIN_REASONING_TOKENS, Math.floor(maxTokens * BROADSIDE_REASONING_BUDGET_FRACTION)) };
|
|
104
|
+
}
|
|
77
105
|
/** Thrown when a confirm hook declines a run. Nothing was submitted. */
|
|
78
106
|
export class BroadsideCancelledError extends Error {
|
|
79
107
|
constructor(message = "Broad-Side submission cancelled. Nothing was submitted.") {
|
|
@@ -1136,7 +1164,7 @@ async function sumFileSizes(targetDir, files) {
|
|
|
1136
1164
|
return total;
|
|
1137
1165
|
}
|
|
1138
1166
|
// ---------- request building ----------
|
|
1139
|
-
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride) {
|
|
1167
|
+
export function buildBatchRequest(lens, info, slice, index, sliceCount, model = BROADSIDE_MODEL, maxTokensOverride, reasoningOverride) {
|
|
1140
1168
|
const moduleTag = sanitizeId(slice.moduleName);
|
|
1141
1169
|
const customId = sliceCount > 1 ? `${lens.id}-${moduleTag}-${index + 1}` : `${lens.id}-${moduleTag}`;
|
|
1142
1170
|
return {
|
|
@@ -1149,6 +1177,9 @@ export function buildBatchRequest(lens, info, slice, index, sliceCount, model =
|
|
|
1149
1177
|
],
|
|
1150
1178
|
response_format: { type: "json_schema", json_schema: SCHEMAS[lens.schemaName] },
|
|
1151
1179
|
max_tokens: maxTokensOverride ?? lens.maxTokens,
|
|
1180
|
+
// Always sent, never inherited: an absent field means the model's
|
|
1181
|
+
// own default, and that default is what truncated the JSON.
|
|
1182
|
+
reasoning: reasoningOverride ?? lens.reasoning ?? defaultReasoningFor(maxTokensOverride ?? lens.maxTokens),
|
|
1152
1183
|
},
|
|
1153
1184
|
};
|
|
1154
1185
|
}
|
|
@@ -1297,6 +1328,24 @@ export async function persistBroadsideRun(broadsideDir, run) {
|
|
|
1297
1328
|
state.runs[index] = run;
|
|
1298
1329
|
});
|
|
1299
1330
|
}
|
|
1331
|
+
/** Read a `reasoning:` block from config.yaml, ignoring anything malformed. */
|
|
1332
|
+
function parseReasoningConfig(raw) {
|
|
1333
|
+
if (raw === false)
|
|
1334
|
+
return { enabled: false };
|
|
1335
|
+
if (raw === true)
|
|
1336
|
+
return { enabled: true };
|
|
1337
|
+
if (!raw || typeof raw !== "object")
|
|
1338
|
+
return null;
|
|
1339
|
+
const value = raw;
|
|
1340
|
+
const out = {};
|
|
1341
|
+
if (typeof value.enabled === "boolean")
|
|
1342
|
+
out.enabled = value.enabled;
|
|
1343
|
+
if (value.effort === "minimal" || value.effort === "low" || value.effort === "medium" || value.effort === "high")
|
|
1344
|
+
out.effort = value.effort;
|
|
1345
|
+
if (typeof value.max_tokens === "number" && value.max_tokens > 0)
|
|
1346
|
+
out.max_tokens = value.max_tokens;
|
|
1347
|
+
return Object.keys(out).length > 0 ? out : null;
|
|
1348
|
+
}
|
|
1300
1349
|
export async function loadBroadsideConfig(broadsideDir) {
|
|
1301
1350
|
const configPath = join(broadsideDir, BROADSIDE_CONFIG_FILE);
|
|
1302
1351
|
let raw = {};
|
|
@@ -1337,6 +1386,10 @@ export async function loadBroadsideConfig(broadsideDir) {
|
|
|
1337
1386
|
? { inputPerM: inputOverride, outputPerM: outputOverride }
|
|
1338
1387
|
: null,
|
|
1339
1388
|
lensModels,
|
|
1389
|
+
// An escape hatch, not a knob to reach for: a model whose reasoning is
|
|
1390
|
+
// worth paying for needs its lens maxTokens raised to cover both the
|
|
1391
|
+
// thinking and the answer, or the JSON truncates exactly as before.
|
|
1392
|
+
reasoning: parseReasoningConfig(raw.reasoning),
|
|
1340
1393
|
incremental: flag("incremental", false),
|
|
1341
1394
|
retryTruncated: flag("retry_truncated", true),
|
|
1342
1395
|
includeSynthesis: flag("include_synthesis", true),
|
|
@@ -1797,7 +1850,7 @@ export async function runBroadsideSubmit(cwd, apiKey, opts = {}) {
|
|
|
1797
1850
|
const { lens, maxTokens, lensModel, lensOutputCap } = priced;
|
|
1798
1851
|
const lensId = lens.id;
|
|
1799
1852
|
const slices = slicesByLens.get(lensId) ?? [];
|
|
1800
|
-
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens));
|
|
1853
|
+
const requests = slices.map((sl, i) => buildBatchRequest(lens, info, sl, i, slices.length, lensModel, maxTokens, config.reasoning ?? undefined));
|
|
1801
1854
|
for (const request of requests)
|
|
1802
1855
|
requestsByCustomId[request.custom_id] = request;
|
|
1803
1856
|
const entry = {
|
|
@@ -11,6 +11,7 @@ import { readFile, readdir } from "node:fs/promises";
|
|
|
11
11
|
import { join } from "node:path";
|
|
12
12
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
13
13
|
import { closeoutFileName, pathExists } from "../../core/index.js";
|
|
14
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
14
15
|
/** Closeout content over this many bytes is truncated before being passed to
|
|
15
16
|
* the rewriter. Keeps the orchestrator-side cost predictable. */
|
|
16
17
|
const CLOSEOUT_BYTE_BUDGET = 8000;
|
|
@@ -143,6 +144,7 @@ async function runRewriterOnce(ctx, prompt) {
|
|
|
143
144
|
const { session } = await createAgentSession({
|
|
144
145
|
cwd,
|
|
145
146
|
agentDir,
|
|
147
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
146
148
|
sessionManager: SessionManager.inMemory(cwd),
|
|
147
149
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
148
150
|
model: ctx.model,
|
|
@@ -13,6 +13,7 @@ import { access } from "node:fs/promises";
|
|
|
13
13
|
import { join, resolve } from "node:path";
|
|
14
14
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
15
15
|
import { canonicalPath, isWithinPath } from "../../core/index.js";
|
|
16
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
16
17
|
import { phaseCompactionExtension } from "./phase-compaction.js";
|
|
17
18
|
// Tools available to the phase sub-agent. Matches the codecarto interception
|
|
18
19
|
// allowlist (SAFE_TOOL_NAMES in extensions/codecarto/index.ts), minus bash.
|
|
@@ -104,6 +105,7 @@ export async function runPhase(ctx, prompt, callbacks = {}, options = {}, signal
|
|
|
104
105
|
const { session } = await createAgentSession({
|
|
105
106
|
cwd,
|
|
106
107
|
agentDir,
|
|
108
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
107
109
|
sessionManager,
|
|
108
110
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
109
111
|
model: ctx.model,
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// Model runtime for codecarto's child sessions (phase sub-agents, the
|
|
2
|
+
// next-phase rewriter, the dashboard narrator).
|
|
3
|
+
//
|
|
4
|
+
// All three child sessions load with `noExtensions: true` so a globally
|
|
5
|
+
// installed codecarto doesn't register its commands and tool guards twice
|
|
6
|
+
// inside its own sub-agent. That isolation has a side effect: providers
|
|
7
|
+
// registered by *other* global extensions via `pi.registerProvider()` — an
|
|
8
|
+
// Ollama Cloud bridge, a company gateway, any custom `streamSimple` provider —
|
|
9
|
+
// are registered onto the parent's ModelRuntime by the resource loader that
|
|
10
|
+
// loaded them. A child that builds a fresh ModelRuntime never sees them, so
|
|
11
|
+
// the parent's selected model resolves to a provider the child does not know,
|
|
12
|
+
// and the session throws `No API key found for <provider>` before its first
|
|
13
|
+
// turn.
|
|
14
|
+
//
|
|
15
|
+
// Carrying the parent's registered provider configs across keeps the child on
|
|
16
|
+
// the same model the user picked without reloading (and re-registering) the
|
|
17
|
+
// extensions themselves.
|
|
18
|
+
import { join } from "node:path";
|
|
19
|
+
import { ModelRuntime } from "@earendil-works/pi-coding-agent";
|
|
20
|
+
export async function createChildModelRuntime(ctx, agentDir) {
|
|
21
|
+
const runtime = await ModelRuntime.create({
|
|
22
|
+
authPath: join(agentDir, "auth.json"),
|
|
23
|
+
modelsPath: join(agentDir, "models.json"),
|
|
24
|
+
});
|
|
25
|
+
for (const providerId of ctx.modelRegistry.getRegisteredProviderIds()) {
|
|
26
|
+
const config = ctx.modelRegistry.getRegisteredProviderConfig(providerId);
|
|
27
|
+
if (!config)
|
|
28
|
+
continue;
|
|
29
|
+
try {
|
|
30
|
+
runtime.registerProvider(providerId, config);
|
|
31
|
+
}
|
|
32
|
+
catch {
|
|
33
|
+
// A provider the child can't accept is not worth failing the phase
|
|
34
|
+
// over: the child either doesn't need it (the user's model comes
|
|
35
|
+
// from a different provider) or fails later with the provider-
|
|
36
|
+
// specific error, which is more useful than one thrown here.
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
// Recompose the provider table so the newly registered providers land in
|
|
40
|
+
// the availability snapshot that `hasConfiguredAuth` reads. Offline: the
|
|
41
|
+
// parent already paid for any network catalog refresh.
|
|
42
|
+
await runtime.refresh({ allowNetwork: false });
|
|
43
|
+
return runtime;
|
|
44
|
+
}
|
|
@@ -14,6 +14,7 @@ import { readFile, readdir, rename, writeFile } from "node:fs/promises";
|
|
|
14
14
|
import { join } from "node:path";
|
|
15
15
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, SettingsManager, } from "@earendil-works/pi-coding-agent";
|
|
16
16
|
import { computeTotals, NARRATION_CACHE_RELATIVE_PATH, loadUsage, pathExists, stringifySimpleYaml, } from "../../core/index.js";
|
|
17
|
+
import { createChildModelRuntime } from "./child-model-runtime.js";
|
|
17
18
|
// Per-closeout byte budget when stuffing the narrator's input. Three
|
|
18
19
|
// closeouts × 4 KB each ≈ 12 KB of prompt context, which is well under any
|
|
19
20
|
// reasonable model's input window.
|
|
@@ -153,6 +154,7 @@ async function runNarratorOnce(ctx, prompt) {
|
|
|
153
154
|
const { session } = await createAgentSession({
|
|
154
155
|
cwd,
|
|
155
156
|
agentDir,
|
|
157
|
+
modelRuntime: await createChildModelRuntime(ctx, agentDir),
|
|
156
158
|
sessionManager: SessionManager.inMemory(cwd),
|
|
157
159
|
settingsManager: SettingsManager.create(cwd, agentDir),
|
|
158
160
|
model: ctx.model,
|
|
@@ -785,7 +785,25 @@ export default function codeCartographerExtension(pi) {
|
|
|
785
785
|
ctx.ui.notify(`Cannot complete ${validation.phaseId}: ${validation.overall}`, "error");
|
|
786
786
|
return;
|
|
787
787
|
}
|
|
788
|
-
|
|
788
|
+
// Completion refuses for reasons the framework words carefully — a
|
|
789
|
+
// missing phase handoff, a carry-forward without `derives_from`, a
|
|
790
|
+
// closure lacking runtime evidence. Those messages are the whole
|
|
791
|
+
// point of the refusal, and this was the one call in this file that
|
|
792
|
+
// let them escape as a rejection instead of showing them. The
|
|
793
|
+
// irony was sharp: /codecarto-next catches this same throw and tells
|
|
794
|
+
// the user to run /codecarto-complete manually, which then threw.
|
|
795
|
+
let completion;
|
|
796
|
+
try {
|
|
797
|
+
completion = await autoCompletePhase(ctx.cwd, validation);
|
|
798
|
+
}
|
|
799
|
+
catch (error) {
|
|
800
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
801
|
+
lastFeedbackLines = [`Completion refused: ${message}`];
|
|
802
|
+
setUiState(ctx, currentState, lastFeedbackLines);
|
|
803
|
+
ctx.ui.notify(message, "error");
|
|
804
|
+
return;
|
|
805
|
+
}
|
|
806
|
+
const { updatedState, closeoutNotice, warnings } = completion;
|
|
789
807
|
lastFeedbackLines = [
|
|
790
808
|
`Completed phase: ${validation.phaseId}`,
|
|
791
809
|
`Validation: ${validation.overall}`,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "codecartographer-pi",
|
|
3
|
-
"version": "0.19.
|
|
3
|
+
"version": "0.19.2",
|
|
4
4
|
"mcpName": "io.github.HuginnIndustries/codecartographer",
|
|
5
5
|
"description": "Turn an unfamiliar codebase into a validated reimplementation spec, then synthesize confirmed specs and a product vision into a traceable plan.",
|
|
6
6
|
"type": "module",
|