@viniciosrab/pi-claude-bridge 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.ts ADDED
@@ -0,0 +1,2706 @@
1
+ import { calculateCost, createAssistantMessageEventStream, type AssistantMessage, type AssistantMessageEventStream, type Context, type ImageContent, type Model, type SimpleStreamOptions, type TextContent, type Tool, type UserMessage } from "@earendil-works/pi-ai";
2
+ import { getModels } from "@earendil-works/pi-ai/compat";
3
+ import { buildSessionContext, compact, generateBranchSummary, keyHint, type BranchSummaryResult, type CompactionEntry, type ExtensionAPI, type ExtensionContext, type ExtensionUIContext } from "@earendil-works/pi-coding-agent";
4
+ import { query, type EffortLevel, type SDKMessage, type SettingSource } from "@anthropic-ai/claude-agent-sdk";
5
+ import type { Base64ImageSource, ContentBlockParam } from "@anthropic-ai/sdk/resources";
6
+ import { Text } from "@earendil-works/pi-tui";
7
+ import { createSession, deleteSession, openSession, repairToolPairing } from "cc-session-io";
8
+ import { appendFileSync, mkdirSync, realpathSync, statSync } from "fs";
9
+ import { homedir } from "os";
10
+ import { dirname, join } from "path";
11
+ import { PROVIDER_ID, messageContentToText, convertPiMessages } from "./convert.js";
12
+ import { applyLongContext, buildModels, claudeCodeModelId, type LongContextSettings, resolveModel as _resolveModel } from "./models.js";
13
+ import { MCP_SERVER_NAME, MCP_TOOL_PREFIX, renderSkillsBlock } from "./skills.js";
14
+ import { verifyWrittenSession as _verifyWrittenSession } from "./session-verify.js";
15
+ import { extractAllToolResults as _extractAllToolResults, type McpResult } from "./extract-tool-results.js";
16
+ import { QueryContext, ctx } from "./query-state.js";
17
+ import { makePromptStream, userMessage, type PromptStream } from "./prompt-stream.js";
18
+ import { claudeCodeSettings, loadConfig, markStartupNoticeShown, providerSettingSourcesOption, type Config } from "./config.js";
19
+ import {
20
+ collectPromptSkills,
21
+ projectPromptCapture,
22
+ sharedPromptCaptures,
23
+ } from "./prompt-capture.js";
24
+ import { collectCarriedAttachments, placeCarriedAttachments, type CarriedAttachment } from "./attachments.js";
25
+ import { createToolServer } from "./mcp-server.js";
26
+ import { buildActionSummary, type ToolCallState } from "./askclaude-ui.js";
27
+ import { askClaudeCallTags, askClaudeToolDescription, buildAskClaudeParams, resolveAskClaudeDefaults, resolveAskClaudeMode, type AskClaudeMode } from "./askclaude-schema.js";
28
+ import { nonSystemMessages, toBridgeContext } from "./transcript.js";
29
+
30
+ // --- Debug logging ---
31
+ // CLAUDE_BRIDGE_DEBUG=1 enables debug logging to ~/.pi/agent/claude-bridge.log
32
+
33
+ const DEBUG = process.env.CLAUDE_BRIDGE_DEBUG === "1";
34
+ const DEBUG_LOG_PATH = process.env.CLAUDE_BRIDGE_DEBUG_PATH || join(homedir(), ".pi", "agent", "claude-bridge.log");
35
+ const DIAG_LOG_PATH = join(homedir(), ".pi", "agent", "claude-bridge-diag.log");
36
+
37
+ // CLAUDE_BRIDGE_RECORD_STREAM=<path> appends every SDK message consumeQuery sees,
38
+ // one JSON object per line. Used by tests/lib/record-sdk-streams.mjs to capture
39
+ // replay fixtures, so unit tests assert against message shapes Claude Code really
40
+ // emitted rather than ones we imagined.
41
+ const RECORD_STREAM_PATH = process.env.CLAUDE_BRIDGE_RECORD_STREAM;
42
+
43
+ // Applied to every Claude Code subprocess the bridge spawns — provider, AskClaude
44
+ // and the compact summary. One place, so a guard is added once rather than three
45
+ // times, and so a missing one is visible.
46
+ //
47
+ // - ENABLE_CLAUDEAI_MCP_SERVERS=0: keep the user's claude.ai-connected MCP servers
48
+ // out of a pi session, which serves its own tools.
49
+ // - DISABLE_AUTO_COMPACT=1: pi owns compaction; CC compacting its own copy would
50
+ // diverge from pi's history, which is the source of truth for every rebuild.
51
+ const CC_CHILD_ENV = {
52
+ ENABLE_CLAUDEAI_MCP_SERVERS: "0",
53
+ DISABLE_AUTO_COMPACT: "1",
54
+ } as const;
55
+
56
+ // Pi owns context files on the provider path, so Claude Code must not load its
57
+ // own on top: otherwise a project CLAUDE.md arrives twice, and the user's
58
+ // ~/.claude/CLAUDE.md — a persona written for a harness that is not the one
59
+ // running — arrives at all, stamped "These instructions OVERRIDE any default
60
+ // behavior" and outranking Pi's own AGENTS.md.
61
+ //
62
+ // Excludes rather than settingSources: the source gate that suppresses CLAUDE.md
63
+ // is the same one that reads settings.json, where Bedrock/Vertex users keep
64
+ // `env` and `apiKeyHelper`. Patterns are matched with picomatch against absolute
65
+ // paths; "**/CLAUDE.md" covers the user, ancestor, project and .claude/ copies,
66
+ // while rules need their own. Managed/policy memory is not excludable by design.
67
+ const CLAUDE_MD_EXCLUDES = ["**/CLAUDE.md", "**/.claude/rules/**"];
68
+
69
+ // Ensure log directories exist when debug is enabled
70
+ if (DEBUG) {
71
+ try {
72
+ mkdirSync(dirname(DEBUG_LOG_PATH), { recursive: true });
73
+ mkdirSync(dirname(DIAG_LOG_PATH), { recursive: true });
74
+ } catch {
75
+ // If directory creation fails, debug functions will throw on first use
76
+ }
77
+ }
78
+
79
+ // Unique per module evaluation — confirms whether subagents share module state
80
+ const moduleInstanceId = Math.random().toString(36).slice(2, 8);
81
+
82
+ function debug(...args: unknown[]) {
83
+ if (!DEBUG) return;
84
+ const ts = new Date().toISOString();
85
+ const fmt = (a: unknown): string => {
86
+ if (typeof a === "string") return a;
87
+ if (a instanceof Error) return `${a.name}: ${a.message}${a.stack ? "\n" + a.stack : ""}`;
88
+ return JSON.stringify(a);
89
+ };
90
+ const msg = args.map(fmt).join(" ");
91
+ appendFileSync(DEBUG_LOG_PATH, `[${ts}] [${moduleInstanceId}] ${msg}\n`);
92
+ }
93
+
94
+ // Per-query CLI debug capture. When CLAUDE_BRIDGE_DEBUG=1, ask the Claude Code
95
+ // CLI subprocess to write its own debug log to a file we choose, and also
96
+ // forward its stderr into our debug stream. Drops straight into the real SDK's
97
+ // Options — see @anthropic-ai/claude-agent-sdk sdk.d.ts:1245 (debug, debugFile,
98
+ // stderr). Without this, CC's internal view of the world is invisible to us
99
+ // and "No conversation found" / empty-error reports are unactionable.
100
+ let nextCliDebugSeq = 1;
101
+ function makeCliDebugOptions(tag: string): { debug?: boolean; debugFile?: string; stderr?: (data: string) => void } {
102
+ if (!DEBUG) return {};
103
+ const seq = nextCliDebugSeq++;
104
+ const ts = new Date().toISOString().replace(/[:.]/g, "-");
105
+ const logDir = join(dirname(DEBUG_LOG_PATH), "cc-cli-logs");
106
+ try { mkdirSync(logDir, { recursive: true }); } catch { /* ignore */ }
107
+ const debugFile = join(logDir, `${ts}-${tag}-${seq}.log`);
108
+ debug(`cli-debug: ${tag} #${seq} → ${debugFile}`);
109
+ return {
110
+ debug: true,
111
+ debugFile,
112
+ stderr: (data: string) => {
113
+ for (const line of data.split(/\r?\n/)) {
114
+ if (line) debug(`[cli-stderr ${tag}#${seq}] ${line}`);
115
+ }
116
+ },
117
+ };
118
+ }
119
+
120
+ /** Unconditional diagnostic dump — for "should never happen" paths */
121
+ function diagDump(label: string, data: Record<string, unknown>) {
122
+ const ts = new Date().toISOString();
123
+ const entry = { ts, moduleInstanceId, label, ...data };
124
+ appendFileSync(DIAG_LOG_PATH, JSON.stringify(entry) + "\n");
125
+ debug(`DIAG: ${label} (see ${DIAG_LOG_PATH})`);
126
+ }
127
+
128
+ // --- Constants ---
129
+
130
+ // Marks which bridge module instance owns the registered provider's stream fn.
131
+ // Full registration policy (first vs later instances, shared vs own registry):
132
+ // see the "--- Provider ---" block in activate() below.
133
+ const ACTIVE_STREAM_SIMPLE_KEY = Symbol.for("claude-bridge:activeStreamSimple");
134
+
135
+ // Claude Code's own builtin tools, for the AskClaude path where CC really runs
136
+ // them. The provider path never sees these — it starts CC with `tools: []`.
137
+ const SDK_TO_PI_TOOL_NAME: Record<string, string> = {
138
+ read: "read", write: "write", edit: "edit", bash: "bash",
139
+ };
140
+
141
+ // MODELS is buildModels(getModels("anthropic")) — projection kept in models.js.
142
+ const MODELS = buildModels(getModels("anthropic"));
143
+ let providerSettings: NonNullable<Config["provider"]> = {};
144
+ let longContextSettings: LongContextSettings = { plan: "pro", longContextExtraUsage: false };
145
+
146
+ function resolveModel(input: string) {
147
+ return _resolveModel(MODELS, input);
148
+ }
149
+
150
+ // --- Error handling ---
151
+
152
+ function errorMessage(err: unknown): string {
153
+ if (err instanceof Error) return err.message;
154
+ if (err && typeof err === "object") {
155
+ const obj = err as Record<string, unknown>;
156
+ if (typeof obj.message === "string") return obj.message;
157
+ if (typeof obj.error === "string") return obj.error;
158
+ try { return JSON.stringify(err); } catch {}
159
+ }
160
+ return String(err);
161
+ }
162
+
163
+ // AskClaude mode presets — controls which CC tools are blocked per mode.
164
+ // Only block tools that can't work (no pi TUI for user interaction).
165
+ // Other CC tools (Agent, SendMessage, RemoteTrigger, Tasks, etc.) are intentionally not blocked.
166
+ const ASKCLAUDE_ALWAYS_BLOCKED = [
167
+ "AskUserQuestion", "EnterPlanMode", "ExitPlanMode",
168
+ "ToolSearch", // probes for blocked tools, wastes tokens
169
+ "ScheduleWakeup", // no harness to fire wakeup from inside a delegated subagent
170
+ ];
171
+ const MODE_DISALLOWED_TOOLS: Record<AskClaudeMode, string[]> = {
172
+ full: ASKCLAUDE_ALWAYS_BLOCKED,
173
+ read: [
174
+ ...ASKCLAUDE_ALWAYS_BLOCKED,
175
+ "Write", "Edit", "Bash", "NotebookEdit",
176
+ "EnterWorktree", "ExitWorktree", "CronCreate", "CronDelete", "TeamCreate", "TeamDelete",
177
+ ],
178
+ none: [
179
+ ...ASKCLAUDE_ALWAYS_BLOCKED,
180
+ "Read", "Write", "Edit", "Glob", "Grep", "Bash", "Agent",
181
+ "NotebookEdit", "EnterWorktree", "ExitWorktree",
182
+ "CronCreate", "CronDelete", "TeamCreate", "TeamDelete",
183
+ "WebFetch", "WebSearch",
184
+ ],
185
+ };
186
+
187
+ // --- Session persistence ---
188
+
189
+ interface SessionState {
190
+ sessionId: string;
191
+ cursor: number;
192
+ cwd: string;
193
+ // The pi session this CC conversation serves, from the provider call's
194
+ // options.sessionId. Attribution for history rewrites: a subagent's
195
+ // session_compact must not force a rebuild of a conversation that belongs
196
+ // to a different pi session. Null until a provider call recorded it.
197
+ piSessionId?: string;
198
+ // Force the next syncSharedSession call down the REBUILD path. Set when
199
+ // pi has mutated its messages array out from under us (compact, tree
200
+ // navigation) or after an abort left the JSONL in an indeterminate state.
201
+ // REBUILD wipes and rewrites the file to match pi's current history.
202
+ needsRebuild?: boolean;
203
+ // Set ONLY where we have just killed a CC subprocess: an abort, or a query
204
+ // discarded because pi rewrote the history under it. The killed subprocess
205
+ // may still be flushing a late "[Request interrupted by user]" record to the
206
+ // session JSONL. Reusing the same sessionId/path would race that orphan write
207
+ // into our fresh file and break CC's parent-uuid chain on the next resume.
208
+ // When this flag is set, REBUILD takes a fresh UUID and skips deleteSession
209
+ // so the orphan writes land on a dead inode. A compact or tree navigation
210
+ // with no query in flight does NOT set this — there's no concurrent CC writer
211
+ // then, so in-place rebuild (preserve UUID, deleteSession + createSession) is safe.
212
+ forceRotate?: boolean;
213
+ }
214
+
215
+ /**
216
+ * Claude Code's `@file` expansions from the session about to be replaced.
217
+ *
218
+ * Must be called before `deleteSession`, which wipes the file they live in —
219
+ * reading after it yields nothing, with no error to notice.
220
+ */
221
+ function readCarriedAttachments(sessionId: string, cwd: string): CarriedAttachment[] {
222
+ try {
223
+ const previous = openSession({ sessionId, projectPath: cwd, claudeDir: process.env.CLAUDE_CONFIG_DIR });
224
+ return collectCarriedAttachments(previous.records);
225
+ } catch (error) {
226
+ // A post-abort rebuild reads a file the killed CC subprocess may have been
227
+ // midway through writing, and cc-session-io parses each line with a bare
228
+ // JSON.parse, so a truncated last line throws. Throwing here would turn a
229
+ // lost attachment into a failed turn; carrying none is exactly what happened
230
+ // before this existed, so the failure mode is bounded by the status quo.
231
+ debug(`WARNING: could not read attachments from session ${sessionId.slice(0, 8)}:`, error);
232
+ return [];
233
+ }
234
+ }
235
+
236
+ /** The session key a pi session's provider calls address. Unattributed calls
237
+ * (no options.sessionId — AskClaude's direct sync, or a host that omits it)
238
+ * share the "(none)" bucket: they cannot be told apart, so they share the
239
+ * pre-existing single-slot semantics. */
240
+ function sessionKey(piSessionId: string | null | undefined): string {
241
+ return piSessionId ?? "(none)";
242
+ }
243
+
244
+ /** Mirror of the CC conversation one pi session's turns are running on. One
245
+ * entry per pi session: a bridge process serves several sessions at once
246
+ * (pi-subagents children run their own AgentSessions), and a single shared
247
+ * slot forced them to fight over it — a length-matching foreign sync could
248
+ * REUSE or rebuild another session's CC file, and the completion capture was
249
+ * last-writer-wins (a foreground child in the parent's first turn permanently
250
+ * reassigned the parent's conversation). Keyed lookup removes the fight: each
251
+ * session's reads, writes and teardown marks touch only its own mirror. */
252
+ const sharedSessions = new Map<string, SessionState>();
253
+
254
+ /** The mirror for `piSessionId`, or null when this session has none yet. */
255
+ function sessionStateFor(piSessionId: string | null | undefined): SessionState | null {
256
+ return sharedSessions.get(sessionKey(piSessionId)) ?? null;
257
+ }
258
+
259
+ /** Replace (or plant) the mirror for `piSessionId`. */
260
+ function setSessionStateFor(piSessionId: string | null | undefined, state: SessionState | null): void {
261
+ if (state === null) sharedSessions.delete(sessionKey(piSessionId));
262
+ else sharedSessions.set(sessionKey(piSessionId), state);
263
+ }
264
+
265
+ // pi replaced one of its sessions' history (compact, tree) rather than appending
266
+ // to it. Read on the tool-result path — the one provider call that never reaches
267
+ // syncSharedSession — so a query parked at a tool boundary is discarded rather
268
+ // than resumed (issue #101). Keyed by pi session id, not process-global: a bridge
269
+ // process serves several pi sessions at once (subagents run their own
270
+ // AgentSessions and can compact mid-run while the parent is parked), and
271
+ // marking across that boundary kills healthy queries (or, worse, rebuilds the
272
+ // parent around a compaction that never touched it).
273
+ //
274
+ // Holds real pi session ids only, never the "(none)" key: no production caller
275
+ // marks with null (the rewrite events attribute via ctx.sessionManager), and
276
+ // every reader guards on a non-null piSessionId — so a "(none)" entry could
277
+ // never be matched or consumed, only leaked.
278
+ const historyRewrittenBySession = new Set<string>();
279
+
280
+ /** Handlers that arm rewrite staleness, one per module instance. Worktree-
281
+ * spawned subagents can load this module fresh (pi's loader cache is keyed on
282
+ * cwd, and a worktree cwd clears it), while the serving instance — whose
283
+ * streamSimple the pi sessions actually call — is whoever registered first.
284
+ * A fresh instance must forward its session's rewrites to the serving one.
285
+ * Symbol.for: one registry per process, like SHARED_CAPTURES_KEY below. */
286
+ const MARK_REBUILD_HOOKS_KEY = Symbol.for("claude-bridge:markRebuildHooks");
287
+ type MarkRebuildHook = (piSession: string | null, event: string) => void;
288
+ const markRebuildHooks: Set<MarkRebuildHook> =
289
+ ((globalThis as Record<symbol, unknown>)[MARK_REBUILD_HOOKS_KEY] as Set<MarkRebuildHook> | undefined) ?? new Set();
290
+ (globalThis as Record<symbol, unknown>)[MARK_REBUILD_HOOKS_KEY] = markRebuildHooks;
291
+
292
+ /** pi mutated its messages array out from under us: force the next
293
+ * syncSharedSession down REBUILD, and arm the discard above. `piSession`
294
+ * is never null from an event handler (each pi session has its own runner).
295
+ * The "(none)" key still covers the mirror for a hypothetical direct caller
296
+ * with no session id, so such a rewrite forces the REBUILD side; the discard
297
+ * set holds real ids only — see the comment on historyRewrittenBySession. */
298
+ function markRebuildForSession(piSession: string | null, event: string): void {
299
+ // The rewriting session's own mirror: the rewrite changed the history it was
300
+ // built from, so its next sync must REBUILD rather than REUSE. Every other
301
+ // session's mirror stays untouched — its conversation was never rewritten.
302
+ const key = sessionKey(piSession);
303
+ const state = sharedSessions.get(key);
304
+ if (!state) {
305
+ debug(`${event}: history rewritten, no session to mark yet`);
306
+ } else {
307
+ sharedSessions.set(key, { ...state, needsRebuild: true });
308
+ debug(`${event}: marking needsRebuild on session ${state.sessionId.slice(0, 8)}`);
309
+ }
310
+ // Arming parked contexts cannot wait for delivery: the entry checks
311
+ // `resultCtx.historyStale`, and a rewrite usually lands *while* the query is
312
+ // parked (compaction runs inside pi's turn loop, not between provider calls).
313
+ if (piSession) historyRewrittenBySession.add(piSession);
314
+ armStaleContexts();
315
+ }
316
+
317
+ /** Copy each armed session's mark onto the parked queries built from it. */
318
+ function armStaleContexts(): void {
319
+ for (const c of activeQueryContexts) {
320
+ if (c.piSessionId && historyRewrittenBySession.has(c.piSessionId)) c.historyStale = true;
321
+ }
322
+ }
323
+
324
+ // This instance enlists. Session ids reaching any hook equal options.sessionId
325
+ // on the serving instance's provider calls, so forwarding is safe: only the
326
+ // owning instance's contexts and SessionState match the key.
327
+ markRebuildHooks.add(markRebuildForSession);
328
+
329
+ /** Event handlers call this: it fans the rewrite out to every module instance
330
+ * in the process, of which exactly one is serving provider traffic for any
331
+ * given pi session. */
332
+ function sponsorMarkRebuildForSession(piSession: string | null, event: string): void {
333
+ for (const hook of markRebuildHooks) hook(piSession, event);
334
+ }
335
+
336
+ // Convert pi messages to Anthropic API format for session import.
337
+ // Lossy: only text, thinking and toolCall blocks survive, and thinking only when
338
+ // Claude Code itself minted the signature. An assistant message whose blocks all
339
+ // filter out keeps its slot with a placeholder, since dropping it can create a
340
+ // tool_result with no preceding tool_use. A turn aborted before anything streamed
341
+ // is dropped instead — it never had content, and inventing one diverges from the
342
+ // prefix Claude Code cached.
343
+ function convertAndImportMessages(
344
+ session: ReturnType<typeof createSession>,
345
+ messages: Context["messages"],
346
+ customToolNameToSdk?: Map<string, string>,
347
+ carried?: readonly CarriedAttachment[],
348
+ ): void {
349
+ const { anthropicMessages, sanitizedIds, dropped } = convertPiMessages(messages, customToolNameToSdk);
350
+
351
+ debug(`convertAndImportMessages: ${messages.length} pi msgs → ${anthropicMessages.length} anthropic msgs`);
352
+ debug(`convertAndImportMessages: imported roles:`, anthropicMessages.map((m, i) => {
353
+ const c = m.content;
354
+ if (typeof c === "string") return `[${i}]${m.role}:text`;
355
+ if (Array.isArray(c)) return `[${i}]${m.role}:${(c).map((b) => b.type).join("+")}`;
356
+ return `[${i}]${m.role}:?`;
357
+ }).join(" "));
358
+ // The roles line above shows only what survived, so a stripped block is
359
+ // indistinguishable there from one that never existed. Name the losses.
360
+ const droppedParts = [
361
+ dropped.thinking ? `${dropped.thinking} thinking (${[...dropped.providers].sort().join(", ")})` : "",
362
+ dropped.abortedTurns ? `${dropped.abortedTurns} aborted turn(s)` : "",
363
+ ...[...dropped.other].map(([type, n]) => `${n} ${type}`),
364
+ ].filter(Boolean);
365
+ if (droppedParts.length > 0) {
366
+ debug(`convertAndImportMessages: dropped ${droppedParts.join(", ")}`);
367
+ }
368
+ if (sanitizedIds.size > 0) {
369
+ debug(`convertAndImportMessages: sanitized ${sanitizedIds.size} tool IDs:`,
370
+ [...sanitizedIds.entries()].map(([orig, clean]) => orig === clean ? orig : `${orig}→${clean}`).join(", "));
371
+ }
372
+ // Pre-repair for debug logging; importMessages also repairs internally (idempotent).
373
+ const repaired = repairToolPairing(anthropicMessages);
374
+ if (repaired.length !== anthropicMessages.length) {
375
+ debug(`convertAndImportMessages: repairToolPairing ${anthropicMessages.length} → ${repaired.length} msgs`);
376
+ }
377
+ // Placement runs against the repaired array because that is the index space
378
+ // importMessages reads. Attachments are links in CC's uuid chain, so they have
379
+ // to be written in order with the messages, not appended afterwards.
380
+ const placed = carried?.length
381
+ ? placeCarriedAttachments(carried, repaired as unknown as { role: string; content: unknown }[])
382
+ : undefined;
383
+ if (placed?.skipped.length) {
384
+ debug(`convertAndImportMessages: dropped ${placed.skipped.length} carried attachment(s): ${placed.skipped.join("; ")}`);
385
+ }
386
+ if (placed?.attachments.length) {
387
+ debug(`convertAndImportMessages: carrying ${placed.attachments.length} attachment(s) across the rebuild`);
388
+ }
389
+ if (repaired.length) {
390
+ session.importMessages(repaired, placed?.attachments.length ? { attachments: placed.attachments } : undefined);
391
+ }
392
+ }
393
+
394
+ // Pi doesn't pass tool results directly — it appends them to the context and calls
395
+ // the provider again. Thin wrapper over extract-tool-results.js that adds per-turn
396
+ // debug logging at the extraction boundary.
397
+ function extractAllToolResults(context: Context): McpResult[] {
398
+ const { results, stopIdx } = _extractAllToolResults(context.messages as unknown as Array<{ role: string; [key: string]: unknown }>);
399
+ debug(`extractAllToolResults: ${results.length} results from ${context.messages.length} msgs, stopped at index ${stopIdx}`);
400
+ debug(`extractAllToolResults: all msg roles:`, context.messages.map((m, i) => `[${i}]${m.role}`).join(" "));
401
+ for (let r = 0; r < results.length; r++) {
402
+ debug(`extractAllToolResults: result[${r}] id=${results[r].toolCallId}${results[r].isError ? " ERROR" : ""} preview:`, JSON.stringify(results[r].content).slice(0, 150));
403
+ }
404
+ return results;
405
+ }
406
+
407
+ /** Index of the first message of the current user turn — the trailing run of
408
+ * user messages that has not been written into the Claude Code session yet.
409
+ * Equals messages.length when the last message is not a user message.
410
+ *
411
+ * Single source of truth for the history/prompt split: everything before this
412
+ * index is replayed as session history, everything from it onward becomes the
413
+ * prompt. Deriving both halves from one index is what keeps a message from
414
+ * landing in both — an extension appending a display-only user message after
415
+ * the real one (see issue #34) makes the turn longer than one message. */
416
+ function turnStart(messages: Context["messages"]): number {
417
+ let i = messages.length;
418
+ while (i > 0 && messages[i - 1].role === "user") i--;
419
+ return i;
420
+ }
421
+
422
+ /** Extract the current user turn as a prompt string. Returns null if the last message is not a user message. */
423
+ function extractUserPrompt(messages: Context["messages"]): string | null {
424
+ const turn = messages.slice(turnStart(messages)) as UserMessage[];
425
+ if (turn.length === 0) return null;
426
+ // Drop empties before joining so an all-empty turn still yields "" and trips
427
+ // the caller's empty-prompt guard rather than sending bare newlines.
428
+ return turn
429
+ .map((m) => (typeof m.content === "string" ? m.content : messageContentToText(m.content)))
430
+ .filter((text) => text)
431
+ .join("\n");
432
+ }
433
+
434
+ /** Extract the current user turn as ContentBlockParam[] (preserving images).
435
+ * Returns null if no images — caller should fall back to string prompt. */
436
+ function extractUserPromptBlocks(messages: Context["messages"]): ContentBlockParam[] | null {
437
+ const turn = messages.slice(turnStart(messages)) as UserMessage[];
438
+ if (turn.length === 0) return null;
439
+
440
+ let hasImage = false;
441
+ const blocks: ContentBlockParam[] = [];
442
+ for (const message of turn) {
443
+ const content: (TextContent | ImageContent)[] = typeof message.content === "string"
444
+ ? [{ type: "text", text: message.content }]
445
+ : message.content;
446
+ // Off-type content violates UserMessage's contract, so fail rather than
447
+ // degrade — but name the shape, since the cause is almost always another
448
+ // extension appending a malformed message, not this file.
449
+ if (!Array.isArray(content)) {
450
+ throw new Error(
451
+ `extractUserPromptBlocks: user message content must be a string or block array, got ${typeof content} — likely a malformed message from another extension`,
452
+ );
453
+ }
454
+ for (const block of content) {
455
+ if (block.type === "text" && block.text) {
456
+ blocks.push({ type: "text", text: block.text });
457
+ } else if (block.type === "image") {
458
+ // Guard before logging: data-less image blocks do occur, and reading
459
+ // .length off the missing field in the debug template would throw
460
+ // before this check ever runs (template args evaluate unconditionally).
461
+ if (!block.data || !block.mimeType) {
462
+ debug(`image block missing data or mimeType, skipping: keys=${Object.keys(block).join(",")}`);
463
+ continue;
464
+ }
465
+ debug(`image block: mimeType=${block.mimeType}, data length=${block.data.length}`);
466
+ hasImage = true;
467
+ blocks.push({
468
+ type: "image",
469
+ source: {
470
+ type: "base64",
471
+ media_type: block.mimeType as Base64ImageSource["media_type"],
472
+ data: block.data,
473
+ },
474
+ });
475
+ }
476
+ }
477
+ }
478
+ debug(`extractUserPromptBlocks: ${turn.length} msgs in turn, ${blocks.length} blocks, types=${blocks.map((b) => b.type).join(",")}`);
479
+ return hasImage ? blocks : null;
480
+ }
481
+
482
+ function newAssistantOutput(model: Model<any>, text: string, stopReason: AssistantMessage["stopReason"], errorMessage?: string): AssistantMessage {
483
+ return {
484
+ role: "assistant",
485
+ content: text ? [{ type: "text", text }] : [],
486
+ api: model.api,
487
+ provider: model.provider,
488
+ model: model.id,
489
+ usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0,
490
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
491
+ stopReason,
492
+ ...(errorMessage ? { errorMessage } : {}),
493
+ timestamp: Date.now(),
494
+ };
495
+ }
496
+
497
+ function extractIsolatedSummaryPrompt(messages: Context["messages"]): string {
498
+ if (messages.length !== 1 || messages[0].role !== "user") {
499
+ throw new Error(
500
+ `isolatedStreamFn: expected exactly 1 user message, got ${messages.length} ` +
501
+ `(${messages.map((m) => m.role).join(",")})`,
502
+ );
503
+ }
504
+ const promptText = extractUserPrompt(messages);
505
+ if (!promptText) throw new Error("isolatedStreamFn: summarization prompt is empty");
506
+ return promptText;
507
+ }
508
+
509
+ /** Failure text for an SDK result, or undefined when it succeeded. CC reports API failures
510
+ * (429 capacity, overload, prompt-too-long) with `is_error` on an otherwise success-shaped
511
+ * result; the dedicated error subtypes carry `errors` instead. */
512
+ function resultErrorText(message: SDKMessage): string | undefined {
513
+ const result = message as SDKMessage & { subtype?: string; is_error?: boolean; result?: string; errors?: unknown; error?: unknown };
514
+ if (result.subtype === "success") return result.is_error ? result.result || "Claude Code reported an error" : undefined;
515
+ if (Array.isArray(result.errors) && result.errors.length) return result.errors.map(String).join("\n");
516
+ if (typeof result.error === "string") return result.error;
517
+ return `Claude Code failed: ${result.subtype ?? "unknown result"}`;
518
+ }
519
+
520
+ /** Name a failure as a rate limit when a rejection preceded it.
521
+ *
522
+ * pi has no typed rate-limit error — `stopReason` is only ever `"error"` and the sole carrier
523
+ * is `errorMessage` — so everything that reacts to a rate limit pattern-matches that string:
524
+ * pi-subagents gates `fallbackModels` on its own pattern list, and key-rotating extensions use
525
+ * their own. Claude Code words a subscription limit as "You're out of extra usage · resets
526
+ * 6:30pm", which matches none of them, so an exhausted quota reads as a fatal error and the
527
+ * fallback chain never runs (issue #58).
528
+ *
529
+ * Leading with "Claude rate limit" rather than appending keeps the phrase in any truncated
530
+ * render, and avoids the `<tool> failed (exit N):` shape that pi-subagents treats as a tool
531
+ * failure and refuses to retry. */
532
+ function describeRateLimitFailure(rejection: { rateLimitType?: string; resetsAt?: number }, failure: string): string {
533
+ const kind = rejection.rateLimitType ? ` (${rejection.rateLimitType})` : "";
534
+ const resets = rejection.resetsAt ? ` — resets ${new Date(rejection.resetsAt * 1000).toLocaleTimeString()}` : ""; // resetsAt: Unix seconds (unit undocumented in the SDK; observed)
535
+ return `Claude rate limit${kind}${resets}: ${failure}`;
536
+ }
537
+
538
+ function isolatedStreamFn(model: Model<any>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream {
539
+ const stream = createAssistantMessageEventStream();
540
+ void runIsolatedSummary(model, context, options, stream);
541
+ return stream;
542
+ }
543
+
544
+ async function runIsolatedSummary(
545
+ model: Model<any>,
546
+ context: Context,
547
+ options: SimpleStreamOptions | undefined,
548
+ stream: AssistantMessageEventStream,
549
+ ): Promise<void> {
550
+ // pi delivers compaction/branch-summary requests as a transcript: the summarization
551
+ // prompt folded into a leading system message ahead of the lone user message
552
+ // (issue #106). toBridgeContext restores the prompt/tools fields the extraction
553
+ // assertion below assumes; the summarization prompt still reaches CC as its systemPrompt.
554
+ context = toBridgeContext(context);
555
+ let sdkQuery: ReturnType<typeof query> | undefined;
556
+ let wasAborted = false;
557
+ const onAbort = () => {
558
+ wasAborted = true;
559
+ void sdkQuery?.interrupt().catch(() => {});
560
+ try { sdkQuery?.close(); } catch {}
561
+ };
562
+
563
+ try {
564
+ // One-off summarizer calls (compaction, branch summary, turn prefix, bug report —
565
+ // anything routed through pi's completeSummarization) are marked cacheRetention:
566
+ // "none". Any of them may appear in a future pi release without a bridge change,
567
+ // so route on the marker, not on which summarizer is calling. Non-summarizer calls
568
+ // must still match the [system,user] compaction shape exactly.
569
+ const isOneOffSummary = options?.cacheRetention === "none";
570
+ const promptText = isOneOffSummary
571
+ ? extractUserPrompt(context.messages)
572
+ : extractIsolatedSummaryPrompt(context.messages);
573
+ if (!promptText) throw new Error("runIsolatedSummary: one-off summary without a user prompt (last message is not user?)");
574
+ const cwd = process.cwd();
575
+ const compactProviderSettings = loadConfig(cwd).provider;
576
+ const claudeExecutable = compactProviderSettings?.pathToClaudeCodeExecutable;
577
+ const cliModel = claudeCodeModelId(model, longContextSettings);
578
+ debug(`compact summary: spawn model=${cliModel} registeredModel=${model.id} promptLen=${promptText.length}`);
579
+
580
+ sdkQuery = query({
581
+ prompt: promptText,
582
+ options: {
583
+ cwd,
584
+ env: { ...process.env, ...CC_CHILD_ENV },
585
+ settings: { autoMemoryEnabled: false },
586
+ tools: [],
587
+ strictMcpConfig: true,
588
+ settingSources: [] as SettingSource[],
589
+ skills: [],
590
+ persistSession: false,
591
+ systemPrompt: context.systemPrompt,
592
+ model: cliModel,
593
+ maxTurns: 1,
594
+ ...(claudeExecutable ? { pathToClaudeCodeExecutable: claudeExecutable } : {}),
595
+ ...makeCliDebugOptions("compact-summary"),
596
+ },
597
+ });
598
+
599
+ if (options?.signal) {
600
+ if (options.signal.aborted) onAbort();
601
+ else options.signal.addEventListener("abort", onAbort, { once: true });
602
+ }
603
+
604
+ let assistantText = "";
605
+ let finalText = "";
606
+ let errorText: string | undefined;
607
+ let firstEventLogged = false;
608
+
609
+ for await (const message of sdkQuery) {
610
+ if (!firstEventLogged) {
611
+ debug(`compact summary: first event type=${message.type}`);
612
+ firstEventLogged = true;
613
+ }
614
+ if (wasAborted) break;
615
+
616
+ if (message.type === "assistant") {
617
+ for (const block of (message as any).message?.content ?? []) {
618
+ if (block.type === "text" && typeof block.text === "string") assistantText += block.text;
619
+ }
620
+ } else if (message.type === "result") {
621
+ logServedContextWindow("compact summary", message, model);
622
+ errorText = resultErrorText(message);
623
+ if (!errorText && message.subtype === "success") finalText = message.result || assistantText;
624
+ }
625
+ }
626
+
627
+ if (wasAborted) {
628
+ const output = newAssistantOutput(model, "", "aborted", "Operation aborted");
629
+ debug("compact summary: aborted");
630
+ stream.push({ type: "error", reason: "aborted", error: output });
631
+ stream.end();
632
+ return;
633
+ }
634
+
635
+ const text = finalText || assistantText;
636
+ if (errorText || !text.trim()) {
637
+ const msg = errorText ?? "Claude Code summary returned empty text";
638
+ debug(`compact summary: error ${msg}`);
639
+ stream.push({ type: "error", reason: "error", error: newAssistantOutput(model, "", "error", msg) });
640
+ stream.end();
641
+ return;
642
+ }
643
+
644
+ debug(`compact summary: done textLen=${text.length}`);
645
+ stream.push({ type: "done", reason: "stop", message: newAssistantOutput(model, text, "stop") });
646
+ stream.end();
647
+ } catch (err) {
648
+ const msg = errorMessage(err);
649
+ debug("runIsolatedSummary threw; pushing terminal error", err);
650
+ stream.push({ type: "error", reason: "error", error: newAssistantOutput(model, "", "error", msg) });
651
+ stream.end();
652
+ } finally {
653
+ options?.signal?.removeEventListener("abort", onAbort);
654
+ try { sdkQuery?.close(); } catch {}
655
+ }
656
+ }
657
+
658
+ function reinjectPriorCompactionFileOps(branchEntries: Array<{ type: string; details?: unknown }>, preparation: { fileOps: { read: Set<string>; edited: Set<string> } }): void {
659
+ const prior = [...branchEntries]
660
+ .reverse()
661
+ .find((entry): entry is CompactionEntry => entry.type === "compaction");
662
+ const details = prior?.details as { readFiles?: unknown; modifiedFiles?: unknown } | undefined;
663
+ if (!Array.isArray(details?.readFiles) || !Array.isArray(details?.modifiedFiles)) return;
664
+ for (const file of details.readFiles) preparation.fileOps.read.add(String(file));
665
+ for (const file of details.modifiedFiles) preparation.fileOps.edited.add(String(file));
666
+ debug(`compact takeover: re-injected prior file ops read=${details.readFiles.length} modified=${details.modifiedFiles.length}`);
667
+ }
668
+
669
+ interface SyncResult {
670
+ sessionId: string | null;
671
+ preserveSharedSession?: boolean;
672
+ }
673
+
674
+ /**
675
+ * Ensure the shared session has all messages up to (but not including) the last user message.
676
+ * Returns session ID to resume from, or null if no resume needed.
677
+ */
678
+ // Read the session file we just wrote and sanity-check it. Warns instead of
679
+ // throwing — CC may be more tolerant than our checks, so a false positive
680
+ // shouldn't block the user. Pure logic is in session-verify.js; this wrapper
681
+ // fans each warning out to debug log + piUI notify + diagDump.
682
+ function verifyWrittenSession(
683
+ jsonlPath: string,
684
+ expectedSessionId: string,
685
+ expectedRecordCount: number,
686
+ cwd: string,
687
+ ): void {
688
+ const warnings = _verifyWrittenSession(jsonlPath, expectedSessionId, expectedRecordCount);
689
+ for (const msg of warnings) {
690
+ debug(`WARNING session verify: ${msg}`);
691
+ piUI?.notify(
692
+ `Session file issue: ${msg}\n` +
693
+ `cwd=${cwd} realpath=${safeRealpath(cwd)} CLAUDE_CONFIG_DIR=${process.env.CLAUDE_CONFIG_DIR ?? "(unset)"}\n` +
694
+ `Please copy and paste this message into a new issue at https://github.com/elidickinson/pi-claude-bridge/issues/new` +
695
+ (DEBUG ? ` and attach ${DEBUG_LOG_PATH}` : ` (rerun with CLAUDE_BRIDGE_DEBUG=1 to capture a debug log)`),
696
+ "warning",
697
+ );
698
+ diagDump("session_verify_fail", { msg, jsonlPath, cwd, realpath: safeRealpath(cwd), claudeConfigDir: process.env.CLAUDE_CONFIG_DIR ?? null });
699
+ }
700
+ }
701
+
702
+ function safeRealpath(p: string): string {
703
+ try { return realpathSync(p); } catch (e) { return `<failed: ${(e as Error).message}>`; }
704
+ }
705
+
706
+ // Diagnostic snapshot of where a session file was just written. Catches the
707
+ // class of bugs where pi writes to ~/.claude/projects/<X> but CC SDK reads
708
+ // from ~/.claude/projects/<Y> (symlinks, CLAUDE_CONFIG_DIR, hash mismatch).
709
+ function debugSessionPaths(label: string, cwd: string, jsonlPath: string): void {
710
+ const realCwd = safeRealpath(cwd);
711
+ let fileSize: number | null = null;
712
+ let fileExists = false;
713
+ try {
714
+ const st = statSync(jsonlPath);
715
+ fileExists = true;
716
+ fileSize = st.size;
717
+ } catch { /* file may not exist yet */ }
718
+ debug(`${label}: cwd=${cwd}`);
719
+ if (realCwd !== cwd) debug(`${label}: realpath(cwd)=${realCwd} (DIFFERS — symlink-resolved path is what CC SDK uses)`);
720
+ debug(`${label}: jsonlPath=${jsonlPath}`);
721
+ debug(`${label}: fileExists=${fileExists}${fileSize != null ? ` size=${fileSize}` : ""}`);
722
+ debug(`${label}: env.CLAUDE_CONFIG_DIR=${process.env.CLAUDE_CONFIG_DIR ?? "(unset)"} HOME=${process.env.HOME ?? "(unset)"}`);
723
+ }
724
+
725
+ // Two semantic paths:
726
+ // REUSE — pi's history is in sync with the existing sharedSession (or drifted
727
+ // only by the trailing final-assistant message that pi appends after
728
+ // streamSimple returns, which CC's own persisted session already has).
729
+ // Returns the existing sessionId. Keeps CC's prompt cache warm.
730
+ // REBUILD — no session yet, or pi's history has diverged (non-trailing
731
+ // missed messages, e.g. another provider took a turn). Wipes the existing
732
+ // session file (if any) and writes a fresh one containing all prior
733
+ // messages, reusing the same sessionId across rebuilds so UUIDs stay
734
+ // stable for the lifetime of pi's session.
735
+ //
736
+ // Why a full rebuild rather than patching:
737
+ // Injecting deltas into an existing session creates a branch that CC's
738
+ // --resume doesn't follow (documented attempt prior to this). A complete
739
+ // overwrite at the same path is simpler and correct.
740
+ //
741
+ // Why reuse the sessionId across rebuilds:
742
+ // CC re-reads the JSONL on every --resume call — no in-process UUID
743
+ // caching. Validated in tests/exp-session-clear.mjs, including the case
744
+ // where CC had appended its own tool_use/tool_result records between
745
+ // rebuilds. Preserving the UUID means stable log correlation across
746
+ // provider switches and no orphaned session files.
747
+ //
748
+ // Log strings still say "Case 1/2/3/4" so existing diagnostics (int-cache.sh,
749
+ // int-session-resume.mjs) keep grepping the same anchors.
750
+ function syncSharedSession(
751
+ messages: Context["messages"],
752
+ cwd: string,
753
+ customToolNameToSdk?: Map<string, string>,
754
+ modelId?: string,
755
+ piSessionId?: string | null,
756
+ ): SyncResult {
757
+ // System messages are pi's transcript representation of prompt and tool state, not
758
+ // conversation history — they are never imported into a CC session, so exclude them from
759
+ // the history space (priorMessages, cursor, missed) everywhere below (issue #106).
760
+ // The mirror this sync coordinates belongs to the syncing pi session alone:
761
+ // every read and write below addresses sessionStateFor(piSessionId), so a
762
+ // foreign session's shape-matching context can never REUSE or rebuild another
763
+ // session's CC file.
764
+ const sharedSession = sessionStateFor(piSessionId);
765
+ const history = nonSystemMessages(messages);
766
+ const priorMessages = history.slice(0, turnStart(history)); // everything before the current user turn
767
+
768
+ // REUSE path
769
+ //
770
+ // Guard on priorMessages.length >= cursor: a shorter incoming context cannot
771
+ // be a continuation of the cached session. This is the general invariant for
772
+ // pi-side history rewrites such as /compact and session_tree: without it,
773
+ // missed = [].slice(cursor) can falsely hit REUSE and resume an unrelated
774
+ // longer CC session. See issue #25.
775
+ if (sharedSession && !sharedSession.needsRebuild && priorMessages.length >= sharedSession.cursor) {
776
+ const missed = priorMessages.slice(sharedSession.cursor);
777
+ const trailingAssistantOnly =
778
+ missed.length === 1 && (missed[0] as { role?: string }).role === "assistant";
779
+ if (missed.length === 0 || trailingAssistantOnly) {
780
+ if (trailingAssistantOnly) {
781
+ setSessionStateFor(piSessionId, { ...sharedSession, cursor: priorMessages.length, cwd });
782
+ debug(`Case 3: advanced cursor past trailing assistant, resuming session ${sharedSession.sessionId.slice(0, 8)}, cursor=${priorMessages.length}`);
783
+ } else {
784
+ debug(`Case 3: resuming session ${sharedSession.sessionId.slice(0, 8)}, cursor=${sharedSession.cursor}`);
785
+ }
786
+ debug(`syncResult: path=reuse sessionId=${sharedSession.sessionId} cursor=${sharedSession?.cursor}`);
787
+ return { sessionId: sharedSession.sessionId };
788
+ }
789
+ }
790
+ // This is what keeps a caller with a pruned or short context from resuming
791
+ // — then overwriting — the bucket's session: shorter-than-cursor means the
792
+ // incoming history cannot be a continuation, so start clean and preserve.
793
+ // Historically this also caught reentrant subagents (a subagent's priors are
794
+ // shorter than the parent's cursor); with per-session mirrors it now catches
795
+ // the pruned-context shapes on a session's own bucket, and the non-isolated
796
+ // AskClaude path on the "(none)" bucket. The captured ephemeral session is
797
+ // deleted once its query completes (see preserveSharedSession in the
798
+ // completion handler).
799
+ //
800
+ // It is NOT, despite an earlier comment here, the isolated compact-summary
801
+ // path: runIsolatedSummary never calls syncSharedSession at all.
802
+ //
803
+ // Only reachable when needsRebuild is false — user-facing history rewrites
804
+ // (/compact, session_tree, /new, fork) always set needsRebuild or clear
805
+ // sharedSession before the next syncSharedSession call.
806
+ if (sharedSession && !sharedSession.needsRebuild && priorMessages.length < sharedSession.cursor) {
807
+ debug(`Case 1 synthetic: clean start for shorter context, preserving shared session ${sharedSession.sessionId.slice(0, 8)}, cursor=${sharedSession.cursor}`);
808
+ debug(`syncResult: path=clean-start preserve-shared sessionId=${sharedSession.sessionId} cursor=${sharedSession.cursor}`);
809
+ return { sessionId: null, preserveSharedSession: true };
810
+ }
811
+
812
+ // REBUILD path
813
+ if (priorMessages.length === 0) {
814
+ debug(`Case 1: clean start, ${history.length} total messages`);
815
+ debug(`syncResult: path=clean-start`);
816
+ return { sessionId: null };
817
+ }
818
+ const previousSessionId = sharedSession?.sessionId;
819
+ const previousCursor = sharedSession?.cursor ?? 0;
820
+ // preserveId: rebuild in place (deleteSession + createSession with the
821
+ // existing UUID), so prompt-cache UUIDs stay stable for log correlation
822
+ // and for any tools that key off them. Skipped when there's a concurrent
823
+ // writer we shouldn't race (forceRotate).
824
+ const preserveId = previousSessionId !== undefined && !sharedSession?.forceRotate;
825
+ // Before deleteSession — it wipes the file these live in.
826
+ const carried = previousSessionId !== undefined ? readCarriedAttachments(previousSessionId, cwd) : [];
827
+ if (preserveId) {
828
+ // Wipe prior jsonl + companion dir (no-op if nothing to wipe).
829
+ deleteSession(previousSessionId!, cwd, process.env.CLAUDE_CONFIG_DIR);
830
+ }
831
+ const session = createSession({
832
+ projectPath: cwd,
833
+ claudeDir: process.env.CLAUDE_CONFIG_DIR,
834
+ ...(preserveId ? { sessionId: previousSessionId } : {}),
835
+ ...(modelId ? { model: modelId } : {}),
836
+ });
837
+ convertAndImportMessages(session, priorMessages, customToolNameToSdk, carried);
838
+ session.save();
839
+ // records, not messages: `messages` filters out the attachment records that
840
+ // carrying an `@file` expansion across a rebuild writes into the same file.
841
+ verifyWrittenSession(session.jsonlPath, session.sessionId, session.records.length, cwd);
842
+ setSessionStateFor(piSessionId, { sessionId: session.sessionId, cursor: priorMessages.length, cwd, piSessionId: piSessionId ?? undefined });
843
+ if (previousSessionId === undefined) {
844
+ debug(`Case 2: first turn with ${priorMessages.length} prior messages → session ${session.sessionId.slice(0, 8)}, ${session.records.length} records`);
845
+ } else if (preserveId) {
846
+ const missedCount = priorMessages.length - previousCursor;
847
+ debug(`Case 4: ${missedCount} missed messages, ${priorMessages.length} total → rewrote session ${session.sessionId.slice(0, 8)} (same id), ${session.records.length} records`);
848
+ } else {
849
+ debug(`Case 4 post-abort: ${priorMessages.length} total → new session ${session.sessionId.slice(0, 8)} (was ${previousSessionId.slice(0, 8)}, rotated to avoid race with orphan writer), ${session.records.length} records`);
850
+ }
851
+ debugSessionPaths(`${session.sessionId.slice(0, 8)}`, cwd, session.jsonlPath);
852
+ debug(`syncResult: path=rebuild sessionId=${session.sessionId} priors=${priorMessages.length} ${previousSessionId === undefined ? "first" : preserveId ? "preserved" : "rotated-post-abort"}`);
853
+ return { sessionId: session.sessionId };
854
+ }
855
+
856
+ // The SDK's query(), or a test double (see setQuery). The compact/summary
857
+ // path calls the real query() directly — its subprocess must never be swapped
858
+ // out from under a real compaction.
859
+ let queryImpl: typeof query = query;
860
+
861
+ // @internal
862
+ export const __test = {
863
+ setQuery(fn: typeof query | null) {
864
+ queryImpl = fn ?? query;
865
+ },
866
+ resetSharedSession(piSessionId?: string | null) {
867
+ // No id: full reset (the pre-map semantics — tests start from a blank slate).
868
+ if (piSessionId === undefined) sharedSessions.clear();
869
+ else setSessionStateFor(piSessionId, null);
870
+ historyRewrittenBySession.clear();
871
+ },
872
+ markRebuildForSession,
873
+ getHistoryRewritten: () => historyRewrittenBySession.size > 0,
874
+ historyRewrittenBySession,
875
+ armStaleContexts,
876
+ discardRewrittenQuery,
877
+ contextForToolResults,
878
+ isQueryAbandoned: (q: object) => abandonedQueries.has(q),
879
+ get activeQueryContexts() {
880
+ return activeQueryContexts;
881
+ },
882
+ setSharedSession(piSessionId: string | null, state: SessionState | null) {
883
+ setSessionStateFor(piSessionId, state);
884
+ },
885
+ getSharedSession(piSessionId: string | null = null) {
886
+ return sessionStateFor(piSessionId);
887
+ },
888
+ setPiUI(ui: ExtensionUIContext | null) {
889
+ piUI = ui;
890
+ },
891
+ toBridgeContext,
892
+ syncSharedSession,
893
+ extractUserPromptBlocks,
894
+ consumeQuery,
895
+ finalizeCurrentStream,
896
+ resultErrorText,
897
+ deliverToolResults,
898
+ drainForAbort,
899
+ CC_CHILD_ENV,
900
+ buildMcpServers,
901
+ branchSummaryOutcome,
902
+ get promptCaptures() {
903
+ return promptCaptures;
904
+ },
905
+ };
906
+
907
+ // --- Provider helpers: tool name mapping ---
908
+
909
+ // AskClaude path: CC runs its own tools, so builtin names are real.
910
+ function mapToolName(name: string): string {
911
+ const normalized = name.toLowerCase();
912
+ const builtin = SDK_TO_PI_TOOL_NAME[normalized];
913
+ if (builtin) return builtin;
914
+ if (normalized.startsWith(MCP_TOOL_PREFIX)) return name.slice(MCP_TOOL_PREFIX.length);
915
+ return name;
916
+ }
917
+
918
+ // Provider path: the query runs with `tools: []`, so the only tools CC can
919
+ // legitimately call are the pi tools we serve over MCP. Any other name is the
920
+ // model hallucinating a builtin (`bash`, `Bash`, `Edit`, an MCP server we don't
921
+ // serve). CC answers those itself with "No such tool available" and retries
922
+ // inside the same query, never dispatching them to our MCP server — so a tool
923
+ // call under such a name must not reach pi. Forwarding one ran a tool CC never
924
+ // dispatched (real side effects) and, because the retry carries a fresh
925
+ // tool_use id, left the handler for the retry with no result to release it:
926
+ // pi's result arrived keyed to the dead id, and both sides deadlocked.
927
+ function piToolNameFor(name: string, customToolNameToPi: Map<string, string>): string | undefined {
928
+ return customToolNameToPi.get(name) ?? customToolNameToPi.get(name.toLowerCase());
929
+ }
930
+
931
+ // Renames for Claude Code SDK param names that differ from pi's native names.
932
+ // Keys not listed here pass through unchanged, so new pi params work automatically.
933
+ const SDK_KEY_RENAMES: Record<string, Record<string, string>> = {
934
+ read: { file_path: "path" },
935
+ write: { file_path: "path" },
936
+ edit: { file_path: "path", old_string: "oldText", new_string: "newText", old_text: "oldText", new_text: "newText" },
937
+ };
938
+
939
+ // Maps SDK tool args to pi tool args via key renaming + pass-through.
940
+ // Pi's own prepareArguments hooks handle any structural transforms (e.g. edit oldText/newText → edits[]).
941
+ function mapToolArgs(
942
+ toolName: string, args: Record<string, unknown> | undefined,
943
+ ): Record<string, unknown> {
944
+ const input = args ?? {};
945
+ const renames = SDK_KEY_RENAMES[toolName.toLowerCase()];
946
+ const result: Record<string, unknown> = {};
947
+ for (const [key, value] of Object.entries(input)) {
948
+ const piKey = renames?.[key] ?? key;
949
+ if (!(piKey in result)) result[piKey] = value; // first alias wins
950
+ }
951
+ // Pi bash has no default timeout; add a safety default
952
+ if (toolName.toLowerCase() === "bash" && result.timeout == null) {
953
+ result.timeout = 120;
954
+ }
955
+ return result;
956
+ }
957
+
958
+ // --- Query state ---
959
+ // QueryContext lives in query-state.js so tests can import it without
960
+ // activating the extension.
961
+
962
+ // Global (not query state):
963
+ let piUI: ExtensionUIContext | null = null;
964
+ let piMode: ExtensionContext["mode"] | null = null;
965
+ const activeQueryContexts = new Set<QueryContext>();
966
+
967
+ // Defaults that silently cost the user something (no Opus 1M on Max, no
968
+ // AskClaude tool) are announced once. Deferred to the first bridge query rather
969
+ // than session_start: the notice persists a flag to the global config, and
970
+ // firing it on startup would write that file for every pi session that merely
971
+ // has this extension installed. One message, because consecutive info notifies
972
+ // overwrite each other in the TUI.
973
+ let pendingNotices: string[] = [];
974
+
975
+ function showStartupNoticeOnce(): void {
976
+ // `hasUI` is true in RPC mode too — it means dialogs are possible, not that a
977
+ // human is watching. Only a terminal user can act on this.
978
+ if (pendingNotices.length === 0 || piMode !== "tui") return;
979
+ const notices = pendingNotices;
980
+ pendingNotices = [];
981
+ const path = markStartupNoticeShown();
982
+ // pi wraps the whole notify string in the theme's dim foreground; the inner reset
983
+ // drops back to the terminal default rather than dim, which is fine here.
984
+ const title = `\x1b[33mWelcome to pi-claude-bridge\x1b[39m — settings live in ${path}`;
985
+ const bullets = [...notices, "This message only appears once. See README.md for more."].map((n) => `• ${n}`);
986
+ piUI?.notify([title, ...bullets, "─".repeat(64)].join("\n"), "info");
987
+ }
988
+
989
+ // Captures of what pi assembled per agent; see src/prompt-capture.ts for why this
990
+ // is keyed rather than held in a single slot. One process-wide instance, shared
991
+ // across every extension module instance: isolated subagents re-evaluate this
992
+ // module, and the pinned stream they all route through resolves against it.
993
+ const promptCaptures = sharedPromptCaptures((diagnostic) => {
994
+ const first = diagnostic.matches[0];
995
+ debug(
996
+ `prompt-capture: no match for ${diagnostic.systemPrompt.length}-char system prompt. `
997
+ + (first
998
+ ? `closest known (${first.key.length}-char) shares its first ${first.firstDivergent} chars and diverges at offset ${first.firstDivergent}: `
999
+ + JSON.stringify(diagnostic.systemPrompt.slice(first.firstDivergent - 40, first.firstDivergent + 60))
1000
+ : "no known captures to compare against."
1001
+ ) + ` known keys=${diagnostic.matches.length}`,
1002
+ );
1003
+ });
1004
+
1005
+ /** Whatever a settled session left behind, named in one greppable line.
1006
+ *
1007
+ * Every one of these should be empty once the last turn ends, and each is a leak
1008
+ * that costs something real: a retained context routes a later orphaned tool result
1009
+ * into the delivery path and returns a stream nobody ends; a pending tool call is an
1010
+ * MCP handler Claude Code is still waiting on; a live prompt stream is an unresolved
1011
+ * ack. The activeQueryContexts leak was present on every single happy-path run and
1012
+ * no test noticed, because nothing asserted that anything ends clean — so assert it
1013
+ * where the real sessions are, and let diag/audit-warnings.mjs scan for it. */
1014
+ function reportLeaks(label: string): void {
1015
+ const pendingCalls = [...activeQueryContexts].reduce((n, c) => n + c.pendingToolCalls.size, 0);
1016
+ const liveStreams = [...activeQueryContexts].filter((c) => c.promptStream !== null).length;
1017
+ if (activeQueryContexts.size === 0 && pendingCalls === 0 && liveStreams === 0) return;
1018
+ debug(
1019
+ `WARNING: ${label} left state behind — contexts=${activeQueryContexts.size} `
1020
+ + `pendingToolCalls=${pendingCalls} promptStreams=${liveStreams}`,
1021
+ );
1022
+ }
1023
+
1024
+ /** What pi's branch summary means for the navigation it was asked for.
1025
+ *
1026
+ * Cancelling on failure matches pi's own path, which rethrows a summary error out
1027
+ * of the navigation rather than moving without one. Separated from the event
1028
+ * handler so this decision is testable without a Claude Code subprocess — driving
1029
+ * `generateBranchSummary` itself would only be testing pi. */
1030
+ function branchSummaryOutcome(result: BranchSummaryResult): { cancel: true } | { summary: { summary: string; details: unknown; usage?: BranchSummaryResult["usage"] } } {
1031
+ if (result.aborted) return { cancel: true };
1032
+ if (result.error) throw new Error(result.error);
1033
+ debug(`session_before_tree: takeover complete summaryLen=${result.summary?.length ?? 0}`);
1034
+ return {
1035
+ summary: {
1036
+ summary: result.summary ?? "",
1037
+ details: { readFiles: result.readFiles ?? [], modifiedFiles: result.modifiedFiles ?? [] },
1038
+ usage: result.usage,
1039
+ },
1040
+ };
1041
+ }
1042
+
1043
+ function contextForToolResults(results: McpResult[]): QueryContext | undefined {
1044
+ for (const result of results) {
1045
+ const id = result.toolCallId;
1046
+ if (!id) continue;
1047
+ for (const queryCtx of activeQueryContexts) {
1048
+ if (queryCtx.pendingToolCalls.has(id) || queryCtx.pendingResults.has(id) || queryCtx.turnToolCallIds.includes(id)) {
1049
+ return queryCtx;
1050
+ }
1051
+ }
1052
+ }
1053
+ return undefined;
1054
+ }
1055
+
1056
+ function resolveMcpTools(context: Context, excludeToolName?: string): {
1057
+ mcpTools: Tool[];
1058
+ customToolNameToSdk: Map<string, string>;
1059
+ customToolNameToPi: Map<string, string>;
1060
+ } {
1061
+ const mcpTools: Tool[] = [];
1062
+ const customToolNameToSdk = new Map<string, string>();
1063
+ const customToolNameToPi = new Map<string, string>();
1064
+
1065
+ if (!context.tools) return { mcpTools, customToolNameToSdk, customToolNameToPi };
1066
+
1067
+ for (const tool of context.tools) {
1068
+ if (tool.name === excludeToolName) continue;
1069
+ const sdkName = `${MCP_TOOL_PREFIX}${tool.name}`;
1070
+ mcpTools.push(tool);
1071
+ customToolNameToSdk.set(tool.name, sdkName);
1072
+ customToolNameToSdk.set(tool.name.toLowerCase(), sdkName);
1073
+ customToolNameToPi.set(sdkName, tool.name);
1074
+ customToolNameToPi.set(sdkName.toLowerCase(), tool.name);
1075
+ }
1076
+
1077
+ return { mcpTools, customToolNameToSdk, customToolNameToPi };
1078
+ }
1079
+
1080
+ // Creates an MCP server that bridges pi tools to the SDK. Each tool handler
1081
+ // blocks on a Promise until pi delivers the tool result via streamSimple.
1082
+ // Handlers receive their toolCallId from Claude's tools/call _meta, so results
1083
+ // are matched by ID end to end.
1084
+ //
1085
+ // The handler and pi's result can arrive in either order, hence the two maps:
1086
+ // a result that lands first waits in `pendingResults` for the handler to claim
1087
+ // it, and a handler that runs first parks its resolver in `pendingToolCalls`.
1088
+ // Handlers close over the captured `queryCtx`, ensuring they operate on the
1089
+ // correct query's state while multiple queries run concurrently.
1090
+ function buildMcpServers(tools: Tool[], queryCtx: QueryContext): Record<string, ReturnType<typeof createToolServer>> | undefined {
1091
+ if (!tools.length) return undefined;
1092
+ const mcpTools = tools.map((tool) => ({
1093
+ name: tool.name,
1094
+ description: tool.description,
1095
+ inputSchema: tool.parameters,
1096
+ handler: async (toolCallId: string) => {
1097
+ if (queryCtx.pendingResults.has(toolCallId)) {
1098
+ const result = queryCtx.pendingResults.get(toolCallId)!;
1099
+ queryCtx.pendingResults.delete(toolCallId);
1100
+ debug(`mcp handler: ${tool.name} [${toolCallId}] → resolved from queue (${queryCtx.pendingResults.size} remaining)`);
1101
+ return result;
1102
+ }
1103
+ debug(`mcp handler: ${tool.name} [${toolCallId}] → waiting`);
1104
+ return new Promise<McpResult>((resolve) => {
1105
+ queryCtx.pendingToolCalls.set(toolCallId, { toolName: tool.name, resolve });
1106
+ });
1107
+ },
1108
+ }));
1109
+ return { [MCP_SERVER_NAME]: createToolServer(MCP_SERVER_NAME, mcpTools) };
1110
+ }
1111
+
1112
+ // --- Usage helpers ---
1113
+
1114
+ function updateUsage(output: AssistantMessage, usage: Record<string, number | undefined>, model: Model<any>): void {
1115
+ if (usage.input_tokens != null) output.usage.input = usage.input_tokens;
1116
+ if (usage.output_tokens != null) output.usage.output = usage.output_tokens;
1117
+ if (usage.cache_read_input_tokens != null) output.usage.cacheRead = usage.cache_read_input_tokens;
1118
+ if (usage.cache_creation_input_tokens != null) output.usage.cacheWrite = usage.cache_creation_input_tokens;
1119
+ // Claude Code may report reasoning/thinking tokens separately from output tokens.
1120
+ const reasoning = usage.reasoning_tokens ?? usage.thinking_tokens;
1121
+ if (reasoning != null) output.usage.reasoning = reasoning;
1122
+ output.usage.totalTokens = output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite;
1123
+ calculateCost(model, output.usage);
1124
+ const promptTokens = output.usage.input + output.usage.cacheRead + output.usage.cacheWrite;
1125
+ const cachePct = promptTokens > 0 ? Math.round(output.usage.cacheRead / promptTokens * 100) : 0;
1126
+ const reasoningText = reasoning != null ? ` reasoning=${reasoning}` : "";
1127
+ debug(`usage: in=${output.usage.input} out=${output.usage.output} cacheRead=${output.usage.cacheRead} cacheWrite=${output.usage.cacheWrite} total=${output.usage.totalTokens}${reasoningText} cachePct=${cachePct}% model=${model.id}`);
1128
+ }
1129
+
1130
+ // Log the *served* context window reported by an SDK result message
1131
+ // (modelUsage[id].contextWindow), which can differ from the window pi
1132
+ // registered (model.contextWindow) when the runtime entitlement doesn't
1133
+ // match the docs — e.g. bare Opus served 200K on Pro, or [1m] not honored.
1134
+ // The result message's modelUsage is otherwise discarded; this makes the
1135
+ // gap observable. See issue #18.
1136
+ function logServedContextWindow(label: string, message: SDKMessage, model: Model<any>): void {
1137
+ const modelUsage = (message as any).modelUsage as Record<string, { contextWindow?: number; maxOutputTokens?: number }> | undefined;
1138
+ if (!modelUsage) return;
1139
+ for (const [k, v] of Object.entries(modelUsage)) {
1140
+ debug(`${label}: served contextWindow=${v.contextWindow ?? "?"} maxOutputTokens=${v.maxOutputTokens ?? "?"} servedModel=${k} registered=${model.contextWindow}`);
1141
+ }
1142
+ }
1143
+
1144
+ // --- Effort level mapping ---
1145
+ // Pi reasoning levels → CC SDK effort levels
1146
+
1147
+ const REASONING_TO_EFFORT: Record<string, EffortLevel> = {
1148
+ minimal: "low", low: "low", medium: "medium", high: "high", xhigh: "max",
1149
+ };
1150
+
1151
+ const VALID_EFFORTS = new Set<string>(["low", "medium", "high", "xhigh", "max"]);
1152
+
1153
+ // --- Provider helpers: misc ---
1154
+
1155
+ function mapStopReason(reason: string | undefined): "stop" | "length" | "toolUse" {
1156
+ switch (reason) {
1157
+ case "tool_use": return "toolUse";
1158
+ case "max_tokens": return "length";
1159
+ case "end_turn": default: return "stop";
1160
+ }
1161
+ }
1162
+
1163
+ function parsePartialJson(input: string, fallback: Record<string, unknown>): Record<string, unknown> {
1164
+ if (!input) return fallback;
1165
+ try { return JSON.parse(input); } catch { return fallback; }
1166
+ }
1167
+
1168
+
1169
+ // --- Provider: streaming function ---
1170
+ //
1171
+ // Push-based streaming with MCP tool bridge:
1172
+ // 1. streamSimple starts a query() and kicks off consumeQuery() in background
1173
+ // 2. consumeQuery() iterates the SDK generator, pushing events to currentPiStream
1174
+ // 3. On tool_use: ends the current pi stream, nulls it out. The MCP handler
1175
+ // blocks the generator naturally — no events arrive until resolved.
1176
+ // 4. Pi executes the tool, calls streamSimple again. We swap in the new stream,
1177
+ // resolve the MCP handler, and the generator unblocks — events flow to new stream.
1178
+ //
1179
+ // Note: resetTurnState clears turnSawStreamEvent while the generator may still
1180
+ // have queued messages from the previous turn. This is safe because step 3 nulls
1181
+ // currentPiStream, so any leftover messages hit the `!ctx().currentPiStream` guard
1182
+ // in consumeQuery and are skipped before resetTurnState runs.
1183
+
1184
+ const completedStreams = new WeakSet<object>();
1185
+
1186
+ function markStreamComplete(stream: AssistantMessageEventStream | null): void {
1187
+ if (stream) completedStreams.add(stream as object);
1188
+ }
1189
+
1190
+ function claimCurrentPiStream(stream: AssistantMessageEventStream, label: string, c: QueryContext): void {
1191
+ if (c.currentPiStream && !completedStreams.has(c.currentPiStream as object)) {
1192
+ debug(`WARNING: currentPiStream overwritten before terminal event (${label}); activeQuery=${Boolean(c.activeQuery)} pendingHandlers=${c.pendingToolCalls.size}`);
1193
+ }
1194
+ c.currentPiStream = stream;
1195
+ }
1196
+
1197
+ function ensureTurnStarted(c: QueryContext): void {
1198
+ if (!c.turnStarted && c.currentPiStream && c.turnOutput) {
1199
+ c.currentPiStream!.push({ type: "start", partial: c.turnOutput });
1200
+ c.turnStarted = true;
1201
+ }
1202
+ }
1203
+
1204
+ function finalizeCurrentStream(c: QueryContext, stopReason?: string): void {
1205
+ if (!c.currentPiStream || !c.turnOutput) return;
1206
+ debug(`provider: finalizeCurrentStream called, stopReason=${stopReason}, turnOutput=${JSON.stringify({stopReason: c.turnOutput!.stopReason, error: c.turnOutput!.errorMessage})}`);
1207
+ if (!c.turnStarted) ensureTurnStarted(c);
1208
+ const stream = c.currentPiStream;
1209
+ if (c.turnOutput.stopReason === "error") {
1210
+ stream!.push({ type: "error", reason: "error", error: c.turnOutput });
1211
+ } else {
1212
+ const reason = stopReason === "length" ? "length" : "stop";
1213
+ stream!.push({ type: "done", reason, message: c.turnOutput });
1214
+ }
1215
+ markStreamComplete(stream);
1216
+ stream!.end();
1217
+ c.currentPiStream = null;
1218
+ }
1219
+
1220
+ /** Maps Anthropic stream events to pi stream events (text, thinking, toolcall).
1221
+ * On message_stop with tool_use: ends currentPiStream so pi can execute the tool. */
1222
+ function processStreamEvent(
1223
+ message: SDKMessage,
1224
+ customToolNameToPi: Map<string, string>,
1225
+ model: Model<any>,
1226
+ c: QueryContext,
1227
+ ): void {
1228
+ if (!c.currentPiStream || !c.turnOutput) return;
1229
+ c.turnSawStreamEvent = true;
1230
+ const event = (message as SDKMessage & { event: any }).event;
1231
+
1232
+ if (event?.type === "message_start") {
1233
+ // Still open from an earlier message_start: Claude Code gave up on that
1234
+ // stream and is retrying it. Its blocks were never completed.
1235
+ if (c.turnStreamOpen) dropAbandonedStreamBlocks(c, `restreamed as ${event.message?.id}`);
1236
+ c.turnToolCallIds = [];
1237
+ c.turnStreamMessageId = event.message?.id;
1238
+ c.turnStreamOpen = true;
1239
+ c.turnStreamBlockStart = c.turnBlocks.length;
1240
+ if (event.message?.usage) updateUsage(c.turnOutput, event.message.usage, model);
1241
+ return;
1242
+ }
1243
+
1244
+ if (event?.type === "content_block_start") {
1245
+ ensureTurnStarted(c);
1246
+ if (event.content_block?.type === "text") {
1247
+ c.turnBlocks.push({ type: "text", text: "", index: event.index });
1248
+ c.currentPiStream!.push({ type: "text_start", contentIndex: c.turnBlocks.length - 1, partial: c.turnOutput });
1249
+ } else if (event.content_block?.type === "thinking") {
1250
+ c.turnBlocks.push({ type: "thinking", thinking: "", thinkingSignature: "", index: event.index });
1251
+ c.currentPiStream!.push({ type: "thinking_start", contentIndex: c.turnBlocks.length - 1, partial: c.turnOutput });
1252
+ } else if (event.content_block?.type === "tool_use") {
1253
+ const piName = piToolNameFor(event.content_block.name, customToolNameToPi);
1254
+ if (!piName) {
1255
+ debug(`processStreamEvent: skipping tool_use for unserved tool ${event.content_block.name} [${event.content_block.id}] — CC rejects it and retries`);
1256
+ return;
1257
+ }
1258
+ c.turnSawToolCall = true;
1259
+ c.turnToolCallIds.push(event.content_block.id);
1260
+ c.turnBlocks.push({
1261
+ type: "toolCall", id: event.content_block.id,
1262
+ name: piName,
1263
+ arguments: (event.content_block.input as Record<string, unknown>) ?? {},
1264
+ partialJson: "", index: event.index,
1265
+ });
1266
+ c.currentPiStream!.push({ type: "toolcall_start", contentIndex: c.turnBlocks.length - 1, partial: c.turnOutput });
1267
+ } else {
1268
+ debug("processStreamEvent: unhandled content_block_start type", event.content_block?.type);
1269
+ }
1270
+ return;
1271
+ }
1272
+
1273
+ if (event?.type === "content_block_delta") {
1274
+ const index = c.turnBlocks.findIndex((b: any) => b.index === event.index);
1275
+ const block = c.turnBlocks[index];
1276
+ if (!block) return;
1277
+ if (event.delta?.type === "text_delta" && block.type === "text") {
1278
+ block.text += event.delta.text;
1279
+ c.currentPiStream!.push({ type: "text_delta", contentIndex: index, delta: event.delta.text, partial: c.turnOutput });
1280
+ } else if (event.delta?.type === "thinking_delta" && block.type === "thinking") {
1281
+ block.thinking += event.delta.thinking;
1282
+ c.currentPiStream!.push({ type: "thinking_delta", contentIndex: index, delta: event.delta.thinking, partial: c.turnOutput });
1283
+ } else if (event.delta?.type === "input_json_delta" && block.type === "toolCall") {
1284
+ block.partialJson += event.delta.partial_json;
1285
+ block.arguments = parsePartialJson(block.partialJson, block.arguments);
1286
+ c.currentPiStream!.push({ type: "toolcall_delta", contentIndex: index, delta: event.delta.partial_json, partial: c.turnOutput });
1287
+ } else if (event.delta?.type === "signature_delta" && block.type === "thinking") {
1288
+ block.thinkingSignature = (block.thinkingSignature ?? "") + event.delta.signature;
1289
+ } else {
1290
+ debug("processStreamEvent: unhandled content_block_delta type", event.delta?.type);
1291
+ }
1292
+ return;
1293
+ }
1294
+
1295
+ if (event?.type === "content_block_stop") {
1296
+ const index = c.turnBlocks.findIndex((b: any) => b.index === event.index);
1297
+ const block = c.turnBlocks[index];
1298
+ if (!block) return;
1299
+ delete block.index;
1300
+ if (block.type === "text") {
1301
+ c.currentPiStream!.push({ type: "text_end", contentIndex: index, content: block.text, partial: c.turnOutput });
1302
+ } else if (block.type === "thinking") {
1303
+ c.currentPiStream!.push({ type: "thinking_end", contentIndex: index, content: block.thinking, partial: c.turnOutput });
1304
+ } else if (block.type === "toolCall") {
1305
+ c.turnSawToolCall = true;
1306
+ block.arguments = mapToolArgs(
1307
+ block.name, parsePartialJson(block.partialJson, block.arguments),
1308
+ );
1309
+ delete block.partialJson;
1310
+ c.currentPiStream!.push({ type: "toolcall_end", contentIndex: index, toolCall: block, partial: c.turnOutput });
1311
+ }
1312
+ return;
1313
+ }
1314
+
1315
+ if (event?.type === "message_delta") {
1316
+ c.turnOutput.stopReason = mapStopReason(event.delta?.stop_reason);
1317
+ if (event.usage) updateUsage(c.turnOutput, event.usage, model);
1318
+ return;
1319
+ }
1320
+
1321
+ if (event?.type === "message_stop") c.turnStreamOpen = false;
1322
+
1323
+ if (event?.type === "message_stop" && c.turnSawToolCall) {
1324
+ // Tool call complete — end this pi stream. The SDK will still yield an
1325
+ // assistant message for this turn, but currentPiStream=null causes
1326
+ // consumeQuery to skip it. The MCP handler blocks the generator until
1327
+ // pi delivers the tool result via the next streamSimple call.
1328
+ c.turnOutput.stopReason = "toolUse";
1329
+ const stream = c.currentPiStream;
1330
+ stream!.push({ type: "done", reason: "toolUse", message: c.turnOutput });
1331
+ markStreamComplete(stream);
1332
+ stream!.end();
1333
+ c.currentPiStream = null;
1334
+
1335
+ // Cursor is updated by the next streamSimple call (tool result delivery path)
1336
+ // which sets cursor = context.messages.length with the post-tool-result context.
1337
+ return;
1338
+ }
1339
+
1340
+ if (event?.type !== "message_stop" && event?.type !== "ping") {
1341
+ debug("processStreamEvent: unhandled event type", event?.type);
1342
+ }
1343
+ }
1344
+
1345
+ /** Remove the blocks a stream Claude Code abandoned mid-message. They never got a
1346
+ * message_stop, so a thinking block has no signature and a tool call is one CC will
1347
+ * never dispatch; left in, pi would run the tool and the turn would wait on a
1348
+ * handler that never comes, or the next request would replay a broken block.
1349
+ * The fallback then restarts those indices. pi's normal provider path tolerates that;
1350
+ * pi-agent-core's experimental harness frame encoder keys blocks by contentIndex and
1351
+ * rejects a repeated start, so it would need a change there to drive this provider. */
1352
+ function dropAbandonedStreamBlocks(c: QueryContext, why: string): void {
1353
+ const dropped = c.turnBlocks.splice(c.turnStreamBlockStart);
1354
+ debug(`dropAbandonedStreamBlocks: ${why}; dropped ${dropped.length} blocks from ${c.turnStreamMessageId} types=${dropped.map((b: any) => b.type).join(",")}`);
1355
+ c.turnToolCallIds = [];
1356
+ c.turnSawToolCall = c.turnBlocks.some((b: any) => b.type === "toolCall");
1357
+ c.turnStreamOpen = false;
1358
+ }
1359
+
1360
+ // The SDK always yields `assistant` messages (completed content blocks) after streaming.
1361
+ // When stream_events already delivered the content, this is a no-op. But after
1362
+ // resetTurnState (e.g. tool result delivery), if the next turn's assistant message
1363
+ // arrives before any stream_events, this is the primary content path. Must maintain
1364
+ // the same stream lifecycle as processStreamEvent — including ending the stream on
1365
+ // tool_use to prevent deadlock with the MCP handler.
1366
+ //
1367
+ // It is also the content path when a stream stalls: Claude Code drops it and asks
1368
+ // again without streaming ("Error streaming, falling back to non-streaming mode"),
1369
+ // and the answer arrives as one assistant message, under a new message id, with no
1370
+ // stream_events of its own. turnSawStreamEvent is already set by the dead stream,
1371
+ // so gating on it alone dropped that message: its tool calls never reached pi, CC
1372
+ // sat in the MCP handler waiting for their results, and the turn hung on "Working"
1373
+ // until the user aborted it.
1374
+ function processAssistantMessage(message: SDKMessage, model: Model<any>, customToolNameToPi: Map<string, string>, c: QueryContext): void {
1375
+ const assistantMsg = (message as any).message;
1376
+ if (!assistantMsg?.content) return;
1377
+ if (c.turnSawStreamEvent) {
1378
+ // Same id was already delivered; a new id is CC's non-streaming fallback.
1379
+ // Drop the stalled stream's partial blocks if it never stopped. Deliberately
1380
+ // deliver even if it stopped, at the risk of duplication if CC renumbers it.
1381
+ const id = assistantMsg.id;
1382
+ if (!id || !c.turnStreamMessageId || id === c.turnStreamMessageId) return;
1383
+ if (c.turnStreamOpen) dropAbandonedStreamBlocks(c, `non-streaming fallback ${id}`);
1384
+ }
1385
+ c.turnToolCallIds = [];
1386
+ debug(`processAssistantMessage fallback: ${assistantMsg.content.length} blocks, types=${assistantMsg.content.map((b: any) => b.type).join(",")}`);
1387
+ for (const block of assistantMsg.content) {
1388
+ if (block.type === "text" && block.text) {
1389
+ ensureTurnStarted(c);
1390
+ c.turnBlocks.push({ type: "text", text: block.text });
1391
+ const idx = c.turnBlocks.length - 1;
1392
+ c.currentPiStream?.push({ type: "text_start", contentIndex: idx, partial: c.turnOutput });
1393
+ c.currentPiStream?.push({ type: "text_delta", contentIndex: idx, delta: block.text, partial: c.turnOutput });
1394
+ c.currentPiStream?.push({ type: "text_end", contentIndex: idx, content: block.text, partial: c.turnOutput });
1395
+ } else if (block.type === "thinking") {
1396
+ ensureTurnStarted(c);
1397
+ c.turnBlocks.push({ type: "thinking", thinking: block.thinking ?? "", thinkingSignature: block.signature ?? "" });
1398
+ const idx = c.turnBlocks.length - 1;
1399
+ c.currentPiStream?.push({ type: "thinking_start", contentIndex: idx, partial: c.turnOutput });
1400
+ if (block.thinking) c.currentPiStream?.push({ type: "thinking_delta", contentIndex: idx, delta: block.thinking, partial: c.turnOutput });
1401
+ c.currentPiStream?.push({ type: "thinking_end", contentIndex: idx, content: block.thinking ?? "", partial: c.turnOutput });
1402
+ } else if (block.type === "tool_use") {
1403
+ const piName = piToolNameFor(block.name, customToolNameToPi);
1404
+ if (!piName) {
1405
+ debug(`processAssistantMessage: skipping tool_use for unserved tool ${block.name} [${block.id}] — CC rejects it and retries`);
1406
+ continue;
1407
+ }
1408
+ ensureTurnStarted(c);
1409
+ c.turnSawToolCall = true;
1410
+ c.turnToolCallIds.push(block.id);
1411
+ c.turnBlocks.push({
1412
+ type: "toolCall", id: block.id,
1413
+ name: piName,
1414
+ arguments: mapToolArgs(piName, block.input),
1415
+ });
1416
+ const idx = c.turnBlocks.length - 1;
1417
+ const toolBlock = c.turnBlocks[idx];
1418
+ c.currentPiStream?.push({ type: "toolcall_start", contentIndex: idx, partial: c.turnOutput });
1419
+ c.currentPiStream?.push({ type: "toolcall_end", contentIndex: idx, toolCall: toolBlock as any, partial: c.turnOutput });
1420
+ } else {
1421
+ debug("processAssistantMessage: unhandled block type", block.type);
1422
+ }
1423
+ }
1424
+ if (assistantMsg.usage && c.turnOutput) updateUsage(c.turnOutput, assistantMsg.usage, model);
1425
+
1426
+ // End the stream on tool_use, same as processStreamEvent's message_stop handler.
1427
+ if (c.turnSawToolCall && c.currentPiStream && c.turnOutput) {
1428
+ c.turnOutput.stopReason = "toolUse";
1429
+ const stream = c.currentPiStream;
1430
+ stream.push({ type: "done", reason: "toolUse", message: c.turnOutput });
1431
+ markStreamComplete(stream);
1432
+ stream.end();
1433
+ c.currentPiStream = null;
1434
+ }
1435
+ }
1436
+
1437
+ /** Background consumer: iterates the SDK generator, pushing events to currentPiStream.
1438
+ * Runs until the query ends. Per turn, the SDK yields stream_events (deltas), then
1439
+ * an assistant message (completed blocks). On tool_use, the stream is ended by
1440
+ * whichever path handles it first (processStreamEvent or processAssistantMessage),
1441
+ * and the MCP handler blocks the generator until pi delivers the tool result. */
1442
+ async function consumeQuery(
1443
+ sdkQuery: ReturnType<typeof query>,
1444
+ customToolNameToPi: Map<string, string>,
1445
+ model: Model<any>,
1446
+ wasAborted: () => boolean,
1447
+ queryCtx: QueryContext,
1448
+ ): Promise<{ capturedSessionId?: string }> {
1449
+ let capturedSessionId: string | undefined;
1450
+
1451
+ for await (const message of sdkQuery) {
1452
+ if (RECORD_STREAM_PATH) appendFileSync(RECORD_STREAM_PATH, `${JSON.stringify(message)}\n`);
1453
+ if (wasAborted()) break;
1454
+ // Everything below the currentPiStream guard is content, which there is
1455
+ // nowhere to put once a turn has ended on a tool call. These three are not
1456
+ // content and must not share that gate:
1457
+ //
1458
+ // - stdin: nothing else closes the CLI's stdin now that the prompt is a
1459
+ // streamed generator (isSingleUserTurn=false), so missing this hangs the query.
1460
+ // - the failure a `result` carries: it is the only record that the turn
1461
+ // failed at all. Behind the guard, a 429 arriving at a tool boundary set
1462
+ // no stopReason, no errorMessage, and logged nothing — the turn simply
1463
+ // ended empty.
1464
+ // - rate-limit events: notifications to the user, which are most likely to
1465
+ // fire during exactly the long tool-using turns the guard was skipping.
1466
+ let resultError: string | undefined;
1467
+ if (message.type === "result") {
1468
+ queryCtx.promptStream?.end();
1469
+ logServedContextWindow("result", message, model);
1470
+ resultError = resultErrorText(message);
1471
+ if (resultError !== undefined) {
1472
+ // Consume the rejection alongside the failure it caused, so a later
1473
+ // unrelated failure on this query doesn't inherit the label.
1474
+ if (queryCtx.rateLimitRejection) {
1475
+ resultError = describeRateLimitFailure(queryCtx.rateLimitRejection, resultError);
1476
+ queryCtx.rateLimitRejection = null;
1477
+ }
1478
+ debug(`consumeQuery: error result, subtype=${message.subtype}, error=${resultError}`);
1479
+ if (queryCtx.turnOutput) {
1480
+ queryCtx.turnOutput.stopReason = "error";
1481
+ queryCtx.turnOutput.errorMessage = resultError;
1482
+ }
1483
+ }
1484
+ }
1485
+ if (message.type === "rate_limit_event") {
1486
+ const info = (message as any).rate_limit_info;
1487
+ debug("consumeQuery: rate_limit_event", JSON.stringify(info).slice(0, 300));
1488
+ if (info?.status === "rejected") {
1489
+ // Held so the failure Claude Code sends next can be named as a rate limit.
1490
+ queryCtx.rateLimitRejection = info;
1491
+ // The "rate limited" notice below supersedes warnings; re-arm so the next
1492
+ // window's warnings fire even if it opens straight into allowed_warning.
1493
+ queryCtx.lastRateLimitWarnStep = null;
1494
+ queryCtx.lastRateLimitWarnThreshold = undefined;
1495
+ // resetsAt is Unix seconds, not milliseconds.
1496
+ const resetsAt = info.resetsAt ? new Date(info.resetsAt * 1000).toLocaleTimeString() : "unknown";
1497
+ piUI?.notify(`Claude rate limited (${info.rateLimitType ?? "unknown"}) — resets at ${resetsAt}`, "warning");
1498
+ } else if (info?.status === "allowed") {
1499
+ // Back under the threshold (window reset) — re-arm the warning dedupe.
1500
+ queryCtx.lastRateLimitWarnStep = null;
1501
+ queryCtx.lastRateLimitWarnThreshold = undefined;
1502
+ } else if (info?.status === "allowed_warning") {
1503
+ // utilization is a fraction (0..1); allowed_warning fires once it crosses surpassedThreshold.
1504
+ const percent = Math.round((info.utilization ?? 0) * 100);
1505
+ // The SDK emits one event per request, so only re-notify when the level
1506
+ // rises past a new 5% step or the threshold changes.
1507
+ const step = Math.floor(percent / 5);
1508
+ const rose = queryCtx.lastRateLimitWarnStep === null || step > queryCtx.lastRateLimitWarnStep;
1509
+ if (rose || info.surpassedThreshold !== queryCtx.lastRateLimitWarnThreshold) {
1510
+ queryCtx.lastRateLimitWarnStep = step;
1511
+ queryCtx.lastRateLimitWarnThreshold = info.surpassedThreshold;
1512
+ piUI?.notify(`Claude rate limit warning: ${percent}% used (${info.rateLimitType ?? ""})`, "warning");
1513
+ }
1514
+ }
1515
+ continue;
1516
+ }
1517
+ if (!queryCtx.currentPiStream || !queryCtx.turnOutput) continue;
1518
+
1519
+ switch (message.type) {
1520
+ case "stream_event":
1521
+ processStreamEvent(message, customToolNameToPi, model, queryCtx);
1522
+ break;
1523
+ case "assistant":
1524
+ processAssistantMessage(message, model, customToolNameToPi, queryCtx);
1525
+ break;
1526
+ case "result": {
1527
+ // The failure itself was recorded above the guard, along with the served
1528
+ // context window. What is left here is the success path: push the result
1529
+ // text when no assistant message already delivered it.
1530
+ if (resultError === undefined && !queryCtx.turnSawStreamEvent && message.subtype === "success") {
1531
+ ensureTurnStarted(queryCtx);
1532
+ const text = message.result || "";
1533
+ queryCtx.turnBlocks.push({ type: "text", text });
1534
+ const idx = queryCtx.turnBlocks.length - 1;
1535
+ queryCtx.currentPiStream?.push({ type: "text_start", contentIndex: idx, partial: queryCtx.turnOutput });
1536
+ queryCtx.currentPiStream?.push({ type: "text_delta", contentIndex: idx, delta: text, partial: queryCtx.turnOutput });
1537
+ queryCtx.currentPiStream?.push({ type: "text_end", contentIndex: idx, content: text, partial: queryCtx.turnOutput });
1538
+ }
1539
+ break;
1540
+ }
1541
+ case "system":
1542
+ if ((message as any).subtype === "init" && (message as any).session_id) {
1543
+ capturedSessionId = (message as any).session_id;
1544
+ }
1545
+ break;
1546
+ case "user":
1547
+ // SDK echo of the user prompt — no stream events to emit. Note it
1548
+ // carries only prompts and tool results: a steer CC drained at a
1549
+ // tool boundary is recorded in its session transcript as a
1550
+ // `queued_command` attachment and never reaches this stream, which
1551
+ // is why the mid-turn steering tripwire has to live in the
1552
+ // integration test.
1553
+ break;
1554
+ default:
1555
+ debug("consumeQuery: unhandled SDK message type", message.type);
1556
+ break;
1557
+ }
1558
+ }
1559
+
1560
+ // DEBUG: trace when consumeQuery exits
1561
+ debug(`consumeQuery: for-await loop exited, wasAborted=${wasAborted()}, capturedSessionId=${capturedSessionId?.slice(0, 8) ?? "none"}`);
1562
+
1563
+ return { capturedSessionId };
1564
+ }
1565
+
1566
+ /** The trailing user turn as content blocks, or null if there isn't one.
1567
+ * Blocks rather than text so image steers keep their images. */
1568
+ function steerBlocks(messages: Context["messages"]): ContentBlockParam[] | null {
1569
+ const blocks = extractUserPromptBlocks(messages);
1570
+ if (blocks) return blocks;
1571
+ const text = extractUserPrompt(messages);
1572
+ return text ? [{ type: "text", text }] : null;
1573
+ }
1574
+
1575
+ /** A steer that never made it into CC's session. The cursor has already counted
1576
+ * it, so count-based sync would skip it forever — rebuild instead, which
1577
+ * re-imports the message from pi's context. */
1578
+ function steerMissedSession(c: QueryContext, text: string): void {
1579
+ c.missedSteer = true;
1580
+ const state = sessionStateFor(c.piSessionId);
1581
+ if (state) setSessionStateFor(c.piSessionId, { ...state, needsRebuild: true });
1582
+ debug(`provider: steer never reached CC, marked query for rebuild: ${text.slice(0, 60)}`);
1583
+ }
1584
+
1585
+ /** Releases this turn's tool results to their MCP handlers, after first pushing
1586
+ * any steer to CC.
1587
+ *
1588
+ * The ordering is mandatory, not an optimization. The steer and the MCP tool
1589
+ * result travel back to CC over the same stdin FIFO. Awaiting the push ack
1590
+ * (which resolves only once the SDK's write to stdin completed) before
1591
+ * resolving any handler guarantees CC enqueues the steer *before* it reads the
1592
+ * tool result, so its post-tool-call drain sees it and acts on it this turn.
1593
+ * Resolve first and the steer misses the drain, silently degrading to
1594
+ * follow-up semantics.
1595
+ *
1596
+ * Both the post-tool-call drain and the FIFO ordering are CC CLI internals,
1597
+ * not SDK contract — tests/int-tool-message.mjs is the tripwire if they move. */
1598
+ async function deliverToolResults(
1599
+ c: QueryContext,
1600
+ results: McpResult[],
1601
+ steer: ContentBlockParam[] | null,
1602
+ contextLength: number,
1603
+ ): Promise<void> {
1604
+ if (steer) {
1605
+ const text = steer.map((b) => (b.type === "text" ? b.text : "[image]")).join("\n");
1606
+ if (!c.promptStream) {
1607
+ debug(`WARNING: steer with no prompt stream, dropping: ${text.slice(0, 60)}`);
1608
+ steerMissedSession(c, text);
1609
+ } else {
1610
+ try {
1611
+ await c.promptStream.push(userMessage(steer, "next"));
1612
+ debug(`provider: steer written to CC stdin before tool result: ${text.slice(0, 60)}`);
1613
+ } catch (error) {
1614
+ // The query is ending — pushing further input would wedge tool-result
1615
+ // delivery, so the steer doesn't reach this query. It is still in
1616
+ // pi's context, and the caller has already advanced the session
1617
+ // cursor past it, so force a rebuild or CC would never see it.
1618
+ debug(`provider: steer push rejected, delivering tool result anyway:`, error);
1619
+ steerMissedSession(c, text);
1620
+ }
1621
+ }
1622
+ }
1623
+
1624
+ debug(`provider: tool results, ${results.length} results, ${c.pendingToolCalls.size} waiting handlers, ctx.msgs=${contextLength}`);
1625
+ for (const result of results) {
1626
+ const id = result.toolCallId;
1627
+ if (id && c.pendingToolCalls.has(id)) {
1628
+ const pending = c.pendingToolCalls.get(id)!;
1629
+ c.pendingToolCalls.delete(id);
1630
+ debug(`provider: resolving ${pending.toolName} [${id}]${result.isError ? " (error)" : ""}`, JSON.stringify(result.content).slice(0, 200));
1631
+ pending.resolve(result);
1632
+ } else if (id) {
1633
+ c.pendingResults.set(id, result);
1634
+ debug(`provider: queued result [${id}] (${c.pendingResults.size} pending)`);
1635
+ } else {
1636
+ debug(`WARNING: tool result without toolCallId, cannot match`);
1637
+ }
1638
+ if (c.pendingToolCalls.size > 0 && c.pendingResults.size > 0) {
1639
+ debug(`BUG: both maps non-empty! handlers=${c.pendingToolCalls.size} results=${c.pendingResults.size}`);
1640
+ }
1641
+ }
1642
+ if (c.pendingToolCalls.size > 0) {
1643
+ debug(`WARNING: ${c.pendingToolCalls.size} MCP handlers still waiting after delivering ${results.length} results`);
1644
+ piUI?.notify(`Claude bridge: ${c.pendingToolCalls.size} tool handler(s) still waiting — provider may be stuck`, "warning");
1645
+ }
1646
+ }
1647
+
1648
+ /** Abort teardown for one query: settle everything that would otherwise be left
1649
+ * awaiting a subprocess we are about to kill. The pump abandons iteration on
1650
+ * abort, so an in-flight prompt-stream push would hang forever and take
1651
+ * tool-result delivery with it. */
1652
+ function drainForAbort(c: QueryContext, promptStream: PromptStream): void {
1653
+ promptStream.fail(new Error("Operation aborted"));
1654
+ c.releasePendingToolCalls("Operation aborted");
1655
+ }
1656
+
1657
+ /** Queries pi's history moved out from under. Their completion must not touch
1658
+ * `sharedSession` or the pi stream: the query that took over the turn has
1659
+ * already rebuilt both from the new history, and this one's session id names the
1660
+ * conversation pi just discarded. */
1661
+ const abandonedQueries = new WeakSet<object>();
1662
+
1663
+ /** The prompt a continuation query is opened with: the pi turn goes on, but its
1664
+ * last message is a tool result rather than a prompt, and query() cannot resume
1665
+ * a session without one. */
1666
+ const CONTINUE_AFTER_REWRITE_PROMPT =
1667
+ "[Your context was compacted. What precedes this is a summary plus the most recent messages, "
1668
+ + "ending with the tool result you were waiting for. Continue the task from there.]";
1669
+
1670
+ /** Drop a Claude Code query parked at a tool boundary whose conversation pi has
1671
+ * since rewritten (/compact, tree navigation).
1672
+ *
1673
+ * Delivering the turn's tool result into that query hands Claude Code the
1674
+ * context pi just shrank: one pi turn is one CC query, and the query keeps its
1675
+ * own context inside the CLI whatever pi does to its transcript. It answers off
1676
+ * the pre-compaction conversation, reports the pre-compaction usage back, and pi
1677
+ * crosses the same threshold at the next boundary — measured as one compaction
1678
+ * per tool call with usage never dropping (issue #101). `needsRebuild` does not
1679
+ * prevent it: only syncSharedSession reads that flag, and tool-result delivery
1680
+ * is the one call that never syncs.
1681
+ *
1682
+ * The caller then takes the fresh-query path, where REBUILD imports pi's
1683
+ * rewritten history — this tool result included, since it is already in that
1684
+ * history — so the turn continues instead of ending here. Nothing is lost by
1685
+ * killing the subprocess: pi owns the only copy of the conversation that counts. */
1686
+ function discardRewrittenQuery(c: QueryContext): void {
1687
+ const discarded = c.activeQuery as { interrupt?: () => Promise<unknown>; close?: () => void } | null;
1688
+ if (discarded) abandonedQueries.add(discarded);
1689
+ c.activeQuery = null;
1690
+ // Leaving the routing set is what stops this result coming straight back here:
1691
+ // contextForToolResults only matches ids against contexts still in it.
1692
+ activeQueryContexts.delete(c);
1693
+ c.turnToolCallIds = [];
1694
+ c.promptStream?.fail(new Error("conversation rewritten"));
1695
+ c.promptStream = null;
1696
+ // Settle the parked handlers before killing the CLI, for drainForAbort's
1697
+ // reason: one left awaiting a dead subprocess never settles.
1698
+ c.releasePendingToolCalls("Context was compacted; this query was discarded.");
1699
+ void discarded?.interrupt?.().catch(() => {});
1700
+ try { discarded?.close?.(); } catch {}
1701
+ // The CLI we just killed may still flush a record into the session JSONL, and
1702
+ // the rebuild is the next thing that happens — so rotate rather than race it,
1703
+ // exactly as after an abort. Only this session's mirror: the discarding query
1704
+ // proves its own conversation is the one being rebuilt around.
1705
+ const state = sessionStateFor(c.piSessionId);
1706
+ if (state) setSessionStateFor(c.piSessionId, { ...state, needsRebuild: true, forceRotate: true });
1707
+ if (c.piSessionId) historyRewrittenBySession.delete(c.piSessionId);
1708
+ debug("provider: history rewritten under a parked query — discarded it, rebuilding from current history");
1709
+ }
1710
+
1711
+ /** Provider entry point. Pi calls this for each new prompt and each tool result.
1712
+ * Two cases: tool result delivery (active query) or fresh query. */
1713
+ function streamClaudeAgentSdk(model: Model<any>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream {
1714
+ showStartupNoticeOnce();
1715
+ // pi hands providers a transcript (prompt/tools folded into system messages) — fold it
1716
+ // back out to the prompt/tools fields every cursor write, syncSharedSession call and
1717
+ // prompt-capture lookup below assumes (issue #106).
1718
+ context = toBridgeContext(context);
1719
+
1720
+ // One-off summarizer calls arrive HERE too, not only via isolatedStreamFn: /bug report
1721
+ // (summarizeForBugReport) routes through agent.streamFunction -> streamSimple, with no
1722
+ // takeover hook. pi marks every one-off summarizer with cacheRetention:"none" in
1723
+ // completeSummarization, so route on the marker: their prompt is never recorded by the
1724
+ // capture boundaries and resolveOrDerive would throw. Hand them to the isolated path
1725
+ // (separate persistSession:false CC process, no session sync needed).
1726
+ if (options?.cacheRetention === "none") {
1727
+ debug(`provider: one-off summarizer call (cacheRetention none) routed to isolated summary, msgs=${context.messages.length}`);
1728
+ return isolatedStreamFn(model, context, options);
1729
+ }
1730
+
1731
+ const stream = createAssistantMessageEventStream();
1732
+
1733
+ // DEBUG: trace followUp message triggering
1734
+ const lastMsgRole = context.messages[context.messages.length - 1]?.role;
1735
+ debug(`provider: streamClaudeAgentSdk called, activeQuery=${!!ctx().activeQuery}, lastMsgRole=${lastMsgRole}, isReentrant=${ctx().activeQuery !== null}`);
1736
+
1737
+ let activeQuery = ctx().activeQuery !== null;
1738
+ const allResults = activeQueryContexts.size > 0 ? extractAllToolResults(context) : [];
1739
+ let resultCtx = allResults.length > 0 ? contextForToolResults(allResults) : undefined;
1740
+
1741
+ // pi rewrote its history while this query sat parked at a tool boundary, so the
1742
+ // query answers about a conversation that no longer exists. Discard it and let
1743
+ // this tool result carry the turn into a fresh query over the rewritten history.
1744
+ // The staleness mark is per pi session: a subagent's compaction (its own
1745
+ // AgentSession, sharing this process) must not discard the parent's parked
1746
+ // query, and vice versa.
1747
+ const rewrittenUnderQuery = Boolean(resultCtx?.historyStale);
1748
+ if (resultCtx && rewrittenUnderQuery) {
1749
+ discardRewrittenQuery(resultCtx);
1750
+ resultCtx = undefined;
1751
+ // Recomputed, not cleared: a reentrant subagent may still hold a query of its own.
1752
+ activeQuery = ctx().activeQuery !== null;
1753
+ }
1754
+
1755
+ const isReentrantUserQuery = activeQuery && lastMsgRole === "user" && allResults.length === 0;
1756
+ if (isReentrantUserQuery) {
1757
+ debug(`provider: active query user-only call treated as reentrant fresh query, waitingHandlers=${ctx().pendingToolCalls.size}, ctx.msgs=${context.messages.length}`);
1758
+ }
1759
+
1760
+ // --- Tool result delivery ---
1761
+ // Pi appends tool results to context and calls back. Extract this turn's results
1762
+ // (everything after the last assistant message) and match against waiting MCP
1763
+ // handlers. Results that arrive before their handler get queued in pendingResults.
1764
+ if (resultCtx) {
1765
+ claimCurrentPiStream(stream, "tool-result", resultCtx);
1766
+ resultCtx.resetTurnState(model);
1767
+ // A rewrite that armed the mark after this query parked gets copied here,
1768
+ // though markRebuildForSession usually reaches it directly.
1769
+ if (!resultCtx.historyStale && resultCtx.piSessionId && historyRewrittenBySession.has(resultCtx.piSessionId)) {
1770
+ resultCtx.historyStale = true;
1771
+ }
1772
+ // User messages (steer/followUp) pi injected into context during the
1773
+ // active query: a steer sent while a tool was executing, drained by pi at
1774
+ // the turn boundary and appended alongside the tool result.
1775
+ const steer = lastMsgRole === "user" ? steerBlocks(context.messages) : null;
1776
+ // Delivery is async because the steer must reach CC's stdin *before* the
1777
+ // tool result does — see deliverToolResults. Detached so the provider
1778
+ // still returns its stream synchronously.
1779
+ void deliverToolResults(resultCtx, allResults, steer, context.messages.length);
1780
+ // The shared cursor tracks the top-level conversation. A reentrant subagent
1781
+ // delivering its own results would drag it to that subagent's message count
1782
+ // — observed pulling a parent from 5 back to 3, which cost the parent's next
1783
+ // turn a full rebuild and a flushed prompt cache.
1784
+ const state = sessionStateFor(resultCtx.piSessionId);
1785
+ if (state) state.cursor = context.messages.length;
1786
+ resultCtx.latestCursor = Math.max(resultCtx.latestCursor, context.messages.length);
1787
+ return stream;
1788
+ }
1789
+
1790
+ // --- Orphaned tool result (e.g. user aborted a tool call) ---
1791
+ // The query is gone but pi still delivered the result. Nothing to do — just
1792
+ // emit end_turn so pi waits for the next real user message. The discard
1793
+ // branch above already siphoned off the stale-query case, which goes on to a
1794
+ // rebuild instead — that one has somewhere to deliver the result to.
1795
+ const lastMsg = context.messages[context.messages.length - 1];
1796
+ if (lastMsg?.role === "toolResult" && !rewrittenUnderQuery) {
1797
+ debug(`provider: orphaned tool result after abort, emitting end_turn`);
1798
+ // With no query in flight anywhere, the top-level session this result
1799
+ // belongs to is the one whose turn just ended: its cursor advances to
1800
+ // count the result (options.sessionId is that session — pi emits the
1801
+ // result event through the same session's streamSimple call).
1802
+ const orphanState = sessionStateFor(options?.sessionId ?? null);
1803
+ if (orphanState && activeQueryContexts.size === 0) orphanState.cursor = context.messages.length;
1804
+ // No query owns this result, so there is no context to reset: resetTurnState
1805
+ // on the top-level ctx() would replace a live parent's turnOutput mid-stream,
1806
+ // stranding the blocks it had already emitted. A throwaway context just
1807
+ // supplies the empty message this turn ends with.
1808
+ const c = new QueryContext();
1809
+ c.resetTurnState(model);
1810
+ queueMicrotask(() => {
1811
+ stream.push({ type: "done", reason: "stop", message: c.turnOutput });
1812
+ markStreamComplete(stream);
1813
+ stream.end();
1814
+ });
1815
+ return stream;
1816
+ }
1817
+
1818
+ // --- Fresh query ---
1819
+
1820
+ // 1. Determine reentrancy. Reentrant queries get their own QueryContext so
1821
+ // background subagents can run concurrently with the parent query.
1822
+ const isReentrant = activeQuery;
1823
+ const queryCtx = isReentrant ? new QueryContext() : ctx();
1824
+ debug(`provider: fresh query setup, isReentrant=${isReentrant}, activeContexts=${activeQueryContexts.size}`);
1825
+
1826
+ // Resolved first: an unaccountable system prompt throws, and doing that before
1827
+ // anything is claimed or reset leaves no half-built query behind — in particular
1828
+ // no stream claimed on the shared context that nobody will ever end.
1829
+ const { mcpTools, customToolNameToSdk, customToolNameToPi } = resolveMcpTools(context, askClaudeToolName);
1830
+ // Build from what Pi loaded for this run, so `--no-context-files` and
1831
+ // `--no-skills` reach Claude Code by leaving nothing to forward. A sub-agent's
1832
+ // custom override embeds its parent's assembled Pi prompt; recursive projection
1833
+ // replaces that exact inherited prompt with its already-safe portable parts.
1834
+ // Derive the key from the transcript replay (toBridgeContext), NOT from the
1835
+ // recorded keys: under a forced prompt the transcript head is projected via
1836
+ // transformContext after turn_start, so ctx.getSystemPrompt() is not the head.
1837
+ const promptCapture = promptCaptures.resolveOrDerive(context.systemPrompt);
1838
+ const systemPromptAppend = promptCapture
1839
+ ? projectPromptCapture(promptCapture, {
1840
+ skillReadTool: mcpTools.some((tool) => tool.name === "read") ? "mcp" : "none",
1841
+ })
1842
+ : undefined;
1843
+
1844
+ // 2. Fresh child context — constructor already gave us clean Maps and empty
1845
+ // arrays. For a reused top-level context, clear explicitly.
1846
+ claimCurrentPiStream(stream, "fresh-query", queryCtx);
1847
+ queryCtx.pendingToolCalls.clear();
1848
+ queryCtx.pendingResults.clear();
1849
+ // Stale ids would let a late result from the previous query route here via
1850
+ // contextForToolResults — which now means pushing its steer into this
1851
+ // query's stdin, not just mismatching a map.
1852
+ queryCtx.turnToolCallIds = [];
1853
+ queryCtx.resetTurnState(model);
1854
+ queryCtx.latestCursor = 0;
1855
+ // The served pi session, for rewrite attribution on delivery (issue #101
1856
+ // follow-up) and on SessionState. A fresh instance of this module inside a
1857
+ // worktree-spawned subagent has its own contexts; each records its own.
1858
+ queryCtx.piSessionId = options?.sessionId ?? null;
1859
+ // A discarded query's replacement reuses this context; without the reset its
1860
+ // first tool result would sit on armed staleness again (the mark is consumed
1861
+ // from the set, not from here) and re-discard a healthy query.
1862
+ queryCtx.historyStale = false;
1863
+ queryCtx.missedSteer = false;
1864
+
1865
+ const cwd = process.cwd();
1866
+ // cliModel is the actual id sent to Claude Code (may carry [1m]); model.id is the
1867
+ // pi-registered id. Log cliModel so debug lines reflect what CC actually received.
1868
+ const cliModel = claudeCodeModelId(model, longContextSettings);
1869
+ // Which pi session this query serves — the attribution key for history
1870
+ // rewrites (session_compact / session_tree) and for SessionState above.
1871
+ const piSessionId = options?.sessionId ?? null;
1872
+ const syncResult = syncSharedSession(context.messages, cwd, customToolNameToSdk, cliModel, piSessionId);
1873
+ // This query starts from the history pi has now: consume this session's
1874
+ // armed rewrite — a sibling pi session's stays armed for its own queries.
1875
+ if (piSessionId) historyRewrittenBySession.delete(piSessionId);
1876
+ const { sessionId: resumeSessionId } = syncResult;
1877
+ const promptBlocks = extractUserPromptBlocks(context.messages);
1878
+ let promptText = extractUserPrompt(context.messages) ?? "";
1879
+
1880
+ // A turn continuing past a discarded query ends at its tool result, not at a
1881
+ // prompt, so say what happened rather than falling into the empty-prompt
1882
+ // recovery below — that one is for a shape we do not expect, and this is one
1883
+ // we do. The rebuilt session already ends with the tool result, placed after
1884
+ // the tool call it answers.
1885
+ if (rewrittenUnderQuery && !promptText && !promptBlocks) {
1886
+ promptText = CONTINUE_AFTER_REWRITE_PROMPT;
1887
+ debug(`provider: continuing the turn after a rewritten history, ${context.messages.length} msgs rebuilt`);
1888
+ }
1889
+
1890
+ // Guard: empty prompt means the last context message isn't a user message.
1891
+ // This should never happen with per-query state — dump diagnostics if it does.
1892
+ if (!promptText && !promptBlocks) {
1893
+ diagDump("empty_prompt", {
1894
+ contextLength: context.messages.length,
1895
+ lastMsgRole: lastMsg?.role,
1896
+ isReentrant,
1897
+ activeQueryContexts: activeQueryContexts.size,
1898
+ activeQueryExists: queryCtx.activeQuery !== null,
1899
+ sharedSession: sessionStateFor(piSessionId) ? { sessionId: sessionStateFor(piSessionId)!.sessionId.slice(0, 8), cursor: sessionStateFor(piSessionId)!.cursor } : (sessionStateFor(null) ? { sessionId: sessionStateFor(null)!.sessionId.slice(0, 8), cursor: sessionStateFor(null)!.cursor } : null),
1900
+ messageRoles: context.messages.map((m, i) => `[${i}]${m.role}`).join(" "),
1901
+ });
1902
+ // Recover: use a continuation prompt so the SDK doesn't send an empty text block
1903
+ promptText = "[continue]";
1904
+ }
1905
+
1906
+ // Always stream the prompt rather than passing a string: a parked input
1907
+ // generator is what lets us write steers to CC's stdin mid-turn. The cost is
1908
+ // that `isSingleUserTurn` is false, so the SDK no longer closes stdin on the
1909
+ // first result — consumeQuery ends the stream explicitly instead, or the
1910
+ // query would never terminate.
1911
+ const promptStream = makePromptStream();
1912
+ void promptStream.push(userMessage(promptBlocks ?? [{ type: "text", text: promptText }]))
1913
+ .catch((error) => debug(`provider: initial prompt push rejected:`, error));
1914
+ queryCtx.promptStream = promptStream;
1915
+ const mcpServers = buildMcpServers(mcpTools, queryCtx);
1916
+
1917
+ // MCP auto-loading suppression: CC reads MCP servers from ~/.claude.json (top-level
1918
+ // + per-project) and .mcp.json. Since pi executes tools (not CC), those are pure
1919
+ // token overhead. --strict-mcp-config tells the binary to use ONLY mcpServers passed
1920
+ // programmatically and ignore filesystem MCP entries — applied unconditionally because
1921
+ // settingSources is left at CC's default (all sources) unless loadClaudeSettings is false.
1922
+ const strictMcpConfigEnabled = providerSettings.strictMcpConfig !== false;
1923
+ const claudeExecutable = providerSettings.pathToClaudeCodeExecutable;
1924
+
1925
+ // Prefer the model's own thinkingLevelMap (per-model overrides — e.g. a map can
1926
+ // route xhigh→xhigh where the generic table maps xhigh→max). pi-ai's catalog
1927
+ // ships a map for most Claude models; the table below covers models without
1928
+ // one, and the levels a map leaves unnamed. A null entry means the level is
1929
+ // unsupported on that model: no effort argument is sent, so Claude Code's own
1930
+ // default applies rather than the generic table's value. Map values are
1931
+ // provider-generic strings, so a map value is trusted only when it names a
1932
+ // level CC accepts.
1933
+ const mapped = options?.reasoning ? model.thinkingLevelMap?.[options.reasoning] : undefined;
1934
+ const effort = options?.reasoning
1935
+ ? mapped === undefined
1936
+ ? REASONING_TO_EFFORT[options.reasoning]
1937
+ : VALID_EFFORTS.has(mapped as EffortLevel) ? mapped as EffortLevel : undefined
1938
+ : undefined;
1939
+
1940
+ const extraArgs: Record<string, string | null> = { model: cliModel };
1941
+ if (strictMcpConfigEnabled) extraArgs["strict-mcp-config"] = null;
1942
+ // Opus 4.7 defaults thinking.display to "omitted" (empty thinking text in stream).
1943
+ // Force summarized so thinking_delta events arrive. See anthropics/claude-agent-sdk-python#830.
1944
+ if (effort) extraArgs["thinking-display"] = "summarized";
1945
+
1946
+ // Suppress claude.ai cloud MCP servers (Figma/Canva/etc. auto-discovered via OAuth
1947
+ // when the user is logged into Anthropic). These are a separate code path from
1948
+ // filesystem MCP and are NOT blocked by --strict-mcp-config or settingSources=undefined.
1949
+ // The native CC binary gates them on env var ENABLE_CLAUDEAI_MCP_SERVERS: setting it
1950
+ // to "0"/"false"/"no"/"off" makes the loader return early before any cloud fetch.
1951
+ // DISABLE_AUTO_COMPACT=1: pi owns context-management and propagates its own
1952
+ // /compact via session_compact (see handler in default export). Letting CC
1953
+ // also autocompact would double-flush the prompt cache and races pi's
1954
+ // threshold with CC's, including CC's anti-thrashing guard (issue #8).
1955
+ // Manual /compact in CC still works (we never invoke it).
1956
+ const childEnv = { ...process.env, ...CC_CHILD_ENV };
1957
+ const queryOptions: NonNullable<Parameters<typeof query>[0]["options"]> = {
1958
+ cwd,
1959
+ env: childEnv,
1960
+ tools: [],
1961
+ permissionMode: "bypassPermissions",
1962
+ includePartialMessages: true,
1963
+ // Opt-out of Claude Code user/project/local settings. Pi already owns hooks and
1964
+ // extensions, so reloading CC's hooks/plugins per turn is pure overhead for users
1965
+ // who don't need settings-sourced env or apiKeyHelper.
1966
+ ...providerSettingSourcesOption(providerSettings),
1967
+ // includeGitInstructions:false drops the gitStatus block from the preset.
1968
+ // That block is the trailing suffix of the cached system block, and a
1969
+ // git-state transition (new file, staging, commit) rewrites it — busting
1970
+ // the prompt cache for the whole conversation from there on (see
1971
+ // diag/probe-git-cache.mjs). The bridge re-invokes CC per turn, so this
1972
+ // hit on every transition. Cost here is nil: the setting also strips
1973
+ // CC's git-workflow guidance from its Bash tool prompt, but the provider
1974
+ // path runs CC with `tools: []`, so those definitions never ship.
1975
+ // AskClaude keeps CC's native tools and its guidance — unaffected.
1976
+ settings: {
1977
+ ...claudeCodeSettings(providerSettings),
1978
+ claudeMdExcludes: CLAUDE_MD_EXCLUDES,
1979
+ includeGitInstructions: false,
1980
+ },
1981
+ systemPrompt: {
1982
+ type: "preset", preset: "claude_code",
1983
+ append: systemPromptAppend ? systemPromptAppend : undefined,
1984
+ },
1985
+ extraArgs,
1986
+ ...(effort ? { effort } : {}),
1987
+ ...(mcpServers ? { mcpServers } : {}),
1988
+ ...(resumeSessionId ? { resume: resumeSessionId } : {}),
1989
+ ...(claudeExecutable ? { pathToClaudeCodeExecutable: claudeExecutable } : {}),
1990
+ ...makeCliDebugOptions("provider"),
1991
+ };
1992
+
1993
+ debug("provider: fresh query",
1994
+ `model=${cliModel} msgs=${context.messages.length} tools=${mcpTools.length}`,
1995
+ `resume=${resumeSessionId?.slice(0, 8) ?? "none"} effort=${effort ?? "default"}`,
1996
+ `ctxFiles=${promptCapture?.contextFiles.length ?? 0} strictMcp=${strictMcpConfigEnabled}`,
1997
+ `prompt=${promptText.slice(0, 60)}${promptBlocks ? " [+images]" : ""}`);
1998
+
1999
+ // 3. Start SDK query and claim it for this context
2000
+ let wasAborted = false;
2001
+ const sdkQuery = queryImpl({ prompt: promptStream.stream, options: queryOptions });
2002
+ queryCtx.activeQuery = sdkQuery;
2003
+ activeQueryContexts.add(queryCtx);
2004
+
2005
+ // 4. Capture context for abort handling
2006
+ const abortCtx = queryCtx;
2007
+
2008
+ const requestAbort = () => {
2009
+ // interrupt() asks the CLI to stop gracefully; close() kills it immediately.
2010
+ // Both are needed — interrupt alone lets the current API call finish.
2011
+ void sdkQuery.interrupt().catch(() => {});
2012
+ try { sdkQuery.close(); } catch {}
2013
+ };
2014
+ const onAbort = () => {
2015
+ wasAborted = true;
2016
+ drainForAbort(abortCtx, promptStream);
2017
+ requestAbort();
2018
+ };
2019
+ if (options?.signal) {
2020
+ if (options.signal.aborted) onAbort();
2021
+ else options.signal.addEventListener("abort", onAbort, { once: true });
2022
+ }
2023
+
2024
+ // Background consumer — runs until query ends
2025
+ consumeQuery(sdkQuery, customToolNameToPi, model, () => wasAborted, queryCtx)
2026
+ .then(async ({ capturedSessionId }) => {
2027
+ debug(`provider: consumeQuery completed, stopReason=${queryCtx.turnOutput?.stopReason}, error=${queryCtx.turnOutput?.errorMessage}, aborted=${wasAborted}`);
2028
+
2029
+ // Discarded out from under: the query continuing the turn owns the context,
2030
+ // the session and the stream. Capturing this one's session id here would put
2031
+ // Claude Code back on the conversation it was discarded for.
2032
+ if (abandonedQueries.has(sdkQuery)) {
2033
+ debug("provider: discarded query completed, leaving session and stream to its replacement");
2034
+ return;
2035
+ }
2036
+
2037
+ // --- Abort detection in normal completion path ---
2038
+ if (wasAborted || options?.signal?.aborted) {
2039
+ // The killed subprocess may flush a late record into this session's
2040
+ // JSONL — its own mirror's next sync must rebuild and rotate.
2041
+ const state = sessionStateFor(queryCtx.piSessionId);
2042
+ if (state) setSessionStateFor(queryCtx.piSessionId, { ...state, needsRebuild: true, forceRotate: true });
2043
+ debug(`provider: abort detected, marked sharedSession needsRebuild + forceRotate`);
2044
+ if (queryCtx.turnOutput) {
2045
+ queryCtx.turnOutput.stopReason = "aborted";
2046
+ queryCtx.turnOutput.errorMessage = "Operation aborted";
2047
+ }
2048
+ const stream = queryCtx.currentPiStream;
2049
+ stream?.push({ type: "error", reason: "aborted", error: queryCtx.turnOutput! });
2050
+ markStreamComplete(stream);
2051
+ stream?.end();
2052
+ queryCtx.currentPiStream = null;
2053
+ return;
2054
+ }
2055
+
2056
+ // --- Capture session ID ---
2057
+ // This query's own mirror — a reentrant subagent completing does not
2058
+ // reassign the parent's conversation to the child's CC file.
2059
+ if (syncResult.preserveSharedSession) {
2060
+ const state = sessionStateFor(queryCtx.piSessionId);
2061
+ if (capturedSessionId && capturedSessionId !== state?.sessionId) {
2062
+ deleteSession(capturedSessionId, cwd, process.env.CLAUDE_CONFIG_DIR);
2063
+ debug(`provider: query done, deleted ephemeral session ${capturedSessionId.slice(0, 8)} to preserve shared session`);
2064
+ }
2065
+ debug(`provider: query done, ignoring captured session ${capturedSessionId?.slice(0, 8) ?? "none"} to preserve shared session`);
2066
+ } else {
2067
+ const state = sessionStateFor(queryCtx.piSessionId);
2068
+ const sessionId = capturedSessionId ?? state?.sessionId;
2069
+ if (sessionId) {
2070
+ const cursor = Math.max(context.messages.length, queryCtx.latestCursor, state?.cursor ?? 0);
2071
+ debug(`provider: query done, session=${sessionId.slice(0, 8)}, cursor=${cursor}`);
2072
+ // A missed steer may precede the first mirror or arrive while this
2073
+ // query is still able to complete. Preserve both rebuild signals.
2074
+ setSessionStateFor(queryCtx.piSessionId, { ...state, sessionId, cursor, cwd, piSessionId: queryCtx.piSessionId ?? undefined, needsRebuild: queryCtx.missedSteer || state?.needsRebuild });
2075
+ }
2076
+ }
2077
+
2078
+ if (!isReentrant && queryCtx.activeQuery === sdkQuery) {
2079
+ debug("provider: clearing activeQuery before final stream completion");
2080
+ queryCtx.activeQuery = null;
2081
+ }
2082
+ finalizeCurrentStream(queryCtx, queryCtx.turnOutput?.stopReason);
2083
+ })
2084
+ .catch((error) => {
2085
+ debug(`provider: query error, model=${cliModel}, aborted=${Boolean(options?.signal?.aborted)}, error=`, error);
2086
+ if (abandonedQueries.has(sdkQuery)) {
2087
+ debug("provider: discarded query ended in error, leaving session and stream to its replacement");
2088
+ return;
2089
+ }
2090
+ if ((wasAborted || options?.signal?.aborted)) {
2091
+ const state = sessionStateFor(queryCtx.piSessionId);
2092
+ if (state) setSessionStateFor(queryCtx.piSessionId, { ...state, needsRebuild: true, forceRotate: true });
2093
+ } else {
2094
+ // Drop this session's mirror: its conversation is in an unknown
2095
+ // state after the error. Other sessions' mirrors stay — one
2096
+ // session's failure says nothing about another's conversation.
2097
+ setSessionStateFor(queryCtx.piSessionId, null);
2098
+ }
2099
+ promptStream.fail(error instanceof Error ? error : new Error(String(error)));
2100
+ if (queryCtx.turnOutput) {
2101
+ queryCtx.turnOutput.stopReason = options?.signal?.aborted ? "aborted" : "error";
2102
+ // The SDK drops its copy of the result text if any message follows the error
2103
+ // result, so prefer the cause consumeQuery recorded off the result itself.
2104
+ queryCtx.turnOutput.errorMessage ??= error instanceof Error ? error.message : String(error);
2105
+ }
2106
+ if (!isReentrant && queryCtx.activeQuery === sdkQuery) {
2107
+ queryCtx.releasePendingToolCalls("Query ended");
2108
+ debug("provider: clearing activeQuery before error stream completion");
2109
+ queryCtx.activeQuery = null;
2110
+ }
2111
+ const stream = queryCtx.currentPiStream;
2112
+ stream?.push({ type: "error", reason: (queryCtx.turnOutput?.stopReason ?? "error") as "aborted" | "error", error: queryCtx.turnOutput! });
2113
+ markStreamComplete(stream);
2114
+ stream?.end();
2115
+ queryCtx.currentPiStream = null;
2116
+ })
2117
+ .finally(() => {
2118
+ if (options?.signal) options.signal.removeEventListener("abort", onAbort);
2119
+ // Settle any ack still parked in the generator — the CLI is gone, so
2120
+ // nothing will resume it. Clear the handle only if a later query
2121
+ // hasn't already claimed the shared context.
2122
+ promptStream.fail(new Error("query ended"));
2123
+ if (queryCtx.promptStream === promptStream) queryCtx.promptStream = null;
2124
+ // A later query claiming this context sets activeQuery to its own handle;
2125
+ // null means the .then/.catch above cleared ours and nothing replaced it.
2126
+ // Testing only for `=== sdkQuery` would never fire on the non-reentrant
2127
+ // path, leaving the top-level context in the routing set forever — where a
2128
+ // later orphaned tool result matches its stale turnToolCallIds and takes
2129
+ // the delivery branch, returning a stream nothing ends.
2130
+ if (queryCtx.activeQuery === sdkQuery || queryCtx.activeQuery === null) {
2131
+ queryCtx.releasePendingToolCalls("Query ended");
2132
+ queryCtx.activeQuery = null;
2133
+ activeQueryContexts.delete(queryCtx);
2134
+ }
2135
+ sdkQuery.close();
2136
+ });
2137
+
2138
+ return stream;
2139
+ }
2140
+
2141
+ // --- AskClaude: prompt and wait ---
2142
+
2143
+ async function promptAndWait(
2144
+ prompt: string,
2145
+ mode: "full" | "read" | "none",
2146
+ toolCalls: Map<string, ToolCallState>,
2147
+ signal?: AbortSignal,
2148
+ options?: {
2149
+ systemPrompt?: string;
2150
+ appendSkills?: boolean;
2151
+ onStreamUpdate?: (responseText: string) => void;
2152
+ model?: string;
2153
+ thinking?: string;
2154
+ isolated?: boolean;
2155
+ context?: Context["messages"];
2156
+ /** pi session the calling tool ran in — AskClaude's conversation continues
2157
+ * the session that called it, so its sync and capture key that mirror. */
2158
+ piSessionId?: string | null;
2159
+ },
2160
+ ): Promise<{ responseText: string; stopReason: string }> {
2161
+ const cwd = process.cwd();
2162
+ const requestedModel = options?.model ?? "opus";
2163
+ const model = resolveModel(requestedModel);
2164
+ const modelId = model?.id ?? requestedModel;
2165
+ const cliModel = model ? claudeCodeModelId(model, longContextSettings) : modelId;
2166
+
2167
+ // Session resume for shared mode — reuse provider's session if it exists,
2168
+ // otherwise create one from pi's context.
2169
+ // Note: doesn't update the mirror's cursor after completion, so the next
2170
+ // provider call will see missed messages and trigger a Case 4 rebuild.
2171
+ // AskClaude has a random-origin context handed to it, but the session it
2172
+ // belongs to is the one whose tool ran — the extension API's execute ctx
2173
+ // carries it — passed here as piSessionId and used for every map access.
2174
+ const askClaudeSessionId = options?.piSessionId ?? null;
2175
+ let resumeSessionId: string | null = null;
2176
+ if (!options?.isolated && options?.context?.length) {
2177
+ const askClaudeState = sessionStateFor(askClaudeSessionId);
2178
+ if (askClaudeState) {
2179
+ // Provider already has a session — just resume from it
2180
+ // Any missed messages from other providers were already handled by the provider's Case 4
2181
+ resumeSessionId = askClaudeState.sessionId;
2182
+ } else {
2183
+ // No provider session yet — create one from pi's context
2184
+ const contextWithPrompt = [...options.context, { role: "user" as const, content: prompt, timestamp: Date.now() }];
2185
+ const sync = syncSharedSession(contextWithPrompt as Context["messages"], cwd, undefined, cliModel, askClaudeSessionId);
2186
+ resumeSessionId = sync.sessionId;
2187
+ }
2188
+ }
2189
+
2190
+ // Mode → disallowed tools
2191
+ const disallowedTools = MODE_DISALLOWED_TOOLS[mode];
2192
+
2193
+ // AskClaude uses Claude Code's native Read tool rather than Pi's MCP bridge.
2194
+ // Same resolver as the provider path: a prompt neither recorded nor derivable
2195
+ // throws here too, rather than silently sending Claude Code no skills.
2196
+ //
2197
+ // Resolved only when the answer would be used. The throw is justified by what a
2198
+ // miss would cost, so where it costs nothing — skills switched off, or no reader
2199
+ // to open a skill file with — an unrelated miss must not fail the call.
2200
+ const skillReadTool = disallowedTools.includes("Read") ? "none" : "native";
2201
+ const skillCapture = options?.appendSkills !== false && skillReadTool !== "none"
2202
+ ? promptCaptures.resolveOrDerive(options?.systemPrompt)
2203
+ : undefined;
2204
+ const skillsBlock = skillCapture
2205
+ ? renderSkillsBlock(collectPromptSkills(skillCapture), skillReadTool)
2206
+ : undefined;
2207
+
2208
+ // Effort
2209
+ const effort = options?.thinking && options.thinking !== "off"
2210
+ ? REASONING_TO_EFFORT[options.thinking] : undefined;
2211
+
2212
+ const claudeExecutable = providerSettings.pathToClaudeCodeExecutable;
2213
+
2214
+ const extraArgs: Record<string, string | null> = {
2215
+ "strict-mcp-config": null,
2216
+ model: cliModel,
2217
+ };
2218
+ if (effort) extraArgs["thinking-display"] = "summarized";
2219
+
2220
+ debug("askClaude:",
2221
+ `mode=${mode} model=${modelId} cliModel=${cliModel} effort=${effort ?? "default"}`,
2222
+ `isolated=${options?.isolated ?? false} resume=${resumeSessionId?.slice(0, 8) ?? "none"}`,
2223
+ `skills=${Boolean(skillsBlock)} promptLen=${prompt.length}`);
2224
+
2225
+ // skills: [] suppresses Claude Code's own skill listing, a system-reminder naming every
2226
+ // skill under the ~/.claude estate. The provider path gets this for free — `tools: []`
2227
+ // removes the Skill tool and the listing with it — but AskClaude runs on CC's native
2228
+ // tools, so it has to be asked for. Pi-side skills still arrive via skillsBlock below,
2229
+ // which is meant to be the only channel.
2230
+ const sdkQuery = query({
2231
+ prompt,
2232
+ options: {
2233
+ cwd,
2234
+ env: { ...process.env, ...CC_CHILD_ENV },
2235
+ permissionMode: "bypassPermissions",
2236
+ settings: { ...claudeCodeSettings(providerSettings), claudeMdExcludes: CLAUDE_MD_EXCLUDES },
2237
+ skills: [],
2238
+ ...(disallowedTools.length ? { disallowedTools } : {}),
2239
+ ...(effort ? { effort } : {}),
2240
+ // Preset unconditionally: omitting it leaves the child on the SDK's bare default,
2241
+ // without the tool and permission guidance the bridge relies on everywhere else.
2242
+ // Whether pi has skills to append is unrelated to whether the child needs that.
2243
+ systemPrompt: { type: "preset", preset: "claude_code", append: skillsBlock },
2244
+ settingSources: ["user", "project"] as SettingSource[],
2245
+ extraArgs,
2246
+ ...(resumeSessionId ? { resume: resumeSessionId } : {}),
2247
+ ...(options?.isolated ? { persistSession: false } : {}),
2248
+ ...(claudeExecutable ? { pathToClaudeCodeExecutable: claudeExecutable } : {}),
2249
+ ...makeCliDebugOptions("askclaude"),
2250
+ },
2251
+ });
2252
+
2253
+ // Abort handling
2254
+ let wasAborted = false;
2255
+ const onAbort = () => {
2256
+ wasAborted = true;
2257
+ sdkQuery.interrupt().catch(() => { try { sdkQuery.close(); } catch {} });
2258
+ };
2259
+ if (signal?.aborted) { onAbort(); throw new Error("Aborted"); }
2260
+ signal?.addEventListener("abort", onAbort, { once: true });
2261
+
2262
+ let responseText = "";
2263
+ let sdkMessageCount = 0;
2264
+ let textDeltaCount = 0;
2265
+ let resultSubtype: string | undefined;
2266
+
2267
+ try {
2268
+ for await (const message of sdkQuery) {
2269
+ if (wasAborted) break;
2270
+ sdkMessageCount++;
2271
+
2272
+ switch (message.type) {
2273
+ case "stream_event": {
2274
+ const event = (message as SDKMessage & { event: any }).event;
2275
+ // Text deltas → accumulate and stream
2276
+ if (event?.type === "content_block_delta" && event.delta?.type === "text_delta") {
2277
+ responseText += event.delta.text;
2278
+ textDeltaCount++;
2279
+ options?.onStreamUpdate?.(responseText);
2280
+ }
2281
+ // Tool call start → track for action summary progress
2282
+ if (event?.type === "content_block_start" && event.content_block?.type === "tool_use") {
2283
+ debug(`askClaude: tool_use start: ${event.content_block.name}`);
2284
+ toolCalls.set(event.content_block.id, {
2285
+ name: mapToolName(event.content_block.name),
2286
+ status: "running",
2287
+ });
2288
+ }
2289
+ break;
2290
+ }
2291
+ case "assistant": {
2292
+ // Update tool calls with full input for action summary
2293
+ for (const block of (message as any).message?.content ?? []) {
2294
+ if (block.type === "tool_use") {
2295
+ toolCalls.set(block.id, {
2296
+ name: mapToolName(block.name),
2297
+ status: "complete",
2298
+ rawInput: block.input,
2299
+ });
2300
+ }
2301
+ }
2302
+ break;
2303
+ }
2304
+ case "result": {
2305
+ resultSubtype = message.subtype;
2306
+ const r = message as any;
2307
+ if (r.usage) {
2308
+ debug(`askClaude: result usage: in=${r.usage.input_tokens} out=${r.usage.output_tokens} cacheRead=${r.usage.cache_read_input_tokens ?? 0} cacheWrite=${r.usage.cache_creation_input_tokens ?? 0} turns=${r.num_turns ?? "?"}`);
2309
+ }
2310
+ // Claude Code reports an API failure with `is_error` on a result whose
2311
+ // subtype is still "success", so without this the error text was returned
2312
+ // as Claude's answer and pi's model read a 429 as content. Throwing hands
2313
+ // it to the tool's own catch, which renders it as an error result.
2314
+ const failure = wasAborted ? undefined : resultErrorText(message);
2315
+ if (failure) throw new Error(failure);
2316
+ if (!responseText && message.subtype === "success" && message.result) {
2317
+ responseText = message.result;
2318
+ }
2319
+ break;
2320
+ }
2321
+ }
2322
+ }
2323
+
2324
+ const stopReason = wasAborted ? "cancelled" : "stop";
2325
+ debug(`askClaude: done`,
2326
+ `stopReason=${stopReason} resultSubtype=${resultSubtype ?? "none"}`,
2327
+ `sdkMessages=${sdkMessageCount} textDeltas=${textDeltaCount} responseLen=${responseText.length}`,
2328
+ `toolCalls=${toolCalls.size}`);
2329
+ return { responseText, stopReason };
2330
+ } finally {
2331
+ signal?.removeEventListener("abort", onAbort);
2332
+ sdkQuery.close();
2333
+ }
2334
+ }
2335
+
2336
+ // --- Extension registration ---
2337
+
2338
+ const PREVIEW_MAX_CHARS = 1000;
2339
+ const PREVIEW_MAX_LINES = 6;
2340
+
2341
+ let askClaudeToolName = "AskClaude";
2342
+
2343
+ export default function (pi: ExtensionAPI) {
2344
+ // Disable non-essential Claude Code traffic (update checks, MCP registry, telemetry)
2345
+ process.env.CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC = "1";
2346
+
2347
+ const config = loadConfig(process.cwd());
2348
+ debug("loadConfig:", JSON.stringify(config));
2349
+ providerSettings = config.provider ?? {};
2350
+ // We need these settings to know if we're eligible for 1M context on certain models
2351
+ // Validate at the boundary: a non-array here would throw inside every
2352
+ // claudeCodeModelId call and brick the extension at activation.
2353
+ const forceTwoHundredK = Array.isArray(providerSettings.forceTwoHundredK)
2354
+ ? providerSettings.forceTwoHundredK.filter((id): id is string => typeof id === "string")
2355
+ : undefined;
2356
+ longContextSettings = {
2357
+ plan: providerSettings.plan ?? "pro",
2358
+ longContextExtraUsage: providerSettings.longContextExtraUsage ?? false,
2359
+ forceTwoHundredK,
2360
+ };
2361
+ const registeredModels = applyLongContext(MODELS, longContextSettings);
2362
+ if (registeredModels.length === 0) {
2363
+ console.error("claude-bridge: no models available from pi-ai's anthropic catalog — update @earendil-works/pi-ai (requires >=0.86.1)");
2364
+ }
2365
+
2366
+ if (!config.startupNoticeShown) {
2367
+ if (config.provider?.plan === undefined) pendingNotices.push('Are you using a Max plan? You need to set provider.plan to "max" to unlock 1M context in Opus.');
2368
+ if (config.askClaude?.enabled === undefined) pendingNotices.push("The AskClaude tool is opt-in only. Set askClaude.enabled to use it.");
2369
+ }
2370
+
2371
+ // Reset shared session on pi session lifecycle events
2372
+ const clearSession = (event: string) => {
2373
+ debug(`${event}: clearing ${sharedSessions.size} shared session${sharedSessions.size === 1 ? "" : "s"}`);
2374
+ // Whole map: children never emit session_shutdown (only runtime teardown
2375
+ // and /reload do), so there is no per-entry removal to do here — the
2376
+ // top-level transition takes every mirror with it.
2377
+ sharedSessions.clear();
2378
+ historyRewrittenBySession.clear();
2379
+
2380
+ // Clear the global streamSimple if this instance registered it.
2381
+ // This allows /reload to work — the old instance clears the flag so
2382
+ // the new instance can register fresh without wrapping stale state.
2383
+ const g = globalThis as Record<symbol, any>;
2384
+ if (g[ACTIVE_STREAM_SIMPLE_KEY] === streamClaudeAgentSdk) {
2385
+ debug(`${event}: clearing ACTIVE_STREAM_SIMPLE_KEY`);
2386
+ g[ACTIVE_STREAM_SIMPLE_KEY] = undefined;
2387
+ }
2388
+ };
2389
+ pi.on("session_start", (event, ctx) => {
2390
+ piUI = ctx.ui;
2391
+ piMode = ctx.mode;
2392
+ if (event.reason === "new" || event.reason === "resume" || event.reason === "fork") {
2393
+ clearSession(`session_start:${event.reason}`);
2394
+ }
2395
+ });
2396
+ // `--system-prompt` replaces pi's default rather than adding to it, but Claude
2397
+ // Code's preset carries its own tool and permission guidance that the bridge
2398
+ // still depends on, so both flags are forwarded as an append.
2399
+ //
2400
+ // The options (custom/append/contextFiles/skills) are pi config, stable across a
2401
+ // turn; only the auto-generated tool list in the rendered prompt varies. Stash them
2402
+ // at before_agent_start so the agent_start recording below can reuse them.
2403
+ type RecordOptions = Parameters<typeof recordSystemPrompt>[2];
2404
+ let lastSystemPromptOptions: RecordOptions | undefined;
2405
+ function recordSystemPrompt(source: string, systemPrompt: string | undefined, options: {
2406
+ customPrompt?: string;
2407
+ appendSystemPrompt?: string;
2408
+ contextFiles?: { path: string; content: string }[];
2409
+ skills?: Parameters<typeof promptCaptures.record>[1]["skills"];
2410
+ selectedTools?: string[];
2411
+ } | undefined) {
2412
+ if (!systemPrompt) return;
2413
+ const hasRead = !options?.selectedTools || options.selectedTools.includes("read");
2414
+ promptCaptures.record(systemPrompt, {
2415
+ custom: options?.customPrompt,
2416
+ append: options?.appendSystemPrompt,
2417
+ contextFiles: options?.contextFiles ?? [],
2418
+ skills: hasRead ? options?.skills ?? [] : [],
2419
+ }, source);
2420
+ }
2421
+ pi.on("before_agent_start", (event) => {
2422
+ lastSystemPromptOptions = event.systemPromptOptions;
2423
+ recordSystemPrompt("before_agent_start", event.systemPrompt, event.systemPromptOptions);
2424
+ });
2425
+ // The prompt the provider actually queries with is the fully-widened one: MCP tool
2426
+ // descriptions merge into the system prompt only after their servers connect, which
2427
+ // is after before_agent_start. ctx.getSystemPrompt() returns that widened prompt by
2428
+ // agent_start (verified: before_agent_start=10,988 chars vs agent_start/query=23,479).
2429
+ // A subagent embeds the widened parent prompt verbatim (pi-subagents reads
2430
+ // ctx.getSystemPrompt() at dispatch), so unless the widened prompt is a capture key
2431
+ // too, the child's turn resolves against nothing, falls to a verbatim side request,
2432
+ // and ships pi's harness — tripping the server's third-party plan-eligibility check
2433
+ // ("out of extra usage"). Recording it here, before the query, restores the match.
2434
+ //
2435
+ // agent_start also captures a handler-returned forceSystemPrompt, which
2436
+ // buildSystemPrompt renders verbatim.
2437
+ pi.on("agent_start", (_event, ctx) => {
2438
+ recordSystemPrompt("agent_start", ctx.getSystemPrompt(), lastSystemPromptOptions);
2439
+ });
2440
+
2441
+ // Mid-run re-renders: turn_start fires before every turn (first turn included)
2442
+ // after the turn's prompt is final: prepareNextTurnWithContext has
2443
+ // re-rendered the options (pi's section-based prompt) and any mid-run
2444
+ // setActiveToolsByName rebuild has already landed. Re-keying at each boundary
2445
+ // the prompt can change at keeps exact-match alive mid-run. The stashed options can
2446
+ // lag a mid-run tool-loadout change, which skews the hasRead skills filter until the
2447
+ // next before_agent_start — accepted: a stale skills list beats failing the turn.
2448
+ pi.on("turn_start", (_event, ctx) => {
2449
+ recordSystemPrompt("turn_start", ctx.getSystemPrompt(), lastSystemPromptOptions);
2450
+ });
2451
+ pi.on("session_shutdown", () => {
2452
+ reportLeaks("session_shutdown");
2453
+ clearSession("session_shutdown");
2454
+ });
2455
+
2456
+ pi.on("session_before_compact", async (event, ctx) => {
2457
+ if (ctx.model?.baseUrl !== "claude-bridge") return undefined;
2458
+ debug(
2459
+ `session_before_compact: takeover reason=${event.reason} willRetry=${event.willRetry} ` +
2460
+ `isSplitTurn=${event.preparation.isSplitTurn} messages=${event.preparation.messagesToSummarize.length} ` +
2461
+ `turnPrefix=${event.preparation.turnPrefixMessages.length}`,
2462
+ );
2463
+ try {
2464
+ reinjectPriorCompactionFileOps(event.branchEntries, event.preparation);
2465
+ const compaction = await compact(
2466
+ event.preparation,
2467
+ ctx.model,
2468
+ undefined,
2469
+ undefined,
2470
+ event.customInstructions,
2471
+ event.signal,
2472
+ undefined,
2473
+ isolatedStreamFn,
2474
+ undefined,
2475
+ );
2476
+ debug(`session_before_compact: takeover complete summaryLen=${compaction.summary.length}`);
2477
+ return { compaction };
2478
+ } catch (err) {
2479
+ const msg = errorMessage(err);
2480
+ debug("session_before_compact: takeover failed; cancelling to avoid native compact fallback", err);
2481
+ ctx.ui?.notify?.(
2482
+ `Claude bridge compact failed (${msg}); cancelled to avoid known hang. Retry, switch model, or reduce context.`,
2483
+ "error",
2484
+ );
2485
+ return { cancel: true };
2486
+ }
2487
+ });
2488
+
2489
+ // pi /compact and session-tree navigation (rewind / fork-at-point /
2490
+ // branch switch) both mutate pi's messages array out from under the
2491
+ // bridge. syncSharedSession's REUSE check would otherwise see
2492
+ // slice(cursor) === [] (or skip entries) and keep --resume'ing a CC
2493
+ // session that no longer matches pi's history. /compact in particular
2494
+ // triggers CC's autocompact-thrashing guard (issue #8). Force the next
2495
+ // call down the REBUILD path so CC sees the current history — and, when a
2496
+ // query is parked at a tool boundary while this fires, discard that query
2497
+ // instead of resuming it (markRebuild, discardRewrittenQuery).
2498
+ //
2499
+ // Attributed to the compacting session (ctx.sessionManager belongs to the
2500
+ // session whose runner fired this), so a subagent compacting while its
2501
+ // parent sits parked on the Agent tool result discards nothing — the parent's
2502
+ // query is live and its conversation untouched. Registered by every instance
2503
+ // of this module; instances sponsoring stale marks forward them to the
2504
+ // serving instance via sponsorMarkRebuildForSession.
2505
+ pi.on("session_compact", (event, ctx) =>
2506
+ sponsorMarkRebuildForSession(ctx.sessionManager.getSessionId(), `session_compact:${event.reason}:willRetry=${event.willRetry}`));
2507
+ pi.on("session_tree", (_event, ctx) => sponsorMarkRebuildForSession(ctx.sessionManager.getSessionId(), "session_tree"));
2508
+
2509
+ // Branch summarization — rewind or fork-at-point with "summarize" — is the other
2510
+ // place pi asks the model for a summary, and unlike compaction it runs through
2511
+ // the *agent's* stream function (agent-session passes `streamFn:
2512
+ // this.agent.streamFunction`). On a bridge model that reaches this provider
2513
+ // carrying pi's internal summarization prompt, which no `before_agent_start`
2514
+ // ever recorded, so the prompt-capture resolver has nothing to resolve it to.
2515
+ // Take it over the way compaction is taken over: the summary runs as its own
2516
+ // Claude Code subprocess, never touching the live session or the resolver.
2517
+ pi.on("session_before_tree", async (event, ctx) => {
2518
+ if (ctx.model?.baseUrl !== "claude-bridge") return undefined;
2519
+ const { entriesToSummarize, userWantsSummary, customInstructions, replaceInstructions } = event.preparation;
2520
+ if (!userWantsSummary || entriesToSummarize.length === 0) return undefined;
2521
+ debug(`session_before_tree: takeover entries=${entriesToSummarize.length} target=${event.preparation.targetId.slice(0, 8)}`);
2522
+ try {
2523
+ const result = await generateBranchSummary(entriesToSummarize, {
2524
+ model: ctx.model,
2525
+ signal: event.signal,
2526
+ customInstructions,
2527
+ replaceInstructions,
2528
+ streamFn: isolatedStreamFn,
2529
+ });
2530
+ return branchSummaryOutcome(result);
2531
+ } catch (err) {
2532
+ debug("session_before_tree: takeover failed; cancelling navigation", err);
2533
+ ctx.ui?.notify?.(
2534
+ `Claude bridge branch summary failed (${errorMessage(err)}); navigation cancelled.`,
2535
+ "error",
2536
+ );
2537
+ return { cancel: true };
2538
+ }
2539
+ });
2540
+
2541
+ // --- Provider ---
2542
+ //
2543
+ // Registration policy across module instances (a subagent session can load
2544
+ // this module fresh): the FIRST instance registers unconditionally at load,
2545
+ // which is what puts claude-bridge models in the picker before any session
2546
+ // starts. Later instances decide at session_start, when ctx.modelRegistry
2547
+ // reveals who owns this session's registry:
2548
+ //
2549
+ // - Registry already has the provider (host passes the parent's registry down,
2550
+ // e.g. pi-subagents >=0.14.3): skip. Re-registering would overwrite the
2551
+ // parent's pinned streamSimple with this instance's fresh — empty-state —
2552
+ // stream fn, and the parent's next tool-result delivery would route into it.
2553
+ // - Registry lacks the provider (host gives the child its own, e.g. older
2554
+ // pi-subagents forks): register, or every claude-bridge/* dispatch in the
2555
+ // child fails with "Model not found" (#91). Even loading the bridge via the
2556
+ // agent's `extensions:` frontmatter didn't help there — the module loaded,
2557
+ // hit the old skip-guard, and the child's registry stayed empty.
2558
+ //
2559
+ // A per-instance stream fn registered into a per-instance registry is
2560
+ // self-consistent: that session's traffic flows through this module state,
2561
+ // which starts clean and serves only that session.
2562
+ //
2563
+ // On session_shutdown (including /reload), clearSession() resets
2564
+ // ACTIVE_STREAM_SIMPLE_KEY so a freshly loaded module can register as first
2565
+ // again.
2566
+
2567
+ const g = globalThis as Record<symbol, any>;
2568
+ const providerConfig = {
2569
+ baseUrl: "claude-bridge",
2570
+ apiKey: "not-used",
2571
+ api: "claude-bridge",
2572
+ models: registeredModels,
2573
+ // Cast: the Provider interface passes a TranscriptContext; the bridge takes plain
2574
+ // Context models (toBridgeContext normalizes at the stream entry points).
2575
+ streamSimple: streamClaudeAgentSdk as any,
2576
+ };
2577
+ if (!g[ACTIVE_STREAM_SIMPLE_KEY]) {
2578
+ // First instance: store our streamSimple and register.
2579
+ g[ACTIVE_STREAM_SIMPLE_KEY] = streamClaudeAgentSdk;
2580
+ pi.registerProvider(PROVIDER_ID, providerConfig);
2581
+ } else {
2582
+ // Later instance: register only if this session's registry lacks the provider.
2583
+ debug(`provider: deferring registration decision to session_start (module=${moduleInstanceId})`);
2584
+ pi.on("session_start", (_event, ctx) => {
2585
+ if (ctx.modelRegistry.getProvider(PROVIDER_ID)) {
2586
+ debug(`provider: registry already has ${PROVIDER_ID}, skipping registration (module=${moduleInstanceId})`);
2587
+ return;
2588
+ }
2589
+ debug(`provider: registry lacks ${PROVIDER_ID}, registering (module=${moduleInstanceId})`);
2590
+ pi.registerProvider(PROVIDER_ID, providerConfig);
2591
+ });
2592
+ }
2593
+
2594
+ // --- AskClaude tool ---
2595
+
2596
+ const askConf = config.askClaude;
2597
+ const askDefaults = resolveAskClaudeDefaults(askConf);
2598
+ askClaudeToolName = askConf?.name ?? "AskClaude";
2599
+
2600
+ if (askConf?.enabled) {
2601
+ const askClaudeParams = buildAskClaudeParams(askDefaults);
2602
+ pi.registerTool<typeof askClaudeParams>({
2603
+ name: askConf?.name ?? "AskClaude",
2604
+ label: askConf?.label ?? "Ask Claude Code",
2605
+ description: askClaudeToolDescription(askDefaults, askConf?.description),
2606
+ parameters: askClaudeParams,
2607
+ renderCall(args, theme) {
2608
+ let text = theme.fg("mdLink", theme.bold("AskClaude "));
2609
+ const tags = askClaudeCallTags(args, askDefaults);
2610
+ if (tags.length) text += `${theme.fg("accent", `[${tags.join(", ")}]`)} `;
2611
+ const truncated = args.prompt.length > PREVIEW_MAX_CHARS ? args.prompt.substring(0, PREVIEW_MAX_CHARS) : args.prompt;
2612
+ const lines = truncated.split("\n").slice(0, PREVIEW_MAX_LINES);
2613
+ text += theme.fg("muted", `"${lines.join("\n")}"`);
2614
+ if (args.prompt.length > PREVIEW_MAX_CHARS || args.prompt.split("\n").length > PREVIEW_MAX_LINES) text += theme.fg("dim", " …");
2615
+ return new Text(text, 0, 0);
2616
+ },
2617
+ renderResult(result, { expanded, isPartial }, theme) {
2618
+ if (isPartial) {
2619
+ const status = result.content[0]?.type === "text" ? result.content[0].text : "working...";
2620
+ return new Text(theme.fg("mdLink", "◉ Claude Code ") + theme.fg("muted", status), 0, 0);
2621
+ }
2622
+
2623
+ const details = result.details as { prompt?: string; executionTime?: number; actions?: string; error?: boolean } | undefined;
2624
+ const body = result.content[0]?.type === "text" ? result.content[0].text : "";
2625
+
2626
+ let text = details?.error
2627
+ ? theme.fg("error", "✗ Claude Code error")
2628
+ : theme.fg("mdLink", "✓ Claude Code");
2629
+
2630
+ if (details?.executionTime) text += ` ${theme.fg("dim", `${(details.executionTime / 1000).toFixed(1)}s`)}`;
2631
+ if (details?.actions) text += ` ${theme.fg("muted", details.actions)}`;
2632
+
2633
+ if (expanded) {
2634
+ if (details?.prompt) text += `\n${theme.fg("dim", `Prompt: ${details.prompt}`)}`;
2635
+ if (details?.prompt && body) text += `\n${theme.fg("dim", "─".repeat(40))}`;
2636
+ if (body) text += `\n${theme.fg("toolOutput", body)}`;
2637
+ } else {
2638
+ const truncated = body.length > PREVIEW_MAX_CHARS ? body.substring(0, PREVIEW_MAX_CHARS) : body;
2639
+ const lines = truncated.split("\n").slice(0, PREVIEW_MAX_LINES);
2640
+ if (lines.length) text += `\n${theme.fg("toolOutput", lines.join("\n"))}`;
2641
+ if (body.length > PREVIEW_MAX_CHARS || body.split("\n").length > PREVIEW_MAX_LINES) text += `\n${theme.fg("dim", `… (${keyHint("app.tools.expand", "to expand")})`)}`;
2642
+
2643
+ }
2644
+
2645
+ return new Text(text, 0, 0);
2646
+ },
2647
+ async execute(_id, params, signal, onUpdate, ctx) {
2648
+ // Guard: circular delegation
2649
+ if (ctx.model?.baseUrl === "claude-bridge") {
2650
+ debug("askClaude: blocked circular delegation (active provider is claude-bridge)");
2651
+ return {
2652
+ content: [{ type: "text" as const, text: "Error: AskClaude cannot be used when the active provider is claude-bridge — you're already running through Claude Code." }],
2653
+ details: { error: true },
2654
+ };
2655
+ }
2656
+
2657
+ const mode = resolveAskClaudeMode(params.mode, askDefaults);
2658
+ const isolated = params.isolated ?? askDefaults.isolated;
2659
+ const toolCalls = new Map<string, ToolCallState>();
2660
+ const start = Date.now();
2661
+
2662
+ const progressInterval = setInterval(() => {
2663
+ const elapsed = ((Date.now() - start) / 1000).toFixed(0);
2664
+ const summary = buildActionSummary(toolCalls);
2665
+ const status = summary ? `${elapsed}s — ${summary}` : `${elapsed}s — working...`;
2666
+ onUpdate?.({
2667
+ content: [{ type: "text", text: status }],
2668
+ details: { prompt: params.prompt, executionTime: Date.now() - start },
2669
+ });
2670
+ }, 1000);
2671
+
2672
+ try {
2673
+ const result = await promptAndWait(params.prompt, mode, toolCalls, signal, {
2674
+ systemPrompt: ctx.getSystemPrompt(),
2675
+ appendSkills: askConf?.appendSkills,
2676
+ model: params.model,
2677
+ thinking: params.thinking,
2678
+ isolated,
2679
+ context: isolated ? undefined : buildSessionContext(ctx.sessionManager.getBranch()).messages as Context["messages"],
2680
+ piSessionId: ctx.sessionManager.getSessionId(),
2681
+ });
2682
+ clearInterval(progressInterval);
2683
+ onUpdate?.({ content: [{ type: "text", text: "" }], details: {} });
2684
+ const executionTime = Date.now() - start;
2685
+ const actions = buildActionSummary(toolCalls);
2686
+
2687
+ const text = actions
2688
+ ? `${result.responseText}\n\n[Claude Code actions: ${actions}]`
2689
+ : result.responseText;
2690
+ return {
2691
+ content: [{ type: "text" as const, text }],
2692
+ details: { prompt: params.prompt, executionTime, actions },
2693
+ };
2694
+ } catch (err) {
2695
+ clearInterval(progressInterval);
2696
+ debug(`askClaude error: mode=${mode}, model=${params.model ?? "default"}, isolated=${isolated}, elapsed=${((Date.now() - start) / 1000).toFixed(1)}s, error=`, err);
2697
+ const msg = errorMessage(err);
2698
+ return {
2699
+ content: [{ type: "text" as const, text: `Error: ${msg}` }],
2700
+ details: { prompt: params.prompt, executionTime: Date.now() - start, error: true },
2701
+ };
2702
+ }
2703
+ },
2704
+ });
2705
+ }
2706
+ }