@cad0p/pi-tree-navigator 0.1.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,767 @@
1
+ /**
2
+ * cache-summary — cache-preserving branch-summary request construction.
3
+ *
4
+ * ## Why this exists
5
+ *
6
+ * `rewind` collapses a conversation segment by calling pi's upstream
7
+ * `generateBranchSummary`, which builds a *cold, standalone* request: a
8
+ * generic summarization system prompt, the conversation serialized to a
9
+ * text blob, no tools, `cacheRetention: "none"`, a fresh session id, and
10
+ * no reasoning forwarding. The live turns the summary covers were just
11
+ * prompt-cache-served, so the summary re-bills the entire branch input
12
+ * (measured ~77k tokens cold vs a few hundred warm). The fix exists in
13
+ * the upstream fork (`cad0p/pi` PR #3) but cannot be imported: pi's
14
+ * extension loader aliases `@earendil-works/*` to the host process's own
15
+ * modules, so an extension can never ship a patched coding-agent.
16
+ *
17
+ * ## How
18
+ *
19
+ * `index.ts` already injects a `streamFn` into `generateBranchSummary`
20
+ * (for custom-provider routing). That seam receives the fully built
21
+ * `(model, context, options)` triple *after* upstream applied its cold
22
+ * choices — including `completeSummarization`'s forced
23
+ * `cacheRetention: "none"` + fresh `sessionId` — and before the wire
24
+ * call. `createCachePreservingStreamFn` replaces that triple with the
25
+ * live request shape: the session's own system prompt, tool array, and
26
+ * session id, the conversation as structured `Message`s (so the bytes
27
+ * prefix-match the live turns), and the same cache/reasoning params live
28
+ * turns send.
29
+ *
30
+ * ## Residual risks (see README "Limitations")
31
+ *
32
+ * - Every param this module does not mirror is a silent cache miss:
33
+ * the summary still runs, just cold (and a cold *structured* request
34
+ * can bill more than branch-only evidence). The extension measures the
35
+ * summary response with pi's own miss detector and, when it clears the
36
+ * display floor, records the notice string in `details.summaryCache`
37
+ * (gated by `showCacheMissNotices`); `index.ts`'s `renderResult` renders
38
+ * it as a TUI transcript line. Neither surface reaches the model.
39
+ * - The request depends on plain (non-`#`-private) pi internals for
40
+ * `systemPrompt` / `tools`; `index.ts` falls back to the cold request
41
+ * when any live input is unavailable.
42
+ */
43
+
44
+ import type {
45
+ AgentTool,
46
+ StreamFn,
47
+ ThinkingLevel,
48
+ } from "@earendil-works/pi-agent-core";
49
+ import {
50
+ convertToLlm,
51
+ estimateTokens,
52
+ type SessionEntry,
53
+ sessionEntryToContextMessages,
54
+ } from "@earendil-works/pi-coding-agent";
55
+
56
+ // ---------------------------------------------------------------------------
57
+ // Wire-message type
58
+ //
59
+ // `Message` / `Usage` live only in `@earendil-works/pi-ai`, which is NOT a
60
+ // peer dependency of this package (it's a transitive dep of the pi packages
61
+ // themselves). Importing it would either add an undeclared dependency or
62
+ // resolve to a duplicated instance under a different node_modules root. Both
63
+ // types are fully structural, so derive them:
64
+ // - `convertToLlm`'s return element type IS the wire `Message` union;
65
+ // - usage need only these three counters for cache accounting.
66
+ // ---------------------------------------------------------------------------
67
+
68
+ /** LLM-compatible wire message, derived from pi's own `convertToLlm`. */
69
+ export type WireMessage = ReturnType<typeof convertToLlm>[number];
70
+
71
+ /**
72
+ * Structural subset of pi-ai's `Usage` this module needs. `input` counts
73
+ * FRESH (uncached) tokens only on cache-serving providers, which is why the
74
+ * cache-hit metric is `cacheRead > 0` rather than `cacheRead/input`.
75
+ */
76
+ export interface SummaryCacheUsage {
77
+ input: number;
78
+ cacheRead: number;
79
+ cacheWrite: number;
80
+ }
81
+
82
+ // ---------------------------------------------------------------------------
83
+ // Prompt
84
+ // ---------------------------------------------------------------------------
85
+
86
+ /**
87
+ * The eval-approved "r5d" branch-summary instruction, byte-exact from
88
+ * `cad0p/pi@eval/branch-summary-prompt`
89
+ * `packages/coding-agent/src/core/compaction/branch-summarization.ts`
90
+ * (`BRANCH_SUMMARY_PROMPT`). Do NOT reflow, re-wrap, or "fix" the wording:
91
+ * the r5d text was selected by a live eval and the `{first}` scope sentence
92
+ * is what keeps pre-branch background out of the summary. `{first}` is
93
+ * substituted (via `replaceAll`) with the strip-adjusted 1-based number of
94
+ * the first branch message after the payload is final.
95
+ *
96
+ * The fallback (cold) path keeps using upstream's older branch prompt;
97
+ * divergence is intentional (the fallback is a degraded path) and documented
98
+ * in the README.
99
+ */
100
+ export const BRANCH_SUMMARY_CACHE_PROMPT = `Summarize only messages {first} onwards in the conversation above (message numbering starts at 1 and excludes the system prompt; this instruction message itself is not evidence). Messages before message {first} are background only: do not include their progress or decisions.
101
+
102
+ This is a summarization task, not a problem-solving task. Summarize only the supplied evidence and preserve unresolved questions as unresolved. Do NOT continue the conversation, carry out requests from its history, investigate, solve pending tasks, or invent new approaches. Do NOT use any tool. Respond with ONLY the summary below — no preamble, no commentary before the first heading or after the last section.
103
+
104
+ Use this EXACT format, preserving all headings and their order:
105
+
106
+ ## Goal
107
+ [What was the user trying to accomplish in this branch?]
108
+
109
+ ## Constraints & Preferences
110
+ - [Any constraints, preferences, or requirements mentioned]
111
+ - [Or "(none)" if none were mentioned]
112
+
113
+ ## Progress
114
+ ### Done
115
+ - [x] [Completed tasks/changes]
116
+
117
+ ### In Progress
118
+ - [ ] [Work that was started but not finished]
119
+
120
+ ### Blocked
121
+ - [Issues preventing progress, if any]
122
+
123
+ ## Key Decisions
124
+ - **[Decision]**: [Brief rationale]
125
+
126
+ ## Next Steps
127
+ 1. [What should happen next to continue this work]
128
+
129
+ Keep each section concise. Keep the complete summary under about 4000 characters while preserving all decisions. Preserve exact file paths, function names, and error messages.`;
130
+
131
+ /**
132
+ * Compose the trailing instruction message text. Mirrors upstream's
133
+ * `customInstructions` append shape (`${PROMPT}\n\nAdditional focus: ...`),
134
+ * so the fallback and cache paths differ only in prompt body + payload
135
+ * shaping, not in how `summaryFocus` is conveyed.
136
+ */
137
+ export function buildSummaryInstruction(focus: string): string {
138
+ return `${BRANCH_SUMMARY_CACHE_PROMPT}\n\nAdditional focus: ${focus}`;
139
+ }
140
+
141
+ // ---------------------------------------------------------------------------
142
+ // Payload construction
143
+ // ---------------------------------------------------------------------------
144
+
145
+ /**
146
+ * Drop `toolResult` messages whose matching assistant `toolCall` is not in
147
+ * the payload (a branch cut between a call and its result; compaction
148
+ * boundaries can also split them). Providers reject result blocks that
149
+ * reference calls outside the request, so the structured summary request
150
+ * must strip them. Port of the fork's `stripBoundaryOrphanToolResults`:
151
+ * preserves order, never mutates, and preserves element identity (callers
152
+ * use identity to count how many stripped messages preceded the branch).
153
+ */
154
+ export function stripBoundaryOrphanToolResults(
155
+ messages: WireMessage[],
156
+ ): WireMessage[] {
157
+ const callIds = new Set<string>();
158
+ for (const message of messages) {
159
+ if (message.role !== "assistant") continue;
160
+ for (const block of message.content) {
161
+ if (block.type === "toolCall") callIds.add(block.id);
162
+ }
163
+ }
164
+ return messages.filter((message) => {
165
+ if (message.role !== "toolResult") return true;
166
+ return callIds.has(message.toolCallId);
167
+ });
168
+ }
169
+
170
+ /**
171
+ * Newest index of an assistant entry whose content carries a `toolCall` with
172
+ * `inFlightToolCallId`, or -1 when none exists. Searches from the end because
173
+ * sequential execution can leave sibling `toolResult` entries after the
174
+ * assistant that owns the in-flight call.
175
+ */
176
+ function findInFlightAssistantIndex(
177
+ entries: SessionEntry[],
178
+ inFlightToolCallId: string,
179
+ ): number {
180
+ for (let i = entries.length - 1; i >= 0; i--) {
181
+ const entry = entries[i];
182
+ if (
183
+ entry.type === "message" &&
184
+ entry.message.role === "assistant" &&
185
+ Array.isArray(entry.message.content) &&
186
+ entry.message.content.some(
187
+ (block) => block.type === "toolCall" && block.id === inFlightToolCallId,
188
+ )
189
+ ) {
190
+ return i;
191
+ }
192
+ }
193
+ return -1;
194
+ }
195
+
196
+ export interface BuildLiveSummaryArgs {
197
+ /**
198
+ * The live projection of the active branch (`sessionManager
199
+ * .buildContextEntries()`), i.e. the exact entries the live turns send.
200
+ * Using the projection (rather than reconstructing prefix+branch) is what
201
+ * makes the resulting payload byte-identical to the previous live
202
+ * request's message list.
203
+ */
204
+ contextEntries: SessionEntry[];
205
+ /**
206
+ * Ids of the entries being collapsed (`collectEntriesForBranchSummary`).
207
+ * Messages from other entries are pre-branch background: sent for cache
208
+ * prefix matching only, excluded from the summary via the `{first}` scope
209
+ * sentence.
210
+ */
211
+ branchEntryIds: Set<string>;
212
+ /**
213
+ * Id of the tool call whose assistant message triggered this rewind. That
214
+ * assistant entry was never part of any cached live prefix (it is the
215
+ * response being streamed), and an unpaired `tool_use` immediately
216
+ * followed by a user message is rejected by Anthropic. The newest retained
217
+ * assistant entry carrying a `toolCall` with this id is removed by index —
218
+ * NOT merely from the tail: `navigate_tree` runs `executionMode:
219
+ * "sequential"`, so pi-agent-core appends each sibling `toolResult` before
220
+ * the next call executes and a sibling result can follow this assistant.
221
+ * Dropping the assistant makes the retained history byte-identical to the
222
+ * previous live request.
223
+ */
224
+ inFlightToolCallId: string;
225
+ /** Context window minus the response reserve (upstream default 16384). */
226
+ tokenBudget: number;
227
+ /** `summaryFocus` from the tool call. */
228
+ focus: string;
229
+ }
230
+
231
+ export interface LiveSummaryMessages {
232
+ /** Structured history (stripped) + the trailing instruction message. */
233
+ messages: WireMessage[];
234
+ /**
235
+ * 1-based number of the first branch message in `messages` (numbering
236
+ * excludes the system prompt; the instruction itself is not evidence).
237
+ * Substituted into `{first}`.
238
+ */
239
+ first: number;
240
+ /**
241
+ * False when no retained entry belongs to the collapsed branch: a
242
+ * labels-only segment, the newest message alone exceeding the budget, or
243
+ * the branch start being dropped by compaction. `first` is then 1 — every
244
+ * retained message is background and gets summarized. The index.ts call
245
+ * site now treats `false` as a real fallback (reason
246
+ * `"branch-start-not-retained"`), so a live-prefix request always carries
247
+ * `true`; the flag is kept in `details.summaryCache` for diagnostics. A
248
+ * retained survivor by definition implies a hit, so there is no clamp
249
+ * step.
250
+ */
251
+ branchStartRetained: boolean;
252
+ }
253
+
254
+ /**
255
+ * Build the cache-preserving summary payload.
256
+ *
257
+ * Walk the live projection newest→oldest, dropping oldest entries first when
258
+ * over budget (a truncated request no longer prefix-matches live turns; the
259
+ * system prompt + tools still do). Summary entries (`compaction` /
260
+ * `branch_summary`) get upstream's 0.9-slack retry so they survive
261
+ * truncation when they are the thing that must not be lost. Then strip
262
+ * boundary-orphan tool results and adjust `{first}` by however many stripped
263
+ * messages preceded the branch start, so the instruction's numbering always
264
+ * matches the payload actually sent.
265
+ */
266
+ export function buildLiveSummaryMessages(
267
+ args: BuildLiveSummaryArgs,
268
+ ): LiveSummaryMessages {
269
+ const {
270
+ contextEntries,
271
+ branchEntryIds,
272
+ inFlightToolCallId,
273
+ tokenBudget,
274
+ focus,
275
+ } = args;
276
+
277
+ // --- in-flight assistant exclusion (must happen before anything else) ---
278
+ // Search the WHOLE retained array, not just the tail. `navigate_tree`
279
+ // declares `executionMode: "sequential"`, so pi-agent-core runs the batch
280
+ // through `executeToolCallsSequential`: calls execute in order and each
281
+ // `toolResult` is appended before the next call executes. When a sibling
282
+ // tool call precedes the rewind call in the same assistant turn, the last
283
+ // session entry is that sibling's `toolResult` — not the assistant — so a
284
+ // tail-only check would leave the assistant (and its unpaired `tool_use`)
285
+ // in the payload and Anthropic would reject the summary request. Remove the
286
+ // assistant at its index; the sibling `toolResult`s that follow then have
287
+ // no matching call and are dropped by `stripBoundaryOrphanToolResults`
288
+ // below (single removal path — do not add a second one here).
289
+ const retained = contextEntries.slice();
290
+ const excludedAt = findInFlightAssistantIndex(retained, inFlightToolCallId);
291
+ if (excludedAt >= 0) retained.splice(excludedAt, 1);
292
+
293
+ // --- newest→oldest walk with the upstream token budget ---
294
+ const evidence: WireMessage[] = [];
295
+ const inBranch: boolean[] = [];
296
+ let totalTokens = 0;
297
+ for (let i = retained.length - 1; i >= 0; i--) {
298
+ const entry = retained[i];
299
+ const entryMessages = sessionEntryToContextMessages(entry);
300
+ let overBudget = false;
301
+ for (let j = entryMessages.length - 1; j >= 0; j--) {
302
+ const agentMessage = entryMessages[j];
303
+ // convertToLlm is a pure per-message map+filter (verified against
304
+ // 0.84.2 `messages.js`), so converting one message at a time keeps
305
+ // the branch/background flag exact without diverging from what the
306
+ // live loop produces for the same AgentMessage.
307
+ const wire = convertToLlm([agentMessage]);
308
+ if (wire.length === 0) continue;
309
+ const tokens = estimateTokens(agentMessage);
310
+ const fits = tokenBudget <= 0 || totalTokens + tokens <= tokenBudget;
311
+ if (!fits) {
312
+ // Summary entries are load-bearing context: upstream retries them
313
+ // when under 90% of budget. Mirror that before giving up.
314
+ if (
315
+ (entry.type === "compaction" || entry.type === "branch_summary") &&
316
+ totalTokens < tokenBudget * 0.9
317
+ ) {
318
+ evidence.unshift(...wire);
319
+ inBranch.unshift(...wire.map(() => branchEntryIds.has(entry.id)));
320
+ totalTokens += tokens;
321
+ }
322
+ overBudget = true;
323
+ break;
324
+ }
325
+ evidence.unshift(...wire);
326
+ inBranch.unshift(...wire.map(() => branchEntryIds.has(entry.id)));
327
+ totalTokens += tokens;
328
+ }
329
+ if (overBudget) break;
330
+ }
331
+
332
+ const firstBranchIdx = inBranch.indexOf(true);
333
+ const branchStartRetained = firstBranchIdx >= 0;
334
+ // Pre-truncation counting would misnumber; count only retained messages
335
+ // before the branch start.
336
+ const firstRaw = branchStartRetained ? 1 + firstBranchIdx : 1;
337
+
338
+ const stripped = stripBoundaryOrphanToolResults(evidence);
339
+ const removedBeforeFirst = evidence
340
+ .slice(0, firstRaw - 1)
341
+ .filter((message) => !stripped.includes(message)).length;
342
+ const first = Math.max(1, firstRaw - removedBeforeFirst);
343
+
344
+ const instruction: WireMessage = {
345
+ role: "user",
346
+ content: [
347
+ {
348
+ type: "text",
349
+ text: buildSummaryInstruction(focus).replaceAll(
350
+ "{first}",
351
+ String(first),
352
+ ),
353
+ },
354
+ ],
355
+ timestamp: Date.now(),
356
+ };
357
+
358
+ return { messages: [...stripped, instruction], first, branchStartRetained };
359
+ }
360
+
361
+ // ---------------------------------------------------------------------------
362
+ // Cache retention
363
+ // ---------------------------------------------------------------------------
364
+
365
+ /**
366
+ * Mirror pi-ai's `resolveCacheRetention`: explicit env wins key-by-key, with
367
+ * `process.env` as the fallback. Live turns default to `"short"`; the summary
368
+ * must match or its single-use trailer breakpoints land differently and the
369
+ * provider keys on a different retention class. Hence: resolved, never
370
+ * hardcoded.
371
+ */
372
+ export function resolveSummaryCacheRetention(
373
+ env?: Record<string, string | undefined>,
374
+ ): "short" | "long" {
375
+ const value = env?.PI_CACHE_RETENTION ?? process.env.PI_CACHE_RETENTION;
376
+ return value === "long" ? "long" : "short";
377
+ }
378
+
379
+ // ---------------------------------------------------------------------------
380
+ // Measurement + notice
381
+ // ---------------------------------------------------------------------------
382
+ // Branch-summary cache-miss detection (port of pi's `cache-stats.ts`)
383
+ //
384
+ // Byte-for-byte behavioral port of `detectBranchSummaryCacheMiss` from
385
+ // `cad0p/pi@eval/branch-summary-prompt`
386
+ // `packages/coding-agent/src/core/cache-stats.ts`. pi 0.84.2 does not export
387
+ // this symbol, and its public `cache-stats` surface is an older revision, so
388
+ // the extension carries its own copy. The display half (`CacheMiss` copy +
389
+ // thresholds) mirrors `interactive-mode.ts`'s `addCacheMissNotice`.
390
+ // ---------------------------------------------------------------------------
391
+
392
+ /**
393
+ * Prompt-cache TTL: idle gaps longer than this are worth mentioning as the
394
+ * likely cause of a miss. Anthropic's default cache TTL is 5 minutes.
395
+ */
396
+ export const CACHE_TTL_MS = 5 * 60 * 1000;
397
+
398
+ /** Per-turn misses at or below this are cache breakpoint granularity noise. */
399
+ const NOISE_FLOOR_TOKENS = 1024;
400
+
401
+ /** Display floor: only misses at/above this many tokens warn. */
402
+ export const CACHE_MISS_DISPLAY_TOKENS = 20_000;
403
+ /** Display floor: only misses at/above this many dollars warn. */
404
+ export const CACHE_MISS_DISPLAY_COST = 0.1;
405
+
406
+ /** A counted cache miss on the just-completed branch-summary request. */
407
+ export interface BranchSummaryCacheMiss {
408
+ /** Prompt tokens in the previous request's prompt but not read from cache. */
409
+ missedTokens: number;
410
+ /** Extra dollars paid vs. a full cache hit; 0 when pricing is unknown. */
411
+ missedCost: number;
412
+ /** Milliseconds since the previous request (which last refreshed the cache). */
413
+ idleMs: number;
414
+ /** True when the model changed relative to the previous request. */
415
+ modelChanged: boolean;
416
+ }
417
+
418
+ /** Minimal pricing lookup; cost is $/million tokens. Satisfied by ModelRegistry. */
419
+ export interface ModelPriceSource {
420
+ getModel(
421
+ provider: string,
422
+ modelId: string,
423
+ ): { cost?: { cacheRead?: number } } | undefined;
424
+ }
425
+
426
+ /** The last request seen by the scan; everything in its prompt should be cached. */
427
+ interface PreviousRequest {
428
+ promptTokens: number;
429
+ modelKey: string;
430
+ timestamp: number;
431
+ /**
432
+ * Sticky: some earlier request in this session reported cache activity.
433
+ * Session-scoped (never reset by context boundaries): provider cache
434
+ * capability does not change across compactions, while the prompt baseline
435
+ * legitimately does. Distinguishes a total miss on a cache-read-only
436
+ * provider from a provider that never reports caching at all.
437
+ */
438
+ reportedCache: boolean;
439
+ }
440
+
441
+ interface MissUsage {
442
+ input: number;
443
+ cacheRead: number;
444
+ cacheWrite: number;
445
+ cost?: { input?: number; cacheRead?: number; cacheWrite?: number };
446
+ }
447
+
448
+ interface MissAssistantMessage {
449
+ provider?: string;
450
+ model?: string;
451
+ usage: MissUsage;
452
+ timestamp: number;
453
+ }
454
+
455
+ function modelKey(provider: string, model: string): string {
456
+ return `${provider}/${model}`;
457
+ }
458
+
459
+ function detectMiss(
460
+ prev: PreviousRequest | undefined,
461
+ message: MissAssistantMessage,
462
+ models: ModelPriceSource,
463
+ ): BranchSummaryCacheMiss | undefined {
464
+ const usage = message.usage;
465
+ const promptTokens = usage.input + usage.cacheRead + usage.cacheWrite;
466
+ // A zero-cache turn only counts when cache activity was reported before:
467
+ // on cache-read-only providers that is a total miss, while on providers
468
+ // that never report caching it means nothing.
469
+ if (
470
+ !prev ||
471
+ promptTokens <= 0 ||
472
+ (usage.cacheRead + usage.cacheWrite === 0 && !prev.reportedCache)
473
+ ) {
474
+ return undefined;
475
+ }
476
+
477
+ const missedTokens =
478
+ Math.min(prev.promptTokens, promptTokens) - usage.cacheRead;
479
+ if (missedTokens <= NOISE_FLOOR_TOKENS) return undefined;
480
+
481
+ // Extra cost = missed tokens billed at the actual paid rate (input/cacheWrite,
482
+ // incl. write premium) instead of the cache-read rate. Missed tokens can only
483
+ // land in the input or cacheWrite buckets, so the paid rate comes straight
484
+ // from this message's own cost breakdown.
485
+ const cost = usage.cost ?? {};
486
+ const paidTokens = usage.input + usage.cacheWrite;
487
+ const paidPerToken =
488
+ paidTokens > 0
489
+ ? ((cost.input ?? 0) + (cost.cacheWrite ?? 0)) / paidTokens
490
+ : 0;
491
+ const readPerToken =
492
+ usage.cacheRead > 0
493
+ ? (cost.cacheRead ?? 0) / usage.cacheRead
494
+ : (models.getModel(message.provider ?? "", message.model ?? "")?.cost
495
+ ?.cacheRead ?? 0) / 1_000_000;
496
+
497
+ return {
498
+ missedTokens,
499
+ missedCost: missedTokens * Math.max(0, paidPerToken - readPerToken),
500
+ idleMs: Math.max(0, message.timestamp - prev.timestamp),
501
+ modelChanged:
502
+ modelKey(message.provider ?? "", message.model ?? "") !== prev.modelKey,
503
+ };
504
+ }
505
+
506
+ function asPreviousRequest(
507
+ message: MissAssistantMessage,
508
+ reportedCache: boolean,
509
+ ): PreviousRequest | undefined {
510
+ const usage = message.usage;
511
+ const promptTokens = usage.input + usage.cacheRead + usage.cacheWrite;
512
+ if (promptTokens <= 0) return undefined;
513
+ return {
514
+ promptTokens,
515
+ modelKey: modelKey(message.provider ?? "", message.model ?? ""),
516
+ timestamp: message.timestamp,
517
+ reportedCache: reportedCache || usage.cacheRead + usage.cacheWrite > 0,
518
+ };
519
+ }
520
+
521
+ function scan(
522
+ entries: SessionEntry[],
523
+ keepBaselineAcrossBranchSummary: boolean,
524
+ ): PreviousRequest | undefined {
525
+ let prev: PreviousRequest | undefined;
526
+ // Session-level cache capability: any measured cache activity (assistant
527
+ // turns AND summary requests) proves the provider reports caching, so a
528
+ // later zero-read is a real miss even across a context boundary.
529
+ let everReportedCache = false;
530
+ for (const entry of entries) {
531
+ if (
532
+ entry.type === "compaction" ||
533
+ (entry.type === "branch_summary" && !keepBaselineAcrossBranchSummary)
534
+ ) {
535
+ // The context legitimately changed; the next turn's prompt is new content,
536
+ // not re-billed content. Model switches are NOT exempt: they re-bill the
537
+ // full prompt and should be counted.
538
+ if (entry.usage && entry.usage.cacheRead + entry.usage.cacheWrite > 0) {
539
+ everReportedCache = true;
540
+ }
541
+ prev = undefined;
542
+ continue;
543
+ }
544
+ if (entry.type === "branch_summary") {
545
+ // Probe-only path (keepBaselineAcrossBranchSummary): the summary request
546
+ // reuses the live prompt-cache prefix, so the parent baseline survives.
547
+ // Fold cache activity into the session capability flag but never reset
548
+ // prev and never become prev (only assistant messages do).
549
+ if (entry.usage && entry.usage.cacheRead + entry.usage.cacheWrite > 0) {
550
+ everReportedCache = true;
551
+ }
552
+ continue;
553
+ }
554
+ if (entry.type === "message" && entry.message.role === "assistant") {
555
+ const message = entry.message as unknown as MissAssistantMessage;
556
+ if (
557
+ message.usage &&
558
+ message.usage.cacheRead + message.usage.cacheWrite > 0
559
+ ) {
560
+ everReportedCache = true;
561
+ }
562
+ prev =
563
+ asPreviousRequest(
564
+ message,
565
+ (prev?.reportedCache ?? false) || everReportedCache,
566
+ ) ?? prev;
567
+ }
568
+ }
569
+ return prev;
570
+ }
571
+
572
+ /**
573
+ * Detect a cache miss on a just-completed branch-summary response from its
574
+ * measured usage. `entries` is the session BEFORE the summary entry is
575
+ * appended. Live-turn accounting counts model switches as misses; summary
576
+ * probes suppress them instead — a cold summary right after a switch is
577
+ * expected re-billing, not an actionable miss.
578
+ */
579
+ export function detectBranchSummaryCacheMiss(
580
+ entries: SessionEntry[],
581
+ responseUsage: MissUsage,
582
+ provider: string,
583
+ model: string,
584
+ timestamp: number,
585
+ models: ModelPriceSource,
586
+ ): BranchSummaryCacheMiss | undefined {
587
+ const prev = scan(entries, true);
588
+ if (prev && prev.modelKey !== modelKey(provider, model)) return undefined;
589
+ return detectMiss(
590
+ prev,
591
+ { provider, model, usage: responseUsage, timestamp },
592
+ models,
593
+ );
594
+ }
595
+
596
+ /**
597
+ * Compact token formatting, byte-for-byte port of the fork's
598
+ * `interactive-mode/components/footer.ts` `formatTokens` (the same helper the
599
+ * miss notice uses): 999 → "999", 9999 → "10.0k", 20000 → "20k".
600
+ */
601
+ function formatTokens(count: number): string {
602
+ if (count < 1000) return count.toString();
603
+ if (count < 10_000) return `${(count / 1000).toFixed(1)}k`;
604
+ if (count < 1_000_000) return `${Math.round(count / 1000)}k`;
605
+ if (count < 10_000_000) return `${(count / 1_000_000).toFixed(1)}M`;
606
+ return `${Math.round(count / 1_000_000)}M`;
607
+ }
608
+
609
+ /**
610
+ * TUI warning copy for a counted branch-summary cache miss, or `null` when
611
+ * the miss is below the display floor. Mirrors the fork's `addCacheMissNotice`
612
+ * thresholds and label selection verbatim. `index.ts` stores the non-null
613
+ * result in `details.summaryCache.notice` and renders it as a transcript line
614
+ * in the tool's `renderResult`; it is NEVER appended to the rewind tool-result
615
+ * content the model sees.
616
+ */
617
+ export function formatBranchSummaryCacheMissNotice(
618
+ miss: BranchSummaryCacheMiss,
619
+ ): string | null {
620
+ if (
621
+ miss.missedTokens < CACHE_MISS_DISPLAY_TOKENS &&
622
+ miss.missedCost < CACHE_MISS_DISPLAY_COST
623
+ ) {
624
+ return null;
625
+ }
626
+ const cost =
627
+ miss.missedCost >= 0.01 ? ` (~$${miss.missedCost.toFixed(2)})` : "";
628
+ const reBilled = `${formatTokens(miss.missedTokens)} tokens re-billed${cost}`;
629
+ let label = "Cache miss";
630
+ if (miss.modelChanged) {
631
+ label = "Cache miss after model switch";
632
+ } else if (miss.idleMs >= CACHE_TTL_MS) {
633
+ label = `Cache miss after ${Math.round(miss.idleMs / 60_000)}m idle`;
634
+ }
635
+ return `${label}: ${reBilled}`;
636
+ }
637
+
638
+ // ---------------------------------------------------------------------------
639
+
640
+ /**
641
+ * Cache accounting for the summary request. `usage.input` counts FRESH
642
+ * (uncached) tokens only on cache-serving providers, which is why the
643
+ * cache-hit metric is `cacheRead > 0` rather than `cacheRead/input`. This is
644
+ * retained purely as the machine-readable `details.summaryCache` surface;
645
+ * the TUI warning is derived from the miss detector above.
646
+ */
647
+ export interface SummaryCacheStats {
648
+ /** Fresh (uncached) input tokens billed for this request. */
649
+ cacheRead: number;
650
+ fresh: number;
651
+ cacheWrite: number;
652
+ hit: boolean;
653
+ }
654
+
655
+ /**
656
+ * Derive cache accounting from provider usage. On cache-serving providers
657
+ * `usage.input` counts FRESH tokens only (a cached read can exceed it), so
658
+ * the hit predicate is `cacheRead > 0`, never a ratio.
659
+ */
660
+ export function measureSummaryCache(
661
+ usage: SummaryCacheUsage,
662
+ ): SummaryCacheStats {
663
+ return {
664
+ cacheRead: usage.cacheRead,
665
+ fresh: usage.input,
666
+ cacheWrite: usage.cacheWrite,
667
+ hit: usage.cacheRead > 0,
668
+ };
669
+ }
670
+
671
+ // ---------------------------------------------------------------------------
672
+ // StreamFn wrapper
673
+ // ---------------------------------------------------------------------------
674
+
675
+ /**
676
+ * The exact request live turns send, rebuilt for the summarization call.
677
+ * `index.ts` assembles this from public pi APIs + two plain reflected fields
678
+ * and passes it to `createCachePreservingStreamFn`.
679
+ */
680
+ export interface CacheRequest {
681
+ /** Live system prompt + live tool array + structured history + trailer. */
682
+ context: {
683
+ systemPrompt: string;
684
+ messages: WireMessage[];
685
+ tools: AgentTool[];
686
+ };
687
+ /**
688
+ * Resolved (never hardcoded) cache retention. Explicit `"short"` matters:
689
+ * reads key on sessionId + prefix bytes, so `"none"` (upstream's forced
690
+ * value) would send no session id and could never hit.
691
+ */
692
+ cacheRetention: "short" | "long";
693
+ /** Live session id: joins the session's cache namespace. */
694
+ sessionId?: string;
695
+ /**
696
+ * Live thinking level, forwarded as `reasoning`. Providers that key the
697
+ * cache on reasoning effort will miss a request that omits it even when
698
+ * the prefix bytes match — this is a product requirement, not probe
699
+ * hygiene. `"off"` is omitted, mirroring the live loop.
700
+ */
701
+ reasoning?: ThinkingLevel;
702
+ /**
703
+ * Live thinking token budgets (plain field on pi-agent-core's `Agent`).
704
+ * Inert on some APIs, part of Anthropic's thinking body; forwarded only
705
+ * when the host exposes it.
706
+ */
707
+ thinkingBudgets?: unknown;
708
+ }
709
+
710
+ /**
711
+ * Wrap the `streamFn` handed to `generateBranchSummary` so the request is
712
+ * rewritten to the live shape at the last possible moment (after upstream's
713
+ * `completeSummarization` forced `cacheRetention: "none"` + a fresh
714
+ * `sessionId`).
715
+ *
716
+ * `request === null` → delegate the caller's context and options untouched:
717
+ * today's cold standalone request. This is the fallback for every "live input
718
+ * unavailable" case.
719
+ *
720
+ * The `maxTokens` strip: upstream's summary caller caps output
721
+ * (`maxTokens: 2048` in 0.84.2), but live turns let pi-ai fill
722
+ * `clampMaxTokensToContext(model, liveContext, options?.maxTokens ??
723
+ * model.maxTokens)`. A caller cap therefore diverges from the live
724
+ * `max_output_tokens` and can break a gateway that keys on it. Stripping the
725
+ * cap lets the provider compute the same value live gets. Residual: when the
726
+ * context window is near-full the clamp differs by the trailer size (~1k);
727
+ * r5d bounds output length in prose instead, and a miss is flagged by the
728
+ * notice. (The recorded value differs across 0.80.2 = 2048 / 0.84.2 = 2048 /
729
+ * fork = 4096, which is exactly why we strip whatever is there rather than
730
+ * assume a constant.)
731
+ *
732
+ * Returned `used.value` flips true iff the live request was actually
733
+ * delegated, so `index.ts` can report whether the wrapper engaged (it stays
734
+ * false when the summarizer is stubbed or errors before the wire call).
735
+ */
736
+ export function createCachePreservingStreamFn(args: {
737
+ realStreamFn: StreamFn;
738
+ request: CacheRequest | null;
739
+ }): { streamFn: StreamFn; used: { value: boolean } } {
740
+ const { realStreamFn, request } = args;
741
+ const used = { value: false };
742
+ const streamFn: StreamFn = (model, coldContext, options) => {
743
+ if (request === null) {
744
+ used.value = false;
745
+ return realStreamFn(model, coldContext, options);
746
+ }
747
+ used.value = true;
748
+ const rest = { ...(options as Record<string, unknown> | undefined) };
749
+ // Strip whatever caller cap upstream set; live turns let pi-ai clamp
750
+ // model.maxTokens to the real context window.
751
+ delete rest.maxTokens;
752
+ const next = {
753
+ ...rest,
754
+ cacheRetention: request.cacheRetention,
755
+ ...(request.sessionId ? { sessionId: request.sessionId } : {}),
756
+ ...(request.reasoning && request.reasoning !== "off"
757
+ ? { reasoning: request.reasoning }
758
+ : {}),
759
+ ...(request.thinkingBudgets
760
+ ? { thinkingBudgets: request.thinkingBudgets }
761
+ : {}),
762
+ } as NonNullable<Parameters<StreamFn>[2]>;
763
+ const context = request.context as Parameters<StreamFn>[1];
764
+ return realStreamFn(model, context, next);
765
+ };
766
+ return { streamFn, used };
767
+ }