@cad0p/pi-tree-navigator 0.1.3 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +50 -3
- package/extensions/navigate-tree/cache-summary.ts +767 -0
- package/extensions/navigate-tree/index.ts +468 -21
- package/package.json +2 -1
|
@@ -0,0 +1,767 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* cache-summary — cache-preserving branch-summary request construction.
|
|
3
|
+
*
|
|
4
|
+
* ## Why this exists
|
|
5
|
+
*
|
|
6
|
+
* `rewind` collapses a conversation segment by calling pi's upstream
|
|
7
|
+
* `generateBranchSummary`, which builds a *cold, standalone* request: a
|
|
8
|
+
* generic summarization system prompt, the conversation serialized to a
|
|
9
|
+
* text blob, no tools, `cacheRetention: "none"`, a fresh session id, and
|
|
10
|
+
* no reasoning forwarding. The live turns the summary covers were just
|
|
11
|
+
* prompt-cache-served, so the summary re-bills the entire branch input
|
|
12
|
+
* (measured ~77k tokens cold vs a few hundred warm). The fix exists in
|
|
13
|
+
* the upstream fork (`cad0p/pi` PR #3) but cannot be imported: pi's
|
|
14
|
+
* extension loader aliases `@earendil-works/*` to the host process's own
|
|
15
|
+
* modules, so an extension can never ship a patched coding-agent.
|
|
16
|
+
*
|
|
17
|
+
* ## How
|
|
18
|
+
*
|
|
19
|
+
* `index.ts` already injects a `streamFn` into `generateBranchSummary`
|
|
20
|
+
* (for custom-provider routing). That seam receives the fully built
|
|
21
|
+
* `(model, context, options)` triple *after* upstream applied its cold
|
|
22
|
+
* choices — including `completeSummarization`'s forced
|
|
23
|
+
* `cacheRetention: "none"` + fresh `sessionId` — and before the wire
|
|
24
|
+
* call. `createCachePreservingStreamFn` replaces that triple with the
|
|
25
|
+
* live request shape: the session's own system prompt, tool array, and
|
|
26
|
+
* session id, the conversation as structured `Message`s (so the bytes
|
|
27
|
+
* prefix-match the live turns), and the same cache/reasoning params live
|
|
28
|
+
* turns send.
|
|
29
|
+
*
|
|
30
|
+
* ## Residual risks (see README "Limitations")
|
|
31
|
+
*
|
|
32
|
+
* - Every param this module does not mirror is a silent cache miss:
|
|
33
|
+
* the summary still runs, just cold (and a cold *structured* request
|
|
34
|
+
* can bill more than branch-only evidence). The extension measures the
|
|
35
|
+
* summary response with pi's own miss detector and, when it clears the
|
|
36
|
+
* display floor, records the notice string in `details.summaryCache`
|
|
37
|
+
* (gated by `showCacheMissNotices`); `index.ts`'s `renderResult` renders
|
|
38
|
+
* it as a TUI transcript line. Neither surface reaches the model.
|
|
39
|
+
* - The request depends on plain (non-`#`-private) pi internals for
|
|
40
|
+
* `systemPrompt` / `tools`; `index.ts` falls back to the cold request
|
|
41
|
+
* when any live input is unavailable.
|
|
42
|
+
*/
|
|
43
|
+
|
|
44
|
+
import type {
|
|
45
|
+
AgentTool,
|
|
46
|
+
StreamFn,
|
|
47
|
+
ThinkingLevel,
|
|
48
|
+
} from "@earendil-works/pi-agent-core";
|
|
49
|
+
import {
|
|
50
|
+
convertToLlm,
|
|
51
|
+
estimateTokens,
|
|
52
|
+
type SessionEntry,
|
|
53
|
+
sessionEntryToContextMessages,
|
|
54
|
+
} from "@earendil-works/pi-coding-agent";
|
|
55
|
+
|
|
56
|
+
// ---------------------------------------------------------------------------
|
|
57
|
+
// Wire-message type
|
|
58
|
+
//
|
|
59
|
+
// `Message` / `Usage` live only in `@earendil-works/pi-ai`, which is NOT a
|
|
60
|
+
// peer dependency of this package (it's a transitive dep of the pi packages
|
|
61
|
+
// themselves). Importing it would either add an undeclared dependency or
|
|
62
|
+
// resolve to a duplicated instance under a different node_modules root. Both
|
|
63
|
+
// types are fully structural, so derive them:
|
|
64
|
+
// - `convertToLlm`'s return element type IS the wire `Message` union;
|
|
65
|
+
// - usage need only these three counters for cache accounting.
|
|
66
|
+
// ---------------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
/** LLM-compatible wire message, derived from pi's own `convertToLlm`. */
|
|
69
|
+
export type WireMessage = ReturnType<typeof convertToLlm>[number];
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Structural subset of pi-ai's `Usage` this module needs. `input` counts
|
|
73
|
+
* FRESH (uncached) tokens only on cache-serving providers, which is why the
|
|
74
|
+
* cache-hit metric is `cacheRead > 0` rather than `cacheRead/input`.
|
|
75
|
+
*/
|
|
76
|
+
export interface SummaryCacheUsage {
|
|
77
|
+
input: number;
|
|
78
|
+
cacheRead: number;
|
|
79
|
+
cacheWrite: number;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// ---------------------------------------------------------------------------
|
|
83
|
+
// Prompt
|
|
84
|
+
// ---------------------------------------------------------------------------
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* The eval-approved "r5d" branch-summary instruction, byte-exact from
|
|
88
|
+
* `cad0p/pi@eval/branch-summary-prompt`
|
|
89
|
+
* `packages/coding-agent/src/core/compaction/branch-summarization.ts`
|
|
90
|
+
* (`BRANCH_SUMMARY_PROMPT`). Do NOT reflow, re-wrap, or "fix" the wording:
|
|
91
|
+
* the r5d text was selected by a live eval and the `{first}` scope sentence
|
|
92
|
+
* is what keeps pre-branch background out of the summary. `{first}` is
|
|
93
|
+
* substituted (via `replaceAll`) with the strip-adjusted 1-based number of
|
|
94
|
+
* the first branch message after the payload is final.
|
|
95
|
+
*
|
|
96
|
+
* The fallback (cold) path keeps using upstream's older branch prompt;
|
|
97
|
+
* divergence is intentional (the fallback is a degraded path) and documented
|
|
98
|
+
* in the README.
|
|
99
|
+
*/
|
|
100
|
+
export const BRANCH_SUMMARY_CACHE_PROMPT = `Summarize only messages {first} onwards in the conversation above (message numbering starts at 1 and excludes the system prompt; this instruction message itself is not evidence). Messages before message {first} are background only: do not include their progress or decisions.
|
|
101
|
+
|
|
102
|
+
This is a summarization task, not a problem-solving task. Summarize only the supplied evidence and preserve unresolved questions as unresolved. Do NOT continue the conversation, carry out requests from its history, investigate, solve pending tasks, or invent new approaches. Do NOT use any tool. Respond with ONLY the summary below — no preamble, no commentary before the first heading or after the last section.
|
|
103
|
+
|
|
104
|
+
Use this EXACT format, preserving all headings and their order:
|
|
105
|
+
|
|
106
|
+
## Goal
|
|
107
|
+
[What was the user trying to accomplish in this branch?]
|
|
108
|
+
|
|
109
|
+
## Constraints & Preferences
|
|
110
|
+
- [Any constraints, preferences, or requirements mentioned]
|
|
111
|
+
- [Or "(none)" if none were mentioned]
|
|
112
|
+
|
|
113
|
+
## Progress
|
|
114
|
+
### Done
|
|
115
|
+
- [x] [Completed tasks/changes]
|
|
116
|
+
|
|
117
|
+
### In Progress
|
|
118
|
+
- [ ] [Work that was started but not finished]
|
|
119
|
+
|
|
120
|
+
### Blocked
|
|
121
|
+
- [Issues preventing progress, if any]
|
|
122
|
+
|
|
123
|
+
## Key Decisions
|
|
124
|
+
- **[Decision]**: [Brief rationale]
|
|
125
|
+
|
|
126
|
+
## Next Steps
|
|
127
|
+
1. [What should happen next to continue this work]
|
|
128
|
+
|
|
129
|
+
Keep each section concise. Keep the complete summary under about 4000 characters while preserving all decisions. Preserve exact file paths, function names, and error messages.`;
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Compose the trailing instruction message text. Mirrors upstream's
|
|
133
|
+
* `customInstructions` append shape (`${PROMPT}\n\nAdditional focus: ...`),
|
|
134
|
+
* so the fallback and cache paths differ only in prompt body + payload
|
|
135
|
+
* shaping, not in how `summaryFocus` is conveyed.
|
|
136
|
+
*/
|
|
137
|
+
export function buildSummaryInstruction(focus: string): string {
|
|
138
|
+
return `${BRANCH_SUMMARY_CACHE_PROMPT}\n\nAdditional focus: ${focus}`;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// ---------------------------------------------------------------------------
|
|
142
|
+
// Payload construction
|
|
143
|
+
// ---------------------------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Drop `toolResult` messages whose matching assistant `toolCall` is not in
|
|
147
|
+
* the payload (a branch cut between a call and its result; compaction
|
|
148
|
+
* boundaries can also split them). Providers reject result blocks that
|
|
149
|
+
* reference calls outside the request, so the structured summary request
|
|
150
|
+
* must strip them. Port of the fork's `stripBoundaryOrphanToolResults`:
|
|
151
|
+
* preserves order, never mutates, and preserves element identity (callers
|
|
152
|
+
* use identity to count how many stripped messages preceded the branch).
|
|
153
|
+
*/
|
|
154
|
+
export function stripBoundaryOrphanToolResults(
|
|
155
|
+
messages: WireMessage[],
|
|
156
|
+
): WireMessage[] {
|
|
157
|
+
const callIds = new Set<string>();
|
|
158
|
+
for (const message of messages) {
|
|
159
|
+
if (message.role !== "assistant") continue;
|
|
160
|
+
for (const block of message.content) {
|
|
161
|
+
if (block.type === "toolCall") callIds.add(block.id);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return messages.filter((message) => {
|
|
165
|
+
if (message.role !== "toolResult") return true;
|
|
166
|
+
return callIds.has(message.toolCallId);
|
|
167
|
+
});
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Newest index of an assistant entry whose content carries a `toolCall` with
|
|
172
|
+
* `inFlightToolCallId`, or -1 when none exists. Searches from the end because
|
|
173
|
+
* sequential execution can leave sibling `toolResult` entries after the
|
|
174
|
+
* assistant that owns the in-flight call.
|
|
175
|
+
*/
|
|
176
|
+
function findInFlightAssistantIndex(
|
|
177
|
+
entries: SessionEntry[],
|
|
178
|
+
inFlightToolCallId: string,
|
|
179
|
+
): number {
|
|
180
|
+
for (let i = entries.length - 1; i >= 0; i--) {
|
|
181
|
+
const entry = entries[i];
|
|
182
|
+
if (
|
|
183
|
+
entry.type === "message" &&
|
|
184
|
+
entry.message.role === "assistant" &&
|
|
185
|
+
Array.isArray(entry.message.content) &&
|
|
186
|
+
entry.message.content.some(
|
|
187
|
+
(block) => block.type === "toolCall" && block.id === inFlightToolCallId,
|
|
188
|
+
)
|
|
189
|
+
) {
|
|
190
|
+
return i;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
return -1;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
export interface BuildLiveSummaryArgs {
|
|
197
|
+
/**
|
|
198
|
+
* The live projection of the active branch (`sessionManager
|
|
199
|
+
* .buildContextEntries()`), i.e. the exact entries the live turns send.
|
|
200
|
+
* Using the projection (rather than reconstructing prefix+branch) is what
|
|
201
|
+
* makes the resulting payload byte-identical to the previous live
|
|
202
|
+
* request's message list.
|
|
203
|
+
*/
|
|
204
|
+
contextEntries: SessionEntry[];
|
|
205
|
+
/**
|
|
206
|
+
* Ids of the entries being collapsed (`collectEntriesForBranchSummary`).
|
|
207
|
+
* Messages from other entries are pre-branch background: sent for cache
|
|
208
|
+
* prefix matching only, excluded from the summary via the `{first}` scope
|
|
209
|
+
* sentence.
|
|
210
|
+
*/
|
|
211
|
+
branchEntryIds: Set<string>;
|
|
212
|
+
/**
|
|
213
|
+
* Id of the tool call whose assistant message triggered this rewind. That
|
|
214
|
+
* assistant entry was never part of any cached live prefix (it is the
|
|
215
|
+
* response being streamed), and an unpaired `tool_use` immediately
|
|
216
|
+
* followed by a user message is rejected by Anthropic. The newest retained
|
|
217
|
+
* assistant entry carrying a `toolCall` with this id is removed by index —
|
|
218
|
+
* NOT merely from the tail: `navigate_tree` runs `executionMode:
|
|
219
|
+
* "sequential"`, so pi-agent-core appends each sibling `toolResult` before
|
|
220
|
+
* the next call executes and a sibling result can follow this assistant.
|
|
221
|
+
* Dropping the assistant makes the retained history byte-identical to the
|
|
222
|
+
* previous live request.
|
|
223
|
+
*/
|
|
224
|
+
inFlightToolCallId: string;
|
|
225
|
+
/** Context window minus the response reserve (upstream default 16384). */
|
|
226
|
+
tokenBudget: number;
|
|
227
|
+
/** `summaryFocus` from the tool call. */
|
|
228
|
+
focus: string;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
export interface LiveSummaryMessages {
|
|
232
|
+
/** Structured history (stripped) + the trailing instruction message. */
|
|
233
|
+
messages: WireMessage[];
|
|
234
|
+
/**
|
|
235
|
+
* 1-based number of the first branch message in `messages` (numbering
|
|
236
|
+
* excludes the system prompt; the instruction itself is not evidence).
|
|
237
|
+
* Substituted into `{first}`.
|
|
238
|
+
*/
|
|
239
|
+
first: number;
|
|
240
|
+
/**
|
|
241
|
+
* False when no retained entry belongs to the collapsed branch: a
|
|
242
|
+
* labels-only segment, the newest message alone exceeding the budget, or
|
|
243
|
+
* the branch start being dropped by compaction. `first` is then 1 — every
|
|
244
|
+
* retained message is background and gets summarized. The index.ts call
|
|
245
|
+
* site now treats `false` as a real fallback (reason
|
|
246
|
+
* `"branch-start-not-retained"`), so a live-prefix request always carries
|
|
247
|
+
* `true`; the flag is kept in `details.summaryCache` for diagnostics. A
|
|
248
|
+
* retained survivor by definition implies a hit, so there is no clamp
|
|
249
|
+
* step.
|
|
250
|
+
*/
|
|
251
|
+
branchStartRetained: boolean;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Build the cache-preserving summary payload.
|
|
256
|
+
*
|
|
257
|
+
* Walk the live projection newest→oldest, dropping oldest entries first when
|
|
258
|
+
* over budget (a truncated request no longer prefix-matches live turns; the
|
|
259
|
+
* system prompt + tools still do). Summary entries (`compaction` /
|
|
260
|
+
* `branch_summary`) get upstream's 0.9-slack retry so they survive
|
|
261
|
+
* truncation when they are the thing that must not be lost. Then strip
|
|
262
|
+
* boundary-orphan tool results and adjust `{first}` by however many stripped
|
|
263
|
+
* messages preceded the branch start, so the instruction's numbering always
|
|
264
|
+
* matches the payload actually sent.
|
|
265
|
+
*/
|
|
266
|
+
export function buildLiveSummaryMessages(
|
|
267
|
+
args: BuildLiveSummaryArgs,
|
|
268
|
+
): LiveSummaryMessages {
|
|
269
|
+
const {
|
|
270
|
+
contextEntries,
|
|
271
|
+
branchEntryIds,
|
|
272
|
+
inFlightToolCallId,
|
|
273
|
+
tokenBudget,
|
|
274
|
+
focus,
|
|
275
|
+
} = args;
|
|
276
|
+
|
|
277
|
+
// --- in-flight assistant exclusion (must happen before anything else) ---
|
|
278
|
+
// Search the WHOLE retained array, not just the tail. `navigate_tree`
|
|
279
|
+
// declares `executionMode: "sequential"`, so pi-agent-core runs the batch
|
|
280
|
+
// through `executeToolCallsSequential`: calls execute in order and each
|
|
281
|
+
// `toolResult` is appended before the next call executes. When a sibling
|
|
282
|
+
// tool call precedes the rewind call in the same assistant turn, the last
|
|
283
|
+
// session entry is that sibling's `toolResult` — not the assistant — so a
|
|
284
|
+
// tail-only check would leave the assistant (and its unpaired `tool_use`)
|
|
285
|
+
// in the payload and Anthropic would reject the summary request. Remove the
|
|
286
|
+
// assistant at its index; the sibling `toolResult`s that follow then have
|
|
287
|
+
// no matching call and are dropped by `stripBoundaryOrphanToolResults`
|
|
288
|
+
// below (single removal path — do not add a second one here).
|
|
289
|
+
const retained = contextEntries.slice();
|
|
290
|
+
const excludedAt = findInFlightAssistantIndex(retained, inFlightToolCallId);
|
|
291
|
+
if (excludedAt >= 0) retained.splice(excludedAt, 1);
|
|
292
|
+
|
|
293
|
+
// --- newest→oldest walk with the upstream token budget ---
|
|
294
|
+
const evidence: WireMessage[] = [];
|
|
295
|
+
const inBranch: boolean[] = [];
|
|
296
|
+
let totalTokens = 0;
|
|
297
|
+
for (let i = retained.length - 1; i >= 0; i--) {
|
|
298
|
+
const entry = retained[i];
|
|
299
|
+
const entryMessages = sessionEntryToContextMessages(entry);
|
|
300
|
+
let overBudget = false;
|
|
301
|
+
for (let j = entryMessages.length - 1; j >= 0; j--) {
|
|
302
|
+
const agentMessage = entryMessages[j];
|
|
303
|
+
// convertToLlm is a pure per-message map+filter (verified against
|
|
304
|
+
// 0.84.2 `messages.js`), so converting one message at a time keeps
|
|
305
|
+
// the branch/background flag exact without diverging from what the
|
|
306
|
+
// live loop produces for the same AgentMessage.
|
|
307
|
+
const wire = convertToLlm([agentMessage]);
|
|
308
|
+
if (wire.length === 0) continue;
|
|
309
|
+
const tokens = estimateTokens(agentMessage);
|
|
310
|
+
const fits = tokenBudget <= 0 || totalTokens + tokens <= tokenBudget;
|
|
311
|
+
if (!fits) {
|
|
312
|
+
// Summary entries are load-bearing context: upstream retries them
|
|
313
|
+
// when under 90% of budget. Mirror that before giving up.
|
|
314
|
+
if (
|
|
315
|
+
(entry.type === "compaction" || entry.type === "branch_summary") &&
|
|
316
|
+
totalTokens < tokenBudget * 0.9
|
|
317
|
+
) {
|
|
318
|
+
evidence.unshift(...wire);
|
|
319
|
+
inBranch.unshift(...wire.map(() => branchEntryIds.has(entry.id)));
|
|
320
|
+
totalTokens += tokens;
|
|
321
|
+
}
|
|
322
|
+
overBudget = true;
|
|
323
|
+
break;
|
|
324
|
+
}
|
|
325
|
+
evidence.unshift(...wire);
|
|
326
|
+
inBranch.unshift(...wire.map(() => branchEntryIds.has(entry.id)));
|
|
327
|
+
totalTokens += tokens;
|
|
328
|
+
}
|
|
329
|
+
if (overBudget) break;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
const firstBranchIdx = inBranch.indexOf(true);
|
|
333
|
+
const branchStartRetained = firstBranchIdx >= 0;
|
|
334
|
+
// Pre-truncation counting would misnumber; count only retained messages
|
|
335
|
+
// before the branch start.
|
|
336
|
+
const firstRaw = branchStartRetained ? 1 + firstBranchIdx : 1;
|
|
337
|
+
|
|
338
|
+
const stripped = stripBoundaryOrphanToolResults(evidence);
|
|
339
|
+
const removedBeforeFirst = evidence
|
|
340
|
+
.slice(0, firstRaw - 1)
|
|
341
|
+
.filter((message) => !stripped.includes(message)).length;
|
|
342
|
+
const first = Math.max(1, firstRaw - removedBeforeFirst);
|
|
343
|
+
|
|
344
|
+
const instruction: WireMessage = {
|
|
345
|
+
role: "user",
|
|
346
|
+
content: [
|
|
347
|
+
{
|
|
348
|
+
type: "text",
|
|
349
|
+
text: buildSummaryInstruction(focus).replaceAll(
|
|
350
|
+
"{first}",
|
|
351
|
+
String(first),
|
|
352
|
+
),
|
|
353
|
+
},
|
|
354
|
+
],
|
|
355
|
+
timestamp: Date.now(),
|
|
356
|
+
};
|
|
357
|
+
|
|
358
|
+
return { messages: [...stripped, instruction], first, branchStartRetained };
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
// ---------------------------------------------------------------------------
|
|
362
|
+
// Cache retention
|
|
363
|
+
// ---------------------------------------------------------------------------
|
|
364
|
+
|
|
365
|
+
/**
|
|
366
|
+
* Mirror pi-ai's `resolveCacheRetention`: explicit env wins key-by-key, with
|
|
367
|
+
* `process.env` as the fallback. Live turns default to `"short"`; the summary
|
|
368
|
+
* must match or its single-use trailer breakpoints land differently and the
|
|
369
|
+
* provider keys on a different retention class. Hence: resolved, never
|
|
370
|
+
* hardcoded.
|
|
371
|
+
*/
|
|
372
|
+
export function resolveSummaryCacheRetention(
|
|
373
|
+
env?: Record<string, string | undefined>,
|
|
374
|
+
): "short" | "long" {
|
|
375
|
+
const value = env?.PI_CACHE_RETENTION ?? process.env.PI_CACHE_RETENTION;
|
|
376
|
+
return value === "long" ? "long" : "short";
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// ---------------------------------------------------------------------------
|
|
380
|
+
// Measurement + notice
|
|
381
|
+
// ---------------------------------------------------------------------------
|
|
382
|
+
// Branch-summary cache-miss detection (port of pi's `cache-stats.ts`)
|
|
383
|
+
//
|
|
384
|
+
// Byte-for-byte behavioral port of `detectBranchSummaryCacheMiss` from
|
|
385
|
+
// `cad0p/pi@eval/branch-summary-prompt`
|
|
386
|
+
// `packages/coding-agent/src/core/cache-stats.ts`. pi 0.84.2 does not export
|
|
387
|
+
// this symbol, and its public `cache-stats` surface is an older revision, so
|
|
388
|
+
// the extension carries its own copy. The display half (`CacheMiss` copy +
|
|
389
|
+
// thresholds) mirrors `interactive-mode.ts`'s `addCacheMissNotice`.
|
|
390
|
+
// ---------------------------------------------------------------------------
|
|
391
|
+
|
|
392
|
+
/**
|
|
393
|
+
* Prompt-cache TTL: idle gaps longer than this are worth mentioning as the
|
|
394
|
+
* likely cause of a miss. Anthropic's default cache TTL is 5 minutes.
|
|
395
|
+
*/
|
|
396
|
+
export const CACHE_TTL_MS = 5 * 60 * 1000;
|
|
397
|
+
|
|
398
|
+
/** Per-turn misses at or below this are cache breakpoint granularity noise. */
|
|
399
|
+
const NOISE_FLOOR_TOKENS = 1024;
|
|
400
|
+
|
|
401
|
+
/** Display floor: only misses at/above this many tokens warn. */
|
|
402
|
+
export const CACHE_MISS_DISPLAY_TOKENS = 20_000;
|
|
403
|
+
/** Display floor: only misses at/above this many dollars warn. */
|
|
404
|
+
export const CACHE_MISS_DISPLAY_COST = 0.1;
|
|
405
|
+
|
|
406
|
+
/** A counted cache miss on the just-completed branch-summary request. */
|
|
407
|
+
export interface BranchSummaryCacheMiss {
|
|
408
|
+
/** Prompt tokens in the previous request's prompt but not read from cache. */
|
|
409
|
+
missedTokens: number;
|
|
410
|
+
/** Extra dollars paid vs. a full cache hit; 0 when pricing is unknown. */
|
|
411
|
+
missedCost: number;
|
|
412
|
+
/** Milliseconds since the previous request (which last refreshed the cache). */
|
|
413
|
+
idleMs: number;
|
|
414
|
+
/** True when the model changed relative to the previous request. */
|
|
415
|
+
modelChanged: boolean;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
/** Minimal pricing lookup; cost is $/million tokens. Satisfied by ModelRegistry. */
|
|
419
|
+
export interface ModelPriceSource {
|
|
420
|
+
getModel(
|
|
421
|
+
provider: string,
|
|
422
|
+
modelId: string,
|
|
423
|
+
): { cost?: { cacheRead?: number } } | undefined;
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
/** The last request seen by the scan; everything in its prompt should be cached. */
|
|
427
|
+
interface PreviousRequest {
|
|
428
|
+
promptTokens: number;
|
|
429
|
+
modelKey: string;
|
|
430
|
+
timestamp: number;
|
|
431
|
+
/**
|
|
432
|
+
* Sticky: some earlier request in this session reported cache activity.
|
|
433
|
+
* Session-scoped (never reset by context boundaries): provider cache
|
|
434
|
+
* capability does not change across compactions, while the prompt baseline
|
|
435
|
+
* legitimately does. Distinguishes a total miss on a cache-read-only
|
|
436
|
+
* provider from a provider that never reports caching at all.
|
|
437
|
+
*/
|
|
438
|
+
reportedCache: boolean;
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
interface MissUsage {
|
|
442
|
+
input: number;
|
|
443
|
+
cacheRead: number;
|
|
444
|
+
cacheWrite: number;
|
|
445
|
+
cost?: { input?: number; cacheRead?: number; cacheWrite?: number };
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
interface MissAssistantMessage {
|
|
449
|
+
provider?: string;
|
|
450
|
+
model?: string;
|
|
451
|
+
usage: MissUsage;
|
|
452
|
+
timestamp: number;
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
function modelKey(provider: string, model: string): string {
|
|
456
|
+
return `${provider}/${model}`;
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
function detectMiss(
|
|
460
|
+
prev: PreviousRequest | undefined,
|
|
461
|
+
message: MissAssistantMessage,
|
|
462
|
+
models: ModelPriceSource,
|
|
463
|
+
): BranchSummaryCacheMiss | undefined {
|
|
464
|
+
const usage = message.usage;
|
|
465
|
+
const promptTokens = usage.input + usage.cacheRead + usage.cacheWrite;
|
|
466
|
+
// A zero-cache turn only counts when cache activity was reported before:
|
|
467
|
+
// on cache-read-only providers that is a total miss, while on providers
|
|
468
|
+
// that never report caching it means nothing.
|
|
469
|
+
if (
|
|
470
|
+
!prev ||
|
|
471
|
+
promptTokens <= 0 ||
|
|
472
|
+
(usage.cacheRead + usage.cacheWrite === 0 && !prev.reportedCache)
|
|
473
|
+
) {
|
|
474
|
+
return undefined;
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
const missedTokens =
|
|
478
|
+
Math.min(prev.promptTokens, promptTokens) - usage.cacheRead;
|
|
479
|
+
if (missedTokens <= NOISE_FLOOR_TOKENS) return undefined;
|
|
480
|
+
|
|
481
|
+
// Extra cost = missed tokens billed at the actual paid rate (input/cacheWrite,
|
|
482
|
+
// incl. write premium) instead of the cache-read rate. Missed tokens can only
|
|
483
|
+
// land in the input or cacheWrite buckets, so the paid rate comes straight
|
|
484
|
+
// from this message's own cost breakdown.
|
|
485
|
+
const cost = usage.cost ?? {};
|
|
486
|
+
const paidTokens = usage.input + usage.cacheWrite;
|
|
487
|
+
const paidPerToken =
|
|
488
|
+
paidTokens > 0
|
|
489
|
+
? ((cost.input ?? 0) + (cost.cacheWrite ?? 0)) / paidTokens
|
|
490
|
+
: 0;
|
|
491
|
+
const readPerToken =
|
|
492
|
+
usage.cacheRead > 0
|
|
493
|
+
? (cost.cacheRead ?? 0) / usage.cacheRead
|
|
494
|
+
: (models.getModel(message.provider ?? "", message.model ?? "")?.cost
|
|
495
|
+
?.cacheRead ?? 0) / 1_000_000;
|
|
496
|
+
|
|
497
|
+
return {
|
|
498
|
+
missedTokens,
|
|
499
|
+
missedCost: missedTokens * Math.max(0, paidPerToken - readPerToken),
|
|
500
|
+
idleMs: Math.max(0, message.timestamp - prev.timestamp),
|
|
501
|
+
modelChanged:
|
|
502
|
+
modelKey(message.provider ?? "", message.model ?? "") !== prev.modelKey,
|
|
503
|
+
};
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
function asPreviousRequest(
|
|
507
|
+
message: MissAssistantMessage,
|
|
508
|
+
reportedCache: boolean,
|
|
509
|
+
): PreviousRequest | undefined {
|
|
510
|
+
const usage = message.usage;
|
|
511
|
+
const promptTokens = usage.input + usage.cacheRead + usage.cacheWrite;
|
|
512
|
+
if (promptTokens <= 0) return undefined;
|
|
513
|
+
return {
|
|
514
|
+
promptTokens,
|
|
515
|
+
modelKey: modelKey(message.provider ?? "", message.model ?? ""),
|
|
516
|
+
timestamp: message.timestamp,
|
|
517
|
+
reportedCache: reportedCache || usage.cacheRead + usage.cacheWrite > 0,
|
|
518
|
+
};
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
function scan(
|
|
522
|
+
entries: SessionEntry[],
|
|
523
|
+
keepBaselineAcrossBranchSummary: boolean,
|
|
524
|
+
): PreviousRequest | undefined {
|
|
525
|
+
let prev: PreviousRequest | undefined;
|
|
526
|
+
// Session-level cache capability: any measured cache activity (assistant
|
|
527
|
+
// turns AND summary requests) proves the provider reports caching, so a
|
|
528
|
+
// later zero-read is a real miss even across a context boundary.
|
|
529
|
+
let everReportedCache = false;
|
|
530
|
+
for (const entry of entries) {
|
|
531
|
+
if (
|
|
532
|
+
entry.type === "compaction" ||
|
|
533
|
+
(entry.type === "branch_summary" && !keepBaselineAcrossBranchSummary)
|
|
534
|
+
) {
|
|
535
|
+
// The context legitimately changed; the next turn's prompt is new content,
|
|
536
|
+
// not re-billed content. Model switches are NOT exempt: they re-bill the
|
|
537
|
+
// full prompt and should be counted.
|
|
538
|
+
if (entry.usage && entry.usage.cacheRead + entry.usage.cacheWrite > 0) {
|
|
539
|
+
everReportedCache = true;
|
|
540
|
+
}
|
|
541
|
+
prev = undefined;
|
|
542
|
+
continue;
|
|
543
|
+
}
|
|
544
|
+
if (entry.type === "branch_summary") {
|
|
545
|
+
// Probe-only path (keepBaselineAcrossBranchSummary): the summary request
|
|
546
|
+
// reuses the live prompt-cache prefix, so the parent baseline survives.
|
|
547
|
+
// Fold cache activity into the session capability flag but never reset
|
|
548
|
+
// prev and never become prev (only assistant messages do).
|
|
549
|
+
if (entry.usage && entry.usage.cacheRead + entry.usage.cacheWrite > 0) {
|
|
550
|
+
everReportedCache = true;
|
|
551
|
+
}
|
|
552
|
+
continue;
|
|
553
|
+
}
|
|
554
|
+
if (entry.type === "message" && entry.message.role === "assistant") {
|
|
555
|
+
const message = entry.message as unknown as MissAssistantMessage;
|
|
556
|
+
if (
|
|
557
|
+
message.usage &&
|
|
558
|
+
message.usage.cacheRead + message.usage.cacheWrite > 0
|
|
559
|
+
) {
|
|
560
|
+
everReportedCache = true;
|
|
561
|
+
}
|
|
562
|
+
prev =
|
|
563
|
+
asPreviousRequest(
|
|
564
|
+
message,
|
|
565
|
+
(prev?.reportedCache ?? false) || everReportedCache,
|
|
566
|
+
) ?? prev;
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
return prev;
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
/**
|
|
573
|
+
* Detect a cache miss on a just-completed branch-summary response from its
|
|
574
|
+
* measured usage. `entries` is the session BEFORE the summary entry is
|
|
575
|
+
* appended. Live-turn accounting counts model switches as misses; summary
|
|
576
|
+
* probes suppress them instead — a cold summary right after a switch is
|
|
577
|
+
* expected re-billing, not an actionable miss.
|
|
578
|
+
*/
|
|
579
|
+
export function detectBranchSummaryCacheMiss(
|
|
580
|
+
entries: SessionEntry[],
|
|
581
|
+
responseUsage: MissUsage,
|
|
582
|
+
provider: string,
|
|
583
|
+
model: string,
|
|
584
|
+
timestamp: number,
|
|
585
|
+
models: ModelPriceSource,
|
|
586
|
+
): BranchSummaryCacheMiss | undefined {
|
|
587
|
+
const prev = scan(entries, true);
|
|
588
|
+
if (prev && prev.modelKey !== modelKey(provider, model)) return undefined;
|
|
589
|
+
return detectMiss(
|
|
590
|
+
prev,
|
|
591
|
+
{ provider, model, usage: responseUsage, timestamp },
|
|
592
|
+
models,
|
|
593
|
+
);
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
/**
|
|
597
|
+
* Compact token formatting, byte-for-byte port of the fork's
|
|
598
|
+
* `interactive-mode/components/footer.ts` `formatTokens` (the same helper the
|
|
599
|
+
* miss notice uses): 999 → "999", 9999 → "10.0k", 20000 → "20k".
|
|
600
|
+
*/
|
|
601
|
+
function formatTokens(count: number): string {
|
|
602
|
+
if (count < 1000) return count.toString();
|
|
603
|
+
if (count < 10_000) return `${(count / 1000).toFixed(1)}k`;
|
|
604
|
+
if (count < 1_000_000) return `${Math.round(count / 1000)}k`;
|
|
605
|
+
if (count < 10_000_000) return `${(count / 1_000_000).toFixed(1)}M`;
|
|
606
|
+
return `${Math.round(count / 1_000_000)}M`;
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
/**
|
|
610
|
+
* TUI warning copy for a counted branch-summary cache miss, or `null` when
|
|
611
|
+
* the miss is below the display floor. Mirrors the fork's `addCacheMissNotice`
|
|
612
|
+
* thresholds and label selection verbatim. `index.ts` stores the non-null
|
|
613
|
+
* result in `details.summaryCache.notice` and renders it as a transcript line
|
|
614
|
+
* in the tool's `renderResult`; it is NEVER appended to the rewind tool-result
|
|
615
|
+
* content the model sees.
|
|
616
|
+
*/
|
|
617
|
+
export function formatBranchSummaryCacheMissNotice(
|
|
618
|
+
miss: BranchSummaryCacheMiss,
|
|
619
|
+
): string | null {
|
|
620
|
+
if (
|
|
621
|
+
miss.missedTokens < CACHE_MISS_DISPLAY_TOKENS &&
|
|
622
|
+
miss.missedCost < CACHE_MISS_DISPLAY_COST
|
|
623
|
+
) {
|
|
624
|
+
return null;
|
|
625
|
+
}
|
|
626
|
+
const cost =
|
|
627
|
+
miss.missedCost >= 0.01 ? ` (~$${miss.missedCost.toFixed(2)})` : "";
|
|
628
|
+
const reBilled = `${formatTokens(miss.missedTokens)} tokens re-billed${cost}`;
|
|
629
|
+
let label = "Cache miss";
|
|
630
|
+
if (miss.modelChanged) {
|
|
631
|
+
label = "Cache miss after model switch";
|
|
632
|
+
} else if (miss.idleMs >= CACHE_TTL_MS) {
|
|
633
|
+
label = `Cache miss after ${Math.round(miss.idleMs / 60_000)}m idle`;
|
|
634
|
+
}
|
|
635
|
+
return `${label}: ${reBilled}`;
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
// ---------------------------------------------------------------------------
|
|
639
|
+
|
|
640
|
+
/**
|
|
641
|
+
* Cache accounting for the summary request. `usage.input` counts FRESH
|
|
642
|
+
* (uncached) tokens only on cache-serving providers, which is why the
|
|
643
|
+
* cache-hit metric is `cacheRead > 0` rather than `cacheRead/input`. This is
|
|
644
|
+
* retained purely as the machine-readable `details.summaryCache` surface;
|
|
645
|
+
* the TUI warning is derived from the miss detector above.
|
|
646
|
+
*/
|
|
647
|
+
export interface SummaryCacheStats {
|
|
648
|
+
/** Fresh (uncached) input tokens billed for this request. */
|
|
649
|
+
cacheRead: number;
|
|
650
|
+
fresh: number;
|
|
651
|
+
cacheWrite: number;
|
|
652
|
+
hit: boolean;
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
/**
|
|
656
|
+
* Derive cache accounting from provider usage. On cache-serving providers
|
|
657
|
+
* `usage.input` counts FRESH tokens only (a cached read can exceed it), so
|
|
658
|
+
* the hit predicate is `cacheRead > 0`, never a ratio.
|
|
659
|
+
*/
|
|
660
|
+
export function measureSummaryCache(
|
|
661
|
+
usage: SummaryCacheUsage,
|
|
662
|
+
): SummaryCacheStats {
|
|
663
|
+
return {
|
|
664
|
+
cacheRead: usage.cacheRead,
|
|
665
|
+
fresh: usage.input,
|
|
666
|
+
cacheWrite: usage.cacheWrite,
|
|
667
|
+
hit: usage.cacheRead > 0,
|
|
668
|
+
};
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
// ---------------------------------------------------------------------------
|
|
672
|
+
// StreamFn wrapper
|
|
673
|
+
// ---------------------------------------------------------------------------
|
|
674
|
+
|
|
675
|
+
/**
|
|
676
|
+
* The exact request live turns send, rebuilt for the summarization call.
|
|
677
|
+
* `index.ts` assembles this from public pi APIs + two plain reflected fields
|
|
678
|
+
* and passes it to `createCachePreservingStreamFn`.
|
|
679
|
+
*/
|
|
680
|
+
export interface CacheRequest {
|
|
681
|
+
/** Live system prompt + live tool array + structured history + trailer. */
|
|
682
|
+
context: {
|
|
683
|
+
systemPrompt: string;
|
|
684
|
+
messages: WireMessage[];
|
|
685
|
+
tools: AgentTool[];
|
|
686
|
+
};
|
|
687
|
+
/**
|
|
688
|
+
* Resolved (never hardcoded) cache retention. Explicit `"short"` matters:
|
|
689
|
+
* reads key on sessionId + prefix bytes, so `"none"` (upstream's forced
|
|
690
|
+
* value) would send no session id and could never hit.
|
|
691
|
+
*/
|
|
692
|
+
cacheRetention: "short" | "long";
|
|
693
|
+
/** Live session id: joins the session's cache namespace. */
|
|
694
|
+
sessionId?: string;
|
|
695
|
+
/**
|
|
696
|
+
* Live thinking level, forwarded as `reasoning`. Providers that key the
|
|
697
|
+
* cache on reasoning effort will miss a request that omits it even when
|
|
698
|
+
* the prefix bytes match — this is a product requirement, not probe
|
|
699
|
+
* hygiene. `"off"` is omitted, mirroring the live loop.
|
|
700
|
+
*/
|
|
701
|
+
reasoning?: ThinkingLevel;
|
|
702
|
+
/**
|
|
703
|
+
* Live thinking token budgets (plain field on pi-agent-core's `Agent`).
|
|
704
|
+
* Inert on some APIs, part of Anthropic's thinking body; forwarded only
|
|
705
|
+
* when the host exposes it.
|
|
706
|
+
*/
|
|
707
|
+
thinkingBudgets?: unknown;
|
|
708
|
+
}
|
|
709
|
+
|
|
710
|
+
/**
|
|
711
|
+
* Wrap the `streamFn` handed to `generateBranchSummary` so the request is
|
|
712
|
+
* rewritten to the live shape at the last possible moment (after upstream's
|
|
713
|
+
* `completeSummarization` forced `cacheRetention: "none"` + a fresh
|
|
714
|
+
* `sessionId`).
|
|
715
|
+
*
|
|
716
|
+
* `request === null` → delegate the caller's context and options untouched:
|
|
717
|
+
* today's cold standalone request. This is the fallback for every "live input
|
|
718
|
+
* unavailable" case.
|
|
719
|
+
*
|
|
720
|
+
* The `maxTokens` strip: upstream's summary caller caps output
|
|
721
|
+
* (`maxTokens: 2048` in 0.84.2), but live turns let pi-ai fill
|
|
722
|
+
* `clampMaxTokensToContext(model, liveContext, options?.maxTokens ??
|
|
723
|
+
* model.maxTokens)`. A caller cap therefore diverges from the live
|
|
724
|
+
* `max_output_tokens` and can break a gateway that keys on it. Stripping the
|
|
725
|
+
* cap lets the provider compute the same value live gets. Residual: when the
|
|
726
|
+
* context window is near-full the clamp differs by the trailer size (~1k);
|
|
727
|
+
* r5d bounds output length in prose instead, and a miss is flagged by the
|
|
728
|
+
* notice. (The recorded value differs across 0.80.2 = 2048 / 0.84.2 = 2048 /
|
|
729
|
+
* fork = 4096, which is exactly why we strip whatever is there rather than
|
|
730
|
+
* assume a constant.)
|
|
731
|
+
*
|
|
732
|
+
* Returned `used.value` flips true iff the live request was actually
|
|
733
|
+
* delegated, so `index.ts` can report whether the wrapper engaged (it stays
|
|
734
|
+
* false when the summarizer is stubbed or errors before the wire call).
|
|
735
|
+
*/
|
|
736
|
+
export function createCachePreservingStreamFn(args: {
|
|
737
|
+
realStreamFn: StreamFn;
|
|
738
|
+
request: CacheRequest | null;
|
|
739
|
+
}): { streamFn: StreamFn; used: { value: boolean } } {
|
|
740
|
+
const { realStreamFn, request } = args;
|
|
741
|
+
const used = { value: false };
|
|
742
|
+
const streamFn: StreamFn = (model, coldContext, options) => {
|
|
743
|
+
if (request === null) {
|
|
744
|
+
used.value = false;
|
|
745
|
+
return realStreamFn(model, coldContext, options);
|
|
746
|
+
}
|
|
747
|
+
used.value = true;
|
|
748
|
+
const rest = { ...(options as Record<string, unknown> | undefined) };
|
|
749
|
+
// Strip whatever caller cap upstream set; live turns let pi-ai clamp
|
|
750
|
+
// model.maxTokens to the real context window.
|
|
751
|
+
delete rest.maxTokens;
|
|
752
|
+
const next = {
|
|
753
|
+
...rest,
|
|
754
|
+
cacheRetention: request.cacheRetention,
|
|
755
|
+
...(request.sessionId ? { sessionId: request.sessionId } : {}),
|
|
756
|
+
...(request.reasoning && request.reasoning !== "off"
|
|
757
|
+
? { reasoning: request.reasoning }
|
|
758
|
+
: {}),
|
|
759
|
+
...(request.thinkingBudgets
|
|
760
|
+
? { thinkingBudgets: request.thinkingBudgets }
|
|
761
|
+
: {}),
|
|
762
|
+
} as NonNullable<Parameters<StreamFn>[2]>;
|
|
763
|
+
const context = request.context as Parameters<StreamFn>[1];
|
|
764
|
+
return realStreamFn(model, context, next);
|
|
765
|
+
};
|
|
766
|
+
return { streamFn, used };
|
|
767
|
+
}
|