pi-zip 0.2.8 → 0.3.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/plan.ts CHANGED
@@ -2,9 +2,10 @@
2
2
  import { classifyRecoverability, type Recover } from "./classify.ts";
3
3
  import { handleFor, makePlaceholderFor, pickKeyLines, RECALL_TOOL, shortArgs } from "./placeholder.ts";
4
4
  import { createHash } from "node:crypto";
5
- import { type Any, PRODUCT, clamp, envInt, textOf, tok4, tokensOf } from "./util.ts";
5
+ import { type Any, COMMIT_CUSTOM, PRODUCT, clamp, envInt, piTokensOf, textOf, tok4, tokensOf } from "./util.ts";
6
6
  import { PH_MARK } from "./placeholder.ts";
7
7
  import { MIN_EXPECT, type Prices } from "./learn.ts";
8
+ import { commandNamesRecall, parseRecallPath, type FileRecall } from "./recallfile.ts";
8
9
 
9
10
  /** Default: when a COLD plan (P(warm) < 0.5) is still above the cap, also fold REREADABLE outputs of the previous user turn, biggest
10
11
  * first. A warm plan does so only at a user return after the declared TTL (PlanOpts.pastTtl): the very returns where the TTL rule
@@ -22,7 +23,31 @@ export const RELAX_PREV_TURN = true;
22
23
  * constant (largest saving with lost <= prod in every stratum); the old rereadable-only filter cost 1.026 / 1.018x at -0.6 lost items. */
23
24
  export const INTURN_AGE = 60;
24
25
 
25
- const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised; folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
26
+ const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised (but see LOOKAHEAD); folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
27
+
28
+ /** F1: a warm (valve) summary passes the same summary gate as a cold one, the call priced in (before: a warm plan skipped it, so a
29
+ * summary the cold gate had just refused was written one request later anyway). PlanOpts.warmSummaryGate. */
30
+ export const WARM_SUMMARY_GATE = true;
31
+ /** F2: a cold plan that does not summarise looks one user turn ahead: if the warm valve would summarise at the next user turn, once the
32
+ * protected window has slid past the previous turn, the cold plan summarises now (the rewrite is free now, not then), its cut allowed
33
+ * inside the previous user turn up to that turn's final assistant answer, which stays verbatim with everything after it. PlanOpts.lookahead. */
34
+ export const LOOKAHEAD = false;
35
+ /** F3: the summary gates price the call the code makes (summary.ts narrative): uncached input of the prefix the narrative reads (the
36
+ * previous summary is not sent: its sections are carried forward by code) plus output of at most the narrative budget, instead of
37
+ * w x the prefix + out x the whole planned summary (which counts the carried-forward previous summary as model output). PlanOpts.narrativePrice. */
38
+ export const NARRATIVE_PRICE = true;
39
+ /** The narrative's token budget (chars/4), summary.ts buildCut: clamp(planned - skeleton, 300, NARRATIVE_MAX). */
40
+ export const NARRATIVE_MAX = 4000;
41
+ /** B2 for folds (formal TRIAGE (a) 6): the outputs a cold plan must protect (the previous user turn's, the mutating ones) are no longer
42
+ * protected one prompt later, when the window slides, and the warm plan then folded them at a warm cache: a rewrite the free cold one
43
+ * dominated. Now a warm plan holds back the blocks that were protected at the last cold return (PlanOpts.hold: that return's prompt)
44
+ * until the next cold return, or until the session has grown by more than HOLD_RELEASE real tokens since it (new information), or the
45
+ * context is at Pi's compaction room; and when even folding them all would not avoid a summary, it is the plan it was before. Chosen
46
+ * over folding them early at the cold return (the user would lose the previous turn's outputs at the prompt that asks about them):
47
+ * formal/TRIAGE.md, B2 for folds. */
48
+ export const FOLD_HOLD = true;
49
+ /** The spec's "no new information" is growth <= 10K real tokens (formal DELTA); the estimate is chars/4 x c, 15% off at worst (TOL_TARGET). */
50
+ export const HOLD_RELEASE = 11_500;
26
51
  const SUMMARY_FLOOR = 1000;
27
52
  const SUMMARY_CAP = 8000;
28
53
  const SUMMARY_RATIO = 0.1;
@@ -55,6 +80,56 @@ export function legacyValve(before: number, after: number, room: number | null):
55
80
  }
56
81
 
57
82
  export const PI_KEEP_RECENT = 20_000; // Pi's compaction keepRecentTokens default
83
+
84
+ /** Pi's compaction keepRecentTokens for this model: compaction.modelOverrides["provider/id"], else compaction.keepRecentTokens, else 20000. */
85
+ export function keepRecentTokensFor(settings: Any, model: Any): number {
86
+ const c = settings?.compaction;
87
+ const key = model ? `${model.provider}/${model.id}` : "";
88
+ for (const v of [c?.modelOverrides?.[key]?.keepRecentTokens, c?.keepRecentTokens]) if (typeof v === "number" && Number.isFinite(v) && v >= 0) return v;
89
+ return PI_KEEP_RECENT;
90
+ }
91
+
92
+ const CUT_POINT_ROLES = new Set(["user", "assistant", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
93
+ const TURN_START_ROLES = new Set(["user", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
94
+
95
+ /**
96
+ * Would Pi's compact() find something to compact in this projection (the branch as it is now), or throw "Nothing to compact (session
97
+ * too small)"? A replica of the emptiness test of Pi's prepareCompaction (core/compaction/compaction.js, not exported): Pi cuts where
98
+ * keepRecent of its chars/4 tokens, counted from the end, begin, at the next cut point; the conversation before that cut (or before the
99
+ * turn it splits) must not be empty. Pi's own extra moves of the cut (over entries without messages, past recovery omissions) never
100
+ * empty it. The caller checks "Already compacted" (the newest branch entry is a compaction) itself.
101
+ */
102
+ export function piCanCompact(entries: Any[], keepRecent: number): boolean {
103
+ const msgs = (e: Any): Any[] => (Array.isArray(e?.messages) ? e.messages : []);
104
+ const isComp = (e: Any) => e?.sourceEntry?.type === "compaction";
105
+ const prev = entries.findIndex((e) => isComp(e) && msgs(e).length > 0);
106
+ const start = prev >= 0 ? prev + 1 : 0, end = entries.length;
107
+ const cuts: number[] = [];
108
+ for (let i = start; i < end; i++) if (!isComp(entries[i]) && msgs(entries[i]).some((m) => CUT_POINT_ROLES.has(m?.role))) cuts.push(i);
109
+ if (!cuts.length) return false;
110
+ let cut = cuts[0], acc = 0;
111
+ for (let i = end - 1; i >= start; i--) {
112
+ const t = msgs(entries[i]).reduce((a, m) => a + piTokensOf(m), 0);
113
+ if (!t) continue;
114
+ acc += t;
115
+ if (acc >= keepRecent) { cut = cuts.find((c) => c >= i) ?? cuts[cuts.length - 1]; break; }
116
+ }
117
+ while (cut > start && !isComp(entries[cut - 1]) && !msgs(entries[cut - 1]).length) cut--;
118
+ if (!entries[cut]?.sourceEntry?.id) return false;
119
+ const startsTurn = (e: Any) => !isComp(e) && msgs(e).some((m) => TURN_START_ROLES.has(m?.role));
120
+ let turn = -1;
121
+ if (!startsTurn(entries[cut])) for (let i = cut; i >= start; i--) if (startsTurn(entries[i])) { turn = i; break; }
122
+ const conversation = (from: number, to: number) => entries.slice(from, to).some((e) => !isComp(e) && msgs(e).some((m) => m?.role !== "system"));
123
+ return turn >= 0 ? conversation(start, turn) || conversation(turn, cut) : conversation(start, cut);
124
+ }
125
+
126
+ /** The projection a run plan would have seen without pi-zip's native compaction: `pre` (taken right before it) plus what was appended
127
+ * after the compaction entry (the prompt and anything sent with it). null when that compaction is no longer the head of `now`. */
128
+ export function spliceNative(pre: Any[], now: Any[], compactionId: string): Any[] | null {
129
+ if (!compactionId || now[0]?.sourceEntry?.id !== compactionId) return null;
130
+ const known = new Set(pre.map((e) => e?.sourceEntry?.id).filter((x) => typeof x === "string"));
131
+ return [...pre, ...now.slice(1).filter((e) => !known.has(e?.sourceEntry?.id))];
132
+ }
58
133
  export const G0 = 2_500; // growth per request before the session's own EWMA has data
59
134
 
60
135
  /** What the law knows at a decision: prices (null = legacy rule), growth per request g (real tokens), P(the cache is warm). */
@@ -88,58 +163,249 @@ export function lawTerms(B: number, A: number, P: number, room: number | null, p
88
163
  export const editAllowed = (B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A): boolean =>
89
164
  lawTerms(B, A, P, room, pr, g, call, T).ok;
90
165
 
91
- // Token scale. Every size in this file is a chars/4 estimate, and chars/4 undercounts real tokens (JSON-heavy tool calls, code,
92
- // identifiers, tool definitions that are not in the text at all). `k` = real tokens per estimated token; every limit (cold cap,
93
- // compaction room, min gain) is compared against k x estimate, so they mean REAL tokens. k is read from the branch itself (the
94
- // usage of the newest assistant message), so it survives a restart and needs no stored state. With no usage to read, DEFAULT_K
95
- // applies: 1.7 is the ratio measured on recorded coding sessions (real first-request context / chars/4 estimate: 1.72 after cold
96
- // returns, 1.71 for previous-prompt peaks), and a too-high k only folds a little deeper, a too-low k leaves the context over the cap.
97
- export const DEFAULT_K = 1.7;
98
- export const K_MIN = 1; // an estimate above the real count is not trusted: never scale down
99
- export const K_MAX = 2.5;
100
-
101
- export interface Calibration {
102
- k: number;
103
- real: number; // input + cacheRead + cacheWrite of the request that produced the newest usable assistant message (0 = none)
104
- est: number; // estimate of what that request carried: system prompt + the view blocks before that message
105
- source: "usage" | "default";
166
+ // Token scale. Every size in this file is a chars/4 estimate of the CONTENT (Pi's own rule); the provider counts real tokens, and the
167
+ // two are not proportional: real = O + c x content.
168
+ // O what every request carries and no edit can shrink: system prompt, tool schemas, the provider's own framing (2-4K on a bare Pi,
169
+ // 29-55K with a large tool set);
170
+ // c real tokens per estimated content token (0.7-1.1 GLM / GPT, 1.5-2.2 Claude; more on CJK-heavy content).
171
+ // pi-zip <= 0.2.9 used ONE ratio, k = real / (system prompt + content) clamped to [1, 2.5]: with O a third of the context k sat on its
172
+ // clamp, and every size after an edit came out 10-15K too small (research round5/affine.md). Both are read from the branch on every
173
+ // decision, so a restart or `pi -p` calibrates exactly like a long-lived process, and nothing is stored:
174
+ // O 1. the cacheRead of the request that first carried the newest summary: everything after the system prompt and tools changed
175
+ // there, so what the provider still read from its cache is exactly O (when the cache was warm at all);
176
+ // 2. else the session's first request minus C0 x its content (an opening prompt is small next to O);
177
+ // 3. else R0 x the chars/4 of the system prompt and the JSON of the declared tools.
178
+ // An O measured under another system prompt or tool set moves by R0 x the change of that chars/4 size (MCP tools come and go).
179
+ // c (real - O) / content of the newest response, clamped to [C_MIN, C_MAX] (a guard for a tiny content estimate; recorded sessions
180
+ // span 0.7-3.9); C0 before the first response. A response followed by edits it did not carry (a warm-valve edit is persisted at
181
+ // turn_end and first sent with the next request: that response's prompt did not shrink) is counted with the originals.
182
+ // Both are per model: only responses of the model the next request goes to count. After a switch, before its first response,
183
+ // O = R0 x the system state and c = C0; after it, with no O evidence of its own, O = c x X, X = O / c of the previous model.
184
+ export const C0 = 1.5;
185
+ export const C_MIN = 0.5;
186
+ export const C_MAX = 4;
187
+ export const R0 = 2;
188
+
189
+ /** The system prompt and the declared tools (Pi API fallback for a transcript that records no system message). */
190
+ export interface SysInfo { text: string; tools: Any[] }
191
+ /** chars/4 of the system prompt plus the JSON of the declared tool definitions (name, description, parameters). */
192
+ export const sysEst = (s: SysInfo): number =>
193
+ tok4(s.text) + (s.tools.length ? tok4(JSON.stringify(s.tools.map((t: Any) => ({ name: t?.name, description: t?.description, parameters: t?.parameters })))) : 0);
194
+
195
+ export interface Scale {
196
+ O: number; // real tokens every request carries before the content (system prompt, tools, framing)
197
+ c: number; // real tokens per estimated (chars/4) content token
198
+ real: number; // input + cacheRead + cacheWrite of the newest usable response (0 = none)
199
+ est: number; // estimated content that request carried
200
+ source: "usage" | "default"; // where c comes from
201
+ oSource: "summary" | "first" | "model" | "default"; // where O comes from ("model": another model's O in content units, X = O' / c')
202
+ }
203
+ /** Sizes taken as real (tests, and the ledger of a plan made without calibration). */
204
+ export const UNIT: Scale = { O: 0, c: 1, real: 0, est: 0, source: "default", oSource: "default" };
205
+ /** Real tokens of a context whose content is estimated at `content`. */
206
+ export const sizeOf = (s: Pick<Scale, "O" | "c">, content: number): number => s.O + s.c * content;
207
+
208
+ /** A response pi-ai drops before sending (an error or an abort): it carries no usable usage and adds nothing to later requests. */
209
+ const dropped = (m: Any): boolean => m?.role === "assistant" && (m.stopReason === "error" || m.stopReason === "aborted");
210
+ /** chars/4 a message adds to every later request: tokensOf, except 0 for an assistant message pi-ai drops before sending. */
211
+ export const wireTokensOf = (m: Any): number => (dropped(m) ? 0 : tokensOf(m));
212
+ /** Prompt size the provider reported for a response that really ran (0 = none, an error or an abort). */
213
+ const promptOf = (m: Any): number => {
214
+ const u = m?.role === "assistant" && !dropped(m) ? m.usage : null;
215
+ return u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
216
+ };
217
+ /** "provider/model" of a response ("" = not recorded: such a response counts for any model). */
218
+ const respKey = (m: Any): string => (m?.provider || m?.model ? `${m.provider ?? ""}/${m.model ?? ""}` : "");
219
+ /** chars/4 of the system state the transcript records in `msgs` (-1: none recorded). */
220
+ function sysOf(msgs: Any[]): number {
221
+ const s = collapseSystem(msgs);
222
+ return s ? sysEst({ text: [textOf(s.content), ...Object.values<string>(s.sections ?? {})].filter((x) => x).join("\n\n"), tools: s.toolsAdded ?? [] }) : -1;
223
+ }
224
+ /** The session's first response from model `key` on the branch: its prompt, the content before it and the system state it was sent with. */
225
+ function firstRequest(branch: Any[], key: string): { real: number; est: number; S: number } | null {
226
+ const sys: Any[] = [];
227
+ let est = 0;
228
+ for (const e of branch) {
229
+ const m = e?.type === "message" ? e.message : e?.type === "custom_message" ? { role: "custom", content: e.content } : null;
230
+ if (!m) continue;
231
+ if (m.role === "system") { sys.push(m); continue; }
232
+ const real = promptOf(m), k = respKey(m);
233
+ if (real > 0 && (!key || !k || k === key)) return { real, est, S: sysOf(sys) };
234
+ est += wireTokensOf(m);
235
+ }
236
+ return null;
106
237
  }
107
238
 
108
- /**
109
- * k from the branch: the newest assistant message with usage says how many real tokens its request carried; the estimate of the
110
- * view blocks before it (the projection, with the folds and cut as persisted, which are exactly what that request carried, I1)
111
- * says how many we would have guessed. Only the INPUT side is used: that message's own output is not part of the request it
112
- * answered, and thinking tokens may or may not be sent back, so including output would add noise to the ratio.
113
- * Skipped: errored messages, messages without input usage, and messages older than a compaction Pi made (not ours): their
114
- * request carried text the projection no longer has.
115
- */
116
- export function calibrate(entries: Any[], sys = 0): Calibration {
117
- const none: Calibration = { k: DEFAULT_K, real: 0, est: 0, source: "default" };
118
- const blocks = buildBlocks(entries);
119
- let foreignCompactionMs = 0;
120
- for (const pe of entries) {
121
- const src = pe.sourceEntry;
122
- if (src?.type === "compaction" && src.details?.by !== PRODUCT) foreignCompactionMs = Math.max(foreignCompactionMs, Date.parse(src.timestamp) || 0);
239
+ /** chars/4 of the context the request behind the response `id` carried, rebuilt from the branch the way Pi projects it: the newest
240
+ * compaction before the response, the entries it keeps, everything after it, and the edits persisted before the response, plus the
241
+ * edits of a run plan persisted right after it (their commit entry says the response itself carried them). null: `id` is not on the
242
+ * branch. Used when a compaction written after the response (Pi's own, another extension's) removed that request's content from the
243
+ * projection, so the projection cannot say what the response measured. */
244
+ function requestEst(branch: Any[], id: string): number | null {
245
+ const at = branch.findIndex((e) => e?.id === id);
246
+ if (at < 0) return null;
247
+ const pre = branch.slice(0, at);
248
+ const edit = new Map<string, Any>();
249
+ for (const e of pre) if (e?.type === "context_edit") edit.set(e.targetId, e.replacement);
250
+ for (let i = at + 1; i < branch.length && !(branch[i]?.type === "message" || branch[i]?.type === "compaction"); i++) {
251
+ const e = branch[i];
252
+ if (e?.type === "custom" && e.customType === COMMIT_CUSTOM) {
253
+ if (e.data?.carried) for (let j = at + 1; j < i; j++) if (branch[j]?.type === "context_edit") edit.set(branch[j].targetId, branch[j].replacement);
254
+ break;
255
+ }
123
256
  }
124
- const before: number[] = []; // estimate of blocks[0..i)
125
- let acc = 0;
126
- for (const b of blocks) { before.push(acc); acc += b.tokens; }
127
- for (let i = blocks.length - 1; i >= 0; i--) {
128
- if (blocks[i].kind !== "assistant") continue;
129
- const m = blocks[i].raw ?? blocks[i].msg;
130
- const u = m?.usage;
131
- const real = u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
132
- if (!(real > 0) || m.stopReason === "error") continue;
133
- if (foreignCompactionMs && typeof m.timestamp === "number" && m.timestamp < foreignCompactionMs) return none;
134
- const est = sys + before[i];
135
- if (!(est > 0)) return none;
136
- return { k: clamp(real / est, K_MIN, K_MAX), real, est, source: "usage" };
257
+ let ci = -1;
258
+ pre.forEach((e, i) => { if (e?.type === "compaction") ci = i; });
259
+ const kept = ci < 0 ? pre : [pre[ci], ...pre.slice(0, ci).slice(Math.max(0, pre.findIndex((e) => e?.id === pre[ci].firstKeptEntryId))).filter((e) => !(e?.type === "message" && e.message?.role === "system")), ...pre.slice(ci + 1)];
260
+ let est = 0;
261
+ for (const e of kept) {
262
+ if (e?.type === "compaction" || e?.type === "branch_summary") est += tokensOf({ role: "compactionSummary", summary: e.summary });
263
+ else if (e?.type === "custom_message") est += tokensOf({ role: "custom", content: edit.has(e.id) ? edit.get(e.id)?.content : e.content });
264
+ else if (e?.type === "message" && e.message) {
265
+ const rep = edit.get(e.id);
266
+ if (rep === null) continue;
267
+ const m = rep && e.message.role !== "system" ? { ...e.message, content: typeof rep.content === "string" && e.message.role !== "user" ? [{ type: "text", text: rep.content }] : rep.content } : e.message;
268
+ est += wireTokensOf(m);
269
+ }
137
270
  }
138
- return none;
271
+ return est;
139
272
  }
140
273
 
274
+ /** O and c from the projection (`entries`) and, for the session's first request, the branch; `sys` stands in for a transcript without
275
+ * system messages. `model` = "provider/id" of the model the next request goes to (default: the newest response's): O and c are
276
+ * properties of the provider's tokenizer, so only that model's responses are evidence. With none, O = R0 x the system state and
277
+ * c = C0; with responses but no O evidence of its own (a model switch mid-session), O keeps the size another model measured in
278
+ * content units, X = O' / c' (the system prompt and tools tokenize like the content around them): real = c (X + E). */
279
+ export function calibrate(entries: Any[], sys?: SysInfo, branch: Any[] = [], model?: string): Scale {
280
+ type Resp = { at: number; est: number; real: number; read: number; nSys: number; ts: number; key: string; id: string };
281
+ const resp: Resp[] = [];
282
+ const sysMsgs: Any[] = [];
283
+ const edits: { at: number; target: string; foreign: boolean }[] = [];
284
+ const marks: { at: number; carried: boolean }[] = []; // pi-zip's commit entries: who first carried the edits right before them (run.ts commit)
285
+ const orig = new Map<string, { at: number; d: number }>(); // edited entries: tokens their originals add back
286
+ const withMsgs: number[] = [0]; // withMsgs[k]: entries before k that carry messages (a commit group has none inside it)
287
+ let est = 0, sysAt = 0; // sysAt: responses before the newest system message
288
+ entries.forEach((pe, at) => {
289
+ const src = pe.sourceEntry, msgs: Any[] = pe.messages ?? [];
290
+ withMsgs.push(withMsgs[at] + (msgs.length ? 1 : 0));
291
+ // an edit with a replacement that is not our placeholder is another extension's (never carried by a request of ours); one without
292
+ // (a bare entry) is judged by the size heuristics below
293
+ if (src?.type === "context_edit") edits.push({ at, target: src.targetId, foreign: "replacement" in src && !textOf(src.replacement?.content).startsWith(PH_MARK) });
294
+ if (src?.type === "custom" && src.customType === COMMIT_CUSTOM) marks.push({ at, carried: !!src.data?.carried });
295
+ if (src?.type === "message" && msgs.length === 1 && msgs[0] !== src.message) orig.set(src.id, { at, d: wireTokensOf(src.message) - wireTokensOf(msgs[0]) });
296
+ for (const m of msgs) {
297
+ if (m?.role === "system") { sysMsgs.push(m); sysAt = resp.length; continue; }
298
+ const real = promptOf(m);
299
+ if (real > 0) resp.push({ at, est, real, read: Number(m.usage.cacheRead) || 0, nSys: sysMsgs.length, ts: Number(m.timestamp) || 0, key: respKey(m), id: src?.id ?? "" });
300
+ est += wireTokensOf(m);
301
+ }
302
+ });
303
+ const sAt = (n: number) => sysOf(sysMsgs.slice(0, n)); // the system state after the first n system messages (computed only where needed)
304
+ const S = sAt(sysMsgs.length);
305
+ const Snow = S >= 0 ? S : sys ? sysEst(sys) : 0;
306
+ const head = entries[0]?.sourceEntry?.type === "compaction" ? entries[0].sourceEntry : null;
307
+ const headMs = head ? Date.parse(head.timestamp) || 0 : 0;
308
+ // a pi-zip summary persisted as a turn_end draft was request-local first (a run plan carries it before its entry exists); one that
309
+ // went in through Pi's compaction before the run (details.via "native") is carried after its entry, exactly like Pi's own
310
+ const local = !!head && head.details?.by === PRODUCT && head.details?.via !== "native";
311
+ // Was the request behind this response sent after the head compaction entry was written? The branch's order says (timestamps tie when
312
+ // a compaction follows a response in the same millisecond); without the branch the timestamps do.
313
+ const pos = new Map<string, number>();
314
+ if (head) branch.forEach((e, i) => { if (typeof e?.id === "string") pos.set(e.id, i); });
315
+ const afterHead = (r: Resp): boolean => {
316
+ const h = pos.get(head?.id), i = pos.get(r.id);
317
+ return h !== undefined && i !== undefined ? i > h : r.ts > headMs;
318
+ };
319
+ const opening = (r: { real: number; est: number } | null | undefined) => !!r && C0 * r.est <= 0.1 * r.real; // content small next to O
320
+
321
+ /** The first commit entry of the group of pi-zip edits that starts at entry `at` (none when a message comes first). */
322
+ const markOf = (at: number) => marks.find((m) => m.at > at && withMsgs[m.at] - withMsgs[at + 1] === 0);
323
+ /** Index of the newest response in `resp` before entry `at`. */
324
+ const respBefore = (at: number) => resp.reduce((k, r, i) => (r.at < at ? i : k), -1);
325
+ /** Tokens the originals add back for response i: edits persisted after it that its own request did not carry (the projection counts
326
+ * their placeholders). A commit entry says who carried them (a run plan: the response before the entry; a valve plan: the next one);
327
+ * with none (a session of an older version) the sizes decide, for the newest response only (`legacy`). */
328
+ const addBack = (i: number, legacy: boolean): number => {
329
+ let d = 0;
330
+ for (const x of edits) {
331
+ const t = orig.get(x.target);
332
+ if (x.at <= resp[i].at || !t || t.at >= resp[i].at) continue;
333
+ const m = x.foreign ? undefined : markOf(x.at);
334
+ if (m ? m.carried && respBefore(m.at) === i : !x.foreign && legacy) continue;
335
+ d += t.d;
336
+ }
337
+ return d;
338
+ };
339
+
340
+ const fit = (want: string, transfer: boolean): Scale => {
341
+ const mine = (r: Resp | undefined): r is Resp => !!r && (!want || !r.key || r.key === want);
342
+ // evidence for O, the newest wins (it needs the smallest correction for system changes since)
343
+ const ev: { at: number; O: number; S: number; src: Scale["oSource"] }[] = [];
344
+ const f = branch.length ? firstRequest(branch, want) : head ? null : resp.find(mine); // the session's first request
345
+ if (f && opening(f)) ev.push({ at: -1, O: f.real - C0 * f.est, S: "S" in f ? f.S : sAt(f.nSys), src: "first" });
346
+ const ci = resp.findIndex((r, i) => i >= sysAt && mine(r)), cur = resp[ci]; // the first response sent with the current system prompt and tools
347
+ // its content as its request carried it: edits persisted after it (a model switch, say, then a run plan's folds) are in `est`
348
+ const curEst = cur ? cur.est + addBack(ci, false) : 0;
349
+ if (cur && (!head || afterHead(cur)) && opening({ real: cur.real, est: curEst })) ev.push({ at: ci, O: cur.real - C0 * curEst, S, src: "first" });
350
+ if (head) {
351
+ // the summary at the head was first carried by the first response after its entry, or, for pi-zip's own request-local summary
352
+ // (a run plan), by the last one before it; Pi's, another extension's or a native pi-zip compaction is never sent before its entry
353
+ const after = resp.findIndex((r) => afterHead(r)), ours = local;
354
+ for (const i of after < 0 ? (ours ? [resp.length - 1] : []) : ours ? [after - 1, after] : [after]) {
355
+ const r = resp[i], prev = resp[i - 1];
356
+ if (mine(r) && r.read > 0 && r.read < 0.8 * r.real && (!prev || r.read < 0.8 * prev.real)) { ev.push({ at: i, O: r.read, S: sAt(r.nSys), src: "summary" }); break; }
357
+ }
358
+ }
359
+ const best = ev.sort((a, b) => b.at - a.at)[0];
360
+ let O = best ? Math.max(0, best.O + (best.S >= 0 && S >= 0 ? R0 * (Snow - best.S) : 0)) : R0 * Snow, oSource: Scale["oSource"] = best?.src ?? "default";
361
+ // c from the newest response of this model
362
+ let n = resp.length - 1;
363
+ while (n >= 0 && !mine(resp[n])) n--;
364
+ if (n < 0) return { O, c: C0, real: 0, est: 0, source: "default", oSource };
365
+ const N = resp[n];
366
+ let estN = N.est;
367
+ const cutLater = !!head && !afterHead(N);
368
+ const later = edits.filter((x) => x.at > N.at);
369
+ const rebuilt = cutLater && !local && !!N.id ? requestEst(branch, N.id) : null; // what its request carried: the compaction after it removed that from the projection
370
+ if (rebuilt !== null && rebuilt > 0) estN = rebuilt;
371
+ else if (later.length || cutLater) {
372
+ // carried (a run plan): the prompt shrank, or a read stopped early; a cache miss alone (no read at all) proves nothing. The
373
+ // commit entries of pi-zip's edits say it outright; the sizes decide only for edits without one (older sessions).
374
+ // Pi's own compaction is never request-local.
375
+ let p = n - 1;
376
+ while (p >= 0 && !mine(resp[p])) p--;
377
+ const P = resp[p];
378
+ const heur = !P || N.real < P.real || (N.read > 0 && N.read < 0.8 * P.real);
379
+ const carried = !(cutLater && !local) && (cutLater && typeof head.details?.carried === "boolean" ? head.details.carried : heur);
380
+ if (!carried && cutLater) return { O, c: C0, real: N.real, est: 0, source: "default", oSource }; // what it carried is summarised away
381
+ estN += addBack(n, heur && carried);
382
+ }
383
+ if (!best && transfer && estN > 0) {
384
+ // no O of its own: the newest other model's O in content units
385
+ // (a compaction may have taken the other model's responses out of the projection: the branch still has them)
386
+ const other = [...resp].reverse().find((r) => r.key && r.key !== want)?.key
387
+ ?? [...branch].reverse().map((e) => (e?.type === "message" && promptOf(e.message) > 0 ? respKey(e.message) : "")).find((k) => k && k !== want);
388
+ const o = other ? fit(other, false) : null;
389
+ if (o && o.oSource !== "default" && o.c > 0) {
390
+ const X = o.O / o.c, c = clamp(N.real / (X + estN), C_MIN, C_MAX);
391
+ return { O: c * X, c, real: N.real, est: estN, source: "usage", oSource: "model" };
392
+ }
393
+ }
394
+ return { O, c: estN > 0 ? clamp((N.real - O) / estN, C_MIN, C_MAX) : C0, real: N.real, est: estN, source: estN > 0 ? "usage" : "default", oSource };
395
+ };
396
+ return fit(model || [...resp].reverse().find((r) => r.key)?.key || "", true);
397
+ }
398
+
399
+ /** The cold cap's target in real tokens of the WHOLE context (system prompt and tools included, as the cap was tuned): max(cap, O + cap/2),
400
+ * never above Pi's compaction room. While O is under half the cap this is the plain whole-context cap. Above that the target is O plus
401
+ * half the cap of conversation: never at or below O, where no edit reaches it (0.2.9's 40K with O = 29-55K summarised at every cold
402
+ * return), and not 40K of conversation on top of O either (the cap on content alone cost +13% live on a 36K prefix, with no quality
403
+ * difference measured). Offline it never loses more items than the whole-context cap (research round5/affine.md section 11). */
404
+ export const CAP_CONTENT_MIN = 0.5;
405
+ export const coldTarget = (cap: number, O: number, room: number | null): number => Math.min(Math.max(cap, O + CAP_CONTENT_MIN * cap), room ?? Infinity);
406
+
141
407
  export const settings = () => ({
142
- coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many tokens
408
+ coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many real tokens of the whole context, but never below O + half of it (coldTarget)
143
409
  foldMin: envInt("FOLD_MIN", 500), // outputs below this many tokens are never folded
144
410
  keepLines: envInt("KEEP_LINES", 8),
145
411
  minGain: envInt("MIN_GAIN", 10_000), // legacy rule only (no prices): a summary must remove at least max(minGain, 15% of the context)
@@ -203,7 +469,7 @@ export function buildBlocks(contextEntries: Any[], steerIds?: Set<string>): Bloc
203
469
  for (const c of msg.content ?? []) if (c?.type === "toolCall") issued.set(c.id, asst);
204
470
  }
205
471
  asstOf[blocks.length] = kind === "toolResult" ? (issued.get(msg.toolCallId) ?? asst) : asst;
206
- blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + tokensOf(m), 0), userTurn, edited, ours, age: 0 });
472
+ blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + wireTokensOf(m), 0), userTurn, edited, ours, age: 0 });
207
473
  }
208
474
  for (const b of blocks) if (b.kind === "toolResult") b.age = asst - asstOf[b.idx];
209
475
  return blocks;
@@ -244,6 +510,7 @@ export interface Cut {
244
510
  costUsd: number; // what the narrative model call cost (0 when unknown)
245
511
  usage?: Any;
246
512
  ms: number; // total production time (the narrative model call included)
513
+ ahead?: boolean; // F2: a lookahead cut (inside the previous user turn, before its final answer; guard.ts)
247
514
  }
248
515
 
249
516
  /** What a run sends from its first request on and persists verbatim at turn_end (F8). */
@@ -254,7 +521,7 @@ export interface RunPlan {
254
521
  ctxBefore: number;
255
522
  ctxAfter: number;
256
523
  ms: number;
257
- k: number; // the token scale the plan was made with (stats and notices of this run use the same one)
524
+ scale: Scale; // the token scale the plan was made with (stats and notices of this run use the same one)
258
525
  persisted: boolean;
259
526
  cutVisibleIdx?: number; // where the kept part starts among the projected non-system messages
260
527
  cutKeptFirstMsg?: Any; // ... and that message itself (a disagreeing request view drops the cut)
@@ -262,6 +529,9 @@ export interface RunPlan {
262
529
  applied?: Set<string>; // entry ids whose fold the latest request view actually carried (what turn_end must persist, no more)
263
530
  cutTs?: number; // timestamp of the request-local summary message: one value per run, so every request of the run is identical
264
531
  untouched?: number; // real tokens before the earliest edited block: what the first request carrying the plan can still read from the cache
532
+ waitMs?: number; // how long the user waited for this plan's summary (0: none, or it was ready)
533
+ noticed?: boolean; // its notice is already in the transcript (shown when the first request carrying it went out)
534
+ native?: Cut; // this run's summary went in through Pi's own compaction before the run: no request-local cut, no compaction draft
265
535
  }
266
536
 
267
537
  export interface PlanOpts {
@@ -270,8 +540,7 @@ export interface PlanOpts {
270
540
  trace?: (LawTerms & { where: "summary" | "plan" })[]; // every law evaluation is pushed here (ledger)
271
541
  reserve?: number; // Pi's compaction reserveTokens (default 16384)
272
542
  steerIds?: Set<string>; // user entries that are mid-run steering or follow-up messages, not new user turns
273
- sys: number; // estimated tokens of the system prompt (same chars/4 scale as the blocks; tool definitions are covered by k)
274
- k?: number; // real tokens per estimated token (calibrate); default 1 = the estimates are taken as real
543
+ scale?: Scale; // real = O + c x content estimate (calibrate); default UNIT = the estimates are taken as real
275
544
  cwd: string;
276
545
  coldCap?: number;
277
546
  foldMin?: number;
@@ -284,7 +553,16 @@ export interface PlanOpts {
284
553
  inturnAge?: number; // default settings().inturnAge (PI_ZIP_INTURN_AGE, 60); 0 = outputs of the protected turns never fold on age
285
554
  noSummary?: boolean; // a summary was already made while this cache stayed warm: no second one unless the context is at/above the compaction room
286
555
  rereadOnly?: boolean; // zip_recall is not available to the model: fold only outputs that can be re-read (classify "rereadable")
556
+ fileRecall?: FileRecall; // zip_recall is hidden but read, grep or bash is active: fold as in full mode, placeholders name the recall file
287
557
  pastTtl?: boolean; // a user return after the declared TTL: a warm plan may relax into the previous user turn too (RELAX_PREV_TURN)
558
+ warmSummaryGate?: boolean; // F1, default WARM_SUMMARY_GATE
559
+ lookahead?: boolean; // F2, default LOOKAHEAD (cold plans only)
560
+ narrativePrice?: boolean; // F3, default NARRATIVE_PRICE
561
+ slide?: number; // internal (F2): plan as if this many more user turns had started (the protected window slid)
562
+ valveB?: number; // internal (F2): the real context the slid warm plan's law sees (the cold plan's folds already applied)
563
+ hold?: { anchor: string }; // B2 for folds: the entry id of the prompt of the last cold return (run.ts holdOf); undefined = nothing held
564
+ holdPass?: boolean; // internal: the held pass of a warm plan with a hold
565
+ warmStyle?: boolean; // internal: a cold plan at 0 < P(warm) < 0.5 that the law refused is planned once more the way a warm plan is (no relax folds unless pastTtl, the summary gated)
288
566
  }
289
567
 
290
568
  export interface PlanResult {
@@ -293,21 +571,33 @@ export interface PlanResult {
293
571
  folds: FoldTarget[];
294
572
  cutIdx: number | null;
295
573
  firstKeptEntryId: string | null;
296
- prefixTokens: number; // estimate units (x k = real), like summaryTokensPlanned
574
+ prefixTokens: number; // content estimate units (x c = real), like summaryTokensPlanned
297
575
  summaryTokensPlanned: number;
298
576
  sumTrigger: "cold" | "valve" | null;
299
- k: number;
300
- ctxTokens: number; // calibrated: k x (sys + blocks)
301
- ctxAfterFolds: number; // calibrated
577
+ scale: Scale;
578
+ ctxTokens: number; // real: O + c x content
579
+ ctxAfterFolds: number; // real
302
580
  userTurns: number;
303
581
  cutVisibleIdx: number;
304
582
  cutKeptFirstMsg: Any;
583
+ fileRecall?: FileRecall; // the summary of this plan names recall files instead of zip_recall (PlanOpts.fileRecall)
584
+ ahead?: boolean; // F2: the cut was decided by the lookahead (it may lie inside the previous user turn, before its final answer)
585
+ }
586
+
587
+ /** Index of the final assistant message of user turn `turn` (the answer the user saw), -1 if none. */
588
+ export function finalAnswerIdx(blocks: Block[], turn: number): number {
589
+ for (let i = blocks.length - 1; i >= 0; i--) {
590
+ const b = blocks[i];
591
+ if (b.userTurn < turn) break;
592
+ if (b.userTurn === turn && b.kind === "assistant" && b.entryId && b.msg?.stopReason !== "error" && b.msg?.stopReason !== "aborted") return i;
593
+ }
594
+ return -1;
305
595
  }
306
596
 
307
- /** Estimated tokens before the earliest edited block (system prompt included): what stays cached through the edit. A cut edits from the start. */
308
- export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean, sys: number): number {
597
+ /** Estimated content before the earliest edited block: what stays cached through the edit with O (real: O + c x it). A cut edits from the start. */
598
+ export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean): number {
309
599
  const ids = new Set(folds.map((t) => t.entryId));
310
- let est = sys;
600
+ let est = 0;
311
601
  if (!cut) for (const b of blocks) { if (b.entryId && ids.has(b.entryId)) break; est += b.tokens; }
312
602
  return est;
313
603
  }
@@ -320,35 +610,49 @@ export const contentKeyOf = (content: Any): string => {
320
610
 
321
611
  /** The planner. Cold: fold everything outside the protected window, relax into the previous turn if still above the cap,
322
612
  * summarise only if folds cannot reach the cap and the law prices the summary call in. Warm: null unless the context is above the
323
- * cold cap and the law fires for the plan (the summary call is sunk there); then the cold plan minus the previous-turn relax (a warm plan
613
+ * cold cap and the law fires for the plan; its summary passes the same summary gate (WARM_SUMMARY_GATE); then the cold plan minus the previous-turn relax (a warm plan
324
614
  * folds the previous user turn by relax only at a return after the declared TTL, o.pastTtl). A cold plan with P(warm) > 0
325
615
  * passes the same law (expected cost). The cap never exceeds Pi's compaction room. Returns null when there is nothing to plan on
326
616
  * or the law says no. */
327
617
  export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
618
+ if (o.hold && !o.holdPass && (o.mode ?? "cold") === "warm") {
619
+ // the hold gives way to a summary: when the plan without it summarises, that is the plan (a summary voids the folds before its
620
+ // cut anyway); otherwise the held plan, with no summary of its own (the next cold return writes it)
621
+ const full = planContext(entries, { ...o, hold: undefined, trace: undefined });
622
+ if (!full) return null;
623
+ if (full.cutIdx !== null) return planContext(entries, { ...o, hold: undefined });
624
+ return planContext(entries, { ...o, noSummary: true, holdPass: true });
625
+ }
328
626
  const s = settings();
329
627
  const mode = o.mode ?? "cold";
330
- // limits are in real tokens; the plan works in estimate units, so they are divided by k once, here
331
- const k = o.k ?? 1;
628
+ const warmLike = mode === "warm" || o.warmStyle === true; // plans the way a warm plan does (the label of the trigger stays the mode's)
629
+ // limits are in real tokens; the plan works in content estimate units (real = O + c x estimate), so they are converted once, here.
630
+ // The cold cap is on the whole context, but leaves at least half of itself for the conversation above O (coldTarget): a cap the
631
+ // fixed prefix alone fills is out of reach, and every cold return then folds all it may and summarises. The compaction room is
632
+ // Pi's trigger on the whole context.
633
+ const { O, c } = o.scale ?? UNIT;
634
+ const real = (est: number) => O + c * est;
332
635
  const room = compactionRoom(o.model, o.reserve);
333
- const coldCap = Math.min(o.coldCap ?? s.coldCap, room ?? Infinity) / k;
636
+ const coldCap = (coldTarget(o.coldCap ?? s.coldCap, O, room) - O) / c;
334
637
  const law: Law = o.law ?? { pr: null, g: G0, pWarm: mode === "cold" ? 0 : 1 };
335
638
  const gate = (where: "summary" | "plan", B: number, A: number, r: number | null, call: number, T: number, Tsuf?: number): boolean => {
336
639
  const t = lawTerms(B, A, law.pWarm, r, law.pr, law.g, call, T);
337
640
  o.trace?.push({ where, ...t, ...(Tsuf !== undefined ? { Tsuf } : {}) });
338
641
  return t.ok;
339
642
  };
340
- const minGain = (o.minGain ?? s.minGain) / k;
643
+ const minGain = o.minGain ?? s.minGain;
341
644
  const foldMin = o.foldMin ?? s.foldMin;
342
645
  const keepLines = o.keepLines ?? s.keepLines;
343
646
  const relax = o.relax ?? RELAX_PREV_TURN;
344
647
  const inturnAge = o.inturnAge ?? s.inturnAge;
345
648
  const blocks = buildBlocks(entries, o.steerIds);
346
649
  if (!blocks.length) return null;
347
- const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0); // the upcoming prompt is a new user turn
650
+ const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0) + (o.slide ?? 0); // the upcoming prompt is a new user turn
651
+ const f1 = o.warmSummaryGate ?? WARM_SUMMARY_GATE, f2 = mode === "cold" && !o.warmStyle && (o.lookahead ?? LOOKAHEAD), f3 = o.narrativePrice ?? NARRATIVE_PRICE;
348
652
  const calls = toolCallIndex(blocks);
349
653
  const recalled = o.recalled ?? new Set<string>();
350
- const ctxEst = o.sys + blocks.reduce((a, b) => a + b.tokens, 0);
351
- if (mode === "warm" && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
654
+ const ctxEst = blocks.reduce((a, b) => a + b.tokens, 0); // content
655
+ if (warmLike && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
352
656
  const trig = mode === "warm" ? "valve" : "cold";
353
657
  const tokOverride = new Map<number, number>();
354
658
  const folds: FoldTarget[] = [];
@@ -368,9 +672,9 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
368
672
  };
369
673
  const addFold = (b: Block, trig: string): boolean => {
370
674
  if (tokOverride.has(b.idx) || recalled.has(handleFor(b.entryId!))) return false; // F4: never refold what the model recalled
371
- const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly);
675
+ const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly, o.fileRecall);
372
676
  if (ph === null) return false;
373
- const phTok = tok4(ph);
677
+ const phTok = tokensOf({ ...b.msg, content: [{ type: "text", text: ph }] }); // the folded message: its id and framing stay
374
678
  if (!(phTok < 0.9 * b.tokens)) return false;
375
679
  tokOverride.set(b.idx, phTok);
376
680
  const call = calls.get(b.msg.toolCallId);
@@ -378,20 +682,41 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
378
682
  return true;
379
683
  };
380
684
  // a zip_recall result IS content the model just asked for: folding it would undo the recall (and loop); never
381
- const isRecall = (b: Block) => (calls.get(b.msg.toolCallId)?.name ?? b.msg.toolName) === RECALL_TOOL;
685
+ // ... and so is a built-in read, grep, find, ls or bash call on a recall file (file recall, when an allowlist hides zip_recall)
686
+ const isRecall = (b: Block) => {
687
+ const call = calls.get(b.msg.toolCallId), name = call?.name ?? b.msg.toolName;
688
+ if (name === RECALL_TOOL) return true;
689
+ if (name === "bash") return commandNamesRecall(call?.args?.command);
690
+ return (name === "read" || name === "grep" || name === "find" || name === "ls") && !!parseRecallPath(call?.args?.path);
691
+ };
382
692
  // Observability: on an automatic (prefix) cache no fold starts inside the first MIN_EXPECT real tokens, so the next response's cacheRead
383
693
  // is an uncensored survival sample (learn.ts sample). Without it every idle return folded into the first 1-7K tokens and its sample was
384
694
  // censored, so the learned curve never saw a return (live GLM bench). Explicit caches are exempt: the provider looks back only ~20
385
695
  // blocks from the last breakpoint, a far head is not read either way. A cut replaces the prefix from the first message: exempt too.
386
- const head = law.pr?.cls === "automatic" ? MIN_EXPECT / k : 0;
696
+ const head = law.pr?.cls === "automatic" ? (MIN_EXPECT - O) / c : 0; // MIN_EXPECT counts from the first token, O included
387
697
  const start: number[] = [];
388
- blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), o.sys);
698
+ blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), 0);
389
699
  const foldable = (b: Block) => b.kind === "toolResult" && !b.edited && !!b.entryId && b.tokens > foldMin && !isRecall(b) && start[b.idx] >= head && (!o.rereadOnly || classify(b) === "rereadable");
390
700
  const protectedTurn = (b: Block) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1;
391
701
  const savings = () => folds.reduce((a, t) => a + t.entryTokens - t.phTokens, 0);
392
- const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b));
702
+ // the blocks protected at the last cold return stay out of a warm plan's folds (see FOLD_HOLD)
703
+ const held: (b: Block) => boolean = (() => {
704
+ const NOT_HELD = () => false;
705
+ if (mode !== "warm" || !o.hold) return NOT_HELD;
706
+ const from = blocks.findIndex((b) => b.entryId === o.hold!.anchor); // gone (a later summary replaced it): nothing is held
707
+ if (from < 0) return NOT_HELD;
708
+ // what the session grew by before this decision: the blocks between the cold return's prompt and the newest response (that
709
+ // response, the tool results after it and a new prompt are what the decision is about, not yet something the cache has seen)
710
+ let lastAsst = blocks.length - 1;
711
+ while (lastAsst > 0 && blocks[lastAsst].kind !== "assistant") lastAsst--;
712
+ const grown = c * blocks.slice(from + 1, Math.max(from + 1, lastAsst)).reduce((a, b) => a + b.tokens, 0);
713
+ if (grown > HOLD_RELEASE || (room !== null && real(ctxEst) >= room)) return NOT_HELD;
714
+ const turn = blocks[from].userTurn;
715
+ return (b: Block) => b.userTurn >= turn - PROTECT_USER_TURNS + 1;
716
+ })();
717
+ const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b) && !held(b));
393
718
  for (const b of cands) addFold(b, trig);
394
- if (relax && (mode === "cold" || o.pastTtl === true)) {
719
+ if (relax && ((mode === "cold" && !o.warmStyle) || o.pastTtl === true)) {
395
720
  // the protected window = the new prompt + the previous user turn; that turn's big reads are what makes a cold return
396
721
  // expensive. Rereadable ones can be recalled exactly: fold them biggest-first until the cap; never the latest turn's own.
397
722
  // Cold plans, and warm plans at a return after the declared TTL; any other warm plan keeps the previous user turn visible.
@@ -418,49 +743,78 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
418
743
  }
419
744
  }
420
745
  let ctxAfterFolds = ctxEst - savings();
421
- // what no summary can remove: the system prompt, the protected turns (after their folds) and the previous summary, which the next
422
- // one carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
746
+ // what no summary can remove (besides O): the protected turns (after their folds) and the previous summary, which the next one
747
+ // carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
423
748
  const prevSum = blocks[0]?.kind === "summary" ? blocks[0].tokens : 0;
424
- const floor = o.sys + prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
749
+ const floor = prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
425
750
  const target = Math.max(coldCap, floor + SUMMARY_FLOOR);
426
- const atRoom = room !== null && k * ctxAfterFolds >= room;
751
+ const atRoom = room !== null && real(ctxAfterFolds) >= room;
427
752
  const sumTrigger: "cold" | "valve" | null = ctxAfterFolds > target && (!o.noSummary || atRoom) ? trig : null;
428
753
  let cutIdx: number | null = null;
429
754
  let prefixTokens = 0;
430
755
  let summaryTokensPlanned = 0;
431
- if (sumTrigger) {
432
- const limit = blocks.findIndex((b) => protectedTurn(b));
756
+ let ahead = false;
757
+ const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
758
+ const pre: number[] = [];
759
+ const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
760
+ // Block index of every fold: the folds a cut voids are those before it
761
+ const foldAt = folds.map((t) => blocks.findIndex((b) => b.entryId === t.entryId));
762
+ /** The whole plan's law (F10): B is the context as it is (or as the slid valve saw it), A what the plan leaves, `call` the summary call
763
+ * a cut makes; `untouched` is the content before the earliest edit (ledger only). A cold plan with some chance of a warm cache is
764
+ * priced by expectation (no prices: a cold plan always fires). T = A (the whole post-edit context, O included, as tuned: the cheaper
765
+ * A - O fires warm edits earlier and loses items offline). */
766
+ const planGate = (after: number, untouched: number, call: number, planned: boolean): boolean =>
767
+ !(warmLike || (planned && law.pr)) || gate("plan", Math.max(o.valveB ?? real(ctxEst), real(after)), real(after), room, call, real(after), c * (after - untouched));
768
+ /** The cut for cut points minCut..maxCut: the first that brings the context to the cap, else the deepest; gated = the summary law
769
+ * (no prices: the legacy gain rule) must pass. When the minimal cut's summary or the whole plan with it is refused, the deeper cuts
770
+ * are tried in turn (the gain grows as D^2, the call only with the prefix read). The whole plan's law prices the summary call too.
771
+ * Applies the first cut both laws accept (folds in the prefix are void) and returns true, or changes nothing. */
772
+ const summarise = (maxCut: number, gated: boolean): boolean => {
433
773
  const cuts: number[] = [];
434
- const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
435
- for (let i = minCut; i < blocks.length; i++) {
436
- if (limit >= 0 && i > limit) break;
437
- if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
438
- }
439
- const pre: number[] = [];
774
+ for (let i = minCut; i < blocks.length && i <= maxCut; i++) if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
775
+ if (!cuts.length) return false;
440
776
  let acc = 0;
441
777
  for (let i = 0; i < blocks.length; i++) {
442
778
  pre[i] = acc;
443
779
  acc += tokOverride.get(i) ?? blocks[i].tokens;
444
780
  }
445
- const total = o.sys + acc;
446
- const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
447
- if (cuts.length) {
448
- let pick = cuts[cuts.length - 1];
449
- for (const c of cuts) if (total - pre[c] + S(pre[c]) <= coldCap) { pick = c; break; }
450
- // cold: the summary call must pay for itself (verdict: call = w X + out S for an uncached call); warm: it is sunk in the plan's law
451
- const X = k * pre[pick], Sr = k * S(pre[pick]);
452
- const gain = !law.pr ? summaryGainOk(pre[pick], S(pre[pick]), total, minGain) : mode === "warm" || gate("summary", X, Sr, null, law.pr.w * X + law.pr.out * Sr, Sr);
453
- if (pre[pick] >= 2 * S(pre[pick]) && gain) {
454
- cutIdx = pick;
455
- prefixTokens = pre[pick];
456
- summaryTokensPlanned = S(pre[pick]);
457
- for (let i = folds.length - 1; i >= 0; i--) {
458
- const bi = blocks.findIndex((b) => b.entryId === folds[i].entryId);
459
- if (bi < pick) { tokOverride.delete(bi); folds.splice(i, 1); } // the prefix is replaced: its folds are void
460
- }
461
- ctxAfterFolds = ctxEst - savings();
781
+ const total = acc;
782
+ let first = cuts.length - 1;
783
+ for (let k = 0; k < cuts.length; k++) if (total - pre[cuts[k]] + S(pre[cuts[k]]) <= coldCap) { first = k; break; }
784
+ for (const pick of cuts.slice(first)) {
785
+ const X = c * pre[pick], Sr = c * S(pre[pick]);
786
+ // the call is the narrative call summary.ts makes (F3, NARRATIVE_PRICE): uncached input of the prefix without the previous
787
+ // summary, which it never reads, + output of at most the narrative budget; without F3 the verdict's w X + out S (the carried
788
+ // summary as output)
789
+ const call = !law.pr ? 0 : f3 ? law.pr.input * c * Math.max(pre[pick] - prevSum, 0) + law.pr.out * c * Math.min(S(pre[pick]), NARRATIVE_MAX) : law.pr.w * X + law.pr.out * Sr;
790
+ if (gated) {
791
+ // the summary call must pay for itself, cold or warm (F1, WARM_SUMMARY_GATE; before, a warm plan skipped the gate)
792
+ const ok = !law.pr ? summaryGainOk(X, Sr, real(total), minGain) : (warmLike && !f1) || gate("summary", X, Sr, null, call, Sr);
793
+ if (!ok) continue;
462
794
  }
795
+ if (!(pre[pick] >= 2 * S(pre[pick]))) continue;
796
+ // what the plan leaves: the context with its folds (those inside the prefix too: the prefix is measured as folded, `pre`) minus
797
+ // what the summary replaces plus the summary. (Counting the voided folds' savings back, as before, overstated it by them.)
798
+ if (!planGate(ctxAfterFolds - (pre[pick] - S(pre[pick])), 0, call, true)) continue;
799
+ cutIdx = pick;
800
+ prefixTokens = pre[pick];
801
+ summaryTokensPlanned = S(pre[pick]);
802
+ for (let i = folds.length - 1; i >= 0; i--) if (foldAt[i] < pick) { tokOverride.delete(foldAt[i]); folds.splice(i, 1); foldAt.splice(i, 1); } // the prefix is replaced: its folds are void
803
+ return true;
463
804
  }
805
+ return false;
806
+ };
807
+ if (sumTrigger) {
808
+ const limit = blocks.findIndex((b) => protectedTurn(b));
809
+ summarise(limit >= 0 ? limit : blocks.length - 1, true);
810
+ }
811
+ if (f2 && cutIdx === null && law.pr && ctxAfterFolds > coldCap && (!o.noSummary || atRoom)) {
812
+ // F2 lookahead: the warm valve's plan at the next user turn (this context, its folds applied, the window slid by one turn). If it
813
+ // would summarise, the summary is written now, at the cold return, where the rewrite costs nothing extra (and at settle it is
814
+ // prepared while the user is away), instead of one turn later as a warm rewrite the user waits for.
815
+ const next = planContext(entries, { ...o, mode: "warm", slide: (o.slide ?? 0) + 1, lookahead: false, pastTtl: false, noSummary: false, trace: undefined, law: { ...law, pWarm: 1 }, valveB: real(ctxAfterFolds) });
816
+ const fa = finalAnswerIdx(blocks, userTurns - 1);
817
+ if (next && next.cutIdx !== null && fa >= minCut && summarise(fa, false)) ahead = true;
464
818
  }
465
819
  // where the kept part starts among the projected non-system messages (the request-local view drops everything before it)
466
820
  let cutVisibleIdx = -1;
@@ -480,13 +834,17 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
480
834
  }
481
835
  }
482
836
  // the untouched prefix: everything before the earliest edited block (a cut edits from the first block on)
483
- const untouched = untouchedEst(blocks, folds, cutIdx !== null, o.sys);
837
+ const untouched = untouchedEst(blocks, folds, cutIdx !== null);
484
838
  const after = ctxAfterFolds - (cutIdx !== null ? prefixTokens - summaryTokensPlanned : 0);
485
839
  const planned = folds.length > 0 || cutIdx !== null;
486
- // F10: a warm rewrite must pay back; a cold plan with some chance of a warm cache is priced by expectation (no prices: cold always fires).
487
- // T = A (the whole post-edit context); the suffix after the earliest edit goes to the ledger only.
488
- if ((mode === "warm" || (planned && law.pr)) && !gate("plan", k * ctxEst, k * after, room, 0, k * after, k * (after - untouched))) return null;
489
- return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger, k, ctxTokens: k * ctxEst, ctxAfterFolds: k * ctxAfterFolds, userTurns, cutVisibleIdx, cutKeptFirstMsg };
840
+ // a plan with a cut passed the whole plan's law with its call inside summarise; folds only (no summary wanted, or none the laws
841
+ // accepted: the deeper cuts and the cut-free plan both stand) is judged here
842
+ if (cutIdx === null && !planGate(after, untouched, 0, planned)) {
843
+ // a cold plan with some chance of a warm cache: the plan a warm cache would make may pay where this one does not
844
+ if (mode === "cold" && !o.warmStyle && law.pr && law.pWarm > 0 && law.pWarm < 0.5) return planContext(entries, { ...o, warmStyle: true });
845
+ return null;
846
+ }
847
+ return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger: cutIdx !== null ? (sumTrigger ?? trig) : sumTrigger, scale: o.scale ?? UNIT, ctxTokens: real(ctxEst), ctxAfterFolds: real(ctxAfterFolds), userTurns, cutVisibleIdx, cutKeptFirstMsg, ...(o.fileRecall ? { fileRecall: o.fileRecall } : {}), ...(ahead ? { ahead } : {}) };
490
848
  }
491
849
 
492
850
  const sameMsg = (a: Any, b: Any): boolean =>
@@ -532,6 +890,11 @@ export function collapseSystem(messages: Any[]): Any | undefined {
532
890
  * message, the summary, then the kept non-system messages. null = nothing to apply. A cut whose kept boundary cannot be
533
891
  * located is dropped (folds still apply); an inconsistent split is never sent.
534
892
  */
893
+ /** Fold targets of a plan made on the view before a native compaction, renumbered for the request Pi builds after it: Pi's summary
894
+ * message first, then the kept part (positions among the non-system messages, as in applyPlanToMessages). */
895
+ export const rebaseFolds = (folds: FoldTarget[], cutVisibleIdx: number): FoldTarget[] =>
896
+ folds.map((t) => ({ ...t, visIdx: t.visIdx >= 0 && cutVisibleIdx >= 0 && t.visIdx >= cutVisibleIdx ? t.visIdx - cutVisibleIdx + 1 : -1 }));
897
+
535
898
  export function applyPlanToMessages(messages: Any[], plan: RunPlan | null): { messages: Any[]; droppedCut: boolean; applied: Set<string> } | null {
536
899
  if (!plan || plan.persisted || (!plan.folds.length && !plan.cut)) return null;
537
900
  const vis: number[] = []; // indices of the non-system messages