pi-zip 0.2.7 → 0.3.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/plan.ts CHANGED
@@ -2,9 +2,10 @@
2
2
  import { classifyRecoverability, type Recover } from "./classify.ts";
3
3
  import { handleFor, makePlaceholderFor, pickKeyLines, RECALL_TOOL, shortArgs } from "./placeholder.ts";
4
4
  import { createHash } from "node:crypto";
5
- import { type Any, PRODUCT, clamp, envInt, textOf, tok4, tokensOf } from "./util.ts";
5
+ import { type Any, COMMIT_CUSTOM, PRODUCT, clamp, envInt, piTokensOf, textOf, tok4, tokensOf } from "./util.ts";
6
6
  import { PH_MARK } from "./placeholder.ts";
7
7
  import { MIN_EXPECT, type Prices } from "./learn.ts";
8
+ import { commandNamesRecall, parseRecallPath, type FileRecall } from "./recallfile.ts";
8
9
 
9
10
  /** Default: when a COLD plan (P(warm) < 0.5) is still above the cap, also fold REREADABLE outputs of the previous user turn, biggest
10
11
  * first. A warm plan does so only at a user return after the declared TTL (PlanOpts.pastTtl): the very returns where the TTL rule
@@ -22,7 +23,21 @@ export const RELAX_PREV_TURN = true;
22
23
  * constant (largest saving with lost <= prod in every stratum); the old rereadable-only filter cost 1.026 / 1.018x at -0.6 lost items. */
23
24
  export const INTURN_AGE = 60;
24
25
 
25
- const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised; folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
26
+ const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised (but see LOOKAHEAD); folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
27
+
28
+ /** F1: a warm (valve) summary passes the same summary gate as a cold one, the call priced in (before: a warm plan skipped it, so a
29
+ * summary the cold gate had just refused was written one request later anyway). PlanOpts.warmSummaryGate. */
30
+ export const WARM_SUMMARY_GATE = true;
31
+ /** F2: a cold plan that does not summarise looks one user turn ahead: if the warm valve would summarise at the next user turn, once the
32
+ * protected window has slid past the previous turn, the cold plan summarises now (the rewrite is free now, not then), its cut allowed
33
+ * inside the previous user turn up to that turn's final assistant answer, which stays verbatim with everything after it. PlanOpts.lookahead. */
34
+ export const LOOKAHEAD = false;
35
+ /** F3: the summary gates price the call the code makes (summary.ts narrative): uncached input of the prefix the narrative reads (the
36
+ * previous summary is not sent: its sections are carried forward by code) plus output of at most the narrative budget, instead of
37
+ * w x the prefix + out x the whole planned summary (which counts the carried-forward previous summary as model output). PlanOpts.narrativePrice. */
38
+ export const NARRATIVE_PRICE = true;
39
+ /** The narrative's token budget (chars/4), summary.ts buildCut: clamp(planned - skeleton, 300, NARRATIVE_MAX). */
40
+ export const NARRATIVE_MAX = 4000;
26
41
  const SUMMARY_FLOOR = 1000;
27
42
  const SUMMARY_CAP = 8000;
28
43
  const SUMMARY_RATIO = 0.1;
@@ -55,6 +70,56 @@ export function legacyValve(before: number, after: number, room: number | null):
55
70
  }
56
71
 
57
72
  export const PI_KEEP_RECENT = 20_000; // Pi's compaction keepRecentTokens default
73
+
74
+ /** Pi's compaction keepRecentTokens for this model: compaction.modelOverrides["provider/id"], else compaction.keepRecentTokens, else 20000. */
75
+ export function keepRecentTokensFor(settings: Any, model: Any): number {
76
+ const c = settings?.compaction;
77
+ const key = model ? `${model.provider}/${model.id}` : "";
78
+ for (const v of [c?.modelOverrides?.[key]?.keepRecentTokens, c?.keepRecentTokens]) if (typeof v === "number" && Number.isFinite(v) && v >= 0) return v;
79
+ return PI_KEEP_RECENT;
80
+ }
81
+
82
+ const CUT_POINT_ROLES = new Set(["user", "assistant", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
83
+ const TURN_START_ROLES = new Set(["user", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
84
+
85
+ /**
86
+ * Would Pi's compact() find something to compact in this projection (the branch as it is now), or throw "Nothing to compact (session
87
+ * too small)"? A replica of the emptiness test of Pi's prepareCompaction (core/compaction/compaction.js, not exported): Pi cuts where
88
+ * keepRecent of its chars/4 tokens, counted from the end, begin, at the next cut point; the conversation before that cut (or before the
89
+ * turn it splits) must not be empty. Pi's own extra moves of the cut (over entries without messages, past recovery omissions) never
90
+ * empty it. The caller checks "Already compacted" (the newest branch entry is a compaction) itself.
91
+ */
92
+ export function piCanCompact(entries: Any[], keepRecent: number): boolean {
93
+ const msgs = (e: Any): Any[] => (Array.isArray(e?.messages) ? e.messages : []);
94
+ const isComp = (e: Any) => e?.sourceEntry?.type === "compaction";
95
+ const prev = entries.findIndex((e) => isComp(e) && msgs(e).length > 0);
96
+ const start = prev >= 0 ? prev + 1 : 0, end = entries.length;
97
+ const cuts: number[] = [];
98
+ for (let i = start; i < end; i++) if (!isComp(entries[i]) && msgs(entries[i]).some((m) => CUT_POINT_ROLES.has(m?.role))) cuts.push(i);
99
+ if (!cuts.length) return false;
100
+ let cut = cuts[0], acc = 0;
101
+ for (let i = end - 1; i >= start; i--) {
102
+ const t = msgs(entries[i]).reduce((a, m) => a + piTokensOf(m), 0);
103
+ if (!t) continue;
104
+ acc += t;
105
+ if (acc >= keepRecent) { cut = cuts.find((c) => c >= i) ?? cuts[cuts.length - 1]; break; }
106
+ }
107
+ while (cut > start && !isComp(entries[cut - 1]) && !msgs(entries[cut - 1]).length) cut--;
108
+ if (!entries[cut]?.sourceEntry?.id) return false;
109
+ const startsTurn = (e: Any) => !isComp(e) && msgs(e).some((m) => TURN_START_ROLES.has(m?.role));
110
+ let turn = -1;
111
+ if (!startsTurn(entries[cut])) for (let i = cut; i >= start; i--) if (startsTurn(entries[i])) { turn = i; break; }
112
+ const conversation = (from: number, to: number) => entries.slice(from, to).some((e) => !isComp(e) && msgs(e).some((m) => m?.role !== "system"));
113
+ return turn >= 0 ? conversation(start, turn) || conversation(turn, cut) : conversation(start, cut);
114
+ }
115
+
116
+ /** The projection a run plan would have seen without pi-zip's native compaction: `pre` (taken right before it) plus what was appended
117
+ * after the compaction entry (the prompt and anything sent with it). null when that compaction is no longer the head of `now`. */
118
+ export function spliceNative(pre: Any[], now: Any[], compactionId: string): Any[] | null {
119
+ if (!compactionId || now[0]?.sourceEntry?.id !== compactionId) return null;
120
+ const known = new Set(pre.map((e) => e?.sourceEntry?.id).filter((x) => typeof x === "string"));
121
+ return [...pre, ...now.slice(1).filter((e) => !known.has(e?.sourceEntry?.id))];
122
+ }
58
123
  export const G0 = 2_500; // growth per request before the session's own EWMA has data
59
124
 
60
125
  /** What the law knows at a decision: prices (null = legacy rule), growth per request g (real tokens), P(the cache is warm). */
@@ -88,58 +153,249 @@ export function lawTerms(B: number, A: number, P: number, room: number | null, p
88
153
  export const editAllowed = (B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A): boolean =>
89
154
  lawTerms(B, A, P, room, pr, g, call, T).ok;
90
155
 
91
- // Token scale. Every size in this file is a chars/4 estimate, and chars/4 undercounts real tokens (JSON-heavy tool calls, code,
92
- // identifiers, tool definitions that are not in the text at all). `k` = real tokens per estimated token; every limit (cold cap,
93
- // compaction room, min gain) is compared against k x estimate, so they mean REAL tokens. k is read from the branch itself (the
94
- // usage of the newest assistant message), so it survives a restart and needs no stored state. With no usage to read, DEFAULT_K
95
- // applies: 1.7 is the ratio measured on recorded coding sessions (real first-request context / chars/4 estimate: 1.72 after cold
96
- // returns, 1.71 for previous-prompt peaks), and a too-high k only folds a little deeper, a too-low k leaves the context over the cap.
97
- export const DEFAULT_K = 1.7;
98
- export const K_MIN = 1; // an estimate above the real count is not trusted: never scale down
99
- export const K_MAX = 2.5;
100
-
101
- export interface Calibration {
102
- k: number;
103
- real: number; // input + cacheRead + cacheWrite of the request that produced the newest usable assistant message (0 = none)
104
- est: number; // estimate of what that request carried: system prompt + the view blocks before that message
105
- source: "usage" | "default";
156
+ // Token scale. Every size in this file is a chars/4 estimate of the CONTENT (Pi's own rule); the provider counts real tokens, and the
157
+ // two are not proportional: real = O + c x content.
158
+ // O what every request carries and no edit can shrink: system prompt, tool schemas, the provider's own framing (2-4K on a bare Pi,
159
+ // 29-55K with a large tool set);
160
+ // c real tokens per estimated content token (0.7-1.1 GLM / GPT, 1.5-2.2 Claude; more on CJK-heavy content).
161
+ // pi-zip <= 0.2.9 used ONE ratio, k = real / (system prompt + content) clamped to [1, 2.5]: with O a third of the context k sat on its
162
+ // clamp, and every size after an edit came out 10-15K too small (research round5/affine.md). Both are read from the branch on every
163
+ // decision, so a restart or `pi -p` calibrates exactly like a long-lived process, and nothing is stored:
164
+ // O 1. the cacheRead of the request that first carried the newest summary: everything after the system prompt and tools changed
165
+ // there, so what the provider still read from its cache is exactly O (when the cache was warm at all);
166
+ // 2. else the session's first request minus C0 x its content (an opening prompt is small next to O);
167
+ // 3. else R0 x the chars/4 of the system prompt and the JSON of the declared tools.
168
+ // An O measured under another system prompt or tool set moves by R0 x the change of that chars/4 size (MCP tools come and go).
169
+ // c (real - O) / content of the newest response, clamped to [C_MIN, C_MAX] (a guard for a tiny content estimate; recorded sessions
170
+ // span 0.7-3.9); C0 before the first response. A response followed by edits it did not carry (a warm-valve edit is persisted at
171
+ // turn_end and first sent with the next request: that response's prompt did not shrink) is counted with the originals.
172
+ // Both are per model: only responses of the model the next request goes to count. After a switch, before its first response,
173
+ // O = R0 x the system state and c = C0; after it, with no O evidence of its own, O = c x X, X = O / c of the previous model.
174
+ export const C0 = 1.5;
175
+ export const C_MIN = 0.5;
176
+ export const C_MAX = 4;
177
+ export const R0 = 2;
178
+
179
+ /** The system prompt and the declared tools (Pi API fallback for a transcript that records no system message). */
180
+ export interface SysInfo { text: string; tools: Any[] }
181
+ /** chars/4 of the system prompt plus the JSON of the declared tool definitions (name, description, parameters). */
182
+ export const sysEst = (s: SysInfo): number =>
183
+ tok4(s.text) + (s.tools.length ? tok4(JSON.stringify(s.tools.map((t: Any) => ({ name: t?.name, description: t?.description, parameters: t?.parameters })))) : 0);
184
+
185
+ export interface Scale {
186
+ O: number; // real tokens every request carries before the content (system prompt, tools, framing)
187
+ c: number; // real tokens per estimated (chars/4) content token
188
+ real: number; // input + cacheRead + cacheWrite of the newest usable response (0 = none)
189
+ est: number; // estimated content that request carried
190
+ source: "usage" | "default"; // where c comes from
191
+ oSource: "summary" | "first" | "model" | "default"; // where O comes from ("model": another model's O in content units, X = O' / c')
192
+ }
193
+ /** Sizes taken as real (tests, and the ledger of a plan made without calibration). */
194
+ export const UNIT: Scale = { O: 0, c: 1, real: 0, est: 0, source: "default", oSource: "default" };
195
+ /** Real tokens of a context whose content is estimated at `content`. */
196
+ export const sizeOf = (s: Pick<Scale, "O" | "c">, content: number): number => s.O + s.c * content;
197
+
198
+ /** A response pi-ai drops before sending (an error or an abort): it carries no usable usage and adds nothing to later requests. */
199
+ const dropped = (m: Any): boolean => m?.role === "assistant" && (m.stopReason === "error" || m.stopReason === "aborted");
200
+ /** chars/4 a message adds to every later request: tokensOf, except 0 for an assistant message pi-ai drops before sending. */
201
+ export const wireTokensOf = (m: Any): number => (dropped(m) ? 0 : tokensOf(m));
202
+ /** Prompt size the provider reported for a response that really ran (0 = none, an error or an abort). */
203
+ const promptOf = (m: Any): number => {
204
+ const u = m?.role === "assistant" && !dropped(m) ? m.usage : null;
205
+ return u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
206
+ };
207
+ /** "provider/model" of a response ("" = not recorded: such a response counts for any model). */
208
+ const respKey = (m: Any): string => (m?.provider || m?.model ? `${m.provider ?? ""}/${m.model ?? ""}` : "");
209
+ /** chars/4 of the system state the transcript records in `msgs` (-1: none recorded). */
210
+ function sysOf(msgs: Any[]): number {
211
+ const s = collapseSystem(msgs);
212
+ return s ? sysEst({ text: [textOf(s.content), ...Object.values<string>(s.sections ?? {})].filter((x) => x).join("\n\n"), tools: s.toolsAdded ?? [] }) : -1;
213
+ }
214
+ /** The session's first response from model `key` on the branch: its prompt, the content before it and the system state it was sent with. */
215
+ function firstRequest(branch: Any[], key: string): { real: number; est: number; S: number } | null {
216
+ const sys: Any[] = [];
217
+ let est = 0;
218
+ for (const e of branch) {
219
+ const m = e?.type === "message" ? e.message : e?.type === "custom_message" ? { role: "custom", content: e.content } : null;
220
+ if (!m) continue;
221
+ if (m.role === "system") { sys.push(m); continue; }
222
+ const real = promptOf(m), k = respKey(m);
223
+ if (real > 0 && (!key || !k || k === key)) return { real, est, S: sysOf(sys) };
224
+ est += wireTokensOf(m);
225
+ }
226
+ return null;
106
227
  }
107
228
 
108
- /**
109
- * k from the branch: the newest assistant message with usage says how many real tokens its request carried; the estimate of the
110
- * view blocks before it (the projection, with the folds and cut as persisted, which are exactly what that request carried, I1)
111
- * says how many we would have guessed. Only the INPUT side is used: that message's own output is not part of the request it
112
- * answered, and thinking tokens may or may not be sent back, so including output would add noise to the ratio.
113
- * Skipped: errored messages, messages without input usage, and messages older than a compaction Pi made (not ours): their
114
- * request carried text the projection no longer has.
115
- */
116
- export function calibrate(entries: Any[], sys = 0): Calibration {
117
- const none: Calibration = { k: DEFAULT_K, real: 0, est: 0, source: "default" };
118
- const blocks = buildBlocks(entries);
119
- let foreignCompactionMs = 0;
120
- for (const pe of entries) {
121
- const src = pe.sourceEntry;
122
- if (src?.type === "compaction" && src.details?.by !== PRODUCT) foreignCompactionMs = Math.max(foreignCompactionMs, Date.parse(src.timestamp) || 0);
229
+ /** chars/4 of the context the request behind the response `id` carried, rebuilt from the branch the way Pi projects it: the newest
230
+ * compaction before the response, the entries it keeps, everything after it, and the edits persisted before the response, plus the
231
+ * edits of a run plan persisted right after it (their commit entry says the response itself carried them). null: `id` is not on the
232
+ * branch. Used when a compaction written after the response (Pi's own, another extension's) removed that request's content from the
233
+ * projection, so the projection cannot say what the response measured. */
234
+ function requestEst(branch: Any[], id: string): number | null {
235
+ const at = branch.findIndex((e) => e?.id === id);
236
+ if (at < 0) return null;
237
+ const pre = branch.slice(0, at);
238
+ const edit = new Map<string, Any>();
239
+ for (const e of pre) if (e?.type === "context_edit") edit.set(e.targetId, e.replacement);
240
+ for (let i = at + 1; i < branch.length && !(branch[i]?.type === "message" || branch[i]?.type === "compaction"); i++) {
241
+ const e = branch[i];
242
+ if (e?.type === "custom" && e.customType === COMMIT_CUSTOM) {
243
+ if (e.data?.carried) for (let j = at + 1; j < i; j++) if (branch[j]?.type === "context_edit") edit.set(branch[j].targetId, branch[j].replacement);
244
+ break;
245
+ }
123
246
  }
124
- const before: number[] = []; // estimate of blocks[0..i)
125
- let acc = 0;
126
- for (const b of blocks) { before.push(acc); acc += b.tokens; }
127
- for (let i = blocks.length - 1; i >= 0; i--) {
128
- if (blocks[i].kind !== "assistant") continue;
129
- const m = blocks[i].raw ?? blocks[i].msg;
130
- const u = m?.usage;
131
- const real = u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
132
- if (!(real > 0) || m.stopReason === "error") continue;
133
- if (foreignCompactionMs && typeof m.timestamp === "number" && m.timestamp < foreignCompactionMs) return none;
134
- const est = sys + before[i];
135
- if (!(est > 0)) return none;
136
- return { k: clamp(real / est, K_MIN, K_MAX), real, est, source: "usage" };
247
+ let ci = -1;
248
+ pre.forEach((e, i) => { if (e?.type === "compaction") ci = i; });
249
+ const kept = ci < 0 ? pre : [pre[ci], ...pre.slice(0, ci).slice(Math.max(0, pre.findIndex((e) => e?.id === pre[ci].firstKeptEntryId))).filter((e) => !(e?.type === "message" && e.message?.role === "system")), ...pre.slice(ci + 1)];
250
+ let est = 0;
251
+ for (const e of kept) {
252
+ if (e?.type === "compaction" || e?.type === "branch_summary") est += tokensOf({ role: "compactionSummary", summary: e.summary });
253
+ else if (e?.type === "custom_message") est += tokensOf({ role: "custom", content: edit.has(e.id) ? edit.get(e.id)?.content : e.content });
254
+ else if (e?.type === "message" && e.message) {
255
+ const rep = edit.get(e.id);
256
+ if (rep === null) continue;
257
+ const m = rep && e.message.role !== "system" ? { ...e.message, content: typeof rep.content === "string" && e.message.role !== "user" ? [{ type: "text", text: rep.content }] : rep.content } : e.message;
258
+ est += wireTokensOf(m);
259
+ }
137
260
  }
138
- return none;
261
+ return est;
262
+ }
263
+
264
+ /** O and c from the projection (`entries`) and, for the session's first request, the branch; `sys` stands in for a transcript without
265
+ * system messages. `model` = "provider/id" of the model the next request goes to (default: the newest response's): O and c are
266
+ * properties of the provider's tokenizer, so only that model's responses are evidence. With none, O = R0 x the system state and
267
+ * c = C0; with responses but no O evidence of its own (a model switch mid-session), O keeps the size another model measured in
268
+ * content units, X = O' / c' (the system prompt and tools tokenize like the content around them): real = c (X + E). */
269
+ export function calibrate(entries: Any[], sys?: SysInfo, branch: Any[] = [], model?: string): Scale {
270
+ type Resp = { at: number; est: number; real: number; read: number; nSys: number; ts: number; key: string; id: string };
271
+ const resp: Resp[] = [];
272
+ const sysMsgs: Any[] = [];
273
+ const edits: { at: number; target: string; foreign: boolean }[] = [];
274
+ const marks: { at: number; carried: boolean }[] = []; // pi-zip's commit entries: who first carried the edits right before them (run.ts commit)
275
+ const orig = new Map<string, { at: number; d: number }>(); // edited entries: tokens their originals add back
276
+ const withMsgs: number[] = [0]; // withMsgs[k]: entries before k that carry messages (a commit group has none inside it)
277
+ let est = 0, sysAt = 0; // sysAt: responses before the newest system message
278
+ entries.forEach((pe, at) => {
279
+ const src = pe.sourceEntry, msgs: Any[] = pe.messages ?? [];
280
+ withMsgs.push(withMsgs[at] + (msgs.length ? 1 : 0));
281
+ // an edit with a replacement that is not our placeholder is another extension's (never carried by a request of ours); one without
282
+ // (a bare entry) is judged by the size heuristics below
283
+ if (src?.type === "context_edit") edits.push({ at, target: src.targetId, foreign: "replacement" in src && !textOf(src.replacement?.content).startsWith(PH_MARK) });
284
+ if (src?.type === "custom" && src.customType === COMMIT_CUSTOM) marks.push({ at, carried: !!src.data?.carried });
285
+ if (src?.type === "message" && msgs.length === 1 && msgs[0] !== src.message) orig.set(src.id, { at, d: wireTokensOf(src.message) - wireTokensOf(msgs[0]) });
286
+ for (const m of msgs) {
287
+ if (m?.role === "system") { sysMsgs.push(m); sysAt = resp.length; continue; }
288
+ const real = promptOf(m);
289
+ if (real > 0) resp.push({ at, est, real, read: Number(m.usage.cacheRead) || 0, nSys: sysMsgs.length, ts: Number(m.timestamp) || 0, key: respKey(m), id: src?.id ?? "" });
290
+ est += wireTokensOf(m);
291
+ }
292
+ });
293
+ const sAt = (n: number) => sysOf(sysMsgs.slice(0, n)); // the system state after the first n system messages (computed only where needed)
294
+ const S = sAt(sysMsgs.length);
295
+ const Snow = S >= 0 ? S : sys ? sysEst(sys) : 0;
296
+ const head = entries[0]?.sourceEntry?.type === "compaction" ? entries[0].sourceEntry : null;
297
+ const headMs = head ? Date.parse(head.timestamp) || 0 : 0;
298
+ // a pi-zip summary persisted as a turn_end draft was request-local first (a run plan carries it before its entry exists); one that
299
+ // went in through Pi's compaction before the run (details.via "native") is carried after its entry, exactly like Pi's own
300
+ const local = !!head && head.details?.by === PRODUCT && head.details?.via !== "native";
301
+ // Was the request behind this response sent after the head compaction entry was written? The branch's order says (timestamps tie when
302
+ // a compaction follows a response in the same millisecond); without the branch the timestamps do.
303
+ const pos = new Map<string, number>();
304
+ if (head) branch.forEach((e, i) => { if (typeof e?.id === "string") pos.set(e.id, i); });
305
+ const afterHead = (r: Resp): boolean => {
306
+ const h = pos.get(head?.id), i = pos.get(r.id);
307
+ return h !== undefined && i !== undefined ? i > h : r.ts > headMs;
308
+ };
309
+ const opening = (r: { real: number; est: number } | null | undefined) => !!r && C0 * r.est <= 0.1 * r.real; // content small next to O
310
+
311
+ /** The first commit entry of the group of pi-zip edits that starts at entry `at` (none when a message comes first). */
312
+ const markOf = (at: number) => marks.find((m) => m.at > at && withMsgs[m.at] - withMsgs[at + 1] === 0);
313
+ /** Index of the newest response in `resp` before entry `at`. */
314
+ const respBefore = (at: number) => resp.reduce((k, r, i) => (r.at < at ? i : k), -1);
315
+ /** Tokens the originals add back for response i: edits persisted after it that its own request did not carry (the projection counts
316
+ * their placeholders). A commit entry says who carried them (a run plan: the response before the entry; a valve plan: the next one);
317
+ * with none (a session of an older version) the sizes decide, for the newest response only (`legacy`). */
318
+ const addBack = (i: number, legacy: boolean): number => {
319
+ let d = 0;
320
+ for (const x of edits) {
321
+ const t = orig.get(x.target);
322
+ if (x.at <= resp[i].at || !t || t.at >= resp[i].at) continue;
323
+ const m = x.foreign ? undefined : markOf(x.at);
324
+ if (m ? m.carried && respBefore(m.at) === i : !x.foreign && legacy) continue;
325
+ d += t.d;
326
+ }
327
+ return d;
328
+ };
329
+
330
+ const fit = (want: string, transfer: boolean): Scale => {
331
+ const mine = (r: Resp | undefined): r is Resp => !!r && (!want || !r.key || r.key === want);
332
+ // evidence for O, the newest wins (it needs the smallest correction for system changes since)
333
+ const ev: { at: number; O: number; S: number; src: Scale["oSource"] }[] = [];
334
+ const f = branch.length ? firstRequest(branch, want) : head ? null : resp.find(mine); // the session's first request
335
+ if (f && opening(f)) ev.push({ at: -1, O: f.real - C0 * f.est, S: "S" in f ? f.S : sAt(f.nSys), src: "first" });
336
+ const ci = resp.findIndex((r, i) => i >= sysAt && mine(r)), cur = resp[ci]; // the first response sent with the current system prompt and tools
337
+ // its content as its request carried it: edits persisted after it (a model switch, say, then a run plan's folds) are in `est`
338
+ const curEst = cur ? cur.est + addBack(ci, false) : 0;
339
+ if (cur && (!head || afterHead(cur)) && opening({ real: cur.real, est: curEst })) ev.push({ at: ci, O: cur.real - C0 * curEst, S, src: "first" });
340
+ if (head) {
341
+ // the summary at the head was first carried by the first response after its entry, or, for pi-zip's own request-local summary
342
+ // (a run plan), by the last one before it; Pi's, another extension's or a native pi-zip compaction is never sent before its entry
343
+ const after = resp.findIndex((r) => afterHead(r)), ours = local;
344
+ for (const i of after < 0 ? (ours ? [resp.length - 1] : []) : ours ? [after - 1, after] : [after]) {
345
+ const r = resp[i], prev = resp[i - 1];
346
+ if (mine(r) && r.read > 0 && r.read < 0.8 * r.real && (!prev || r.read < 0.8 * prev.real)) { ev.push({ at: i, O: r.read, S: sAt(r.nSys), src: "summary" }); break; }
347
+ }
348
+ }
349
+ const best = ev.sort((a, b) => b.at - a.at)[0];
350
+ let O = best ? Math.max(0, best.O + (best.S >= 0 && S >= 0 ? R0 * (Snow - best.S) : 0)) : R0 * Snow, oSource: Scale["oSource"] = best?.src ?? "default";
351
+ // c from the newest response of this model
352
+ let n = resp.length - 1;
353
+ while (n >= 0 && !mine(resp[n])) n--;
354
+ if (n < 0) return { O, c: C0, real: 0, est: 0, source: "default", oSource };
355
+ const N = resp[n];
356
+ let estN = N.est;
357
+ const cutLater = !!head && !afterHead(N);
358
+ const later = edits.filter((x) => x.at > N.at);
359
+ const rebuilt = cutLater && !local && !!N.id ? requestEst(branch, N.id) : null; // what its request carried: the compaction after it removed that from the projection
360
+ if (rebuilt !== null && rebuilt > 0) estN = rebuilt;
361
+ else if (later.length || cutLater) {
362
+ // carried (a run plan): the prompt shrank, or a read stopped early; a cache miss alone (no read at all) proves nothing. The
363
+ // commit entries of pi-zip's edits say it outright; the sizes decide only for edits without one (older sessions).
364
+ // Pi's own compaction is never request-local.
365
+ let p = n - 1;
366
+ while (p >= 0 && !mine(resp[p])) p--;
367
+ const P = resp[p];
368
+ const heur = !P || N.real < P.real || (N.read > 0 && N.read < 0.8 * P.real);
369
+ const carried = !(cutLater && !local) && (cutLater && typeof head.details?.carried === "boolean" ? head.details.carried : heur);
370
+ if (!carried && cutLater) return { O, c: C0, real: N.real, est: 0, source: "default", oSource }; // what it carried is summarised away
371
+ estN += addBack(n, heur && carried);
372
+ }
373
+ if (!best && transfer && estN > 0) {
374
+ // no O of its own: the newest other model's O in content units
375
+ // (a compaction may have taken the other model's responses out of the projection: the branch still has them)
376
+ const other = [...resp].reverse().find((r) => r.key && r.key !== want)?.key
377
+ ?? [...branch].reverse().map((e) => (e?.type === "message" && promptOf(e.message) > 0 ? respKey(e.message) : "")).find((k) => k && k !== want);
378
+ const o = other ? fit(other, false) : null;
379
+ if (o && o.oSource !== "default" && o.c > 0) {
380
+ const X = o.O / o.c, c = clamp(N.real / (X + estN), C_MIN, C_MAX);
381
+ return { O: c * X, c, real: N.real, est: estN, source: "usage", oSource: "model" };
382
+ }
383
+ }
384
+ return { O, c: estN > 0 ? clamp((N.real - O) / estN, C_MIN, C_MAX) : C0, real: N.real, est: estN, source: estN > 0 ? "usage" : "default", oSource };
385
+ };
386
+ return fit(model || [...resp].reverse().find((r) => r.key)?.key || "", true);
139
387
  }
140
388
 
389
+ /** The cold cap's target in real tokens of the WHOLE context (system prompt and tools included, as the cap was tuned): max(cap, O + cap/2),
390
+ * never above Pi's compaction room. While O is under half the cap this is the plain whole-context cap. Above that the target is O plus
391
+ * half the cap of conversation: never at or below O, where no edit reaches it (0.2.9's 40K with O = 29-55K summarised at every cold
392
+ * return), and not 40K of conversation on top of O either (the cap on content alone cost +13% live on a 36K prefix, with no quality
393
+ * difference measured). Offline it never loses more items than the whole-context cap (research round5/affine.md section 11). */
394
+ export const CAP_CONTENT_MIN = 0.5;
395
+ export const coldTarget = (cap: number, O: number, room: number | null): number => Math.min(Math.max(cap, O + CAP_CONTENT_MIN * cap), room ?? Infinity);
396
+
141
397
  export const settings = () => ({
142
- coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many tokens
398
+ coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many real tokens of the whole context, but never below O + half of it (coldTarget)
143
399
  foldMin: envInt("FOLD_MIN", 500), // outputs below this many tokens are never folded
144
400
  keepLines: envInt("KEEP_LINES", 8),
145
401
  minGain: envInt("MIN_GAIN", 10_000), // legacy rule only (no prices): a summary must remove at least max(minGain, 15% of the context)
@@ -203,7 +459,7 @@ export function buildBlocks(contextEntries: Any[], steerIds?: Set<string>): Bloc
203
459
  for (const c of msg.content ?? []) if (c?.type === "toolCall") issued.set(c.id, asst);
204
460
  }
205
461
  asstOf[blocks.length] = kind === "toolResult" ? (issued.get(msg.toolCallId) ?? asst) : asst;
206
- blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + tokensOf(m), 0), userTurn, edited, ours, age: 0 });
462
+ blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + wireTokensOf(m), 0), userTurn, edited, ours, age: 0 });
207
463
  }
208
464
  for (const b of blocks) if (b.kind === "toolResult") b.age = asst - asstOf[b.idx];
209
465
  return blocks;
@@ -244,6 +500,7 @@ export interface Cut {
244
500
  costUsd: number; // what the narrative model call cost (0 when unknown)
245
501
  usage?: Any;
246
502
  ms: number; // total production time (the narrative model call included)
503
+ ahead?: boolean; // F2: a lookahead cut (inside the previous user turn, before its final answer; guard.ts)
247
504
  }
248
505
 
249
506
  /** What a run sends from its first request on and persists verbatim at turn_end (F8). */
@@ -254,7 +511,7 @@ export interface RunPlan {
254
511
  ctxBefore: number;
255
512
  ctxAfter: number;
256
513
  ms: number;
257
- k: number; // the token scale the plan was made with (stats and notices of this run use the same one)
514
+ scale: Scale; // the token scale the plan was made with (stats and notices of this run use the same one)
258
515
  persisted: boolean;
259
516
  cutVisibleIdx?: number; // where the kept part starts among the projected non-system messages
260
517
  cutKeptFirstMsg?: Any; // ... and that message itself (a disagreeing request view drops the cut)
@@ -262,6 +519,9 @@ export interface RunPlan {
262
519
  applied?: Set<string>; // entry ids whose fold the latest request view actually carried (what turn_end must persist, no more)
263
520
  cutTs?: number; // timestamp of the request-local summary message: one value per run, so every request of the run is identical
264
521
  untouched?: number; // real tokens before the earliest edited block: what the first request carrying the plan can still read from the cache
522
+ waitMs?: number; // how long the user waited for this plan's summary (0: none, or it was ready)
523
+ noticed?: boolean; // its notice is already in the transcript (shown when the first request carrying it went out)
524
+ native?: Cut; // this run's summary went in through Pi's own compaction before the run: no request-local cut, no compaction draft
265
525
  }
266
526
 
267
527
  export interface PlanOpts {
@@ -270,8 +530,7 @@ export interface PlanOpts {
270
530
  trace?: (LawTerms & { where: "summary" | "plan" })[]; // every law evaluation is pushed here (ledger)
271
531
  reserve?: number; // Pi's compaction reserveTokens (default 16384)
272
532
  steerIds?: Set<string>; // user entries that are mid-run steering or follow-up messages, not new user turns
273
- sys: number; // estimated tokens of the system prompt (same chars/4 scale as the blocks; tool definitions are covered by k)
274
- k?: number; // real tokens per estimated token (calibrate); default 1 = the estimates are taken as real
533
+ scale?: Scale; // real = O + c x content estimate (calibrate); default UNIT = the estimates are taken as real
275
534
  cwd: string;
276
535
  coldCap?: number;
277
536
  foldMin?: number;
@@ -284,7 +543,14 @@ export interface PlanOpts {
284
543
  inturnAge?: number; // default settings().inturnAge (PI_ZIP_INTURN_AGE, 60); 0 = outputs of the protected turns never fold on age
285
544
  noSummary?: boolean; // a summary was already made while this cache stayed warm: no second one unless the context is at/above the compaction room
286
545
  rereadOnly?: boolean; // zip_recall is not available to the model: fold only outputs that can be re-read (classify "rereadable")
546
+ fileRecall?: FileRecall; // zip_recall is hidden but read, grep or bash is active: fold as in full mode, placeholders name the recall file
287
547
  pastTtl?: boolean; // a user return after the declared TTL: a warm plan may relax into the previous user turn too (RELAX_PREV_TURN)
548
+ warmSummaryGate?: boolean; // F1, default WARM_SUMMARY_GATE
549
+ lookahead?: boolean; // F2, default LOOKAHEAD (cold plans only)
550
+ narrativePrice?: boolean; // F3, default NARRATIVE_PRICE
551
+ slide?: number; // internal (F2): plan as if this many more user turns had started (the protected window slid)
552
+ valveB?: number; // internal (F2): the real context the slid warm plan's law sees (the cold plan's folds already applied)
553
+ warmStyle?: boolean; // internal: a cold plan at 0 < P(warm) < 0.5 that the law refused is planned once more the way a warm plan is (no relax folds unless pastTtl, the summary gated)
288
554
  }
289
555
 
290
556
  export interface PlanResult {
@@ -293,21 +559,33 @@ export interface PlanResult {
293
559
  folds: FoldTarget[];
294
560
  cutIdx: number | null;
295
561
  firstKeptEntryId: string | null;
296
- prefixTokens: number; // estimate units (x k = real), like summaryTokensPlanned
562
+ prefixTokens: number; // content estimate units (x c = real), like summaryTokensPlanned
297
563
  summaryTokensPlanned: number;
298
564
  sumTrigger: "cold" | "valve" | null;
299
- k: number;
300
- ctxTokens: number; // calibrated: k x (sys + blocks)
301
- ctxAfterFolds: number; // calibrated
565
+ scale: Scale;
566
+ ctxTokens: number; // real: O + c x content
567
+ ctxAfterFolds: number; // real
302
568
  userTurns: number;
303
569
  cutVisibleIdx: number;
304
570
  cutKeptFirstMsg: Any;
571
+ fileRecall?: FileRecall; // the summary of this plan names recall files instead of zip_recall (PlanOpts.fileRecall)
572
+ ahead?: boolean; // F2: the cut was decided by the lookahead (it may lie inside the previous user turn, before its final answer)
573
+ }
574
+
575
+ /** Index of the final assistant message of user turn `turn` (the answer the user saw), -1 if none. */
576
+ export function finalAnswerIdx(blocks: Block[], turn: number): number {
577
+ for (let i = blocks.length - 1; i >= 0; i--) {
578
+ const b = blocks[i];
579
+ if (b.userTurn < turn) break;
580
+ if (b.userTurn === turn && b.kind === "assistant" && b.entryId && b.msg?.stopReason !== "error" && b.msg?.stopReason !== "aborted") return i;
581
+ }
582
+ return -1;
305
583
  }
306
584
 
307
- /** Estimated tokens before the earliest edited block (system prompt included): what stays cached through the edit. A cut edits from the start. */
308
- export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean, sys: number): number {
585
+ /** Estimated content before the earliest edited block: what stays cached through the edit with O (real: O + c x it). A cut edits from the start. */
586
+ export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean): number {
309
587
  const ids = new Set(folds.map((t) => t.entryId));
310
- let est = sys;
588
+ let est = 0;
311
589
  if (!cut) for (const b of blocks) { if (b.entryId && ids.has(b.entryId)) break; est += b.tokens; }
312
590
  return est;
313
591
  }
@@ -320,35 +598,41 @@ export const contentKeyOf = (content: Any): string => {
320
598
 
321
599
  /** The planner. Cold: fold everything outside the protected window, relax into the previous turn if still above the cap,
322
600
  * summarise only if folds cannot reach the cap and the law prices the summary call in. Warm: null unless the context is above the
323
- * cold cap and the law fires for the plan (the summary call is sunk there); then the cold plan minus the previous-turn relax (a warm plan
601
+ * cold cap and the law fires for the plan; its summary passes the same summary gate (WARM_SUMMARY_GATE); then the cold plan minus the previous-turn relax (a warm plan
324
602
  * folds the previous user turn by relax only at a return after the declared TTL, o.pastTtl). A cold plan with P(warm) > 0
325
603
  * passes the same law (expected cost). The cap never exceeds Pi's compaction room. Returns null when there is nothing to plan on
326
604
  * or the law says no. */
327
605
  export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
328
606
  const s = settings();
329
607
  const mode = o.mode ?? "cold";
330
- // limits are in real tokens; the plan works in estimate units, so they are divided by k once, here
331
- const k = o.k ?? 1;
608
+ const warmLike = mode === "warm" || o.warmStyle === true; // plans the way a warm plan does (the label of the trigger stays the mode's)
609
+ // limits are in real tokens; the plan works in content estimate units (real = O + c x estimate), so they are converted once, here.
610
+ // The cold cap is on the whole context, but leaves at least half of itself for the conversation above O (coldTarget): a cap the
611
+ // fixed prefix alone fills is out of reach, and every cold return then folds all it may and summarises. The compaction room is
612
+ // Pi's trigger on the whole context.
613
+ const { O, c } = o.scale ?? UNIT;
614
+ const real = (est: number) => O + c * est;
332
615
  const room = compactionRoom(o.model, o.reserve);
333
- const coldCap = Math.min(o.coldCap ?? s.coldCap, room ?? Infinity) / k;
616
+ const coldCap = (coldTarget(o.coldCap ?? s.coldCap, O, room) - O) / c;
334
617
  const law: Law = o.law ?? { pr: null, g: G0, pWarm: mode === "cold" ? 0 : 1 };
335
618
  const gate = (where: "summary" | "plan", B: number, A: number, r: number | null, call: number, T: number, Tsuf?: number): boolean => {
336
619
  const t = lawTerms(B, A, law.pWarm, r, law.pr, law.g, call, T);
337
620
  o.trace?.push({ where, ...t, ...(Tsuf !== undefined ? { Tsuf } : {}) });
338
621
  return t.ok;
339
622
  };
340
- const minGain = (o.minGain ?? s.minGain) / k;
623
+ const minGain = o.minGain ?? s.minGain;
341
624
  const foldMin = o.foldMin ?? s.foldMin;
342
625
  const keepLines = o.keepLines ?? s.keepLines;
343
626
  const relax = o.relax ?? RELAX_PREV_TURN;
344
627
  const inturnAge = o.inturnAge ?? s.inturnAge;
345
628
  const blocks = buildBlocks(entries, o.steerIds);
346
629
  if (!blocks.length) return null;
347
- const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0); // the upcoming prompt is a new user turn
630
+ const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0) + (o.slide ?? 0); // the upcoming prompt is a new user turn
631
+ const f1 = o.warmSummaryGate ?? WARM_SUMMARY_GATE, f2 = mode === "cold" && !o.warmStyle && (o.lookahead ?? LOOKAHEAD), f3 = o.narrativePrice ?? NARRATIVE_PRICE;
348
632
  const calls = toolCallIndex(blocks);
349
633
  const recalled = o.recalled ?? new Set<string>();
350
- const ctxEst = o.sys + blocks.reduce((a, b) => a + b.tokens, 0);
351
- if (mode === "warm" && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
634
+ const ctxEst = blocks.reduce((a, b) => a + b.tokens, 0); // content
635
+ if (warmLike && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
352
636
  const trig = mode === "warm" ? "valve" : "cold";
353
637
  const tokOverride = new Map<number, number>();
354
638
  const folds: FoldTarget[] = [];
@@ -368,9 +652,9 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
368
652
  };
369
653
  const addFold = (b: Block, trig: string): boolean => {
370
654
  if (tokOverride.has(b.idx) || recalled.has(handleFor(b.entryId!))) return false; // F4: never refold what the model recalled
371
- const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly);
655
+ const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly, o.fileRecall);
372
656
  if (ph === null) return false;
373
- const phTok = tok4(ph);
657
+ const phTok = tokensOf({ ...b.msg, content: [{ type: "text", text: ph }] }); // the folded message: its id and framing stay
374
658
  if (!(phTok < 0.9 * b.tokens)) return false;
375
659
  tokOverride.set(b.idx, phTok);
376
660
  const call = calls.get(b.msg.toolCallId);
@@ -378,20 +662,26 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
378
662
  return true;
379
663
  };
380
664
  // a zip_recall result IS content the model just asked for: folding it would undo the recall (and loop); never
381
- const isRecall = (b: Block) => (calls.get(b.msg.toolCallId)?.name ?? b.msg.toolName) === RECALL_TOOL;
665
+ // ... and so is a built-in read, grep, find, ls or bash call on a recall file (file recall, when an allowlist hides zip_recall)
666
+ const isRecall = (b: Block) => {
667
+ const call = calls.get(b.msg.toolCallId), name = call?.name ?? b.msg.toolName;
668
+ if (name === RECALL_TOOL) return true;
669
+ if (name === "bash") return commandNamesRecall(call?.args?.command);
670
+ return (name === "read" || name === "grep" || name === "find" || name === "ls") && !!parseRecallPath(call?.args?.path);
671
+ };
382
672
  // Observability: on an automatic (prefix) cache no fold starts inside the first MIN_EXPECT real tokens, so the next response's cacheRead
383
673
  // is an uncensored survival sample (learn.ts sample). Without it every idle return folded into the first 1-7K tokens and its sample was
384
674
  // censored, so the learned curve never saw a return (live GLM bench). Explicit caches are exempt: the provider looks back only ~20
385
675
  // blocks from the last breakpoint, a far head is not read either way. A cut replaces the prefix from the first message: exempt too.
386
- const head = law.pr?.cls === "automatic" ? MIN_EXPECT / k : 0;
676
+ const head = law.pr?.cls === "automatic" ? (MIN_EXPECT - O) / c : 0; // MIN_EXPECT counts from the first token, O included
387
677
  const start: number[] = [];
388
- blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), o.sys);
678
+ blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), 0);
389
679
  const foldable = (b: Block) => b.kind === "toolResult" && !b.edited && !!b.entryId && b.tokens > foldMin && !isRecall(b) && start[b.idx] >= head && (!o.rereadOnly || classify(b) === "rereadable");
390
680
  const protectedTurn = (b: Block) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1;
391
681
  const savings = () => folds.reduce((a, t) => a + t.entryTokens - t.phTokens, 0);
392
682
  const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b));
393
683
  for (const b of cands) addFold(b, trig);
394
- if (relax && (mode === "cold" || o.pastTtl === true)) {
684
+ if (relax && ((mode === "cold" && !o.warmStyle) || o.pastTtl === true)) {
395
685
  // the protected window = the new prompt + the previous user turn; that turn's big reads are what makes a cold return
396
686
  // expensive. Rereadable ones can be recalled exactly: fold them biggest-first until the cap; never the latest turn's own.
397
687
  // Cold plans, and warm plans at a return after the declared TTL; any other warm plan keeps the previous user turn visible.
@@ -418,49 +708,78 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
418
708
  }
419
709
  }
420
710
  let ctxAfterFolds = ctxEst - savings();
421
- // what no summary can remove: the system prompt, the protected turns (after their folds) and the previous summary, which the next
422
- // one carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
711
+ // what no summary can remove (besides O): the protected turns (after their folds) and the previous summary, which the next one
712
+ // carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
423
713
  const prevSum = blocks[0]?.kind === "summary" ? blocks[0].tokens : 0;
424
- const floor = o.sys + prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
714
+ const floor = prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
425
715
  const target = Math.max(coldCap, floor + SUMMARY_FLOOR);
426
- const atRoom = room !== null && k * ctxAfterFolds >= room;
716
+ const atRoom = room !== null && real(ctxAfterFolds) >= room;
427
717
  const sumTrigger: "cold" | "valve" | null = ctxAfterFolds > target && (!o.noSummary || atRoom) ? trig : null;
428
718
  let cutIdx: number | null = null;
429
719
  let prefixTokens = 0;
430
720
  let summaryTokensPlanned = 0;
431
- if (sumTrigger) {
432
- const limit = blocks.findIndex((b) => protectedTurn(b));
721
+ let ahead = false;
722
+ const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
723
+ const pre: number[] = [];
724
+ const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
725
+ // Block index of every fold: the folds a cut voids are those before it
726
+ const foldAt = folds.map((t) => blocks.findIndex((b) => b.entryId === t.entryId));
727
+ /** The whole plan's law (F10): B is the context as it is (or as the slid valve saw it), A what the plan leaves, `call` the summary call
728
+ * a cut makes; `untouched` is the content before the earliest edit (ledger only). A cold plan with some chance of a warm cache is
729
+ * priced by expectation (no prices: a cold plan always fires). T = A (the whole post-edit context, O included, as tuned: the cheaper
730
+ * A - O fires warm edits earlier and loses items offline). */
731
+ const planGate = (after: number, untouched: number, call: number, planned: boolean): boolean =>
732
+ !(warmLike || (planned && law.pr)) || gate("plan", Math.max(o.valveB ?? real(ctxEst), real(after)), real(after), room, call, real(after), c * (after - untouched));
733
+ /** The cut for cut points minCut..maxCut: the first that brings the context to the cap, else the deepest; gated = the summary law
734
+ * (no prices: the legacy gain rule) must pass. When the minimal cut's summary or the whole plan with it is refused, the deeper cuts
735
+ * are tried in turn (the gain grows as D^2, the call only with the prefix read). The whole plan's law prices the summary call too.
736
+ * Applies the first cut both laws accept (folds in the prefix are void) and returns true, or changes nothing. */
737
+ const summarise = (maxCut: number, gated: boolean): boolean => {
433
738
  const cuts: number[] = [];
434
- const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
435
- for (let i = minCut; i < blocks.length; i++) {
436
- if (limit >= 0 && i > limit) break;
437
- if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
438
- }
439
- const pre: number[] = [];
739
+ for (let i = minCut; i < blocks.length && i <= maxCut; i++) if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
740
+ if (!cuts.length) return false;
440
741
  let acc = 0;
441
742
  for (let i = 0; i < blocks.length; i++) {
442
743
  pre[i] = acc;
443
744
  acc += tokOverride.get(i) ?? blocks[i].tokens;
444
745
  }
445
- const total = o.sys + acc;
446
- const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
447
- if (cuts.length) {
448
- let pick = cuts[cuts.length - 1];
449
- for (const c of cuts) if (total - pre[c] + S(pre[c]) <= coldCap) { pick = c; break; }
450
- // cold: the summary call must pay for itself (verdict: call = w X + out S for an uncached call); warm: it is sunk in the plan's law
451
- const X = k * pre[pick], Sr = k * S(pre[pick]);
452
- const gain = !law.pr ? summaryGainOk(pre[pick], S(pre[pick]), total, minGain) : mode === "warm" || gate("summary", X, Sr, null, law.pr.w * X + law.pr.out * Sr, Sr);
453
- if (pre[pick] >= 2 * S(pre[pick]) && gain) {
454
- cutIdx = pick;
455
- prefixTokens = pre[pick];
456
- summaryTokensPlanned = S(pre[pick]);
457
- for (let i = folds.length - 1; i >= 0; i--) {
458
- const bi = blocks.findIndex((b) => b.entryId === folds[i].entryId);
459
- if (bi < pick) { tokOverride.delete(bi); folds.splice(i, 1); } // the prefix is replaced: its folds are void
460
- }
461
- ctxAfterFolds = ctxEst - savings();
746
+ const total = acc;
747
+ let first = cuts.length - 1;
748
+ for (let k = 0; k < cuts.length; k++) if (total - pre[cuts[k]] + S(pre[cuts[k]]) <= coldCap) { first = k; break; }
749
+ for (const pick of cuts.slice(first)) {
750
+ const X = c * pre[pick], Sr = c * S(pre[pick]);
751
+ // the call is the narrative call summary.ts makes (F3, NARRATIVE_PRICE): uncached input of the prefix without the previous
752
+ // summary, which it never reads, + output of at most the narrative budget; without F3 the verdict's w X + out S (the carried
753
+ // summary as output)
754
+ const call = !law.pr ? 0 : f3 ? law.pr.input * c * Math.max(pre[pick] - prevSum, 0) + law.pr.out * c * Math.min(S(pre[pick]), NARRATIVE_MAX) : law.pr.w * X + law.pr.out * Sr;
755
+ if (gated) {
756
+ // the summary call must pay for itself, cold or warm (F1, WARM_SUMMARY_GATE; before, a warm plan skipped the gate)
757
+ const ok = !law.pr ? summaryGainOk(X, Sr, real(total), minGain) : (warmLike && !f1) || gate("summary", X, Sr, null, call, Sr);
758
+ if (!ok) continue;
462
759
  }
760
+ if (!(pre[pick] >= 2 * S(pre[pick]))) continue;
761
+ // what the plan leaves: the context with its folds (those inside the prefix too: the prefix is measured as folded, `pre`) minus
762
+ // what the summary replaces plus the summary. (Counting the voided folds' savings back, as before, overstated it by them.)
763
+ if (!planGate(ctxAfterFolds - (pre[pick] - S(pre[pick])), 0, call, true)) continue;
764
+ cutIdx = pick;
765
+ prefixTokens = pre[pick];
766
+ summaryTokensPlanned = S(pre[pick]);
767
+ for (let i = folds.length - 1; i >= 0; i--) if (foldAt[i] < pick) { tokOverride.delete(foldAt[i]); folds.splice(i, 1); foldAt.splice(i, 1); } // the prefix is replaced: its folds are void
768
+ return true;
463
769
  }
770
+ return false;
771
+ };
772
+ if (sumTrigger) {
773
+ const limit = blocks.findIndex((b) => protectedTurn(b));
774
+ summarise(limit >= 0 ? limit : blocks.length - 1, true);
775
+ }
776
+ if (f2 && cutIdx === null && law.pr && ctxAfterFolds > coldCap && (!o.noSummary || atRoom)) {
777
+ // F2 lookahead: the warm valve's plan at the next user turn (this context, its folds applied, the window slid by one turn). If it
778
+ // would summarise, the summary is written now, at the cold return, where the rewrite costs nothing extra (and at settle it is
779
+ // prepared while the user is away), instead of one turn later as a warm rewrite the user waits for.
780
+ const next = planContext(entries, { ...o, mode: "warm", slide: (o.slide ?? 0) + 1, lookahead: false, pastTtl: false, noSummary: false, trace: undefined, law: { ...law, pWarm: 1 }, valveB: real(ctxAfterFolds) });
781
+ const fa = finalAnswerIdx(blocks, userTurns - 1);
782
+ if (next && next.cutIdx !== null && fa >= minCut && summarise(fa, false)) ahead = true;
464
783
  }
465
784
  // where the kept part starts among the projected non-system messages (the request-local view drops everything before it)
466
785
  let cutVisibleIdx = -1;
@@ -480,13 +799,17 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
480
799
  }
481
800
  }
482
801
  // the untouched prefix: everything before the earliest edited block (a cut edits from the first block on)
483
- const untouched = untouchedEst(blocks, folds, cutIdx !== null, o.sys);
802
+ const untouched = untouchedEst(blocks, folds, cutIdx !== null);
484
803
  const after = ctxAfterFolds - (cutIdx !== null ? prefixTokens - summaryTokensPlanned : 0);
485
804
  const planned = folds.length > 0 || cutIdx !== null;
486
- // F10: a warm rewrite must pay back; a cold plan with some chance of a warm cache is priced by expectation (no prices: cold always fires).
487
- // T = A (the whole post-edit context); the suffix after the earliest edit goes to the ledger only.
488
- if ((mode === "warm" || (planned && law.pr)) && !gate("plan", k * ctxEst, k * after, room, 0, k * after, k * (after - untouched))) return null;
489
- return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger, k, ctxTokens: k * ctxEst, ctxAfterFolds: k * ctxAfterFolds, userTurns, cutVisibleIdx, cutKeptFirstMsg };
805
+ // a plan with a cut passed the whole plan's law with its call inside summarise; folds only (no summary wanted, or none the laws
806
+ // accepted: the deeper cuts and the cut-free plan both stand) is judged here
807
+ if (cutIdx === null && !planGate(after, untouched, 0, planned)) {
808
+ // a cold plan with some chance of a warm cache: the plan a warm cache would make may pay where this one does not
809
+ if (mode === "cold" && !o.warmStyle && law.pr && law.pWarm > 0 && law.pWarm < 0.5) return planContext(entries, { ...o, warmStyle: true });
810
+ return null;
811
+ }
812
+ return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger: cutIdx !== null ? (sumTrigger ?? trig) : sumTrigger, scale: o.scale ?? UNIT, ctxTokens: real(ctxEst), ctxAfterFolds: real(ctxAfterFolds), userTurns, cutVisibleIdx, cutKeptFirstMsg, ...(o.fileRecall ? { fileRecall: o.fileRecall } : {}), ...(ahead ? { ahead } : {}) };
490
813
  }
491
814
 
492
815
  const sameMsg = (a: Any, b: Any): boolean =>
@@ -532,6 +855,11 @@ export function collapseSystem(messages: Any[]): Any | undefined {
532
855
  * message, the summary, then the kept non-system messages. null = nothing to apply. A cut whose kept boundary cannot be
533
856
  * located is dropped (folds still apply); an inconsistent split is never sent.
534
857
  */
858
+ /** Fold targets of a plan made on the view before a native compaction, renumbered for the request Pi builds after it: Pi's summary
859
+ * message first, then the kept part (positions among the non-system messages, as in applyPlanToMessages). */
860
+ export const rebaseFolds = (folds: FoldTarget[], cutVisibleIdx: number): FoldTarget[] =>
861
+ folds.map((t) => ({ ...t, visIdx: t.visIdx >= 0 && cutVisibleIdx >= 0 && t.visIdx >= cutVisibleIdx ? t.visIdx - cutVisibleIdx + 1 : -1 }));
862
+
535
863
  export function applyPlanToMessages(messages: Any[], plan: RunPlan | null): { messages: Any[]; droppedCut: boolean; applied: Set<string> } | null {
536
864
  if (!plan || plan.persisted || (!plan.folds.length && !plan.cut)) return null;
537
865
  const vis: number[] = []; // indices of the non-system messages