pi-zip 0.2.7 → 0.3.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -42
- package/package.json +1 -1
- package/src/cache.ts +50 -12
- package/src/guard.ts +3 -2
- package/src/index.ts +45 -8
- package/src/learn.ts +18 -8
- package/src/notice.ts +91 -33
- package/src/placeholder.ts +12 -4
- package/src/plan.ts +434 -106
- package/src/recall.ts +111 -27
- package/src/recallfile.ts +165 -0
- package/src/run.ts +782 -116
- package/src/state.ts +53 -0
- package/src/summary.ts +22 -10
- package/src/ui.ts +107 -44
- package/src/util.ts +57 -2
package/src/plan.ts
CHANGED
|
@@ -2,9 +2,10 @@
|
|
|
2
2
|
import { classifyRecoverability, type Recover } from "./classify.ts";
|
|
3
3
|
import { handleFor, makePlaceholderFor, pickKeyLines, RECALL_TOOL, shortArgs } from "./placeholder.ts";
|
|
4
4
|
import { createHash } from "node:crypto";
|
|
5
|
-
import { type Any, PRODUCT, clamp, envInt, textOf, tok4, tokensOf } from "./util.ts";
|
|
5
|
+
import { type Any, COMMIT_CUSTOM, PRODUCT, clamp, envInt, piTokensOf, textOf, tok4, tokensOf } from "./util.ts";
|
|
6
6
|
import { PH_MARK } from "./placeholder.ts";
|
|
7
7
|
import { MIN_EXPECT, type Prices } from "./learn.ts";
|
|
8
|
+
import { commandNamesRecall, parseRecallPath, type FileRecall } from "./recallfile.ts";
|
|
8
9
|
|
|
9
10
|
/** Default: when a COLD plan (P(warm) < 0.5) is still above the cap, also fold REREADABLE outputs of the previous user turn, biggest
|
|
10
11
|
* first. A warm plan does so only at a user return after the declared TTL (PlanOpts.pastTtl): the very returns where the TTL rule
|
|
@@ -22,7 +23,21 @@ export const RELAX_PREV_TURN = true;
|
|
|
22
23
|
* constant (largest saving with lost <= prod in every stratum); the old rereadable-only filter cost 1.026 / 1.018x at -0.6 lost items. */
|
|
23
24
|
export const INTURN_AGE = 60;
|
|
24
25
|
|
|
25
|
-
const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised; folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
|
|
26
|
+
const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised (but see LOOKAHEAD); folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
|
|
27
|
+
|
|
28
|
+
/** F1: a warm (valve) summary passes the same summary gate as a cold one, the call priced in (before: a warm plan skipped it, so a
|
|
29
|
+
* summary the cold gate had just refused was written one request later anyway). PlanOpts.warmSummaryGate. */
|
|
30
|
+
export const WARM_SUMMARY_GATE = true;
|
|
31
|
+
/** F2: a cold plan that does not summarise looks one user turn ahead: if the warm valve would summarise at the next user turn, once the
|
|
32
|
+
* protected window has slid past the previous turn, the cold plan summarises now (the rewrite is free now, not then), its cut allowed
|
|
33
|
+
* inside the previous user turn up to that turn's final assistant answer, which stays verbatim with everything after it. PlanOpts.lookahead. */
|
|
34
|
+
export const LOOKAHEAD = false;
|
|
35
|
+
/** F3: the summary gates price the call the code makes (summary.ts narrative): uncached input of the prefix the narrative reads (the
|
|
36
|
+
* previous summary is not sent: its sections are carried forward by code) plus output of at most the narrative budget, instead of
|
|
37
|
+
* w x the prefix + out x the whole planned summary (which counts the carried-forward previous summary as model output). PlanOpts.narrativePrice. */
|
|
38
|
+
export const NARRATIVE_PRICE = true;
|
|
39
|
+
/** The narrative's token budget (chars/4), summary.ts buildCut: clamp(planned - skeleton, 300, NARRATIVE_MAX). */
|
|
40
|
+
export const NARRATIVE_MAX = 4000;
|
|
26
41
|
const SUMMARY_FLOOR = 1000;
|
|
27
42
|
const SUMMARY_CAP = 8000;
|
|
28
43
|
const SUMMARY_RATIO = 0.1;
|
|
@@ -55,6 +70,56 @@ export function legacyValve(before: number, after: number, room: number | null):
|
|
|
55
70
|
}
|
|
56
71
|
|
|
57
72
|
export const PI_KEEP_RECENT = 20_000; // Pi's compaction keepRecentTokens default
|
|
73
|
+
|
|
74
|
+
/** Pi's compaction keepRecentTokens for this model: compaction.modelOverrides["provider/id"], else compaction.keepRecentTokens, else 20000. */
|
|
75
|
+
export function keepRecentTokensFor(settings: Any, model: Any): number {
|
|
76
|
+
const c = settings?.compaction;
|
|
77
|
+
const key = model ? `${model.provider}/${model.id}` : "";
|
|
78
|
+
for (const v of [c?.modelOverrides?.[key]?.keepRecentTokens, c?.keepRecentTokens]) if (typeof v === "number" && Number.isFinite(v) && v >= 0) return v;
|
|
79
|
+
return PI_KEEP_RECENT;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const CUT_POINT_ROLES = new Set(["user", "assistant", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
|
|
83
|
+
const TURN_START_ROLES = new Set(["user", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Would Pi's compact() find something to compact in this projection (the branch as it is now), or throw "Nothing to compact (session
|
|
87
|
+
* too small)"? A replica of the emptiness test of Pi's prepareCompaction (core/compaction/compaction.js, not exported): Pi cuts where
|
|
88
|
+
* keepRecent of its chars/4 tokens, counted from the end, begin, at the next cut point; the conversation before that cut (or before the
|
|
89
|
+
* turn it splits) must not be empty. Pi's own extra moves of the cut (over entries without messages, past recovery omissions) never
|
|
90
|
+
* empty it. The caller checks "Already compacted" (the newest branch entry is a compaction) itself.
|
|
91
|
+
*/
|
|
92
|
+
export function piCanCompact(entries: Any[], keepRecent: number): boolean {
|
|
93
|
+
const msgs = (e: Any): Any[] => (Array.isArray(e?.messages) ? e.messages : []);
|
|
94
|
+
const isComp = (e: Any) => e?.sourceEntry?.type === "compaction";
|
|
95
|
+
const prev = entries.findIndex((e) => isComp(e) && msgs(e).length > 0);
|
|
96
|
+
const start = prev >= 0 ? prev + 1 : 0, end = entries.length;
|
|
97
|
+
const cuts: number[] = [];
|
|
98
|
+
for (let i = start; i < end; i++) if (!isComp(entries[i]) && msgs(entries[i]).some((m) => CUT_POINT_ROLES.has(m?.role))) cuts.push(i);
|
|
99
|
+
if (!cuts.length) return false;
|
|
100
|
+
let cut = cuts[0], acc = 0;
|
|
101
|
+
for (let i = end - 1; i >= start; i--) {
|
|
102
|
+
const t = msgs(entries[i]).reduce((a, m) => a + piTokensOf(m), 0);
|
|
103
|
+
if (!t) continue;
|
|
104
|
+
acc += t;
|
|
105
|
+
if (acc >= keepRecent) { cut = cuts.find((c) => c >= i) ?? cuts[cuts.length - 1]; break; }
|
|
106
|
+
}
|
|
107
|
+
while (cut > start && !isComp(entries[cut - 1]) && !msgs(entries[cut - 1]).length) cut--;
|
|
108
|
+
if (!entries[cut]?.sourceEntry?.id) return false;
|
|
109
|
+
const startsTurn = (e: Any) => !isComp(e) && msgs(e).some((m) => TURN_START_ROLES.has(m?.role));
|
|
110
|
+
let turn = -1;
|
|
111
|
+
if (!startsTurn(entries[cut])) for (let i = cut; i >= start; i--) if (startsTurn(entries[i])) { turn = i; break; }
|
|
112
|
+
const conversation = (from: number, to: number) => entries.slice(from, to).some((e) => !isComp(e) && msgs(e).some((m) => m?.role !== "system"));
|
|
113
|
+
return turn >= 0 ? conversation(start, turn) || conversation(turn, cut) : conversation(start, cut);
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** The projection a run plan would have seen without pi-zip's native compaction: `pre` (taken right before it) plus what was appended
|
|
117
|
+
* after the compaction entry (the prompt and anything sent with it). null when that compaction is no longer the head of `now`. */
|
|
118
|
+
export function spliceNative(pre: Any[], now: Any[], compactionId: string): Any[] | null {
|
|
119
|
+
if (!compactionId || now[0]?.sourceEntry?.id !== compactionId) return null;
|
|
120
|
+
const known = new Set(pre.map((e) => e?.sourceEntry?.id).filter((x) => typeof x === "string"));
|
|
121
|
+
return [...pre, ...now.slice(1).filter((e) => !known.has(e?.sourceEntry?.id))];
|
|
122
|
+
}
|
|
58
123
|
export const G0 = 2_500; // growth per request before the session's own EWMA has data
|
|
59
124
|
|
|
60
125
|
/** What the law knows at a decision: prices (null = legacy rule), growth per request g (real tokens), P(the cache is warm). */
|
|
@@ -88,58 +153,249 @@ export function lawTerms(B: number, A: number, P: number, room: number | null, p
|
|
|
88
153
|
export const editAllowed = (B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A): boolean =>
|
|
89
154
|
lawTerms(B, A, P, room, pr, g, call, T).ok;
|
|
90
155
|
|
|
91
|
-
// Token scale. Every size in this file is a chars/4 estimate
|
|
92
|
-
//
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
156
|
+
// Token scale. Every size in this file is a chars/4 estimate of the CONTENT (Pi's own rule); the provider counts real tokens, and the
|
|
157
|
+
// two are not proportional: real = O + c x content.
|
|
158
|
+
// O what every request carries and no edit can shrink: system prompt, tool schemas, the provider's own framing (2-4K on a bare Pi,
|
|
159
|
+
// 29-55K with a large tool set);
|
|
160
|
+
// c real tokens per estimated content token (0.7-1.1 GLM / GPT, 1.5-2.2 Claude; more on CJK-heavy content).
|
|
161
|
+
// pi-zip <= 0.2.9 used ONE ratio, k = real / (system prompt + content) clamped to [1, 2.5]: with O a third of the context k sat on its
|
|
162
|
+
// clamp, and every size after an edit came out 10-15K too small (research round5/affine.md). Both are read from the branch on every
|
|
163
|
+
// decision, so a restart or `pi -p` calibrates exactly like a long-lived process, and nothing is stored:
|
|
164
|
+
// O 1. the cacheRead of the request that first carried the newest summary: everything after the system prompt and tools changed
|
|
165
|
+
// there, so what the provider still read from its cache is exactly O (when the cache was warm at all);
|
|
166
|
+
// 2. else the session's first request minus C0 x its content (an opening prompt is small next to O);
|
|
167
|
+
// 3. else R0 x the chars/4 of the system prompt and the JSON of the declared tools.
|
|
168
|
+
// An O measured under another system prompt or tool set moves by R0 x the change of that chars/4 size (MCP tools come and go).
|
|
169
|
+
// c (real - O) / content of the newest response, clamped to [C_MIN, C_MAX] (a guard for a tiny content estimate; recorded sessions
|
|
170
|
+
// span 0.7-3.9); C0 before the first response. A response followed by edits it did not carry (a warm-valve edit is persisted at
|
|
171
|
+
// turn_end and first sent with the next request: that response's prompt did not shrink) is counted with the originals.
|
|
172
|
+
// Both are per model: only responses of the model the next request goes to count. After a switch, before its first response,
|
|
173
|
+
// O = R0 x the system state and c = C0; after it, with no O evidence of its own, O = c x X, X = O / c of the previous model.
|
|
174
|
+
export const C0 = 1.5;
|
|
175
|
+
export const C_MIN = 0.5;
|
|
176
|
+
export const C_MAX = 4;
|
|
177
|
+
export const R0 = 2;
|
|
178
|
+
|
|
179
|
+
/** The system prompt and the declared tools (Pi API fallback for a transcript that records no system message). */
|
|
180
|
+
export interface SysInfo { text: string; tools: Any[] }
|
|
181
|
+
/** chars/4 of the system prompt plus the JSON of the declared tool definitions (name, description, parameters). */
|
|
182
|
+
export const sysEst = (s: SysInfo): number =>
|
|
183
|
+
tok4(s.text) + (s.tools.length ? tok4(JSON.stringify(s.tools.map((t: Any) => ({ name: t?.name, description: t?.description, parameters: t?.parameters })))) : 0);
|
|
184
|
+
|
|
185
|
+
export interface Scale {
|
|
186
|
+
O: number; // real tokens every request carries before the content (system prompt, tools, framing)
|
|
187
|
+
c: number; // real tokens per estimated (chars/4) content token
|
|
188
|
+
real: number; // input + cacheRead + cacheWrite of the newest usable response (0 = none)
|
|
189
|
+
est: number; // estimated content that request carried
|
|
190
|
+
source: "usage" | "default"; // where c comes from
|
|
191
|
+
oSource: "summary" | "first" | "model" | "default"; // where O comes from ("model": another model's O in content units, X = O' / c')
|
|
192
|
+
}
|
|
193
|
+
/** Sizes taken as real (tests, and the ledger of a plan made without calibration). */
|
|
194
|
+
export const UNIT: Scale = { O: 0, c: 1, real: 0, est: 0, source: "default", oSource: "default" };
|
|
195
|
+
/** Real tokens of a context whose content is estimated at `content`. */
|
|
196
|
+
export const sizeOf = (s: Pick<Scale, "O" | "c">, content: number): number => s.O + s.c * content;
|
|
197
|
+
|
|
198
|
+
/** A response pi-ai drops before sending (an error or an abort): it carries no usable usage and adds nothing to later requests. */
|
|
199
|
+
const dropped = (m: Any): boolean => m?.role === "assistant" && (m.stopReason === "error" || m.stopReason === "aborted");
|
|
200
|
+
/** chars/4 a message adds to every later request: tokensOf, except 0 for an assistant message pi-ai drops before sending. */
|
|
201
|
+
export const wireTokensOf = (m: Any): number => (dropped(m) ? 0 : tokensOf(m));
|
|
202
|
+
/** Prompt size the provider reported for a response that really ran (0 = none, an error or an abort). */
|
|
203
|
+
const promptOf = (m: Any): number => {
|
|
204
|
+
const u = m?.role === "assistant" && !dropped(m) ? m.usage : null;
|
|
205
|
+
return u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
|
|
206
|
+
};
|
|
207
|
+
/** "provider/model" of a response ("" = not recorded: such a response counts for any model). */
|
|
208
|
+
const respKey = (m: Any): string => (m?.provider || m?.model ? `${m.provider ?? ""}/${m.model ?? ""}` : "");
|
|
209
|
+
/** chars/4 of the system state the transcript records in `msgs` (-1: none recorded). */
|
|
210
|
+
function sysOf(msgs: Any[]): number {
|
|
211
|
+
const s = collapseSystem(msgs);
|
|
212
|
+
return s ? sysEst({ text: [textOf(s.content), ...Object.values<string>(s.sections ?? {})].filter((x) => x).join("\n\n"), tools: s.toolsAdded ?? [] }) : -1;
|
|
213
|
+
}
|
|
214
|
+
/** The session's first response from model `key` on the branch: its prompt, the content before it and the system state it was sent with. */
|
|
215
|
+
function firstRequest(branch: Any[], key: string): { real: number; est: number; S: number } | null {
|
|
216
|
+
const sys: Any[] = [];
|
|
217
|
+
let est = 0;
|
|
218
|
+
for (const e of branch) {
|
|
219
|
+
const m = e?.type === "message" ? e.message : e?.type === "custom_message" ? { role: "custom", content: e.content } : null;
|
|
220
|
+
if (!m) continue;
|
|
221
|
+
if (m.role === "system") { sys.push(m); continue; }
|
|
222
|
+
const real = promptOf(m), k = respKey(m);
|
|
223
|
+
if (real > 0 && (!key || !k || k === key)) return { real, est, S: sysOf(sys) };
|
|
224
|
+
est += wireTokensOf(m);
|
|
225
|
+
}
|
|
226
|
+
return null;
|
|
106
227
|
}
|
|
107
228
|
|
|
108
|
-
/**
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
const
|
|
118
|
-
const
|
|
119
|
-
let
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
229
|
+
/** chars/4 of the context the request behind the response `id` carried, rebuilt from the branch the way Pi projects it: the newest
|
|
230
|
+
* compaction before the response, the entries it keeps, everything after it, and the edits persisted before the response, plus the
|
|
231
|
+
* edits of a run plan persisted right after it (their commit entry says the response itself carried them). null: `id` is not on the
|
|
232
|
+
* branch. Used when a compaction written after the response (Pi's own, another extension's) removed that request's content from the
|
|
233
|
+
* projection, so the projection cannot say what the response measured. */
|
|
234
|
+
function requestEst(branch: Any[], id: string): number | null {
|
|
235
|
+
const at = branch.findIndex((e) => e?.id === id);
|
|
236
|
+
if (at < 0) return null;
|
|
237
|
+
const pre = branch.slice(0, at);
|
|
238
|
+
const edit = new Map<string, Any>();
|
|
239
|
+
for (const e of pre) if (e?.type === "context_edit") edit.set(e.targetId, e.replacement);
|
|
240
|
+
for (let i = at + 1; i < branch.length && !(branch[i]?.type === "message" || branch[i]?.type === "compaction"); i++) {
|
|
241
|
+
const e = branch[i];
|
|
242
|
+
if (e?.type === "custom" && e.customType === COMMIT_CUSTOM) {
|
|
243
|
+
if (e.data?.carried) for (let j = at + 1; j < i; j++) if (branch[j]?.type === "context_edit") edit.set(branch[j].targetId, branch[j].replacement);
|
|
244
|
+
break;
|
|
245
|
+
}
|
|
123
246
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
247
|
+
let ci = -1;
|
|
248
|
+
pre.forEach((e, i) => { if (e?.type === "compaction") ci = i; });
|
|
249
|
+
const kept = ci < 0 ? pre : [pre[ci], ...pre.slice(0, ci).slice(Math.max(0, pre.findIndex((e) => e?.id === pre[ci].firstKeptEntryId))).filter((e) => !(e?.type === "message" && e.message?.role === "system")), ...pre.slice(ci + 1)];
|
|
250
|
+
let est = 0;
|
|
251
|
+
for (const e of kept) {
|
|
252
|
+
if (e?.type === "compaction" || e?.type === "branch_summary") est += tokensOf({ role: "compactionSummary", summary: e.summary });
|
|
253
|
+
else if (e?.type === "custom_message") est += tokensOf({ role: "custom", content: edit.has(e.id) ? edit.get(e.id)?.content : e.content });
|
|
254
|
+
else if (e?.type === "message" && e.message) {
|
|
255
|
+
const rep = edit.get(e.id);
|
|
256
|
+
if (rep === null) continue;
|
|
257
|
+
const m = rep && e.message.role !== "system" ? { ...e.message, content: typeof rep.content === "string" && e.message.role !== "user" ? [{ type: "text", text: rep.content }] : rep.content } : e.message;
|
|
258
|
+
est += wireTokensOf(m);
|
|
259
|
+
}
|
|
137
260
|
}
|
|
138
|
-
return
|
|
261
|
+
return est;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/** O and c from the projection (`entries`) and, for the session's first request, the branch; `sys` stands in for a transcript without
|
|
265
|
+
* system messages. `model` = "provider/id" of the model the next request goes to (default: the newest response's): O and c are
|
|
266
|
+
* properties of the provider's tokenizer, so only that model's responses are evidence. With none, O = R0 x the system state and
|
|
267
|
+
* c = C0; with responses but no O evidence of its own (a model switch mid-session), O keeps the size another model measured in
|
|
268
|
+
* content units, X = O' / c' (the system prompt and tools tokenize like the content around them): real = c (X + E). */
|
|
269
|
+
export function calibrate(entries: Any[], sys?: SysInfo, branch: Any[] = [], model?: string): Scale {
|
|
270
|
+
type Resp = { at: number; est: number; real: number; read: number; nSys: number; ts: number; key: string; id: string };
|
|
271
|
+
const resp: Resp[] = [];
|
|
272
|
+
const sysMsgs: Any[] = [];
|
|
273
|
+
const edits: { at: number; target: string; foreign: boolean }[] = [];
|
|
274
|
+
const marks: { at: number; carried: boolean }[] = []; // pi-zip's commit entries: who first carried the edits right before them (run.ts commit)
|
|
275
|
+
const orig = new Map<string, { at: number; d: number }>(); // edited entries: tokens their originals add back
|
|
276
|
+
const withMsgs: number[] = [0]; // withMsgs[k]: entries before k that carry messages (a commit group has none inside it)
|
|
277
|
+
let est = 0, sysAt = 0; // sysAt: responses before the newest system message
|
|
278
|
+
entries.forEach((pe, at) => {
|
|
279
|
+
const src = pe.sourceEntry, msgs: Any[] = pe.messages ?? [];
|
|
280
|
+
withMsgs.push(withMsgs[at] + (msgs.length ? 1 : 0));
|
|
281
|
+
// an edit with a replacement that is not our placeholder is another extension's (never carried by a request of ours); one without
|
|
282
|
+
// (a bare entry) is judged by the size heuristics below
|
|
283
|
+
if (src?.type === "context_edit") edits.push({ at, target: src.targetId, foreign: "replacement" in src && !textOf(src.replacement?.content).startsWith(PH_MARK) });
|
|
284
|
+
if (src?.type === "custom" && src.customType === COMMIT_CUSTOM) marks.push({ at, carried: !!src.data?.carried });
|
|
285
|
+
if (src?.type === "message" && msgs.length === 1 && msgs[0] !== src.message) orig.set(src.id, { at, d: wireTokensOf(src.message) - wireTokensOf(msgs[0]) });
|
|
286
|
+
for (const m of msgs) {
|
|
287
|
+
if (m?.role === "system") { sysMsgs.push(m); sysAt = resp.length; continue; }
|
|
288
|
+
const real = promptOf(m);
|
|
289
|
+
if (real > 0) resp.push({ at, est, real, read: Number(m.usage.cacheRead) || 0, nSys: sysMsgs.length, ts: Number(m.timestamp) || 0, key: respKey(m), id: src?.id ?? "" });
|
|
290
|
+
est += wireTokensOf(m);
|
|
291
|
+
}
|
|
292
|
+
});
|
|
293
|
+
const sAt = (n: number) => sysOf(sysMsgs.slice(0, n)); // the system state after the first n system messages (computed only where needed)
|
|
294
|
+
const S = sAt(sysMsgs.length);
|
|
295
|
+
const Snow = S >= 0 ? S : sys ? sysEst(sys) : 0;
|
|
296
|
+
const head = entries[0]?.sourceEntry?.type === "compaction" ? entries[0].sourceEntry : null;
|
|
297
|
+
const headMs = head ? Date.parse(head.timestamp) || 0 : 0;
|
|
298
|
+
// a pi-zip summary persisted as a turn_end draft was request-local first (a run plan carries it before its entry exists); one that
|
|
299
|
+
// went in through Pi's compaction before the run (details.via "native") is carried after its entry, exactly like Pi's own
|
|
300
|
+
const local = !!head && head.details?.by === PRODUCT && head.details?.via !== "native";
|
|
301
|
+
// Was the request behind this response sent after the head compaction entry was written? The branch's order says (timestamps tie when
|
|
302
|
+
// a compaction follows a response in the same millisecond); without the branch the timestamps do.
|
|
303
|
+
const pos = new Map<string, number>();
|
|
304
|
+
if (head) branch.forEach((e, i) => { if (typeof e?.id === "string") pos.set(e.id, i); });
|
|
305
|
+
const afterHead = (r: Resp): boolean => {
|
|
306
|
+
const h = pos.get(head?.id), i = pos.get(r.id);
|
|
307
|
+
return h !== undefined && i !== undefined ? i > h : r.ts > headMs;
|
|
308
|
+
};
|
|
309
|
+
const opening = (r: { real: number; est: number } | null | undefined) => !!r && C0 * r.est <= 0.1 * r.real; // content small next to O
|
|
310
|
+
|
|
311
|
+
/** The first commit entry of the group of pi-zip edits that starts at entry `at` (none when a message comes first). */
|
|
312
|
+
const markOf = (at: number) => marks.find((m) => m.at > at && withMsgs[m.at] - withMsgs[at + 1] === 0);
|
|
313
|
+
/** Index of the newest response in `resp` before entry `at`. */
|
|
314
|
+
const respBefore = (at: number) => resp.reduce((k, r, i) => (r.at < at ? i : k), -1);
|
|
315
|
+
/** Tokens the originals add back for response i: edits persisted after it that its own request did not carry (the projection counts
|
|
316
|
+
* their placeholders). A commit entry says who carried them (a run plan: the response before the entry; a valve plan: the next one);
|
|
317
|
+
* with none (a session of an older version) the sizes decide, for the newest response only (`legacy`). */
|
|
318
|
+
const addBack = (i: number, legacy: boolean): number => {
|
|
319
|
+
let d = 0;
|
|
320
|
+
for (const x of edits) {
|
|
321
|
+
const t = orig.get(x.target);
|
|
322
|
+
if (x.at <= resp[i].at || !t || t.at >= resp[i].at) continue;
|
|
323
|
+
const m = x.foreign ? undefined : markOf(x.at);
|
|
324
|
+
if (m ? m.carried && respBefore(m.at) === i : !x.foreign && legacy) continue;
|
|
325
|
+
d += t.d;
|
|
326
|
+
}
|
|
327
|
+
return d;
|
|
328
|
+
};
|
|
329
|
+
|
|
330
|
+
const fit = (want: string, transfer: boolean): Scale => {
|
|
331
|
+
const mine = (r: Resp | undefined): r is Resp => !!r && (!want || !r.key || r.key === want);
|
|
332
|
+
// evidence for O, the newest wins (it needs the smallest correction for system changes since)
|
|
333
|
+
const ev: { at: number; O: number; S: number; src: Scale["oSource"] }[] = [];
|
|
334
|
+
const f = branch.length ? firstRequest(branch, want) : head ? null : resp.find(mine); // the session's first request
|
|
335
|
+
if (f && opening(f)) ev.push({ at: -1, O: f.real - C0 * f.est, S: "S" in f ? f.S : sAt(f.nSys), src: "first" });
|
|
336
|
+
const ci = resp.findIndex((r, i) => i >= sysAt && mine(r)), cur = resp[ci]; // the first response sent with the current system prompt and tools
|
|
337
|
+
// its content as its request carried it: edits persisted after it (a model switch, say, then a run plan's folds) are in `est`
|
|
338
|
+
const curEst = cur ? cur.est + addBack(ci, false) : 0;
|
|
339
|
+
if (cur && (!head || afterHead(cur)) && opening({ real: cur.real, est: curEst })) ev.push({ at: ci, O: cur.real - C0 * curEst, S, src: "first" });
|
|
340
|
+
if (head) {
|
|
341
|
+
// the summary at the head was first carried by the first response after its entry, or, for pi-zip's own request-local summary
|
|
342
|
+
// (a run plan), by the last one before it; Pi's, another extension's or a native pi-zip compaction is never sent before its entry
|
|
343
|
+
const after = resp.findIndex((r) => afterHead(r)), ours = local;
|
|
344
|
+
for (const i of after < 0 ? (ours ? [resp.length - 1] : []) : ours ? [after - 1, after] : [after]) {
|
|
345
|
+
const r = resp[i], prev = resp[i - 1];
|
|
346
|
+
if (mine(r) && r.read > 0 && r.read < 0.8 * r.real && (!prev || r.read < 0.8 * prev.real)) { ev.push({ at: i, O: r.read, S: sAt(r.nSys), src: "summary" }); break; }
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
const best = ev.sort((a, b) => b.at - a.at)[0];
|
|
350
|
+
let O = best ? Math.max(0, best.O + (best.S >= 0 && S >= 0 ? R0 * (Snow - best.S) : 0)) : R0 * Snow, oSource: Scale["oSource"] = best?.src ?? "default";
|
|
351
|
+
// c from the newest response of this model
|
|
352
|
+
let n = resp.length - 1;
|
|
353
|
+
while (n >= 0 && !mine(resp[n])) n--;
|
|
354
|
+
if (n < 0) return { O, c: C0, real: 0, est: 0, source: "default", oSource };
|
|
355
|
+
const N = resp[n];
|
|
356
|
+
let estN = N.est;
|
|
357
|
+
const cutLater = !!head && !afterHead(N);
|
|
358
|
+
const later = edits.filter((x) => x.at > N.at);
|
|
359
|
+
const rebuilt = cutLater && !local && !!N.id ? requestEst(branch, N.id) : null; // what its request carried: the compaction after it removed that from the projection
|
|
360
|
+
if (rebuilt !== null && rebuilt > 0) estN = rebuilt;
|
|
361
|
+
else if (later.length || cutLater) {
|
|
362
|
+
// carried (a run plan): the prompt shrank, or a read stopped early; a cache miss alone (no read at all) proves nothing. The
|
|
363
|
+
// commit entries of pi-zip's edits say it outright; the sizes decide only for edits without one (older sessions).
|
|
364
|
+
// Pi's own compaction is never request-local.
|
|
365
|
+
let p = n - 1;
|
|
366
|
+
while (p >= 0 && !mine(resp[p])) p--;
|
|
367
|
+
const P = resp[p];
|
|
368
|
+
const heur = !P || N.real < P.real || (N.read > 0 && N.read < 0.8 * P.real);
|
|
369
|
+
const carried = !(cutLater && !local) && (cutLater && typeof head.details?.carried === "boolean" ? head.details.carried : heur);
|
|
370
|
+
if (!carried && cutLater) return { O, c: C0, real: N.real, est: 0, source: "default", oSource }; // what it carried is summarised away
|
|
371
|
+
estN += addBack(n, heur && carried);
|
|
372
|
+
}
|
|
373
|
+
if (!best && transfer && estN > 0) {
|
|
374
|
+
// no O of its own: the newest other model's O in content units
|
|
375
|
+
// (a compaction may have taken the other model's responses out of the projection: the branch still has them)
|
|
376
|
+
const other = [...resp].reverse().find((r) => r.key && r.key !== want)?.key
|
|
377
|
+
?? [...branch].reverse().map((e) => (e?.type === "message" && promptOf(e.message) > 0 ? respKey(e.message) : "")).find((k) => k && k !== want);
|
|
378
|
+
const o = other ? fit(other, false) : null;
|
|
379
|
+
if (o && o.oSource !== "default" && o.c > 0) {
|
|
380
|
+
const X = o.O / o.c, c = clamp(N.real / (X + estN), C_MIN, C_MAX);
|
|
381
|
+
return { O: c * X, c, real: N.real, est: estN, source: "usage", oSource: "model" };
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
return { O, c: estN > 0 ? clamp((N.real - O) / estN, C_MIN, C_MAX) : C0, real: N.real, est: estN, source: estN > 0 ? "usage" : "default", oSource };
|
|
385
|
+
};
|
|
386
|
+
return fit(model || [...resp].reverse().find((r) => r.key)?.key || "", true);
|
|
139
387
|
}
|
|
140
388
|
|
|
389
|
+
/** The cold cap's target in real tokens of the WHOLE context (system prompt and tools included, as the cap was tuned): max(cap, O + cap/2),
|
|
390
|
+
* never above Pi's compaction room. While O is under half the cap this is the plain whole-context cap. Above that the target is O plus
|
|
391
|
+
* half the cap of conversation: never at or below O, where no edit reaches it (0.2.9's 40K with O = 29-55K summarised at every cold
|
|
392
|
+
* return), and not 40K of conversation on top of O either (the cap on content alone cost +13% live on a 36K prefix, with no quality
|
|
393
|
+
* difference measured). Offline it never loses more items than the whole-context cap (research round5/affine.md section 11). */
|
|
394
|
+
export const CAP_CONTENT_MIN = 0.5;
|
|
395
|
+
export const coldTarget = (cap: number, O: number, room: number | null): number => Math.min(Math.max(cap, O + CAP_CONTENT_MIN * cap), room ?? Infinity);
|
|
396
|
+
|
|
141
397
|
export const settings = () => ({
|
|
142
|
-
coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many tokens
|
|
398
|
+
coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many real tokens of the whole context, but never below O + half of it (coldTarget)
|
|
143
399
|
foldMin: envInt("FOLD_MIN", 500), // outputs below this many tokens are never folded
|
|
144
400
|
keepLines: envInt("KEEP_LINES", 8),
|
|
145
401
|
minGain: envInt("MIN_GAIN", 10_000), // legacy rule only (no prices): a summary must remove at least max(minGain, 15% of the context)
|
|
@@ -203,7 +459,7 @@ export function buildBlocks(contextEntries: Any[], steerIds?: Set<string>): Bloc
|
|
|
203
459
|
for (const c of msg.content ?? []) if (c?.type === "toolCall") issued.set(c.id, asst);
|
|
204
460
|
}
|
|
205
461
|
asstOf[blocks.length] = kind === "toolResult" ? (issued.get(msg.toolCallId) ?? asst) : asst;
|
|
206
|
-
blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a +
|
|
462
|
+
blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + wireTokensOf(m), 0), userTurn, edited, ours, age: 0 });
|
|
207
463
|
}
|
|
208
464
|
for (const b of blocks) if (b.kind === "toolResult") b.age = asst - asstOf[b.idx];
|
|
209
465
|
return blocks;
|
|
@@ -244,6 +500,7 @@ export interface Cut {
|
|
|
244
500
|
costUsd: number; // what the narrative model call cost (0 when unknown)
|
|
245
501
|
usage?: Any;
|
|
246
502
|
ms: number; // total production time (the narrative model call included)
|
|
503
|
+
ahead?: boolean; // F2: a lookahead cut (inside the previous user turn, before its final answer; guard.ts)
|
|
247
504
|
}
|
|
248
505
|
|
|
249
506
|
/** What a run sends from its first request on and persists verbatim at turn_end (F8). */
|
|
@@ -254,7 +511,7 @@ export interface RunPlan {
|
|
|
254
511
|
ctxBefore: number;
|
|
255
512
|
ctxAfter: number;
|
|
256
513
|
ms: number;
|
|
257
|
-
|
|
514
|
+
scale: Scale; // the token scale the plan was made with (stats and notices of this run use the same one)
|
|
258
515
|
persisted: boolean;
|
|
259
516
|
cutVisibleIdx?: number; // where the kept part starts among the projected non-system messages
|
|
260
517
|
cutKeptFirstMsg?: Any; // ... and that message itself (a disagreeing request view drops the cut)
|
|
@@ -262,6 +519,9 @@ export interface RunPlan {
|
|
|
262
519
|
applied?: Set<string>; // entry ids whose fold the latest request view actually carried (what turn_end must persist, no more)
|
|
263
520
|
cutTs?: number; // timestamp of the request-local summary message: one value per run, so every request of the run is identical
|
|
264
521
|
untouched?: number; // real tokens before the earliest edited block: what the first request carrying the plan can still read from the cache
|
|
522
|
+
waitMs?: number; // how long the user waited for this plan's summary (0: none, or it was ready)
|
|
523
|
+
noticed?: boolean; // its notice is already in the transcript (shown when the first request carrying it went out)
|
|
524
|
+
native?: Cut; // this run's summary went in through Pi's own compaction before the run: no request-local cut, no compaction draft
|
|
265
525
|
}
|
|
266
526
|
|
|
267
527
|
export interface PlanOpts {
|
|
@@ -270,8 +530,7 @@ export interface PlanOpts {
|
|
|
270
530
|
trace?: (LawTerms & { where: "summary" | "plan" })[]; // every law evaluation is pushed here (ledger)
|
|
271
531
|
reserve?: number; // Pi's compaction reserveTokens (default 16384)
|
|
272
532
|
steerIds?: Set<string>; // user entries that are mid-run steering or follow-up messages, not new user turns
|
|
273
|
-
|
|
274
|
-
k?: number; // real tokens per estimated token (calibrate); default 1 = the estimates are taken as real
|
|
533
|
+
scale?: Scale; // real = O + c x content estimate (calibrate); default UNIT = the estimates are taken as real
|
|
275
534
|
cwd: string;
|
|
276
535
|
coldCap?: number;
|
|
277
536
|
foldMin?: number;
|
|
@@ -284,7 +543,14 @@ export interface PlanOpts {
|
|
|
284
543
|
inturnAge?: number; // default settings().inturnAge (PI_ZIP_INTURN_AGE, 60); 0 = outputs of the protected turns never fold on age
|
|
285
544
|
noSummary?: boolean; // a summary was already made while this cache stayed warm: no second one unless the context is at/above the compaction room
|
|
286
545
|
rereadOnly?: boolean; // zip_recall is not available to the model: fold only outputs that can be re-read (classify "rereadable")
|
|
546
|
+
fileRecall?: FileRecall; // zip_recall is hidden but read, grep or bash is active: fold as in full mode, placeholders name the recall file
|
|
287
547
|
pastTtl?: boolean; // a user return after the declared TTL: a warm plan may relax into the previous user turn too (RELAX_PREV_TURN)
|
|
548
|
+
warmSummaryGate?: boolean; // F1, default WARM_SUMMARY_GATE
|
|
549
|
+
lookahead?: boolean; // F2, default LOOKAHEAD (cold plans only)
|
|
550
|
+
narrativePrice?: boolean; // F3, default NARRATIVE_PRICE
|
|
551
|
+
slide?: number; // internal (F2): plan as if this many more user turns had started (the protected window slid)
|
|
552
|
+
valveB?: number; // internal (F2): the real context the slid warm plan's law sees (the cold plan's folds already applied)
|
|
553
|
+
warmStyle?: boolean; // internal: a cold plan at 0 < P(warm) < 0.5 that the law refused is planned once more the way a warm plan is (no relax folds unless pastTtl, the summary gated)
|
|
288
554
|
}
|
|
289
555
|
|
|
290
556
|
export interface PlanResult {
|
|
@@ -293,21 +559,33 @@ export interface PlanResult {
|
|
|
293
559
|
folds: FoldTarget[];
|
|
294
560
|
cutIdx: number | null;
|
|
295
561
|
firstKeptEntryId: string | null;
|
|
296
|
-
prefixTokens: number; // estimate units (x
|
|
562
|
+
prefixTokens: number; // content estimate units (x c = real), like summaryTokensPlanned
|
|
297
563
|
summaryTokensPlanned: number;
|
|
298
564
|
sumTrigger: "cold" | "valve" | null;
|
|
299
|
-
|
|
300
|
-
ctxTokens: number; //
|
|
301
|
-
ctxAfterFolds: number; //
|
|
565
|
+
scale: Scale;
|
|
566
|
+
ctxTokens: number; // real: O + c x content
|
|
567
|
+
ctxAfterFolds: number; // real
|
|
302
568
|
userTurns: number;
|
|
303
569
|
cutVisibleIdx: number;
|
|
304
570
|
cutKeptFirstMsg: Any;
|
|
571
|
+
fileRecall?: FileRecall; // the summary of this plan names recall files instead of zip_recall (PlanOpts.fileRecall)
|
|
572
|
+
ahead?: boolean; // F2: the cut was decided by the lookahead (it may lie inside the previous user turn, before its final answer)
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
/** Index of the final assistant message of user turn `turn` (the answer the user saw), -1 if none. */
|
|
576
|
+
export function finalAnswerIdx(blocks: Block[], turn: number): number {
|
|
577
|
+
for (let i = blocks.length - 1; i >= 0; i--) {
|
|
578
|
+
const b = blocks[i];
|
|
579
|
+
if (b.userTurn < turn) break;
|
|
580
|
+
if (b.userTurn === turn && b.kind === "assistant" && b.entryId && b.msg?.stopReason !== "error" && b.msg?.stopReason !== "aborted") return i;
|
|
581
|
+
}
|
|
582
|
+
return -1;
|
|
305
583
|
}
|
|
306
584
|
|
|
307
|
-
/** Estimated
|
|
308
|
-
export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean
|
|
585
|
+
/** Estimated content before the earliest edited block: what stays cached through the edit with O (real: O + c x it). A cut edits from the start. */
|
|
586
|
+
export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean): number {
|
|
309
587
|
const ids = new Set(folds.map((t) => t.entryId));
|
|
310
|
-
let est =
|
|
588
|
+
let est = 0;
|
|
311
589
|
if (!cut) for (const b of blocks) { if (b.entryId && ids.has(b.entryId)) break; est += b.tokens; }
|
|
312
590
|
return est;
|
|
313
591
|
}
|
|
@@ -320,35 +598,41 @@ export const contentKeyOf = (content: Any): string => {
|
|
|
320
598
|
|
|
321
599
|
/** The planner. Cold: fold everything outside the protected window, relax into the previous turn if still above the cap,
|
|
322
600
|
* summarise only if folds cannot reach the cap and the law prices the summary call in. Warm: null unless the context is above the
|
|
323
|
-
* cold cap and the law fires for the plan
|
|
601
|
+
* cold cap and the law fires for the plan; its summary passes the same summary gate (WARM_SUMMARY_GATE); then the cold plan minus the previous-turn relax (a warm plan
|
|
324
602
|
* folds the previous user turn by relax only at a return after the declared TTL, o.pastTtl). A cold plan with P(warm) > 0
|
|
325
603
|
* passes the same law (expected cost). The cap never exceeds Pi's compaction room. Returns null when there is nothing to plan on
|
|
326
604
|
* or the law says no. */
|
|
327
605
|
export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
328
606
|
const s = settings();
|
|
329
607
|
const mode = o.mode ?? "cold";
|
|
330
|
-
|
|
331
|
-
|
|
608
|
+
const warmLike = mode === "warm" || o.warmStyle === true; // plans the way a warm plan does (the label of the trigger stays the mode's)
|
|
609
|
+
// limits are in real tokens; the plan works in content estimate units (real = O + c x estimate), so they are converted once, here.
|
|
610
|
+
// The cold cap is on the whole context, but leaves at least half of itself for the conversation above O (coldTarget): a cap the
|
|
611
|
+
// fixed prefix alone fills is out of reach, and every cold return then folds all it may and summarises. The compaction room is
|
|
612
|
+
// Pi's trigger on the whole context.
|
|
613
|
+
const { O, c } = o.scale ?? UNIT;
|
|
614
|
+
const real = (est: number) => O + c * est;
|
|
332
615
|
const room = compactionRoom(o.model, o.reserve);
|
|
333
|
-
const coldCap =
|
|
616
|
+
const coldCap = (coldTarget(o.coldCap ?? s.coldCap, O, room) - O) / c;
|
|
334
617
|
const law: Law = o.law ?? { pr: null, g: G0, pWarm: mode === "cold" ? 0 : 1 };
|
|
335
618
|
const gate = (where: "summary" | "plan", B: number, A: number, r: number | null, call: number, T: number, Tsuf?: number): boolean => {
|
|
336
619
|
const t = lawTerms(B, A, law.pWarm, r, law.pr, law.g, call, T);
|
|
337
620
|
o.trace?.push({ where, ...t, ...(Tsuf !== undefined ? { Tsuf } : {}) });
|
|
338
621
|
return t.ok;
|
|
339
622
|
};
|
|
340
|
-
const minGain =
|
|
623
|
+
const minGain = o.minGain ?? s.minGain;
|
|
341
624
|
const foldMin = o.foldMin ?? s.foldMin;
|
|
342
625
|
const keepLines = o.keepLines ?? s.keepLines;
|
|
343
626
|
const relax = o.relax ?? RELAX_PREV_TURN;
|
|
344
627
|
const inturnAge = o.inturnAge ?? s.inturnAge;
|
|
345
628
|
const blocks = buildBlocks(entries, o.steerIds);
|
|
346
629
|
if (!blocks.length) return null;
|
|
347
|
-
const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0); // the upcoming prompt is a new user turn
|
|
630
|
+
const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0) + (o.slide ?? 0); // the upcoming prompt is a new user turn
|
|
631
|
+
const f1 = o.warmSummaryGate ?? WARM_SUMMARY_GATE, f2 = mode === "cold" && !o.warmStyle && (o.lookahead ?? LOOKAHEAD), f3 = o.narrativePrice ?? NARRATIVE_PRICE;
|
|
348
632
|
const calls = toolCallIndex(blocks);
|
|
349
633
|
const recalled = o.recalled ?? new Set<string>();
|
|
350
|
-
const ctxEst =
|
|
351
|
-
if (
|
|
634
|
+
const ctxEst = blocks.reduce((a, b) => a + b.tokens, 0); // content
|
|
635
|
+
if (warmLike && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
|
|
352
636
|
const trig = mode === "warm" ? "valve" : "cold";
|
|
353
637
|
const tokOverride = new Map<number, number>();
|
|
354
638
|
const folds: FoldTarget[] = [];
|
|
@@ -368,9 +652,9 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
368
652
|
};
|
|
369
653
|
const addFold = (b: Block, trig: string): boolean => {
|
|
370
654
|
if (tokOverride.has(b.idx) || recalled.has(handleFor(b.entryId!))) return false; // F4: never refold what the model recalled
|
|
371
|
-
const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly);
|
|
655
|
+
const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly, o.fileRecall);
|
|
372
656
|
if (ph === null) return false;
|
|
373
|
-
const phTok =
|
|
657
|
+
const phTok = tokensOf({ ...b.msg, content: [{ type: "text", text: ph }] }); // the folded message: its id and framing stay
|
|
374
658
|
if (!(phTok < 0.9 * b.tokens)) return false;
|
|
375
659
|
tokOverride.set(b.idx, phTok);
|
|
376
660
|
const call = calls.get(b.msg.toolCallId);
|
|
@@ -378,20 +662,26 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
378
662
|
return true;
|
|
379
663
|
};
|
|
380
664
|
// a zip_recall result IS content the model just asked for: folding it would undo the recall (and loop); never
|
|
381
|
-
|
|
665
|
+
// ... and so is a built-in read, grep, find, ls or bash call on a recall file (file recall, when an allowlist hides zip_recall)
|
|
666
|
+
const isRecall = (b: Block) => {
|
|
667
|
+
const call = calls.get(b.msg.toolCallId), name = call?.name ?? b.msg.toolName;
|
|
668
|
+
if (name === RECALL_TOOL) return true;
|
|
669
|
+
if (name === "bash") return commandNamesRecall(call?.args?.command);
|
|
670
|
+
return (name === "read" || name === "grep" || name === "find" || name === "ls") && !!parseRecallPath(call?.args?.path);
|
|
671
|
+
};
|
|
382
672
|
// Observability: on an automatic (prefix) cache no fold starts inside the first MIN_EXPECT real tokens, so the next response's cacheRead
|
|
383
673
|
// is an uncensored survival sample (learn.ts sample). Without it every idle return folded into the first 1-7K tokens and its sample was
|
|
384
674
|
// censored, so the learned curve never saw a return (live GLM bench). Explicit caches are exempt: the provider looks back only ~20
|
|
385
675
|
// blocks from the last breakpoint, a far head is not read either way. A cut replaces the prefix from the first message: exempt too.
|
|
386
|
-
const head = law.pr?.cls === "automatic" ? MIN_EXPECT /
|
|
676
|
+
const head = law.pr?.cls === "automatic" ? (MIN_EXPECT - O) / c : 0; // MIN_EXPECT counts from the first token, O included
|
|
387
677
|
const start: number[] = [];
|
|
388
|
-
blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens),
|
|
678
|
+
blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), 0);
|
|
389
679
|
const foldable = (b: Block) => b.kind === "toolResult" && !b.edited && !!b.entryId && b.tokens > foldMin && !isRecall(b) && start[b.idx] >= head && (!o.rereadOnly || classify(b) === "rereadable");
|
|
390
680
|
const protectedTurn = (b: Block) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1;
|
|
391
681
|
const savings = () => folds.reduce((a, t) => a + t.entryTokens - t.phTokens, 0);
|
|
392
682
|
const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b));
|
|
393
683
|
for (const b of cands) addFold(b, trig);
|
|
394
|
-
if (relax && (mode === "cold" || o.pastTtl === true)) {
|
|
684
|
+
if (relax && ((mode === "cold" && !o.warmStyle) || o.pastTtl === true)) {
|
|
395
685
|
// the protected window = the new prompt + the previous user turn; that turn's big reads are what makes a cold return
|
|
396
686
|
// expensive. Rereadable ones can be recalled exactly: fold them biggest-first until the cap; never the latest turn's own.
|
|
397
687
|
// Cold plans, and warm plans at a return after the declared TTL; any other warm plan keeps the previous user turn visible.
|
|
@@ -418,49 +708,78 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
418
708
|
}
|
|
419
709
|
}
|
|
420
710
|
let ctxAfterFolds = ctxEst - savings();
|
|
421
|
-
// what no summary can remove
|
|
422
|
-
//
|
|
711
|
+
// what no summary can remove (besides O): the protected turns (after their folds) and the previous summary, which the next one
|
|
712
|
+
// carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
|
|
423
713
|
const prevSum = blocks[0]?.kind === "summary" ? blocks[0].tokens : 0;
|
|
424
|
-
const floor =
|
|
714
|
+
const floor = prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
|
|
425
715
|
const target = Math.max(coldCap, floor + SUMMARY_FLOOR);
|
|
426
|
-
const atRoom = room !== null &&
|
|
716
|
+
const atRoom = room !== null && real(ctxAfterFolds) >= room;
|
|
427
717
|
const sumTrigger: "cold" | "valve" | null = ctxAfterFolds > target && (!o.noSummary || atRoom) ? trig : null;
|
|
428
718
|
let cutIdx: number | null = null;
|
|
429
719
|
let prefixTokens = 0;
|
|
430
720
|
let summaryTokensPlanned = 0;
|
|
431
|
-
|
|
432
|
-
|
|
721
|
+
let ahead = false;
|
|
722
|
+
const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
|
|
723
|
+
const pre: number[] = [];
|
|
724
|
+
const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
|
|
725
|
+
// Block index of every fold: the folds a cut voids are those before it
|
|
726
|
+
const foldAt = folds.map((t) => blocks.findIndex((b) => b.entryId === t.entryId));
|
|
727
|
+
/** The whole plan's law (F10): B is the context as it is (or as the slid valve saw it), A what the plan leaves, `call` the summary call
|
|
728
|
+
* a cut makes; `untouched` is the content before the earliest edit (ledger only). A cold plan with some chance of a warm cache is
|
|
729
|
+
* priced by expectation (no prices: a cold plan always fires). T = A (the whole post-edit context, O included, as tuned: the cheaper
|
|
730
|
+
* A - O fires warm edits earlier and loses items offline). */
|
|
731
|
+
const planGate = (after: number, untouched: number, call: number, planned: boolean): boolean =>
|
|
732
|
+
!(warmLike || (planned && law.pr)) || gate("plan", Math.max(o.valveB ?? real(ctxEst), real(after)), real(after), room, call, real(after), c * (after - untouched));
|
|
733
|
+
/** The cut for cut points minCut..maxCut: the first that brings the context to the cap, else the deepest; gated = the summary law
|
|
734
|
+
* (no prices: the legacy gain rule) must pass. When the minimal cut's summary or the whole plan with it is refused, the deeper cuts
|
|
735
|
+
* are tried in turn (the gain grows as D^2, the call only with the prefix read). The whole plan's law prices the summary call too.
|
|
736
|
+
* Applies the first cut both laws accept (folds in the prefix are void) and returns true, or changes nothing. */
|
|
737
|
+
const summarise = (maxCut: number, gated: boolean): boolean => {
|
|
433
738
|
const cuts: number[] = [];
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
if (limit >= 0 && i > limit) break;
|
|
437
|
-
if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
|
|
438
|
-
}
|
|
439
|
-
const pre: number[] = [];
|
|
739
|
+
for (let i = minCut; i < blocks.length && i <= maxCut; i++) if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
|
|
740
|
+
if (!cuts.length) return false;
|
|
440
741
|
let acc = 0;
|
|
441
742
|
for (let i = 0; i < blocks.length; i++) {
|
|
442
743
|
pre[i] = acc;
|
|
443
744
|
acc += tokOverride.get(i) ?? blocks[i].tokens;
|
|
444
745
|
}
|
|
445
|
-
const total =
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
//
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
const bi = blocks.findIndex((b) => b.entryId === folds[i].entryId);
|
|
459
|
-
if (bi < pick) { tokOverride.delete(bi); folds.splice(i, 1); } // the prefix is replaced: its folds are void
|
|
460
|
-
}
|
|
461
|
-
ctxAfterFolds = ctxEst - savings();
|
|
746
|
+
const total = acc;
|
|
747
|
+
let first = cuts.length - 1;
|
|
748
|
+
for (let k = 0; k < cuts.length; k++) if (total - pre[cuts[k]] + S(pre[cuts[k]]) <= coldCap) { first = k; break; }
|
|
749
|
+
for (const pick of cuts.slice(first)) {
|
|
750
|
+
const X = c * pre[pick], Sr = c * S(pre[pick]);
|
|
751
|
+
// the call is the narrative call summary.ts makes (F3, NARRATIVE_PRICE): uncached input of the prefix without the previous
|
|
752
|
+
// summary, which it never reads, + output of at most the narrative budget; without F3 the verdict's w X + out S (the carried
|
|
753
|
+
// summary as output)
|
|
754
|
+
const call = !law.pr ? 0 : f3 ? law.pr.input * c * Math.max(pre[pick] - prevSum, 0) + law.pr.out * c * Math.min(S(pre[pick]), NARRATIVE_MAX) : law.pr.w * X + law.pr.out * Sr;
|
|
755
|
+
if (gated) {
|
|
756
|
+
// the summary call must pay for itself, cold or warm (F1, WARM_SUMMARY_GATE; before, a warm plan skipped the gate)
|
|
757
|
+
const ok = !law.pr ? summaryGainOk(X, Sr, real(total), minGain) : (warmLike && !f1) || gate("summary", X, Sr, null, call, Sr);
|
|
758
|
+
if (!ok) continue;
|
|
462
759
|
}
|
|
760
|
+
if (!(pre[pick] >= 2 * S(pre[pick]))) continue;
|
|
761
|
+
// what the plan leaves: the context with its folds (those inside the prefix too: the prefix is measured as folded, `pre`) minus
|
|
762
|
+
// what the summary replaces plus the summary. (Counting the voided folds' savings back, as before, overstated it by them.)
|
|
763
|
+
if (!planGate(ctxAfterFolds - (pre[pick] - S(pre[pick])), 0, call, true)) continue;
|
|
764
|
+
cutIdx = pick;
|
|
765
|
+
prefixTokens = pre[pick];
|
|
766
|
+
summaryTokensPlanned = S(pre[pick]);
|
|
767
|
+
for (let i = folds.length - 1; i >= 0; i--) if (foldAt[i] < pick) { tokOverride.delete(foldAt[i]); folds.splice(i, 1); foldAt.splice(i, 1); } // the prefix is replaced: its folds are void
|
|
768
|
+
return true;
|
|
463
769
|
}
|
|
770
|
+
return false;
|
|
771
|
+
};
|
|
772
|
+
if (sumTrigger) {
|
|
773
|
+
const limit = blocks.findIndex((b) => protectedTurn(b));
|
|
774
|
+
summarise(limit >= 0 ? limit : blocks.length - 1, true);
|
|
775
|
+
}
|
|
776
|
+
if (f2 && cutIdx === null && law.pr && ctxAfterFolds > coldCap && (!o.noSummary || atRoom)) {
|
|
777
|
+
// F2 lookahead: the warm valve's plan at the next user turn (this context, its folds applied, the window slid by one turn). If it
|
|
778
|
+
// would summarise, the summary is written now, at the cold return, where the rewrite costs nothing extra (and at settle it is
|
|
779
|
+
// prepared while the user is away), instead of one turn later as a warm rewrite the user waits for.
|
|
780
|
+
const next = planContext(entries, { ...o, mode: "warm", slide: (o.slide ?? 0) + 1, lookahead: false, pastTtl: false, noSummary: false, trace: undefined, law: { ...law, pWarm: 1 }, valveB: real(ctxAfterFolds) });
|
|
781
|
+
const fa = finalAnswerIdx(blocks, userTurns - 1);
|
|
782
|
+
if (next && next.cutIdx !== null && fa >= minCut && summarise(fa, false)) ahead = true;
|
|
464
783
|
}
|
|
465
784
|
// where the kept part starts among the projected non-system messages (the request-local view drops everything before it)
|
|
466
785
|
let cutVisibleIdx = -1;
|
|
@@ -480,13 +799,17 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
480
799
|
}
|
|
481
800
|
}
|
|
482
801
|
// the untouched prefix: everything before the earliest edited block (a cut edits from the first block on)
|
|
483
|
-
const untouched = untouchedEst(blocks, folds, cutIdx !== null
|
|
802
|
+
const untouched = untouchedEst(blocks, folds, cutIdx !== null);
|
|
484
803
|
const after = ctxAfterFolds - (cutIdx !== null ? prefixTokens - summaryTokensPlanned : 0);
|
|
485
804
|
const planned = folds.length > 0 || cutIdx !== null;
|
|
486
|
-
//
|
|
487
|
-
//
|
|
488
|
-
if (
|
|
489
|
-
|
|
805
|
+
// a plan with a cut passed the whole plan's law with its call inside summarise; folds only (no summary wanted, or none the laws
|
|
806
|
+
// accepted: the deeper cuts and the cut-free plan both stand) is judged here
|
|
807
|
+
if (cutIdx === null && !planGate(after, untouched, 0, planned)) {
|
|
808
|
+
// a cold plan with some chance of a warm cache: the plan a warm cache would make may pay where this one does not
|
|
809
|
+
if (mode === "cold" && !o.warmStyle && law.pr && law.pWarm > 0 && law.pWarm < 0.5) return planContext(entries, { ...o, warmStyle: true });
|
|
810
|
+
return null;
|
|
811
|
+
}
|
|
812
|
+
return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger: cutIdx !== null ? (sumTrigger ?? trig) : sumTrigger, scale: o.scale ?? UNIT, ctxTokens: real(ctxEst), ctxAfterFolds: real(ctxAfterFolds), userTurns, cutVisibleIdx, cutKeptFirstMsg, ...(o.fileRecall ? { fileRecall: o.fileRecall } : {}), ...(ahead ? { ahead } : {}) };
|
|
490
813
|
}
|
|
491
814
|
|
|
492
815
|
const sameMsg = (a: Any, b: Any): boolean =>
|
|
@@ -532,6 +855,11 @@ export function collapseSystem(messages: Any[]): Any | undefined {
|
|
|
532
855
|
* message, the summary, then the kept non-system messages. null = nothing to apply. A cut whose kept boundary cannot be
|
|
533
856
|
* located is dropped (folds still apply); an inconsistent split is never sent.
|
|
534
857
|
*/
|
|
858
|
+
/** Fold targets of a plan made on the view before a native compaction, renumbered for the request Pi builds after it: Pi's summary
|
|
859
|
+
* message first, then the kept part (positions among the non-system messages, as in applyPlanToMessages). */
|
|
860
|
+
export const rebaseFolds = (folds: FoldTarget[], cutVisibleIdx: number): FoldTarget[] =>
|
|
861
|
+
folds.map((t) => ({ ...t, visIdx: t.visIdx >= 0 && cutVisibleIdx >= 0 && t.visIdx >= cutVisibleIdx ? t.visIdx - cutVisibleIdx + 1 : -1 }));
|
|
862
|
+
|
|
535
863
|
export function applyPlanToMessages(messages: Any[], plan: RunPlan | null): { messages: Any[]; droppedCut: boolean; applied: Set<string> } | null {
|
|
536
864
|
if (!plan || plan.persisted || (!plan.folds.length && !plan.cut)) return null;
|
|
537
865
|
const vis: number[] = []; // indices of the non-system messages
|