pi-zip 0.2.8 → 0.3.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +72 -42
- package/package.json +1 -1
- package/src/cache.ts +50 -12
- package/src/guard.ts +3 -2
- package/src/index.ts +30 -8
- package/src/learn.ts +18 -8
- package/src/notice.ts +91 -33
- package/src/placeholder.ts +12 -4
- package/src/plan.ts +470 -107
- package/src/recall.ts +111 -27
- package/src/recallfile.ts +165 -0
- package/src/run.ts +807 -122
- package/src/state.ts +53 -0
- package/src/summary.ts +22 -10
- package/src/ui.ts +214 -46
- package/src/util.ts +62 -2
package/src/plan.ts
CHANGED
|
@@ -2,9 +2,10 @@
|
|
|
2
2
|
import { classifyRecoverability, type Recover } from "./classify.ts";
|
|
3
3
|
import { handleFor, makePlaceholderFor, pickKeyLines, RECALL_TOOL, shortArgs } from "./placeholder.ts";
|
|
4
4
|
import { createHash } from "node:crypto";
|
|
5
|
-
import { type Any, PRODUCT, clamp, envInt, textOf, tok4, tokensOf } from "./util.ts";
|
|
5
|
+
import { type Any, COMMIT_CUSTOM, PRODUCT, clamp, envInt, piTokensOf, textOf, tok4, tokensOf } from "./util.ts";
|
|
6
6
|
import { PH_MARK } from "./placeholder.ts";
|
|
7
7
|
import { MIN_EXPECT, type Prices } from "./learn.ts";
|
|
8
|
+
import { commandNamesRecall, parseRecallPath, type FileRecall } from "./recallfile.ts";
|
|
8
9
|
|
|
9
10
|
/** Default: when a COLD plan (P(warm) < 0.5) is still above the cap, also fold REREADABLE outputs of the previous user turn, biggest
|
|
10
11
|
* first. A warm plan does so only at a user return after the declared TTL (PlanOpts.pastTtl): the very returns where the TTL rule
|
|
@@ -22,7 +23,31 @@ export const RELAX_PREV_TURN = true;
|
|
|
22
23
|
* constant (largest saving with lost <= prod in every stratum); the old rereadable-only filter cost 1.026 / 1.018x at -0.6 lost items. */
|
|
23
24
|
export const INTURN_AGE = 60;
|
|
24
25
|
|
|
25
|
-
const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised; folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
|
|
26
|
+
const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised (but see LOOKAHEAD); folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
|
|
27
|
+
|
|
28
|
+
/** F1: a warm (valve) summary passes the same summary gate as a cold one, the call priced in (before: a warm plan skipped it, so a
|
|
29
|
+
* summary the cold gate had just refused was written one request later anyway). PlanOpts.warmSummaryGate. */
|
|
30
|
+
export const WARM_SUMMARY_GATE = true;
|
|
31
|
+
/** F2: a cold plan that does not summarise looks one user turn ahead: if the warm valve would summarise at the next user turn, once the
|
|
32
|
+
* protected window has slid past the previous turn, the cold plan summarises now (the rewrite is free now, not then), its cut allowed
|
|
33
|
+
* inside the previous user turn up to that turn's final assistant answer, which stays verbatim with everything after it. PlanOpts.lookahead. */
|
|
34
|
+
export const LOOKAHEAD = false;
|
|
35
|
+
/** F3: the summary gates price the call the code makes (summary.ts narrative): uncached input of the prefix the narrative reads (the
|
|
36
|
+
* previous summary is not sent: its sections are carried forward by code) plus output of at most the narrative budget, instead of
|
|
37
|
+
* w x the prefix + out x the whole planned summary (which counts the carried-forward previous summary as model output). PlanOpts.narrativePrice. */
|
|
38
|
+
export const NARRATIVE_PRICE = true;
|
|
39
|
+
/** The narrative's token budget (chars/4), summary.ts buildCut: clamp(planned - skeleton, 300, NARRATIVE_MAX). */
|
|
40
|
+
export const NARRATIVE_MAX = 4000;
|
|
41
|
+
/** B2 for folds (formal TRIAGE (a) 6): the outputs a cold plan must protect (the previous user turn's, the mutating ones) are no longer
|
|
42
|
+
* protected one prompt later, when the window slides, and the warm plan then folded them at a warm cache: a rewrite the free cold one
|
|
43
|
+
* dominated. Now a warm plan holds back the blocks that were protected at the last cold return (PlanOpts.hold: that return's prompt)
|
|
44
|
+
* until the next cold return, or until the session has grown by more than HOLD_RELEASE real tokens since it (new information), or the
|
|
45
|
+
* context is at Pi's compaction room; and when even folding them all would not avoid a summary, it is the plan it was before. Chosen
|
|
46
|
+
* over folding them early at the cold return (the user would lose the previous turn's outputs at the prompt that asks about them):
|
|
47
|
+
* formal/TRIAGE.md, B2 for folds. */
|
|
48
|
+
export const FOLD_HOLD = true;
|
|
49
|
+
/** The spec's "no new information" is growth <= 10K real tokens (formal DELTA); the estimate is chars/4 x c, 15% off at worst (TOL_TARGET). */
|
|
50
|
+
export const HOLD_RELEASE = 11_500;
|
|
26
51
|
const SUMMARY_FLOOR = 1000;
|
|
27
52
|
const SUMMARY_CAP = 8000;
|
|
28
53
|
const SUMMARY_RATIO = 0.1;
|
|
@@ -55,6 +80,56 @@ export function legacyValve(before: number, after: number, room: number | null):
|
|
|
55
80
|
}
|
|
56
81
|
|
|
57
82
|
export const PI_KEEP_RECENT = 20_000; // Pi's compaction keepRecentTokens default
|
|
83
|
+
|
|
84
|
+
/** Pi's compaction keepRecentTokens for this model: compaction.modelOverrides["provider/id"], else compaction.keepRecentTokens, else 20000. */
|
|
85
|
+
export function keepRecentTokensFor(settings: Any, model: Any): number {
|
|
86
|
+
const c = settings?.compaction;
|
|
87
|
+
const key = model ? `${model.provider}/${model.id}` : "";
|
|
88
|
+
for (const v of [c?.modelOverrides?.[key]?.keepRecentTokens, c?.keepRecentTokens]) if (typeof v === "number" && Number.isFinite(v) && v >= 0) return v;
|
|
89
|
+
return PI_KEEP_RECENT;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const CUT_POINT_ROLES = new Set(["user", "assistant", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
|
|
93
|
+
const TURN_START_ROLES = new Set(["user", "bashExecution", "custom", "branchSummary", "compactionSummary"]);
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Would Pi's compact() find something to compact in this projection (the branch as it is now), or throw "Nothing to compact (session
|
|
97
|
+
* too small)"? A replica of the emptiness test of Pi's prepareCompaction (core/compaction/compaction.js, not exported): Pi cuts where
|
|
98
|
+
* keepRecent of its chars/4 tokens, counted from the end, begin, at the next cut point; the conversation before that cut (or before the
|
|
99
|
+
* turn it splits) must not be empty. Pi's own extra moves of the cut (over entries without messages, past recovery omissions) never
|
|
100
|
+
* empty it. The caller checks "Already compacted" (the newest branch entry is a compaction) itself.
|
|
101
|
+
*/
|
|
102
|
+
export function piCanCompact(entries: Any[], keepRecent: number): boolean {
|
|
103
|
+
const msgs = (e: Any): Any[] => (Array.isArray(e?.messages) ? e.messages : []);
|
|
104
|
+
const isComp = (e: Any) => e?.sourceEntry?.type === "compaction";
|
|
105
|
+
const prev = entries.findIndex((e) => isComp(e) && msgs(e).length > 0);
|
|
106
|
+
const start = prev >= 0 ? prev + 1 : 0, end = entries.length;
|
|
107
|
+
const cuts: number[] = [];
|
|
108
|
+
for (let i = start; i < end; i++) if (!isComp(entries[i]) && msgs(entries[i]).some((m) => CUT_POINT_ROLES.has(m?.role))) cuts.push(i);
|
|
109
|
+
if (!cuts.length) return false;
|
|
110
|
+
let cut = cuts[0], acc = 0;
|
|
111
|
+
for (let i = end - 1; i >= start; i--) {
|
|
112
|
+
const t = msgs(entries[i]).reduce((a, m) => a + piTokensOf(m), 0);
|
|
113
|
+
if (!t) continue;
|
|
114
|
+
acc += t;
|
|
115
|
+
if (acc >= keepRecent) { cut = cuts.find((c) => c >= i) ?? cuts[cuts.length - 1]; break; }
|
|
116
|
+
}
|
|
117
|
+
while (cut > start && !isComp(entries[cut - 1]) && !msgs(entries[cut - 1]).length) cut--;
|
|
118
|
+
if (!entries[cut]?.sourceEntry?.id) return false;
|
|
119
|
+
const startsTurn = (e: Any) => !isComp(e) && msgs(e).some((m) => TURN_START_ROLES.has(m?.role));
|
|
120
|
+
let turn = -1;
|
|
121
|
+
if (!startsTurn(entries[cut])) for (let i = cut; i >= start; i--) if (startsTurn(entries[i])) { turn = i; break; }
|
|
122
|
+
const conversation = (from: number, to: number) => entries.slice(from, to).some((e) => !isComp(e) && msgs(e).some((m) => m?.role !== "system"));
|
|
123
|
+
return turn >= 0 ? conversation(start, turn) || conversation(turn, cut) : conversation(start, cut);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** The projection a run plan would have seen without pi-zip's native compaction: `pre` (taken right before it) plus what was appended
|
|
127
|
+
* after the compaction entry (the prompt and anything sent with it). null when that compaction is no longer the head of `now`. */
|
|
128
|
+
export function spliceNative(pre: Any[], now: Any[], compactionId: string): Any[] | null {
|
|
129
|
+
if (!compactionId || now[0]?.sourceEntry?.id !== compactionId) return null;
|
|
130
|
+
const known = new Set(pre.map((e) => e?.sourceEntry?.id).filter((x) => typeof x === "string"));
|
|
131
|
+
return [...pre, ...now.slice(1).filter((e) => !known.has(e?.sourceEntry?.id))];
|
|
132
|
+
}
|
|
58
133
|
export const G0 = 2_500; // growth per request before the session's own EWMA has data
|
|
59
134
|
|
|
60
135
|
/** What the law knows at a decision: prices (null = legacy rule), growth per request g (real tokens), P(the cache is warm). */
|
|
@@ -88,58 +163,249 @@ export function lawTerms(B: number, A: number, P: number, room: number | null, p
|
|
|
88
163
|
export const editAllowed = (B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A): boolean =>
|
|
89
164
|
lawTerms(B, A, P, room, pr, g, call, T).ok;
|
|
90
165
|
|
|
91
|
-
// Token scale. Every size in this file is a chars/4 estimate
|
|
92
|
-
//
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
166
|
+
// Token scale. Every size in this file is a chars/4 estimate of the CONTENT (Pi's own rule); the provider counts real tokens, and the
|
|
167
|
+
// two are not proportional: real = O + c x content.
|
|
168
|
+
// O what every request carries and no edit can shrink: system prompt, tool schemas, the provider's own framing (2-4K on a bare Pi,
|
|
169
|
+
// 29-55K with a large tool set);
|
|
170
|
+
// c real tokens per estimated content token (0.7-1.1 GLM / GPT, 1.5-2.2 Claude; more on CJK-heavy content).
|
|
171
|
+
// pi-zip <= 0.2.9 used ONE ratio, k = real / (system prompt + content) clamped to [1, 2.5]: with O a third of the context k sat on its
|
|
172
|
+
// clamp, and every size after an edit came out 10-15K too small (research round5/affine.md). Both are read from the branch on every
|
|
173
|
+
// decision, so a restart or `pi -p` calibrates exactly like a long-lived process, and nothing is stored:
|
|
174
|
+
// O 1. the cacheRead of the request that first carried the newest summary: everything after the system prompt and tools changed
|
|
175
|
+
// there, so what the provider still read from its cache is exactly O (when the cache was warm at all);
|
|
176
|
+
// 2. else the session's first request minus C0 x its content (an opening prompt is small next to O);
|
|
177
|
+
// 3. else R0 x the chars/4 of the system prompt and the JSON of the declared tools.
|
|
178
|
+
// An O measured under another system prompt or tool set moves by R0 x the change of that chars/4 size (MCP tools come and go).
|
|
179
|
+
// c (real - O) / content of the newest response, clamped to [C_MIN, C_MAX] (a guard for a tiny content estimate; recorded sessions
|
|
180
|
+
// span 0.7-3.9); C0 before the first response. A response followed by edits it did not carry (a warm-valve edit is persisted at
|
|
181
|
+
// turn_end and first sent with the next request: that response's prompt did not shrink) is counted with the originals.
|
|
182
|
+
// Both are per model: only responses of the model the next request goes to count. After a switch, before its first response,
|
|
183
|
+
// O = R0 x the system state and c = C0; after it, with no O evidence of its own, O = c x X, X = O / c of the previous model.
|
|
184
|
+
export const C0 = 1.5;
|
|
185
|
+
export const C_MIN = 0.5;
|
|
186
|
+
export const C_MAX = 4;
|
|
187
|
+
export const R0 = 2;
|
|
188
|
+
|
|
189
|
+
/** The system prompt and the declared tools (Pi API fallback for a transcript that records no system message). */
|
|
190
|
+
export interface SysInfo { text: string; tools: Any[] }
|
|
191
|
+
/** chars/4 of the system prompt plus the JSON of the declared tool definitions (name, description, parameters). */
|
|
192
|
+
export const sysEst = (s: SysInfo): number =>
|
|
193
|
+
tok4(s.text) + (s.tools.length ? tok4(JSON.stringify(s.tools.map((t: Any) => ({ name: t?.name, description: t?.description, parameters: t?.parameters })))) : 0);
|
|
194
|
+
|
|
195
|
+
export interface Scale {
|
|
196
|
+
O: number; // real tokens every request carries before the content (system prompt, tools, framing)
|
|
197
|
+
c: number; // real tokens per estimated (chars/4) content token
|
|
198
|
+
real: number; // input + cacheRead + cacheWrite of the newest usable response (0 = none)
|
|
199
|
+
est: number; // estimated content that request carried
|
|
200
|
+
source: "usage" | "default"; // where c comes from
|
|
201
|
+
oSource: "summary" | "first" | "model" | "default"; // where O comes from ("model": another model's O in content units, X = O' / c')
|
|
202
|
+
}
|
|
203
|
+
/** Sizes taken as real (tests, and the ledger of a plan made without calibration). */
|
|
204
|
+
export const UNIT: Scale = { O: 0, c: 1, real: 0, est: 0, source: "default", oSource: "default" };
|
|
205
|
+
/** Real tokens of a context whose content is estimated at `content`. */
|
|
206
|
+
export const sizeOf = (s: Pick<Scale, "O" | "c">, content: number): number => s.O + s.c * content;
|
|
207
|
+
|
|
208
|
+
/** A response pi-ai drops before sending (an error or an abort): it carries no usable usage and adds nothing to later requests. */
|
|
209
|
+
const dropped = (m: Any): boolean => m?.role === "assistant" && (m.stopReason === "error" || m.stopReason === "aborted");
|
|
210
|
+
/** chars/4 a message adds to every later request: tokensOf, except 0 for an assistant message pi-ai drops before sending. */
|
|
211
|
+
export const wireTokensOf = (m: Any): number => (dropped(m) ? 0 : tokensOf(m));
|
|
212
|
+
/** Prompt size the provider reported for a response that really ran (0 = none, an error or an abort). */
|
|
213
|
+
const promptOf = (m: Any): number => {
|
|
214
|
+
const u = m?.role === "assistant" && !dropped(m) ? m.usage : null;
|
|
215
|
+
return u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
|
|
216
|
+
};
|
|
217
|
+
/** "provider/model" of a response ("" = not recorded: such a response counts for any model). */
|
|
218
|
+
const respKey = (m: Any): string => (m?.provider || m?.model ? `${m.provider ?? ""}/${m.model ?? ""}` : "");
|
|
219
|
+
/** chars/4 of the system state the transcript records in `msgs` (-1: none recorded). */
|
|
220
|
+
function sysOf(msgs: Any[]): number {
|
|
221
|
+
const s = collapseSystem(msgs);
|
|
222
|
+
return s ? sysEst({ text: [textOf(s.content), ...Object.values<string>(s.sections ?? {})].filter((x) => x).join("\n\n"), tools: s.toolsAdded ?? [] }) : -1;
|
|
223
|
+
}
|
|
224
|
+
/** The session's first response from model `key` on the branch: its prompt, the content before it and the system state it was sent with. */
|
|
225
|
+
function firstRequest(branch: Any[], key: string): { real: number; est: number; S: number } | null {
|
|
226
|
+
const sys: Any[] = [];
|
|
227
|
+
let est = 0;
|
|
228
|
+
for (const e of branch) {
|
|
229
|
+
const m = e?.type === "message" ? e.message : e?.type === "custom_message" ? { role: "custom", content: e.content } : null;
|
|
230
|
+
if (!m) continue;
|
|
231
|
+
if (m.role === "system") { sys.push(m); continue; }
|
|
232
|
+
const real = promptOf(m), k = respKey(m);
|
|
233
|
+
if (real > 0 && (!key || !k || k === key)) return { real, est, S: sysOf(sys) };
|
|
234
|
+
est += wireTokensOf(m);
|
|
235
|
+
}
|
|
236
|
+
return null;
|
|
106
237
|
}
|
|
107
238
|
|
|
108
|
-
/**
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
const
|
|
118
|
-
const
|
|
119
|
-
let
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
239
|
+
/** chars/4 of the context the request behind the response `id` carried, rebuilt from the branch the way Pi projects it: the newest
|
|
240
|
+
* compaction before the response, the entries it keeps, everything after it, and the edits persisted before the response, plus the
|
|
241
|
+
* edits of a run plan persisted right after it (their commit entry says the response itself carried them). null: `id` is not on the
|
|
242
|
+
* branch. Used when a compaction written after the response (Pi's own, another extension's) removed that request's content from the
|
|
243
|
+
* projection, so the projection cannot say what the response measured. */
|
|
244
|
+
function requestEst(branch: Any[], id: string): number | null {
|
|
245
|
+
const at = branch.findIndex((e) => e?.id === id);
|
|
246
|
+
if (at < 0) return null;
|
|
247
|
+
const pre = branch.slice(0, at);
|
|
248
|
+
const edit = new Map<string, Any>();
|
|
249
|
+
for (const e of pre) if (e?.type === "context_edit") edit.set(e.targetId, e.replacement);
|
|
250
|
+
for (let i = at + 1; i < branch.length && !(branch[i]?.type === "message" || branch[i]?.type === "compaction"); i++) {
|
|
251
|
+
const e = branch[i];
|
|
252
|
+
if (e?.type === "custom" && e.customType === COMMIT_CUSTOM) {
|
|
253
|
+
if (e.data?.carried) for (let j = at + 1; j < i; j++) if (branch[j]?.type === "context_edit") edit.set(branch[j].targetId, branch[j].replacement);
|
|
254
|
+
break;
|
|
255
|
+
}
|
|
123
256
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
257
|
+
let ci = -1;
|
|
258
|
+
pre.forEach((e, i) => { if (e?.type === "compaction") ci = i; });
|
|
259
|
+
const kept = ci < 0 ? pre : [pre[ci], ...pre.slice(0, ci).slice(Math.max(0, pre.findIndex((e) => e?.id === pre[ci].firstKeptEntryId))).filter((e) => !(e?.type === "message" && e.message?.role === "system")), ...pre.slice(ci + 1)];
|
|
260
|
+
let est = 0;
|
|
261
|
+
for (const e of kept) {
|
|
262
|
+
if (e?.type === "compaction" || e?.type === "branch_summary") est += tokensOf({ role: "compactionSummary", summary: e.summary });
|
|
263
|
+
else if (e?.type === "custom_message") est += tokensOf({ role: "custom", content: edit.has(e.id) ? edit.get(e.id)?.content : e.content });
|
|
264
|
+
else if (e?.type === "message" && e.message) {
|
|
265
|
+
const rep = edit.get(e.id);
|
|
266
|
+
if (rep === null) continue;
|
|
267
|
+
const m = rep && e.message.role !== "system" ? { ...e.message, content: typeof rep.content === "string" && e.message.role !== "user" ? [{ type: "text", text: rep.content }] : rep.content } : e.message;
|
|
268
|
+
est += wireTokensOf(m);
|
|
269
|
+
}
|
|
137
270
|
}
|
|
138
|
-
return
|
|
271
|
+
return est;
|
|
139
272
|
}
|
|
140
273
|
|
|
274
|
+
/** O and c from the projection (`entries`) and, for the session's first request, the branch; `sys` stands in for a transcript without
|
|
275
|
+
* system messages. `model` = "provider/id" of the model the next request goes to (default: the newest response's): O and c are
|
|
276
|
+
* properties of the provider's tokenizer, so only that model's responses are evidence. With none, O = R0 x the system state and
|
|
277
|
+
* c = C0; with responses but no O evidence of its own (a model switch mid-session), O keeps the size another model measured in
|
|
278
|
+
* content units, X = O' / c' (the system prompt and tools tokenize like the content around them): real = c (X + E). */
|
|
279
|
+
export function calibrate(entries: Any[], sys?: SysInfo, branch: Any[] = [], model?: string): Scale {
|
|
280
|
+
type Resp = { at: number; est: number; real: number; read: number; nSys: number; ts: number; key: string; id: string };
|
|
281
|
+
const resp: Resp[] = [];
|
|
282
|
+
const sysMsgs: Any[] = [];
|
|
283
|
+
const edits: { at: number; target: string; foreign: boolean }[] = [];
|
|
284
|
+
const marks: { at: number; carried: boolean }[] = []; // pi-zip's commit entries: who first carried the edits right before them (run.ts commit)
|
|
285
|
+
const orig = new Map<string, { at: number; d: number }>(); // edited entries: tokens their originals add back
|
|
286
|
+
const withMsgs: number[] = [0]; // withMsgs[k]: entries before k that carry messages (a commit group has none inside it)
|
|
287
|
+
let est = 0, sysAt = 0; // sysAt: responses before the newest system message
|
|
288
|
+
entries.forEach((pe, at) => {
|
|
289
|
+
const src = pe.sourceEntry, msgs: Any[] = pe.messages ?? [];
|
|
290
|
+
withMsgs.push(withMsgs[at] + (msgs.length ? 1 : 0));
|
|
291
|
+
// an edit with a replacement that is not our placeholder is another extension's (never carried by a request of ours); one without
|
|
292
|
+
// (a bare entry) is judged by the size heuristics below
|
|
293
|
+
if (src?.type === "context_edit") edits.push({ at, target: src.targetId, foreign: "replacement" in src && !textOf(src.replacement?.content).startsWith(PH_MARK) });
|
|
294
|
+
if (src?.type === "custom" && src.customType === COMMIT_CUSTOM) marks.push({ at, carried: !!src.data?.carried });
|
|
295
|
+
if (src?.type === "message" && msgs.length === 1 && msgs[0] !== src.message) orig.set(src.id, { at, d: wireTokensOf(src.message) - wireTokensOf(msgs[0]) });
|
|
296
|
+
for (const m of msgs) {
|
|
297
|
+
if (m?.role === "system") { sysMsgs.push(m); sysAt = resp.length; continue; }
|
|
298
|
+
const real = promptOf(m);
|
|
299
|
+
if (real > 0) resp.push({ at, est, real, read: Number(m.usage.cacheRead) || 0, nSys: sysMsgs.length, ts: Number(m.timestamp) || 0, key: respKey(m), id: src?.id ?? "" });
|
|
300
|
+
est += wireTokensOf(m);
|
|
301
|
+
}
|
|
302
|
+
});
|
|
303
|
+
const sAt = (n: number) => sysOf(sysMsgs.slice(0, n)); // the system state after the first n system messages (computed only where needed)
|
|
304
|
+
const S = sAt(sysMsgs.length);
|
|
305
|
+
const Snow = S >= 0 ? S : sys ? sysEst(sys) : 0;
|
|
306
|
+
const head = entries[0]?.sourceEntry?.type === "compaction" ? entries[0].sourceEntry : null;
|
|
307
|
+
const headMs = head ? Date.parse(head.timestamp) || 0 : 0;
|
|
308
|
+
// a pi-zip summary persisted as a turn_end draft was request-local first (a run plan carries it before its entry exists); one that
|
|
309
|
+
// went in through Pi's compaction before the run (details.via "native") is carried after its entry, exactly like Pi's own
|
|
310
|
+
const local = !!head && head.details?.by === PRODUCT && head.details?.via !== "native";
|
|
311
|
+
// Was the request behind this response sent after the head compaction entry was written? The branch's order says (timestamps tie when
|
|
312
|
+
// a compaction follows a response in the same millisecond); without the branch the timestamps do.
|
|
313
|
+
const pos = new Map<string, number>();
|
|
314
|
+
if (head) branch.forEach((e, i) => { if (typeof e?.id === "string") pos.set(e.id, i); });
|
|
315
|
+
const afterHead = (r: Resp): boolean => {
|
|
316
|
+
const h = pos.get(head?.id), i = pos.get(r.id);
|
|
317
|
+
return h !== undefined && i !== undefined ? i > h : r.ts > headMs;
|
|
318
|
+
};
|
|
319
|
+
const opening = (r: { real: number; est: number } | null | undefined) => !!r && C0 * r.est <= 0.1 * r.real; // content small next to O
|
|
320
|
+
|
|
321
|
+
/** The first commit entry of the group of pi-zip edits that starts at entry `at` (none when a message comes first). */
|
|
322
|
+
const markOf = (at: number) => marks.find((m) => m.at > at && withMsgs[m.at] - withMsgs[at + 1] === 0);
|
|
323
|
+
/** Index of the newest response in `resp` before entry `at`. */
|
|
324
|
+
const respBefore = (at: number) => resp.reduce((k, r, i) => (r.at < at ? i : k), -1);
|
|
325
|
+
/** Tokens the originals add back for response i: edits persisted after it that its own request did not carry (the projection counts
|
|
326
|
+
* their placeholders). A commit entry says who carried them (a run plan: the response before the entry; a valve plan: the next one);
|
|
327
|
+
* with none (a session of an older version) the sizes decide, for the newest response only (`legacy`). */
|
|
328
|
+
const addBack = (i: number, legacy: boolean): number => {
|
|
329
|
+
let d = 0;
|
|
330
|
+
for (const x of edits) {
|
|
331
|
+
const t = orig.get(x.target);
|
|
332
|
+
if (x.at <= resp[i].at || !t || t.at >= resp[i].at) continue;
|
|
333
|
+
const m = x.foreign ? undefined : markOf(x.at);
|
|
334
|
+
if (m ? m.carried && respBefore(m.at) === i : !x.foreign && legacy) continue;
|
|
335
|
+
d += t.d;
|
|
336
|
+
}
|
|
337
|
+
return d;
|
|
338
|
+
};
|
|
339
|
+
|
|
340
|
+
const fit = (want: string, transfer: boolean): Scale => {
|
|
341
|
+
const mine = (r: Resp | undefined): r is Resp => !!r && (!want || !r.key || r.key === want);
|
|
342
|
+
// evidence for O, the newest wins (it needs the smallest correction for system changes since)
|
|
343
|
+
const ev: { at: number; O: number; S: number; src: Scale["oSource"] }[] = [];
|
|
344
|
+
const f = branch.length ? firstRequest(branch, want) : head ? null : resp.find(mine); // the session's first request
|
|
345
|
+
if (f && opening(f)) ev.push({ at: -1, O: f.real - C0 * f.est, S: "S" in f ? f.S : sAt(f.nSys), src: "first" });
|
|
346
|
+
const ci = resp.findIndex((r, i) => i >= sysAt && mine(r)), cur = resp[ci]; // the first response sent with the current system prompt and tools
|
|
347
|
+
// its content as its request carried it: edits persisted after it (a model switch, say, then a run plan's folds) are in `est`
|
|
348
|
+
const curEst = cur ? cur.est + addBack(ci, false) : 0;
|
|
349
|
+
if (cur && (!head || afterHead(cur)) && opening({ real: cur.real, est: curEst })) ev.push({ at: ci, O: cur.real - C0 * curEst, S, src: "first" });
|
|
350
|
+
if (head) {
|
|
351
|
+
// the summary at the head was first carried by the first response after its entry, or, for pi-zip's own request-local summary
|
|
352
|
+
// (a run plan), by the last one before it; Pi's, another extension's or a native pi-zip compaction is never sent before its entry
|
|
353
|
+
const after = resp.findIndex((r) => afterHead(r)), ours = local;
|
|
354
|
+
for (const i of after < 0 ? (ours ? [resp.length - 1] : []) : ours ? [after - 1, after] : [after]) {
|
|
355
|
+
const r = resp[i], prev = resp[i - 1];
|
|
356
|
+
if (mine(r) && r.read > 0 && r.read < 0.8 * r.real && (!prev || r.read < 0.8 * prev.real)) { ev.push({ at: i, O: r.read, S: sAt(r.nSys), src: "summary" }); break; }
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
const best = ev.sort((a, b) => b.at - a.at)[0];
|
|
360
|
+
let O = best ? Math.max(0, best.O + (best.S >= 0 && S >= 0 ? R0 * (Snow - best.S) : 0)) : R0 * Snow, oSource: Scale["oSource"] = best?.src ?? "default";
|
|
361
|
+
// c from the newest response of this model
|
|
362
|
+
let n = resp.length - 1;
|
|
363
|
+
while (n >= 0 && !mine(resp[n])) n--;
|
|
364
|
+
if (n < 0) return { O, c: C0, real: 0, est: 0, source: "default", oSource };
|
|
365
|
+
const N = resp[n];
|
|
366
|
+
let estN = N.est;
|
|
367
|
+
const cutLater = !!head && !afterHead(N);
|
|
368
|
+
const later = edits.filter((x) => x.at > N.at);
|
|
369
|
+
const rebuilt = cutLater && !local && !!N.id ? requestEst(branch, N.id) : null; // what its request carried: the compaction after it removed that from the projection
|
|
370
|
+
if (rebuilt !== null && rebuilt > 0) estN = rebuilt;
|
|
371
|
+
else if (later.length || cutLater) {
|
|
372
|
+
// carried (a run plan): the prompt shrank, or a read stopped early; a cache miss alone (no read at all) proves nothing. The
|
|
373
|
+
// commit entries of pi-zip's edits say it outright; the sizes decide only for edits without one (older sessions).
|
|
374
|
+
// Pi's own compaction is never request-local.
|
|
375
|
+
let p = n - 1;
|
|
376
|
+
while (p >= 0 && !mine(resp[p])) p--;
|
|
377
|
+
const P = resp[p];
|
|
378
|
+
const heur = !P || N.real < P.real || (N.read > 0 && N.read < 0.8 * P.real);
|
|
379
|
+
const carried = !(cutLater && !local) && (cutLater && typeof head.details?.carried === "boolean" ? head.details.carried : heur);
|
|
380
|
+
if (!carried && cutLater) return { O, c: C0, real: N.real, est: 0, source: "default", oSource }; // what it carried is summarised away
|
|
381
|
+
estN += addBack(n, heur && carried);
|
|
382
|
+
}
|
|
383
|
+
if (!best && transfer && estN > 0) {
|
|
384
|
+
// no O of its own: the newest other model's O in content units
|
|
385
|
+
// (a compaction may have taken the other model's responses out of the projection: the branch still has them)
|
|
386
|
+
const other = [...resp].reverse().find((r) => r.key && r.key !== want)?.key
|
|
387
|
+
?? [...branch].reverse().map((e) => (e?.type === "message" && promptOf(e.message) > 0 ? respKey(e.message) : "")).find((k) => k && k !== want);
|
|
388
|
+
const o = other ? fit(other, false) : null;
|
|
389
|
+
if (o && o.oSource !== "default" && o.c > 0) {
|
|
390
|
+
const X = o.O / o.c, c = clamp(N.real / (X + estN), C_MIN, C_MAX);
|
|
391
|
+
return { O: c * X, c, real: N.real, est: estN, source: "usage", oSource: "model" };
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
return { O, c: estN > 0 ? clamp((N.real - O) / estN, C_MIN, C_MAX) : C0, real: N.real, est: estN, source: estN > 0 ? "usage" : "default", oSource };
|
|
395
|
+
};
|
|
396
|
+
return fit(model || [...resp].reverse().find((r) => r.key)?.key || "", true);
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
/** The cold cap's target in real tokens of the WHOLE context (system prompt and tools included, as the cap was tuned): max(cap, O + cap/2),
|
|
400
|
+
* never above Pi's compaction room. While O is under half the cap this is the plain whole-context cap. Above that the target is O plus
|
|
401
|
+
* half the cap of conversation: never at or below O, where no edit reaches it (0.2.9's 40K with O = 29-55K summarised at every cold
|
|
402
|
+
* return), and not 40K of conversation on top of O either (the cap on content alone cost +13% live on a 36K prefix, with no quality
|
|
403
|
+
* difference measured). Offline it never loses more items than the whole-context cap (research round5/affine.md section 11). */
|
|
404
|
+
export const CAP_CONTENT_MIN = 0.5;
|
|
405
|
+
export const coldTarget = (cap: number, O: number, room: number | null): number => Math.min(Math.max(cap, O + CAP_CONTENT_MIN * cap), room ?? Infinity);
|
|
406
|
+
|
|
141
407
|
export const settings = () => ({
|
|
142
|
-
coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many tokens
|
|
408
|
+
coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many real tokens of the whole context, but never below O + half of it (coldTarget)
|
|
143
409
|
foldMin: envInt("FOLD_MIN", 500), // outputs below this many tokens are never folded
|
|
144
410
|
keepLines: envInt("KEEP_LINES", 8),
|
|
145
411
|
minGain: envInt("MIN_GAIN", 10_000), // legacy rule only (no prices): a summary must remove at least max(minGain, 15% of the context)
|
|
@@ -203,7 +469,7 @@ export function buildBlocks(contextEntries: Any[], steerIds?: Set<string>): Bloc
|
|
|
203
469
|
for (const c of msg.content ?? []) if (c?.type === "toolCall") issued.set(c.id, asst);
|
|
204
470
|
}
|
|
205
471
|
asstOf[blocks.length] = kind === "toolResult" ? (issued.get(msg.toolCallId) ?? asst) : asst;
|
|
206
|
-
blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a +
|
|
472
|
+
blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + wireTokensOf(m), 0), userTurn, edited, ours, age: 0 });
|
|
207
473
|
}
|
|
208
474
|
for (const b of blocks) if (b.kind === "toolResult") b.age = asst - asstOf[b.idx];
|
|
209
475
|
return blocks;
|
|
@@ -244,6 +510,7 @@ export interface Cut {
|
|
|
244
510
|
costUsd: number; // what the narrative model call cost (0 when unknown)
|
|
245
511
|
usage?: Any;
|
|
246
512
|
ms: number; // total production time (the narrative model call included)
|
|
513
|
+
ahead?: boolean; // F2: a lookahead cut (inside the previous user turn, before its final answer; guard.ts)
|
|
247
514
|
}
|
|
248
515
|
|
|
249
516
|
/** What a run sends from its first request on and persists verbatim at turn_end (F8). */
|
|
@@ -254,7 +521,7 @@ export interface RunPlan {
|
|
|
254
521
|
ctxBefore: number;
|
|
255
522
|
ctxAfter: number;
|
|
256
523
|
ms: number;
|
|
257
|
-
|
|
524
|
+
scale: Scale; // the token scale the plan was made with (stats and notices of this run use the same one)
|
|
258
525
|
persisted: boolean;
|
|
259
526
|
cutVisibleIdx?: number; // where the kept part starts among the projected non-system messages
|
|
260
527
|
cutKeptFirstMsg?: Any; // ... and that message itself (a disagreeing request view drops the cut)
|
|
@@ -262,6 +529,9 @@ export interface RunPlan {
|
|
|
262
529
|
applied?: Set<string>; // entry ids whose fold the latest request view actually carried (what turn_end must persist, no more)
|
|
263
530
|
cutTs?: number; // timestamp of the request-local summary message: one value per run, so every request of the run is identical
|
|
264
531
|
untouched?: number; // real tokens before the earliest edited block: what the first request carrying the plan can still read from the cache
|
|
532
|
+
waitMs?: number; // how long the user waited for this plan's summary (0: none, or it was ready)
|
|
533
|
+
noticed?: boolean; // its notice is already in the transcript (shown when the first request carrying it went out)
|
|
534
|
+
native?: Cut; // this run's summary went in through Pi's own compaction before the run: no request-local cut, no compaction draft
|
|
265
535
|
}
|
|
266
536
|
|
|
267
537
|
export interface PlanOpts {
|
|
@@ -270,8 +540,7 @@ export interface PlanOpts {
|
|
|
270
540
|
trace?: (LawTerms & { where: "summary" | "plan" })[]; // every law evaluation is pushed here (ledger)
|
|
271
541
|
reserve?: number; // Pi's compaction reserveTokens (default 16384)
|
|
272
542
|
steerIds?: Set<string>; // user entries that are mid-run steering or follow-up messages, not new user turns
|
|
273
|
-
|
|
274
|
-
k?: number; // real tokens per estimated token (calibrate); default 1 = the estimates are taken as real
|
|
543
|
+
scale?: Scale; // real = O + c x content estimate (calibrate); default UNIT = the estimates are taken as real
|
|
275
544
|
cwd: string;
|
|
276
545
|
coldCap?: number;
|
|
277
546
|
foldMin?: number;
|
|
@@ -284,7 +553,16 @@ export interface PlanOpts {
|
|
|
284
553
|
inturnAge?: number; // default settings().inturnAge (PI_ZIP_INTURN_AGE, 60); 0 = outputs of the protected turns never fold on age
|
|
285
554
|
noSummary?: boolean; // a summary was already made while this cache stayed warm: no second one unless the context is at/above the compaction room
|
|
286
555
|
rereadOnly?: boolean; // zip_recall is not available to the model: fold only outputs that can be re-read (classify "rereadable")
|
|
556
|
+
fileRecall?: FileRecall; // zip_recall is hidden but read, grep or bash is active: fold as in full mode, placeholders name the recall file
|
|
287
557
|
pastTtl?: boolean; // a user return after the declared TTL: a warm plan may relax into the previous user turn too (RELAX_PREV_TURN)
|
|
558
|
+
warmSummaryGate?: boolean; // F1, default WARM_SUMMARY_GATE
|
|
559
|
+
lookahead?: boolean; // F2, default LOOKAHEAD (cold plans only)
|
|
560
|
+
narrativePrice?: boolean; // F3, default NARRATIVE_PRICE
|
|
561
|
+
slide?: number; // internal (F2): plan as if this many more user turns had started (the protected window slid)
|
|
562
|
+
valveB?: number; // internal (F2): the real context the slid warm plan's law sees (the cold plan's folds already applied)
|
|
563
|
+
hold?: { anchor: string }; // B2 for folds: the entry id of the prompt of the last cold return (run.ts holdOf); undefined = nothing held
|
|
564
|
+
holdPass?: boolean; // internal: the held pass of a warm plan with a hold
|
|
565
|
+
warmStyle?: boolean; // internal: a cold plan at 0 < P(warm) < 0.5 that the law refused is planned once more the way a warm plan is (no relax folds unless pastTtl, the summary gated)
|
|
288
566
|
}
|
|
289
567
|
|
|
290
568
|
export interface PlanResult {
|
|
@@ -293,21 +571,33 @@ export interface PlanResult {
|
|
|
293
571
|
folds: FoldTarget[];
|
|
294
572
|
cutIdx: number | null;
|
|
295
573
|
firstKeptEntryId: string | null;
|
|
296
|
-
prefixTokens: number; // estimate units (x
|
|
574
|
+
prefixTokens: number; // content estimate units (x c = real), like summaryTokensPlanned
|
|
297
575
|
summaryTokensPlanned: number;
|
|
298
576
|
sumTrigger: "cold" | "valve" | null;
|
|
299
|
-
|
|
300
|
-
ctxTokens: number; //
|
|
301
|
-
ctxAfterFolds: number; //
|
|
577
|
+
scale: Scale;
|
|
578
|
+
ctxTokens: number; // real: O + c x content
|
|
579
|
+
ctxAfterFolds: number; // real
|
|
302
580
|
userTurns: number;
|
|
303
581
|
cutVisibleIdx: number;
|
|
304
582
|
cutKeptFirstMsg: Any;
|
|
583
|
+
fileRecall?: FileRecall; // the summary of this plan names recall files instead of zip_recall (PlanOpts.fileRecall)
|
|
584
|
+
ahead?: boolean; // F2: the cut was decided by the lookahead (it may lie inside the previous user turn, before its final answer)
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
/** Index of the final assistant message of user turn `turn` (the answer the user saw), -1 if none. */
|
|
588
|
+
export function finalAnswerIdx(blocks: Block[], turn: number): number {
|
|
589
|
+
for (let i = blocks.length - 1; i >= 0; i--) {
|
|
590
|
+
const b = blocks[i];
|
|
591
|
+
if (b.userTurn < turn) break;
|
|
592
|
+
if (b.userTurn === turn && b.kind === "assistant" && b.entryId && b.msg?.stopReason !== "error" && b.msg?.stopReason !== "aborted") return i;
|
|
593
|
+
}
|
|
594
|
+
return -1;
|
|
305
595
|
}
|
|
306
596
|
|
|
307
|
-
/** Estimated
|
|
308
|
-
export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean
|
|
597
|
+
/** Estimated content before the earliest edited block: what stays cached through the edit with O (real: O + c x it). A cut edits from the start. */
|
|
598
|
+
export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean): number {
|
|
309
599
|
const ids = new Set(folds.map((t) => t.entryId));
|
|
310
|
-
let est =
|
|
600
|
+
let est = 0;
|
|
311
601
|
if (!cut) for (const b of blocks) { if (b.entryId && ids.has(b.entryId)) break; est += b.tokens; }
|
|
312
602
|
return est;
|
|
313
603
|
}
|
|
@@ -320,35 +610,49 @@ export const contentKeyOf = (content: Any): string => {
|
|
|
320
610
|
|
|
321
611
|
/** The planner. Cold: fold everything outside the protected window, relax into the previous turn if still above the cap,
|
|
322
612
|
* summarise only if folds cannot reach the cap and the law prices the summary call in. Warm: null unless the context is above the
|
|
323
|
-
* cold cap and the law fires for the plan
|
|
613
|
+
* cold cap and the law fires for the plan; its summary passes the same summary gate (WARM_SUMMARY_GATE); then the cold plan minus the previous-turn relax (a warm plan
|
|
324
614
|
* folds the previous user turn by relax only at a return after the declared TTL, o.pastTtl). A cold plan with P(warm) > 0
|
|
325
615
|
* passes the same law (expected cost). The cap never exceeds Pi's compaction room. Returns null when there is nothing to plan on
|
|
326
616
|
* or the law says no. */
|
|
327
617
|
export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
618
|
+
if (o.hold && !o.holdPass && (o.mode ?? "cold") === "warm") {
|
|
619
|
+
// the hold gives way to a summary: when the plan without it summarises, that is the plan (a summary voids the folds before its
|
|
620
|
+
// cut anyway); otherwise the held plan, with no summary of its own (the next cold return writes it)
|
|
621
|
+
const full = planContext(entries, { ...o, hold: undefined, trace: undefined });
|
|
622
|
+
if (!full) return null;
|
|
623
|
+
if (full.cutIdx !== null) return planContext(entries, { ...o, hold: undefined });
|
|
624
|
+
return planContext(entries, { ...o, noSummary: true, holdPass: true });
|
|
625
|
+
}
|
|
328
626
|
const s = settings();
|
|
329
627
|
const mode = o.mode ?? "cold";
|
|
330
|
-
|
|
331
|
-
|
|
628
|
+
const warmLike = mode === "warm" || o.warmStyle === true; // plans the way a warm plan does (the label of the trigger stays the mode's)
|
|
629
|
+
// limits are in real tokens; the plan works in content estimate units (real = O + c x estimate), so they are converted once, here.
|
|
630
|
+
// The cold cap is on the whole context, but leaves at least half of itself for the conversation above O (coldTarget): a cap the
|
|
631
|
+
// fixed prefix alone fills is out of reach, and every cold return then folds all it may and summarises. The compaction room is
|
|
632
|
+
// Pi's trigger on the whole context.
|
|
633
|
+
const { O, c } = o.scale ?? UNIT;
|
|
634
|
+
const real = (est: number) => O + c * est;
|
|
332
635
|
const room = compactionRoom(o.model, o.reserve);
|
|
333
|
-
const coldCap =
|
|
636
|
+
const coldCap = (coldTarget(o.coldCap ?? s.coldCap, O, room) - O) / c;
|
|
334
637
|
const law: Law = o.law ?? { pr: null, g: G0, pWarm: mode === "cold" ? 0 : 1 };
|
|
335
638
|
const gate = (where: "summary" | "plan", B: number, A: number, r: number | null, call: number, T: number, Tsuf?: number): boolean => {
|
|
336
639
|
const t = lawTerms(B, A, law.pWarm, r, law.pr, law.g, call, T);
|
|
337
640
|
o.trace?.push({ where, ...t, ...(Tsuf !== undefined ? { Tsuf } : {}) });
|
|
338
641
|
return t.ok;
|
|
339
642
|
};
|
|
340
|
-
const minGain =
|
|
643
|
+
const minGain = o.minGain ?? s.minGain;
|
|
341
644
|
const foldMin = o.foldMin ?? s.foldMin;
|
|
342
645
|
const keepLines = o.keepLines ?? s.keepLines;
|
|
343
646
|
const relax = o.relax ?? RELAX_PREV_TURN;
|
|
344
647
|
const inturnAge = o.inturnAge ?? s.inturnAge;
|
|
345
648
|
const blocks = buildBlocks(entries, o.steerIds);
|
|
346
649
|
if (!blocks.length) return null;
|
|
347
|
-
const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0); // the upcoming prompt is a new user turn
|
|
650
|
+
const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0) + (o.slide ?? 0); // the upcoming prompt is a new user turn
|
|
651
|
+
const f1 = o.warmSummaryGate ?? WARM_SUMMARY_GATE, f2 = mode === "cold" && !o.warmStyle && (o.lookahead ?? LOOKAHEAD), f3 = o.narrativePrice ?? NARRATIVE_PRICE;
|
|
348
652
|
const calls = toolCallIndex(blocks);
|
|
349
653
|
const recalled = o.recalled ?? new Set<string>();
|
|
350
|
-
const ctxEst =
|
|
351
|
-
if (
|
|
654
|
+
const ctxEst = blocks.reduce((a, b) => a + b.tokens, 0); // content
|
|
655
|
+
if (warmLike && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
|
|
352
656
|
const trig = mode === "warm" ? "valve" : "cold";
|
|
353
657
|
const tokOverride = new Map<number, number>();
|
|
354
658
|
const folds: FoldTarget[] = [];
|
|
@@ -368,9 +672,9 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
368
672
|
};
|
|
369
673
|
const addFold = (b: Block, trig: string): boolean => {
|
|
370
674
|
if (tokOverride.has(b.idx) || recalled.has(handleFor(b.entryId!))) return false; // F4: never refold what the model recalled
|
|
371
|
-
const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly);
|
|
675
|
+
const ph = makePlaceholderFor(b, calls, keepLines, !!o.rereadOnly, o.fileRecall);
|
|
372
676
|
if (ph === null) return false;
|
|
373
|
-
const phTok =
|
|
677
|
+
const phTok = tokensOf({ ...b.msg, content: [{ type: "text", text: ph }] }); // the folded message: its id and framing stay
|
|
374
678
|
if (!(phTok < 0.9 * b.tokens)) return false;
|
|
375
679
|
tokOverride.set(b.idx, phTok);
|
|
376
680
|
const call = calls.get(b.msg.toolCallId);
|
|
@@ -378,20 +682,41 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
378
682
|
return true;
|
|
379
683
|
};
|
|
380
684
|
// a zip_recall result IS content the model just asked for: folding it would undo the recall (and loop); never
|
|
381
|
-
|
|
685
|
+
// ... and so is a built-in read, grep, find, ls or bash call on a recall file (file recall, when an allowlist hides zip_recall)
|
|
686
|
+
const isRecall = (b: Block) => {
|
|
687
|
+
const call = calls.get(b.msg.toolCallId), name = call?.name ?? b.msg.toolName;
|
|
688
|
+
if (name === RECALL_TOOL) return true;
|
|
689
|
+
if (name === "bash") return commandNamesRecall(call?.args?.command);
|
|
690
|
+
return (name === "read" || name === "grep" || name === "find" || name === "ls") && !!parseRecallPath(call?.args?.path);
|
|
691
|
+
};
|
|
382
692
|
// Observability: on an automatic (prefix) cache no fold starts inside the first MIN_EXPECT real tokens, so the next response's cacheRead
|
|
383
693
|
// is an uncensored survival sample (learn.ts sample). Without it every idle return folded into the first 1-7K tokens and its sample was
|
|
384
694
|
// censored, so the learned curve never saw a return (live GLM bench). Explicit caches are exempt: the provider looks back only ~20
|
|
385
695
|
// blocks from the last breakpoint, a far head is not read either way. A cut replaces the prefix from the first message: exempt too.
|
|
386
|
-
const head = law.pr?.cls === "automatic" ? MIN_EXPECT /
|
|
696
|
+
const head = law.pr?.cls === "automatic" ? (MIN_EXPECT - O) / c : 0; // MIN_EXPECT counts from the first token, O included
|
|
387
697
|
const start: number[] = [];
|
|
388
|
-
blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens),
|
|
698
|
+
blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), 0);
|
|
389
699
|
const foldable = (b: Block) => b.kind === "toolResult" && !b.edited && !!b.entryId && b.tokens > foldMin && !isRecall(b) && start[b.idx] >= head && (!o.rereadOnly || classify(b) === "rereadable");
|
|
390
700
|
const protectedTurn = (b: Block) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1;
|
|
391
701
|
const savings = () => folds.reduce((a, t) => a + t.entryTokens - t.phTokens, 0);
|
|
392
|
-
|
|
702
|
+
// the blocks protected at the last cold return stay out of a warm plan's folds (see FOLD_HOLD)
|
|
703
|
+
const held: (b: Block) => boolean = (() => {
|
|
704
|
+
const NOT_HELD = () => false;
|
|
705
|
+
if (mode !== "warm" || !o.hold) return NOT_HELD;
|
|
706
|
+
const from = blocks.findIndex((b) => b.entryId === o.hold!.anchor); // gone (a later summary replaced it): nothing is held
|
|
707
|
+
if (from < 0) return NOT_HELD;
|
|
708
|
+
// what the session grew by before this decision: the blocks between the cold return's prompt and the newest response (that
|
|
709
|
+
// response, the tool results after it and a new prompt are what the decision is about, not yet something the cache has seen)
|
|
710
|
+
let lastAsst = blocks.length - 1;
|
|
711
|
+
while (lastAsst > 0 && blocks[lastAsst].kind !== "assistant") lastAsst--;
|
|
712
|
+
const grown = c * blocks.slice(from + 1, Math.max(from + 1, lastAsst)).reduce((a, b) => a + b.tokens, 0);
|
|
713
|
+
if (grown > HOLD_RELEASE || (room !== null && real(ctxEst) >= room)) return NOT_HELD;
|
|
714
|
+
const turn = blocks[from].userTurn;
|
|
715
|
+
return (b: Block) => b.userTurn >= turn - PROTECT_USER_TURNS + 1;
|
|
716
|
+
})();
|
|
717
|
+
const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b) && !held(b));
|
|
393
718
|
for (const b of cands) addFold(b, trig);
|
|
394
|
-
if (relax && (mode === "cold" || o.pastTtl === true)) {
|
|
719
|
+
if (relax && ((mode === "cold" && !o.warmStyle) || o.pastTtl === true)) {
|
|
395
720
|
// the protected window = the new prompt + the previous user turn; that turn's big reads are what makes a cold return
|
|
396
721
|
// expensive. Rereadable ones can be recalled exactly: fold them biggest-first until the cap; never the latest turn's own.
|
|
397
722
|
// Cold plans, and warm plans at a return after the declared TTL; any other warm plan keeps the previous user turn visible.
|
|
@@ -418,49 +743,78 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
418
743
|
}
|
|
419
744
|
}
|
|
420
745
|
let ctxAfterFolds = ctxEst - savings();
|
|
421
|
-
// what no summary can remove
|
|
422
|
-
//
|
|
746
|
+
// what no summary can remove (besides O): the protected turns (after their folds) and the previous summary, which the next one
|
|
747
|
+
// carries forward; the cap is never chased below it (a context that is mostly this floor would be re-summarised for nothing)
|
|
423
748
|
const prevSum = blocks[0]?.kind === "summary" ? blocks[0].tokens : 0;
|
|
424
|
-
const floor =
|
|
749
|
+
const floor = prevSum + blocks.reduce((a, b) => a + (protectedTurn(b) ? (tokOverride.get(b.idx) ?? b.tokens) : 0), 0);
|
|
425
750
|
const target = Math.max(coldCap, floor + SUMMARY_FLOOR);
|
|
426
|
-
const atRoom = room !== null &&
|
|
751
|
+
const atRoom = room !== null && real(ctxAfterFolds) >= room;
|
|
427
752
|
const sumTrigger: "cold" | "valve" | null = ctxAfterFolds > target && (!o.noSummary || atRoom) ? trig : null;
|
|
428
753
|
let cutIdx: number | null = null;
|
|
429
754
|
let prefixTokens = 0;
|
|
430
755
|
let summaryTokensPlanned = 0;
|
|
431
|
-
|
|
432
|
-
|
|
756
|
+
let ahead = false;
|
|
757
|
+
const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
|
|
758
|
+
const pre: number[] = [];
|
|
759
|
+
const S = (x: number) => Math.max(clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP), prevSum); // the next summary carries the previous one forward
|
|
760
|
+
// Block index of every fold: the folds a cut voids are those before it
|
|
761
|
+
const foldAt = folds.map((t) => blocks.findIndex((b) => b.entryId === t.entryId));
|
|
762
|
+
/** The whole plan's law (F10): B is the context as it is (or as the slid valve saw it), A what the plan leaves, `call` the summary call
|
|
763
|
+
* a cut makes; `untouched` is the content before the earliest edit (ledger only). A cold plan with some chance of a warm cache is
|
|
764
|
+
* priced by expectation (no prices: a cold plan always fires). T = A (the whole post-edit context, O included, as tuned: the cheaper
|
|
765
|
+
* A - O fires warm edits earlier and loses items offline). */
|
|
766
|
+
const planGate = (after: number, untouched: number, call: number, planned: boolean): boolean =>
|
|
767
|
+
!(warmLike || (planned && law.pr)) || gate("plan", Math.max(o.valveB ?? real(ctxEst), real(after)), real(after), room, call, real(after), c * (after - untouched));
|
|
768
|
+
/** The cut for cut points minCut..maxCut: the first that brings the context to the cap, else the deepest; gated = the summary law
|
|
769
|
+
* (no prices: the legacy gain rule) must pass. When the minimal cut's summary or the whole plan with it is refused, the deeper cuts
|
|
770
|
+
* are tried in turn (the gain grows as D^2, the call only with the prefix read). The whole plan's law prices the summary call too.
|
|
771
|
+
* Applies the first cut both laws accept (folds in the prefix are void) and returns true, or changes nothing. */
|
|
772
|
+
const summarise = (maxCut: number, gated: boolean): boolean => {
|
|
433
773
|
const cuts: number[] = [];
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
if (limit >= 0 && i > limit) break;
|
|
437
|
-
if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
|
|
438
|
-
}
|
|
439
|
-
const pre: number[] = [];
|
|
774
|
+
for (let i = minCut; i < blocks.length && i <= maxCut; i++) if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
|
|
775
|
+
if (!cuts.length) return false;
|
|
440
776
|
let acc = 0;
|
|
441
777
|
for (let i = 0; i < blocks.length; i++) {
|
|
442
778
|
pre[i] = acc;
|
|
443
779
|
acc += tokOverride.get(i) ?? blocks[i].tokens;
|
|
444
780
|
}
|
|
445
|
-
const total =
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
//
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
const bi = blocks.findIndex((b) => b.entryId === folds[i].entryId);
|
|
459
|
-
if (bi < pick) { tokOverride.delete(bi); folds.splice(i, 1); } // the prefix is replaced: its folds are void
|
|
460
|
-
}
|
|
461
|
-
ctxAfterFolds = ctxEst - savings();
|
|
781
|
+
const total = acc;
|
|
782
|
+
let first = cuts.length - 1;
|
|
783
|
+
for (let k = 0; k < cuts.length; k++) if (total - pre[cuts[k]] + S(pre[cuts[k]]) <= coldCap) { first = k; break; }
|
|
784
|
+
for (const pick of cuts.slice(first)) {
|
|
785
|
+
const X = c * pre[pick], Sr = c * S(pre[pick]);
|
|
786
|
+
// the call is the narrative call summary.ts makes (F3, NARRATIVE_PRICE): uncached input of the prefix without the previous
|
|
787
|
+
// summary, which it never reads, + output of at most the narrative budget; without F3 the verdict's w X + out S (the carried
|
|
788
|
+
// summary as output)
|
|
789
|
+
const call = !law.pr ? 0 : f3 ? law.pr.input * c * Math.max(pre[pick] - prevSum, 0) + law.pr.out * c * Math.min(S(pre[pick]), NARRATIVE_MAX) : law.pr.w * X + law.pr.out * Sr;
|
|
790
|
+
if (gated) {
|
|
791
|
+
// the summary call must pay for itself, cold or warm (F1, WARM_SUMMARY_GATE; before, a warm plan skipped the gate)
|
|
792
|
+
const ok = !law.pr ? summaryGainOk(X, Sr, real(total), minGain) : (warmLike && !f1) || gate("summary", X, Sr, null, call, Sr);
|
|
793
|
+
if (!ok) continue;
|
|
462
794
|
}
|
|
795
|
+
if (!(pre[pick] >= 2 * S(pre[pick]))) continue;
|
|
796
|
+
// what the plan leaves: the context with its folds (those inside the prefix too: the prefix is measured as folded, `pre`) minus
|
|
797
|
+
// what the summary replaces plus the summary. (Counting the voided folds' savings back, as before, overstated it by them.)
|
|
798
|
+
if (!planGate(ctxAfterFolds - (pre[pick] - S(pre[pick])), 0, call, true)) continue;
|
|
799
|
+
cutIdx = pick;
|
|
800
|
+
prefixTokens = pre[pick];
|
|
801
|
+
summaryTokensPlanned = S(pre[pick]);
|
|
802
|
+
for (let i = folds.length - 1; i >= 0; i--) if (foldAt[i] < pick) { tokOverride.delete(foldAt[i]); folds.splice(i, 1); foldAt.splice(i, 1); } // the prefix is replaced: its folds are void
|
|
803
|
+
return true;
|
|
463
804
|
}
|
|
805
|
+
return false;
|
|
806
|
+
};
|
|
807
|
+
if (sumTrigger) {
|
|
808
|
+
const limit = blocks.findIndex((b) => protectedTurn(b));
|
|
809
|
+
summarise(limit >= 0 ? limit : blocks.length - 1, true);
|
|
810
|
+
}
|
|
811
|
+
if (f2 && cutIdx === null && law.pr && ctxAfterFolds > coldCap && (!o.noSummary || atRoom)) {
|
|
812
|
+
// F2 lookahead: the warm valve's plan at the next user turn (this context, its folds applied, the window slid by one turn). If it
|
|
813
|
+
// would summarise, the summary is written now, at the cold return, where the rewrite costs nothing extra (and at settle it is
|
|
814
|
+
// prepared while the user is away), instead of one turn later as a warm rewrite the user waits for.
|
|
815
|
+
const next = planContext(entries, { ...o, mode: "warm", slide: (o.slide ?? 0) + 1, lookahead: false, pastTtl: false, noSummary: false, trace: undefined, law: { ...law, pWarm: 1 }, valveB: real(ctxAfterFolds) });
|
|
816
|
+
const fa = finalAnswerIdx(blocks, userTurns - 1);
|
|
817
|
+
if (next && next.cutIdx !== null && fa >= minCut && summarise(fa, false)) ahead = true;
|
|
464
818
|
}
|
|
465
819
|
// where the kept part starts among the projected non-system messages (the request-local view drops everything before it)
|
|
466
820
|
let cutVisibleIdx = -1;
|
|
@@ -480,13 +834,17 @@ export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
|
480
834
|
}
|
|
481
835
|
}
|
|
482
836
|
// the untouched prefix: everything before the earliest edited block (a cut edits from the first block on)
|
|
483
|
-
const untouched = untouchedEst(blocks, folds, cutIdx !== null
|
|
837
|
+
const untouched = untouchedEst(blocks, folds, cutIdx !== null);
|
|
484
838
|
const after = ctxAfterFolds - (cutIdx !== null ? prefixTokens - summaryTokensPlanned : 0);
|
|
485
839
|
const planned = folds.length > 0 || cutIdx !== null;
|
|
486
|
-
//
|
|
487
|
-
//
|
|
488
|
-
if (
|
|
489
|
-
|
|
840
|
+
// a plan with a cut passed the whole plan's law with its call inside summarise; folds only (no summary wanted, or none the laws
|
|
841
|
+
// accepted: the deeper cuts and the cut-free plan both stand) is judged here
|
|
842
|
+
if (cutIdx === null && !planGate(after, untouched, 0, planned)) {
|
|
843
|
+
// a cold plan with some chance of a warm cache: the plan a warm cache would make may pay where this one does not
|
|
844
|
+
if (mode === "cold" && !o.warmStyle && law.pr && law.pWarm > 0 && law.pWarm < 0.5) return planContext(entries, { ...o, warmStyle: true });
|
|
845
|
+
return null;
|
|
846
|
+
}
|
|
847
|
+
return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger: cutIdx !== null ? (sumTrigger ?? trig) : sumTrigger, scale: o.scale ?? UNIT, ctxTokens: real(ctxEst), ctxAfterFolds: real(ctxAfterFolds), userTurns, cutVisibleIdx, cutKeptFirstMsg, ...(o.fileRecall ? { fileRecall: o.fileRecall } : {}), ...(ahead ? { ahead } : {}) };
|
|
490
848
|
}
|
|
491
849
|
|
|
492
850
|
const sameMsg = (a: Any, b: Any): boolean =>
|
|
@@ -532,6 +890,11 @@ export function collapseSystem(messages: Any[]): Any | undefined {
|
|
|
532
890
|
* message, the summary, then the kept non-system messages. null = nothing to apply. A cut whose kept boundary cannot be
|
|
533
891
|
* located is dropped (folds still apply); an inconsistent split is never sent.
|
|
534
892
|
*/
|
|
893
|
+
/** Fold targets of a plan made on the view before a native compaction, renumbered for the request Pi builds after it: Pi's summary
|
|
894
|
+
* message first, then the kept part (positions among the non-system messages, as in applyPlanToMessages). */
|
|
895
|
+
export const rebaseFolds = (folds: FoldTarget[], cutVisibleIdx: number): FoldTarget[] =>
|
|
896
|
+
folds.map((t) => ({ ...t, visIdx: t.visIdx >= 0 && cutVisibleIdx >= 0 && t.visIdx >= cutVisibleIdx ? t.visIdx - cutVisibleIdx + 1 : -1 }));
|
|
897
|
+
|
|
535
898
|
export function applyPlanToMessages(messages: Any[], plan: RunPlan | null): { messages: Any[]; droppedCut: boolean; applied: Set<string> } | null {
|
|
536
899
|
if (!plan || plan.persisted || (!plan.folds.length && !plan.cut)) return null;
|
|
537
900
|
const vis: number[] = []; // indices of the non-system messages
|