pi-zip 0.0.0-stage → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +139 -2
- package/package.json +38 -4
- package/src/cache.ts +116 -0
- package/src/classify.ts +77 -0
- package/src/guard.ts +192 -0
- package/src/index.ts +20 -0
- package/src/learn.ts +141 -0
- package/src/notice.ts +108 -0
- package/src/placeholder.ts +135 -0
- package/src/plan.ts +588 -0
- package/src/recall.ts +199 -0
- package/src/run.ts +649 -0
- package/src/summary.ts +500 -0
- package/src/util.ts +50 -0
package/src/plan.ts
ADDED
|
@@ -0,0 +1,588 @@
|
|
|
1
|
+
// Pure planning: session projection -> which outputs to fold, whether to summarise, where to cut (F5-F10).
|
|
2
|
+
import { classifyRecoverability, type Recover } from "./classify.ts";
|
|
3
|
+
import { handleFor, makePlaceholderFor, pickKeyLines, RECALL_TOOL, shortArgs } from "./placeholder.ts";
|
|
4
|
+
import { createHash } from "node:crypto";
|
|
5
|
+
import { type Any, PRODUCT, clamp, envInt, textOf, tok4, tokensOf } from "./util.ts";
|
|
6
|
+
import { PH_MARK } from "./placeholder.ts";
|
|
7
|
+
import { MIN_EXPECT, type Prices } from "./learn.ts";
|
|
8
|
+
|
|
9
|
+
/** Default: when a COLD plan (P(warm) < 0.5) is still above the cap, also fold REREADABLE outputs of the previous user turn, biggest
|
|
10
|
+
* first. A warm plan does so only at a user return after the declared TTL (PlanOpts.pastTtl): the very returns where the TTL rule
|
|
11
|
+
* always folded that turn, so the quality stays the TTL rule's while the law, with the learned P(warm), still decides whether the
|
|
12
|
+
* rewrite pays (a provider whose cache outlives its declared TTL otherwise keeps the whole previous turn at every return: live GLM
|
|
13
|
+
* 1.15x the TTL rule's bill). Any other warm plan never does: the user comes back to a warm cache and refers to the turn just
|
|
14
|
+
* finished (final model: +4.0 / +5.8 lost items with it). false = no relax fold of the previous user turn, but the
|
|
15
|
+
* in-turn rule (INTURN_AGE) still folds old outputs there; set PI_ZIP_INTURN_AGE=0 as well for a fully protected previous turn. */
|
|
16
|
+
export const RELAX_PREV_TURN = true;
|
|
17
|
+
|
|
18
|
+
/** Default age, in assistant requests, from which ANY foldable output (every recoverability class) inside the protected user turns may
|
|
19
|
+
* still be folded. A sub-agent session is one user turn with hundreds of tool rounds, so without this its candidate set is empty for
|
|
20
|
+
* its whole life (offline sim on 295 recorded sessions: 1.37x the billion-context bill). 0 disables the rule (PI_ZIP_INTURN_AGE).
|
|
21
|
+
* Plans run at turn_end, before the next request, so `age >= 60` here equals the sim's `b <= v - 61`. 60 is the final model's quality
|
|
22
|
+
* constant (largest saving with lost <= prod in every stratum); the old rereadable-only filter cost 1.026 / 1.018x at -0.6 lost items. */
|
|
23
|
+
export const INTURN_AGE = 60;
|
|
24
|
+
|
|
25
|
+
const PROTECT_USER_TURNS = 2; // the latest user turn and the one before it are never summarised; folded only by relax (cold plan, previous turn, rereadable) or in-turn (old enough)
|
|
26
|
+
const SUMMARY_FLOOR = 1000;
|
|
27
|
+
const SUMMARY_CAP = 8000;
|
|
28
|
+
const SUMMARY_RATIO = 0.1;
|
|
29
|
+
|
|
30
|
+
export const DEFAULT_RESERVE_TOKENS = 16_384; // Pi's compaction reserve default
|
|
31
|
+
export const COMPACT_MARGIN = 8_192; // estimates are chars/4: stay clear of the trigger
|
|
32
|
+
const MIN_TARGET = 4_096;
|
|
33
|
+
|
|
34
|
+
/** Pi's compaction reserve for this model: compaction.modelOverrides["provider/id"], else compaction.reserveTokens, else 16384. */
|
|
35
|
+
export function reserveTokensFor(settings: Any, model: Any): number {
|
|
36
|
+
const c = settings?.compaction;
|
|
37
|
+
const key = model ? `${model.provider}/${model.id}` : "";
|
|
38
|
+
for (const v of [c?.modelOverrides?.[key]?.reserveTokens, c?.reserveTokens]) if (typeof v === "number" && Number.isFinite(v) && v >= 0) return v;
|
|
39
|
+
return DEFAULT_RESERVE_TOKENS;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** Highest context size that stays clear of Pi's compaction trigger; null when the window is unknown. */
|
|
43
|
+
export function compactionRoom(model: Any, reserve = DEFAULT_RESERVE_TOKENS): number | null {
|
|
44
|
+
const w = Number(model?.contextWindow);
|
|
45
|
+
return w > 0 ? Math.max(MIN_TARGET, w - reserve - COMPACT_MARGIN) : null;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// Legacy rule, only when nothing at all is known about prices (cache class unknown: no response from this model yet): a warm edit must cut at least half
|
|
49
|
+
// of the context, or, inside Pi's compaction danger zone (before >= room), merely reduce it.
|
|
50
|
+
export const VALVE_MIN_REDUCTION = 0.5;
|
|
51
|
+
export function legacyValve(before: number, after: number, room: number | null): boolean {
|
|
52
|
+
if (!(after < before)) return false;
|
|
53
|
+
if (room !== null && before >= room) return true;
|
|
54
|
+
return after <= (1 - VALVE_MIN_REDUCTION) * before;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export const PI_KEEP_RECENT = 20_000; // Pi's compaction keepRecentTokens default
|
|
58
|
+
export const G0 = 2_500; // growth per request before the session's own EWMA has data
|
|
59
|
+
|
|
60
|
+
/** What the law knows at a decision: prices (null = legacy rule), growth per request g (real tokens), P(the cache is warm). */
|
|
61
|
+
export interface Law { pr: Prices | null; g: number; pWarm: number }
|
|
62
|
+
export interface LawTerms { ok: boolean; B: number; A: number; T: number; P: number; K: number; eta: number; phi: number; Tsuf?: number }
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The round-5 law (verdict section 8): rewrite the cache only when what the edit saves pays for the rewrite it causes.
|
|
66
|
+
* Phi = [ r D^2/(2g) + eta D ] / K >= 1, D = B - A, K = (w - r) (P T - (1 - P) D) + call
|
|
67
|
+
* B, A = real context before / after the edit. T = the rewrite base; the planner passes T = A (final model: pricing only the suffix after
|
|
68
|
+
* the earliest edit, ~0.42 A, fires warm edits earlier and costs +0.9 / +4.6 lost items for -0.6 / -0.3% $; the suffix is logged as Tsuf,
|
|
69
|
+
* a measurement only). T is clamped to A. P = P(warm) at this gap: 1 = the verdict's warm valve, 0 = known dead (K < 0: always fire). The expected form is
|
|
70
|
+
* linear in P because a warm next request costs (w - r) T - r D more with the edit and a cold one w D less, plus the r D read credit.
|
|
71
|
+
* eta = 0 below Pi's compaction room, else Pi's own price per token of room (window term). Prices per token; no prices = legacy rule.
|
|
72
|
+
*/
|
|
73
|
+
export function lawTerms(B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A, keep = PI_KEEP_RECENT): LawTerms {
|
|
74
|
+
const t: LawTerms = { ok: false, B, A, T, P, K: NaN, eta: 0, phi: NaN };
|
|
75
|
+
if (!(A < B)) return t;
|
|
76
|
+
if (!pr || !(pr.r > 0)) return { ...t, ok: legacyValve(B, A, room) };
|
|
77
|
+
const { r, w, out } = pr;
|
|
78
|
+
const D = B - A, K = (w - r) * (P * Math.min(T, A) - (1 - P) * D) + call;
|
|
79
|
+
let eta = 0;
|
|
80
|
+
if (room !== null && B >= room) {
|
|
81
|
+
const S = Math.min(8_000, Math.max(1_000, 0.1 * (B - keep))), Api = keep + S;
|
|
82
|
+
eta = (w * (B - keep) + out * S + (w - r) * Api) / Math.max(B - Api, 1);
|
|
83
|
+
}
|
|
84
|
+
const gain = (r * D * D) / (2 * Math.max(g, 1)) + eta * D;
|
|
85
|
+
return { ...t, ok: K <= 0 || gain >= K, K, eta, phi: K > 0 ? gain / K : Infinity };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export const editAllowed = (B: number, A: number, P: number, room: number | null, pr: Prices | null, g: number, call = 0, T = A): boolean =>
|
|
89
|
+
lawTerms(B, A, P, room, pr, g, call, T).ok;
|
|
90
|
+
|
|
91
|
+
// Token scale. Every size in this file is a chars/4 estimate, and chars/4 undercounts real tokens (JSON-heavy tool calls, code,
|
|
92
|
+
// identifiers, tool definitions that are not in the text at all). `k` = real tokens per estimated token; every limit (cold cap,
|
|
93
|
+
// compaction room, min gain) is compared against k x estimate, so they mean REAL tokens. k is read from the branch itself (the
|
|
94
|
+
// usage of the newest assistant message), so it survives a restart and needs no stored state. With no usage to read, DEFAULT_K
|
|
95
|
+
// applies: 1.7 is the ratio measured on recorded coding sessions (real first-request context / chars/4 estimate: 1.72 after cold
|
|
96
|
+
// returns, 1.71 for previous-prompt peaks), and a too-high k only folds a little deeper, a too-low k leaves the context over the cap.
|
|
97
|
+
export const DEFAULT_K = 1.7;
|
|
98
|
+
export const K_MIN = 1; // an estimate above the real count is not trusted: never scale down
|
|
99
|
+
export const K_MAX = 2.5;
|
|
100
|
+
|
|
101
|
+
export interface Calibration {
|
|
102
|
+
k: number;
|
|
103
|
+
real: number; // input + cacheRead + cacheWrite of the request that produced the newest usable assistant message (0 = none)
|
|
104
|
+
est: number; // estimate of what that request carried: system prompt + the view blocks before that message
|
|
105
|
+
source: "usage" | "default";
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* k from the branch: the newest assistant message with usage says how many real tokens its request carried; the estimate of the
|
|
110
|
+
* view blocks before it (the projection, with the folds and cut as persisted, which are exactly what that request carried, I1)
|
|
111
|
+
* says how many we would have guessed. Only the INPUT side is used: that message's own output is not part of the request it
|
|
112
|
+
* answered, and thinking tokens may or may not be sent back, so including output would add noise to the ratio.
|
|
113
|
+
* Skipped: errored messages, messages without input usage, and messages older than a compaction Pi made (not ours): their
|
|
114
|
+
* request carried text the projection no longer has.
|
|
115
|
+
*/
|
|
116
|
+
export function calibrate(entries: Any[], sys = 0): Calibration {
|
|
117
|
+
const none: Calibration = { k: DEFAULT_K, real: 0, est: 0, source: "default" };
|
|
118
|
+
const blocks = buildBlocks(entries);
|
|
119
|
+
let foreignCompactionMs = 0;
|
|
120
|
+
for (const pe of entries) {
|
|
121
|
+
const src = pe.sourceEntry;
|
|
122
|
+
if (src?.type === "compaction" && src.details?.by !== PRODUCT) foreignCompactionMs = Math.max(foreignCompactionMs, Date.parse(src.timestamp) || 0);
|
|
123
|
+
}
|
|
124
|
+
const before: number[] = []; // estimate of blocks[0..i)
|
|
125
|
+
let acc = 0;
|
|
126
|
+
for (const b of blocks) { before.push(acc); acc += b.tokens; }
|
|
127
|
+
for (let i = blocks.length - 1; i >= 0; i--) {
|
|
128
|
+
if (blocks[i].kind !== "assistant") continue;
|
|
129
|
+
const m = blocks[i].raw ?? blocks[i].msg;
|
|
130
|
+
const u = m?.usage;
|
|
131
|
+
const real = u ? (Number(u.input) || 0) + (Number(u.cacheRead) || 0) + (Number(u.cacheWrite) || 0) : 0;
|
|
132
|
+
if (!(real > 0) || m.stopReason === "error") continue;
|
|
133
|
+
if (foreignCompactionMs && typeof m.timestamp === "number" && m.timestamp < foreignCompactionMs) return none;
|
|
134
|
+
const est = sys + before[i];
|
|
135
|
+
if (!(est > 0)) return none;
|
|
136
|
+
return { k: clamp(real / est, K_MIN, K_MAX), real, est, source: "usage" };
|
|
137
|
+
}
|
|
138
|
+
return none;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
export const settings = () => ({
|
|
142
|
+
coldCap: envInt("COLD_CAP", 40_000), // cold: fold, then summarise, down to this many tokens
|
|
143
|
+
foldMin: envInt("FOLD_MIN", 500), // outputs below this many tokens are never folded
|
|
144
|
+
keepLines: envInt("KEEP_LINES", 8),
|
|
145
|
+
minGain: envInt("MIN_GAIN", 10_000), // legacy rule only (no prices): a summary must remove at least max(minGain, 15% of the context)
|
|
146
|
+
inturnAge: envInt("INTURN_AGE", INTURN_AGE), // outputs (any class) of the protected turns at least this many assistant requests old may fold; 0 = never
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
/** Legacy summary gate (no prices): a summary only pays when it removes a real share of the context. */
|
|
150
|
+
export function summaryGainOk(prefixTokens: number, summaryTokens: number, totalTokens: number, minGain = settings().minGain): boolean {
|
|
151
|
+
return prefixTokens - summaryTokens >= Math.max(minGain, 0.15 * totalTokens);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// ---------------------------------------------------------------------------------------------------------------
|
|
155
|
+
// blocks: the model-visible context, one block per projected session entry
|
|
156
|
+
// ---------------------------------------------------------------------------------------------------------------
|
|
157
|
+
export interface Block {
|
|
158
|
+
idx: number;
|
|
159
|
+
entryId: string | null;
|
|
160
|
+
kind: "summary" | "user" | "assistant" | "toolResult" | "other";
|
|
161
|
+
msg: Any; // projected message (after edits)
|
|
162
|
+
raw: Any; // original message (before edits), message entries only
|
|
163
|
+
tokens: number;
|
|
164
|
+
userTurn: number; // user messages so far (inclusive)
|
|
165
|
+
edited: boolean; // some context_edit changed this entry
|
|
166
|
+
ours: boolean; // ... and it is one of our placeholders
|
|
167
|
+
age: number; // toolResult: assistant requests that came after the one that issued the call (0 = answered by the newest request); other kinds 0
|
|
168
|
+
}
|
|
169
|
+
export type Calls = Map<string, { name: string; args: Any }>;
|
|
170
|
+
|
|
171
|
+
/** User turns so far: the real prompts. Steering and follow-up messages typed during a run belong to that run's turn. */
|
|
172
|
+
export const countUserTurns = (blocks: Block[]): number => blocks.reduce((a, b) => Math.max(a, b.userTurn), 0);
|
|
173
|
+
|
|
174
|
+
export function buildBlocks(contextEntries: Any[], steerIds?: Set<string>): Block[] {
|
|
175
|
+
const blocks: Block[] = [];
|
|
176
|
+
let userTurn = 0;
|
|
177
|
+
let asst = 0; // assistant messages so far
|
|
178
|
+
const issued = new Map<string, number>(); // tool call id -> ordinal of the latest assistant message that issued it (ids may be reused)
|
|
179
|
+
const asstOf: number[] = []; // block idx -> ordinal of the issuing assistant message (toolResult blocks only)
|
|
180
|
+
for (const pe of contextEntries) {
|
|
181
|
+
const src = pe.sourceEntry;
|
|
182
|
+
const msgs: Any[] = pe.messages ?? [];
|
|
183
|
+
if (!msgs.length) continue;
|
|
184
|
+
let kind: Block["kind"] = "other";
|
|
185
|
+
let msg = msgs[msgs.length - 1];
|
|
186
|
+
let raw: Any;
|
|
187
|
+
let edited = false;
|
|
188
|
+
if (src.type === "compaction") {
|
|
189
|
+
kind = "summary";
|
|
190
|
+
msg = msgs.find((m) => m.role === "compactionSummary") ?? msg;
|
|
191
|
+
} else if (src.type === "message" && msgs.length === 1) {
|
|
192
|
+
raw = src.message;
|
|
193
|
+
msg = msgs[0];
|
|
194
|
+
const r = msg.role;
|
|
195
|
+
kind = r === "user" ? "user" : r === "assistant" ? "assistant" : r === "toolResult" ? "toolResult" : "other";
|
|
196
|
+
edited = msg !== raw && msg.content !== raw.content;
|
|
197
|
+
}
|
|
198
|
+
if (kind === "user" && !(src.id && steerIds?.has(src.id))) userTurn++;
|
|
199
|
+
const ours = edited && kind === "toolResult" && textOf(msg.content).startsWith(PH_MARK);
|
|
200
|
+
if (kind === "assistant") {
|
|
201
|
+
// aborted / final-error messages stay in the projection but pi-ai drops them before sending: they add no age (their calls still count as issued)
|
|
202
|
+
if (msg.stopReason !== "error" && msg.stopReason !== "aborted") asst++;
|
|
203
|
+
for (const c of msg.content ?? []) if (c?.type === "toolCall") issued.set(c.id, asst);
|
|
204
|
+
}
|
|
205
|
+
asstOf[blocks.length] = kind === "toolResult" ? (issued.get(msg.toolCallId) ?? asst) : asst;
|
|
206
|
+
blocks.push({ idx: blocks.length, entryId: src.id ?? null, kind, msg, raw, tokens: msgs.reduce((a, m) => a + tokensOf(m), 0), userTurn, edited, ours, age: 0 });
|
|
207
|
+
}
|
|
208
|
+
for (const b of blocks) if (b.kind === "toolResult") b.age = asst - asstOf[b.idx];
|
|
209
|
+
return blocks;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
export function toolCallIndex(blocks: Block[]): Calls {
|
|
213
|
+
const m: Calls = new Map();
|
|
214
|
+
for (const b of blocks) if (b.kind === "assistant") for (const c of b.msg.content ?? []) if (c?.type === "toolCall") m.set(c.id, { name: c.name, args: c.arguments });
|
|
215
|
+
return m;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// ---------------------------------------------------------------------------------------------------------------
|
|
219
|
+
// plan types
|
|
220
|
+
// ---------------------------------------------------------------------------------------------------------------
|
|
221
|
+
export interface FoldTarget {
|
|
222
|
+
entryId: string;
|
|
223
|
+
toolCallId: string;
|
|
224
|
+
visIdx: number; // index among the non-system projected messages when planned (-1 = unknown)
|
|
225
|
+
contentKey: string; // hash + length of the original text: some servers reuse tool call ids, so the id alone does not identify a result
|
|
226
|
+
tool: string;
|
|
227
|
+
args: string;
|
|
228
|
+
entryTokens: number;
|
|
229
|
+
phTokens: number;
|
|
230
|
+
ph: string; // placeholder text: byte-identical in the request-local view and the persisted context_edit
|
|
231
|
+
trig: string;
|
|
232
|
+
recover?: Recover;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
export interface Cut {
|
|
236
|
+
firstKeptEntryId: string;
|
|
237
|
+
text: string; // summary text: byte-identical in the request-local view and the persisted compaction
|
|
238
|
+
trigger: "cold" | "valve";
|
|
239
|
+
count: number; // user requests carried verbatim in the summary
|
|
240
|
+
prefixTokens: number;
|
|
241
|
+
summaryTokens: number;
|
|
242
|
+
llmOk: boolean;
|
|
243
|
+
llmError: string | null;
|
|
244
|
+
costUsd: number; // what the narrative model call cost (0 when unknown)
|
|
245
|
+
usage?: Any;
|
|
246
|
+
ms: number; // total production time (the narrative model call included)
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/** What a run sends from its first request on and persists verbatim at turn_end (F8). */
|
|
250
|
+
export interface RunPlan {
|
|
251
|
+
source: "settle" | "runstart" | "valve";
|
|
252
|
+
folds: FoldTarget[];
|
|
253
|
+
cut: Cut | null;
|
|
254
|
+
ctxBefore: number;
|
|
255
|
+
ctxAfter: number;
|
|
256
|
+
ms: number;
|
|
257
|
+
k: number; // the token scale the plan was made with (stats and notices of this run use the same one)
|
|
258
|
+
persisted: boolean;
|
|
259
|
+
cutVisibleIdx?: number; // where the kept part starts among the projected non-system messages
|
|
260
|
+
cutKeptFirstMsg?: Any; // ... and that message itself (a disagreeing request view drops the cut)
|
|
261
|
+
sent?: boolean; // a request view carrying this plan has gone out
|
|
262
|
+
applied?: Set<string>; // entry ids whose fold the latest request view actually carried (what turn_end must persist, no more)
|
|
263
|
+
cutTs?: number; // timestamp of the request-local summary message: one value per run, so every request of the run is identical
|
|
264
|
+
untouched?: number; // real tokens before the earliest edited block: what the first request carrying the plan can still read from the cache
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
export interface PlanOpts {
|
|
268
|
+
mode?: "cold" | "warm"; // cold: cache believed gone (default). warm: only when the context is above the cold cap and the law fires; then the cold plan without the previous-turn relax (kept only when pastTtl).
|
|
269
|
+
law?: Law; // prices, g and P(warm); default: no prices (legacy rule), P = 0 cold / 1 warm
|
|
270
|
+
trace?: (LawTerms & { where: "summary" | "plan" })[]; // every law evaluation is pushed here (ledger)
|
|
271
|
+
reserve?: number; // Pi's compaction reserveTokens (default 16384)
|
|
272
|
+
steerIds?: Set<string>; // user entries that are mid-run steering or follow-up messages, not new user turns
|
|
273
|
+
sys: number; // estimated tokens of the system prompt (same chars/4 scale as the blocks; tool definitions are covered by k)
|
|
274
|
+
k?: number; // real tokens per estimated token (calibrate); default 1 = the estimates are taken as real
|
|
275
|
+
cwd: string;
|
|
276
|
+
coldCap?: number;
|
|
277
|
+
foldMin?: number;
|
|
278
|
+
keepLines?: number;
|
|
279
|
+
recalled?: Set<string>;
|
|
280
|
+
model?: Any;
|
|
281
|
+
promptPending?: boolean; // true at settle: the upcoming prompt is NOT in `entries` yet but counts as a new user turn
|
|
282
|
+
relax?: boolean; // default RELAX_PREV_TURN
|
|
283
|
+
minGain?: number;
|
|
284
|
+
inturnAge?: number; // default settings().inturnAge (PI_ZIP_INTURN_AGE, 60); 0 = outputs of the protected turns never fold on age
|
|
285
|
+
pastTtl?: boolean; // a user return after the declared TTL: a warm plan may relax into the previous user turn too (RELAX_PREV_TURN)
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
export interface PlanResult {
|
|
289
|
+
blocks: Block[];
|
|
290
|
+
calls: Calls;
|
|
291
|
+
folds: FoldTarget[];
|
|
292
|
+
cutIdx: number | null;
|
|
293
|
+
firstKeptEntryId: string | null;
|
|
294
|
+
prefixTokens: number; // estimate units (x k = real), like summaryTokensPlanned
|
|
295
|
+
summaryTokensPlanned: number;
|
|
296
|
+
sumTrigger: "cold" | "valve" | null;
|
|
297
|
+
k: number;
|
|
298
|
+
ctxTokens: number; // calibrated: k x (sys + blocks)
|
|
299
|
+
ctxAfterFolds: number; // calibrated
|
|
300
|
+
userTurns: number;
|
|
301
|
+
cutVisibleIdx: number;
|
|
302
|
+
cutKeptFirstMsg: Any;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/** Estimated tokens before the earliest edited block (system prompt included): what stays cached through the edit. A cut edits from the start. */
|
|
306
|
+
export function untouchedEst(blocks: Block[], folds: FoldTarget[], cut: boolean, sys: number): number {
|
|
307
|
+
const ids = new Set(folds.map((t) => t.entryId));
|
|
308
|
+
let est = sys;
|
|
309
|
+
if (!cut) for (const b of blocks) { if (b.entryId && ids.has(b.entryId)) break; est += b.tokens; }
|
|
310
|
+
return est;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** Identity of a tool result's text, independent of its tool call id. */
|
|
314
|
+
export const contentKeyOf = (content: Any): string => {
|
|
315
|
+
const t = textOf(content);
|
|
316
|
+
return `${createHash("sha1").update(t).digest("hex")}:${t.length}`;
|
|
317
|
+
};
|
|
318
|
+
|
|
319
|
+
/** The planner. Cold: fold everything outside the protected window, relax into the previous turn if still above the cap,
|
|
320
|
+
* summarise only if folds cannot reach the cap and the law prices the summary call in. Warm: null unless the context is above the
|
|
321
|
+
* cold cap and the law fires for the plan (the summary call is sunk there); then the cold plan minus the previous-turn relax (a warm plan
|
|
322
|
+
* folds the previous user turn by relax only at a return after the declared TTL, o.pastTtl). A cold plan with P(warm) > 0
|
|
323
|
+
* passes the same law (expected cost). The cap never exceeds Pi's compaction room. Returns null when there is nothing to plan on
|
|
324
|
+
* or the law says no. */
|
|
325
|
+
export function planContext(entries: Any[], o: PlanOpts): PlanResult | null {
|
|
326
|
+
const s = settings();
|
|
327
|
+
const mode = o.mode ?? "cold";
|
|
328
|
+
// limits are in real tokens; the plan works in estimate units, so they are divided by k once, here
|
|
329
|
+
const k = o.k ?? 1;
|
|
330
|
+
const room = compactionRoom(o.model, o.reserve);
|
|
331
|
+
const coldCap = Math.min(o.coldCap ?? s.coldCap, room ?? Infinity) / k;
|
|
332
|
+
const law: Law = o.law ?? { pr: null, g: G0, pWarm: mode === "cold" ? 0 : 1 };
|
|
333
|
+
const gate = (where: "summary" | "plan", B: number, A: number, r: number | null, call: number, T: number, Tsuf?: number): boolean => {
|
|
334
|
+
const t = lawTerms(B, A, law.pWarm, r, law.pr, law.g, call, T);
|
|
335
|
+
o.trace?.push({ where, ...t, ...(Tsuf !== undefined ? { Tsuf } : {}) });
|
|
336
|
+
return t.ok;
|
|
337
|
+
};
|
|
338
|
+
const minGain = (o.minGain ?? s.minGain) / k;
|
|
339
|
+
const foldMin = o.foldMin ?? s.foldMin;
|
|
340
|
+
const keepLines = o.keepLines ?? s.keepLines;
|
|
341
|
+
const relax = o.relax ?? RELAX_PREV_TURN;
|
|
342
|
+
const inturnAge = o.inturnAge ?? s.inturnAge;
|
|
343
|
+
const blocks = buildBlocks(entries, o.steerIds);
|
|
344
|
+
if (!blocks.length) return null;
|
|
345
|
+
const userTurns = countUserTurns(blocks) + (o.promptPending ? 1 : 0); // the upcoming prompt is a new user turn
|
|
346
|
+
const calls = toolCallIndex(blocks);
|
|
347
|
+
const recalled = o.recalled ?? new Set<string>();
|
|
348
|
+
const ctxEst = o.sys + blocks.reduce((a, b) => a + b.tokens, 0);
|
|
349
|
+
if (mode === "warm" && !(ctxEst > coldCap)) return null; // I6: warm cache below the cold cap -> never edit
|
|
350
|
+
const trig = mode === "warm" ? "valve" : "cold";
|
|
351
|
+
const tokOverride = new Map<number, number>();
|
|
352
|
+
const folds: FoldTarget[] = [];
|
|
353
|
+
const visOfBlock: number[] = []; // block idx -> index among the non-system projected messages (the numbering a request view uses)
|
|
354
|
+
{
|
|
355
|
+
let vis = 0;
|
|
356
|
+
for (const pe of entries) {
|
|
357
|
+
const msgs: Any[] = pe.messages ?? [];
|
|
358
|
+
if (!msgs.length) continue;
|
|
359
|
+
visOfBlock.push(vis);
|
|
360
|
+
for (const m of msgs) if (String(m?.role ?? "") !== "system") vis++;
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
const classify = (b: Block): Recover => {
|
|
364
|
+
const call = calls.get(b.msg.toolCallId);
|
|
365
|
+
return classifyRecoverability(call?.name ?? b.msg.toolName ?? "", call?.args, textOf((b.raw ?? b.msg).content), o.cwd);
|
|
366
|
+
};
|
|
367
|
+
const addFold = (b: Block, trig: string): boolean => {
|
|
368
|
+
if (tokOverride.has(b.idx) || recalled.has(handleFor(b.entryId!))) return false; // F4: never refold what the model recalled
|
|
369
|
+
const ph = makePlaceholderFor(b, calls, keepLines);
|
|
370
|
+
if (ph === null) return false;
|
|
371
|
+
const phTok = tok4(ph);
|
|
372
|
+
if (!(phTok < 0.9 * b.tokens)) return false;
|
|
373
|
+
tokOverride.set(b.idx, phTok);
|
|
374
|
+
const call = calls.get(b.msg.toolCallId);
|
|
375
|
+
folds.push({ entryId: b.entryId!, toolCallId: b.msg.toolCallId, visIdx: visOfBlock[b.idx] ?? -1, contentKey: contentKeyOf(b.msg.content), tool: call?.name ?? b.msg.toolName ?? "tool", args: call ? shortArgs(call.args) : "", entryTokens: b.tokens, phTokens: phTok, ph, trig, recover: classify(b) });
|
|
376
|
+
return true;
|
|
377
|
+
};
|
|
378
|
+
// a zip_recall result IS content the model just asked for: folding it would undo the recall (and loop); never
|
|
379
|
+
const isRecall = (b: Block) => (calls.get(b.msg.toolCallId)?.name ?? b.msg.toolName) === RECALL_TOOL;
|
|
380
|
+
// Observability: on an automatic (prefix) cache no fold starts inside the first MIN_EXPECT real tokens, so the next response's cacheRead
|
|
381
|
+
// is an uncensored survival sample (learn.ts sample). Without it every idle return folded into the first 1-7K tokens and its sample was
|
|
382
|
+
// censored, so the learned curve never saw a return (live GLM bench). Explicit caches are exempt: the provider looks back only ~20
|
|
383
|
+
// blocks from the last breakpoint, a far head is not read either way. A cut replaces the prefix from the first message: exempt too.
|
|
384
|
+
const head = law.pr?.cls === "automatic" ? MIN_EXPECT / k : 0;
|
|
385
|
+
const start: number[] = [];
|
|
386
|
+
blocks.reduce((acc, b) => ((start[b.idx] = acc), acc + b.tokens), o.sys);
|
|
387
|
+
const foldable = (b: Block) => b.kind === "toolResult" && !b.edited && !!b.entryId && b.tokens > foldMin && !isRecall(b) && start[b.idx] >= head;
|
|
388
|
+
const protectedTurn = (b: Block) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1;
|
|
389
|
+
const savings = () => folds.reduce((a, t) => a + t.entryTokens - t.phTokens, 0);
|
|
390
|
+
const cands = blocks.filter((b) => foldable(b) && !protectedTurn(b));
|
|
391
|
+
for (const b of cands) addFold(b, trig);
|
|
392
|
+
if (relax && (mode === "cold" || o.pastTtl === true)) {
|
|
393
|
+
// the protected window = the new prompt + the previous user turn; that turn's big reads are what makes a cold return
|
|
394
|
+
// expensive. Rereadable ones can be recalled exactly: fold them biggest-first until the cap; never the latest turn's own.
|
|
395
|
+
// Cold plans, and warm plans at a return after the declared TTL; any other warm plan keeps the previous user turn visible.
|
|
396
|
+
let est = ctxEst - savings();
|
|
397
|
+
if (est > coldCap) {
|
|
398
|
+
const prev = blocks.filter((b) => foldable(b) && protectedTurn(b) && b.userTurn < userTurns && classify(b) === "rereadable");
|
|
399
|
+
for (const b of prev.sort((x, y) => y.tokens - x.tokens)) {
|
|
400
|
+
if (est <= coldCap) break;
|
|
401
|
+
if (addFold(b, `${trig}(relax)`)) est -= b.tokens - (tokOverride.get(b.idx) ?? 0);
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
if (inturnAge > 0) {
|
|
406
|
+
// in-turn: a protected turn (the new prompt's, the previous one, or the one a mid-run cold return / valve is inside) still holds
|
|
407
|
+
// outputs that are many requests old; they fold biggest-first until the cap, whatever their recoverability class (every fold
|
|
408
|
+
// stays recallable). Age = assistant requests since the call was issued (the newest outputs, age < inturnAge, always stay).
|
|
409
|
+
let est = ctxEst - savings();
|
|
410
|
+
if (est > coldCap) {
|
|
411
|
+
const aged = blocks.filter((b) => foldable(b) && protectedTurn(b) && b.age >= inturnAge);
|
|
412
|
+
for (const b of aged.sort((x, y) => y.tokens - x.tokens)) {
|
|
413
|
+
if (est <= coldCap) break;
|
|
414
|
+
if (addFold(b, `${trig}(inturn)`)) est -= b.tokens - (tokOverride.get(b.idx) ?? 0);
|
|
415
|
+
}
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
let ctxAfterFolds = ctxEst - savings();
|
|
419
|
+
const sumTrigger: "cold" | "valve" | null = ctxAfterFolds > coldCap ? trig : null;
|
|
420
|
+
let cutIdx: number | null = null;
|
|
421
|
+
let prefixTokens = 0;
|
|
422
|
+
let summaryTokensPlanned = 0;
|
|
423
|
+
if (sumTrigger) {
|
|
424
|
+
const limit = blocks.findIndex((b) => protectedTurn(b));
|
|
425
|
+
const cuts: number[] = [];
|
|
426
|
+
const minCut = blocks[0].kind === "summary" ? 2 : 1; // never re-summarise a prefix that is only the previous summary
|
|
427
|
+
for (let i = minCut; i < blocks.length; i++) {
|
|
428
|
+
if (limit >= 0 && i > limit) break;
|
|
429
|
+
if (blocks[i].entryId && (blocks[i].kind === "user" || blocks[i].kind === "assistant")) cuts.push(i);
|
|
430
|
+
}
|
|
431
|
+
const pre: number[] = [];
|
|
432
|
+
let acc = 0;
|
|
433
|
+
for (let i = 0; i < blocks.length; i++) {
|
|
434
|
+
pre[i] = acc;
|
|
435
|
+
acc += tokOverride.get(i) ?? blocks[i].tokens;
|
|
436
|
+
}
|
|
437
|
+
const total = o.sys + acc;
|
|
438
|
+
const S = (x: number) => clamp(SUMMARY_RATIO * x, SUMMARY_FLOOR, SUMMARY_CAP);
|
|
439
|
+
if (cuts.length) {
|
|
440
|
+
let pick = cuts[cuts.length - 1];
|
|
441
|
+
for (const c of cuts) if (total - pre[c] + S(pre[c]) <= coldCap) { pick = c; break; }
|
|
442
|
+
// cold: the summary call must pay for itself (verdict: call = w X + out S for an uncached call); warm: it is sunk in the plan's law
|
|
443
|
+
const X = k * pre[pick], Sr = k * S(pre[pick]);
|
|
444
|
+
const gain = !law.pr ? summaryGainOk(pre[pick], S(pre[pick]), total, minGain) : mode === "warm" || gate("summary", X, Sr, null, law.pr.w * X + law.pr.out * Sr, Sr);
|
|
445
|
+
if (pre[pick] >= 2 * S(pre[pick]) && gain) {
|
|
446
|
+
cutIdx = pick;
|
|
447
|
+
prefixTokens = pre[pick];
|
|
448
|
+
summaryTokensPlanned = S(pre[pick]);
|
|
449
|
+
for (let i = folds.length - 1; i >= 0; i--) {
|
|
450
|
+
const bi = blocks.findIndex((b) => b.entryId === folds[i].entryId);
|
|
451
|
+
if (bi < pick) { tokOverride.delete(bi); folds.splice(i, 1); } // the prefix is replaced: its folds are void
|
|
452
|
+
}
|
|
453
|
+
ctxAfterFolds = ctxEst - savings();
|
|
454
|
+
}
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
// where the kept part starts among the projected non-system messages (the request-local view drops everything before it)
|
|
458
|
+
let cutVisibleIdx = -1;
|
|
459
|
+
let cutKeptFirstMsg: Any;
|
|
460
|
+
if (cutIdx !== null) {
|
|
461
|
+
let vis = 0;
|
|
462
|
+
let bi = 0;
|
|
463
|
+
for (const pe of entries) {
|
|
464
|
+
const msgs: Any[] = pe.messages ?? [];
|
|
465
|
+
if (!msgs.length) continue;
|
|
466
|
+
for (const m of msgs) {
|
|
467
|
+
if (String(m?.role ?? "") === "system") continue;
|
|
468
|
+
if (bi === cutIdx && cutVisibleIdx < 0) { cutVisibleIdx = vis; cutKeptFirstMsg = m; }
|
|
469
|
+
vis++;
|
|
470
|
+
}
|
|
471
|
+
bi++;
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
// the untouched prefix: everything before the earliest edited block (a cut edits from the first block on)
|
|
475
|
+
const untouched = untouchedEst(blocks, folds, cutIdx !== null, o.sys);
|
|
476
|
+
const after = ctxAfterFolds - (cutIdx !== null ? prefixTokens - summaryTokensPlanned : 0);
|
|
477
|
+
const planned = folds.length > 0 || cutIdx !== null;
|
|
478
|
+
// F10: a warm rewrite must pay back; a cold plan with some chance of a warm cache is priced by expectation (no prices: cold always fires).
|
|
479
|
+
// T = A (the whole post-edit context); the suffix after the earliest edit goes to the ledger only.
|
|
480
|
+
if ((mode === "warm" || (planned && law.pr)) && !gate("plan", k * ctxEst, k * after, room, 0, k * after, k * (after - untouched))) return null;
|
|
481
|
+
return { blocks, calls, folds, cutIdx, firstKeptEntryId: cutIdx !== null ? blocks[cutIdx].entryId : null, prefixTokens, summaryTokensPlanned, sumTrigger, k, ctxTokens: k * ctxEst, ctxAfterFolds: k * ctxAfterFolds, userTurns, cutVisibleIdx, cutKeptFirstMsg };
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
const sameMsg = (a: Any, b: Any): boolean =>
|
|
485
|
+
!!a && !!b && String(a?.role ?? "") === String(b?.role ?? "") && JSON.stringify(a?.content ?? "") === JSON.stringify(b?.content ?? "");
|
|
486
|
+
|
|
487
|
+
const isSystem = (m: Any) => String(m?.role ?? "") === "system";
|
|
488
|
+
|
|
489
|
+
/**
|
|
490
|
+
* Replay every system message into the one leading message Pi puts at the head of a compacted projection
|
|
491
|
+
* (same rules as pi-ai getCurrentSystemMessage; the integration test checks the two agree).
|
|
492
|
+
*/
|
|
493
|
+
export function collapseSystem(messages: Any[]): Any | undefined {
|
|
494
|
+
const content: string[] = [];
|
|
495
|
+
const sections = new Map<string, string>();
|
|
496
|
+
const tools = new Map<string, Any>();
|
|
497
|
+
let timestamp: number | undefined;
|
|
498
|
+
for (const m of messages) {
|
|
499
|
+
if (!isSystem(m)) continue;
|
|
500
|
+
timestamp ??= m.timestamp;
|
|
501
|
+
const text = textOf(m.content);
|
|
502
|
+
if (text.length > 0) content.push(text);
|
|
503
|
+
for (const [name, value] of Object.entries(m.sections ?? {})) {
|
|
504
|
+
if (value === null) sections.delete(name);
|
|
505
|
+
else sections.set(name, value as string);
|
|
506
|
+
}
|
|
507
|
+
for (const t of m.toolsRemoved ?? []) tools.delete(t.name);
|
|
508
|
+
for (const t of m.toolsAdded ?? []) tools.set(t.name, t);
|
|
509
|
+
}
|
|
510
|
+
if (timestamp === undefined && tools.size === 0) return undefined;
|
|
511
|
+
return {
|
|
512
|
+
role: "system",
|
|
513
|
+
content: content.join("\n\n"),
|
|
514
|
+
...(sections.size > 0 ? { sections: Object.fromEntries(sections) } : {}),
|
|
515
|
+
...(tools.size > 0 ? { toolsAdded: [...tools.values()] } : {}),
|
|
516
|
+
timestamp: timestamp ?? 0,
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
/**
|
|
521
|
+
* Request-local view of a plan over the COMPLETE transcript (the `context_with_system` form): fold targets are replaced
|
|
522
|
+
* in place, system messages stay exactly where they are (so the request is byte-identical before and after turn_end
|
|
523
|
+
* persists the same edits, I1). With a cut, the view is what Pi projects after the compaction: the replayed system
|
|
524
|
+
* message, the summary, then the kept non-system messages. null = nothing to apply. A cut whose kept boundary cannot be
|
|
525
|
+
* located is dropped (folds still apply); an inconsistent split is never sent.
|
|
526
|
+
*/
|
|
527
|
+
export function applyPlanToMessages(messages: Any[], plan: RunPlan | null): { messages: Any[]; droppedCut: boolean; applied: Set<string> } | null {
|
|
528
|
+
if (!plan || plan.persisted || (!plan.folds.length && !plan.cut)) return null;
|
|
529
|
+
const vis: number[] = []; // indices of the non-system messages
|
|
530
|
+
messages.forEach((m, i) => { if (!isSystem(m)) vis.push(i); });
|
|
531
|
+
let cutAt = -1; // index into `messages`
|
|
532
|
+
let droppedCut = false;
|
|
533
|
+
if (plan.cut) {
|
|
534
|
+
const v = plan.cutVisibleIdx ?? -1;
|
|
535
|
+
cutAt = v >= 0 && v < vis.length ? vis[v] : -1;
|
|
536
|
+
if (cutAt < 0 || !sameMsg(messages[cutAt], plan.cutKeptFirstMsg)) {
|
|
537
|
+
plan.cut = null;
|
|
538
|
+
cutAt = -1;
|
|
539
|
+
droppedCut = true;
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
// match each result to its target: by position when the position still holds the same result, else by (call id, content) when that is unique
|
|
543
|
+
const used = new Set<FoldTarget>();
|
|
544
|
+
const applied = new Set<string>();
|
|
545
|
+
const byKey = new Map<string, FoldTarget[]>();
|
|
546
|
+
const key = (id: string, ck: string) => `${id}|${ck}`;
|
|
547
|
+
for (const t of plan.folds) {
|
|
548
|
+
const k = key(t.toolCallId, t.contentKey);
|
|
549
|
+
byKey.set(k, [...(byKey.get(k) ?? []), t]);
|
|
550
|
+
}
|
|
551
|
+
const seenKey = new Map<string, number>(); // how many results in THIS request carry each (id, content): more than one = ambiguous
|
|
552
|
+
for (const m of messages) {
|
|
553
|
+
if (m?.role !== "toolResult" || !byKey.has(key(m.toolCallId, contentKeyOf(m.content)))) continue;
|
|
554
|
+
const k = key(m.toolCallId, contentKeyOf(m.content));
|
|
555
|
+
seenKey.set(k, (seenKey.get(k) ?? 0) + 1);
|
|
556
|
+
}
|
|
557
|
+
const byPos = new Map<number, FoldTarget>();
|
|
558
|
+
for (const t of plan.folds) if (t.visIdx >= 0) byPos.set(t.visIdx, t);
|
|
559
|
+
let changed = false;
|
|
560
|
+
let v = -1;
|
|
561
|
+
const folded = messages.map((m: Any, i: number) => {
|
|
562
|
+
if (isSystem(m)) return m;
|
|
563
|
+
v++;
|
|
564
|
+
if (cutAt >= 0 && i < cutAt) return m; // replaced by the compaction message below
|
|
565
|
+
if (m?.role !== "toolResult" || textOf(m.content).startsWith(PH_MARK)) return m; // not ours, or the persisted edit already folded it
|
|
566
|
+
const ck = contentKeyOf(m.content);
|
|
567
|
+
let t = byPos.get(v);
|
|
568
|
+
if (!t || used.has(t) || t.toolCallId !== m.toolCallId || t.contentKey !== ck) {
|
|
569
|
+
const k = key(m.toolCallId, ck);
|
|
570
|
+
const c = byKey.get(k);
|
|
571
|
+
t = c && c.length === 1 && seenKey.get(k) === 1 && !used.has(c[0]) ? c[0] : undefined;
|
|
572
|
+
}
|
|
573
|
+
if (!t) return m;
|
|
574
|
+
used.add(t);
|
|
575
|
+
applied.add(t.entryId);
|
|
576
|
+
changed = true;
|
|
577
|
+
return { ...m, content: [{ type: "text", text: t.ph }] };
|
|
578
|
+
});
|
|
579
|
+
if (cutAt >= 0) {
|
|
580
|
+
const c = plan.cut!;
|
|
581
|
+
const head = collapseSystem(messages);
|
|
582
|
+
const kept = folded.slice(cutAt).filter((m: Any) => !isSystem(m));
|
|
583
|
+
return { messages: [...(head ? [head] : []), { role: "compactionSummary", summary: c.text, tokensBefore: c.prefixTokens, timestamp: plan.cutTs ??= Date.now() }, ...kept], droppedCut, applied };
|
|
584
|
+
}
|
|
585
|
+
return changed || droppedCut ? { messages: folded, droppedCut, applied } : null;
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
export { pickKeyLines };
|