pi-jev-lens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,151 @@
1
+ import { TypeSafeClient } from "@typesafe-ai/sdk";
2
+ import type { Config } from "./config.ts";
3
+ import type { Probabilities } from "./types.ts";
4
+ import { head, tail, truncate } from "./text.ts";
5
+
6
+ /** Everything jev sees about one tool result. Built identically by the extension and the replay harness. */
7
+ export interface ItemState {
8
+ task: { first_user_request: string; latest_user_message: string };
9
+ item: {
10
+ tool: string;
11
+ args: string;
12
+ is_error: boolean;
13
+ total_chars: number;
14
+ output_head: string;
15
+ output_tail: string;
16
+ };
17
+ after: {
18
+ assistant_text: string;
19
+ next_tool_calls: { name: string; args: string }[];
20
+ };
21
+ }
22
+
23
+ export interface TextState {
24
+ task: { first_user_request: string };
25
+ message: string;
26
+ role: "user" | "agent";
27
+ }
28
+
29
+ export interface Classifier {
30
+ classifyToolResult(state: ItemState, signal?: AbortSignal): Promise<Probabilities>;
31
+ classifyText(state: TextState, signal?: AbortSignal): Promise<number>;
32
+ }
33
+
34
+ export function buildItemState(
35
+ cfg: Pick<Config, "stateHeadChars" | "stateTailChars">,
36
+ input: {
37
+ firstUser: string;
38
+ latestUser: string;
39
+ toolName: string;
40
+ args: unknown;
41
+ isError: boolean;
42
+ output: string;
43
+ afterText: string;
44
+ afterCalls: { name: string; arguments: unknown }[];
45
+ },
46
+ ): ItemState {
47
+ return {
48
+ task: {
49
+ first_user_request: truncate(input.firstUser, 600),
50
+ latest_user_message: truncate(input.latestUser, 400),
51
+ },
52
+ item: {
53
+ tool: input.toolName,
54
+ args: truncate(JSON.stringify(input.args ?? {}), 300),
55
+ is_error: input.isError,
56
+ total_chars: input.output.length,
57
+ output_head: head(input.output, cfg.stateHeadChars),
58
+ output_tail: input.output.length > cfg.stateHeadChars ? tail(input.output, cfg.stateTailChars) : "",
59
+ },
60
+ after: {
61
+ assistant_text: truncate(input.afterText, 800),
62
+ next_tool_calls: input.afterCalls.slice(0, 8).map((c) => ({ name: c.name, args: truncate(JSON.stringify(c.arguments ?? {}), 150) })),
63
+ },
64
+ };
65
+ }
66
+
67
+ export const TOOL_RESULT_QUESTIONS = {
68
+ needed: {
69
+ type: "noul" as const,
70
+ instructions:
71
+ "`item` is the output of a tool the coding agent ran while working on `task`. `after` shows what the agent said and which tools it called right after seeing this output. Will the agent still need the full text of `item.output_head` and `item.output_tail` verbatim in its upcoming steps?",
72
+ criteria: {
73
+ true: "The agent is still working on what this output shows: it will edit, quote, compare against, or reason over specific lines of it; or it has not acted on it yet; or the output holds details (line numbers, exact error text, exact code) it will need again.",
74
+ false: "The agent already acted on it (edited the file, fixed the error, answered from it), moved on to a different area, the output was a dead end or irrelevant, or it is cheap to regenerate by running the same tool again.",
75
+ },
76
+ },
77
+ outcome_only: {
78
+ type: "noul" as const,
79
+ instructions:
80
+ "Is the useful information in `item` limited to its outcome, such as success or failure, the final status lines, an error message, or a count, so that the middle of the output could be dropped without losing anything the agent needs?",
81
+ criteria: {
82
+ true: "Command output, logs, install or build noise, test runs where only the pass/fail summary or the failing case matters.",
83
+ false: "Source code, file contents, search results, directory listings, or any output where specific lines in the middle carry the information.",
84
+ },
85
+ },
86
+ durable: {
87
+ type: "noul" as const,
88
+ instructions:
89
+ "Does `item` reveal a stable fact about this project (its structure, conventions, how to build, test or run it, a known pitfall) or about the user's preferences, that would still be true and useful in a future, unrelated session?",
90
+ criteria: {
91
+ true: "Build or test commands that work, project layout, conventions, configuration quirks, recurring gotchas.",
92
+ false: "Task-specific content, transient state, one-off command output, file contents that change with every edit.",
93
+ },
94
+ },
95
+ };
96
+
97
+ export const TEXT_QUESTIONS = {
98
+ durable_user: {
99
+ type: "noul" as const,
100
+ instructions:
101
+ "`message` was written by the user to a coding agent. Does it state a preference, standing instruction, or fact about the user or the project that should be remembered in future sessions, rather than a one-off task instruction?",
102
+ criteria: {
103
+ true: "Coding style preferences, tools or workflows the user wants used, facts about the project's purpose, constraints that will keep applying.",
104
+ false: "A task for right now, a question, feedback about one specific change, small talk.",
105
+ },
106
+ },
107
+ durable_agent: {
108
+ type: "noul" as const,
109
+ instructions:
110
+ "`message` was written by a coding agent. Does it state a conclusion, decision, or discovered fact about the project that will remain true and be useful in future unrelated sessions?",
111
+ criteria: {
112
+ true: "How the project is structured, where things live, what command runs the tests, a root cause that explains recurring behaviour, a design decision that was made.",
113
+ false: "Progress narration, a plan for the current task, a question to the user, a summary of edits just made.",
114
+ },
115
+ },
116
+ };
117
+
118
+ export class JevClassifier implements Classifier {
119
+ private client: TypeSafeClient;
120
+ private model: string;
121
+ constructor(cfg: Pick<Config, "apiKey" | "model">) {
122
+ this.client = new TypeSafeClient({ apiKey: cfg.apiKey });
123
+ this.model = cfg.model;
124
+ }
125
+ async classifyToolResult(state: ItemState, signal?: AbortSignal): Promise<Probabilities> {
126
+ const r = await this.client.systemOne({ state: state as never, questions: TOOL_RESULT_QUESTIONS, model: this.model }, { signal, timeout: 15000 });
127
+ return { needed: r.answers.needed.noul, outcomeOnly: r.answers.outcome_only.noul, durable: r.answers.durable.noul };
128
+ }
129
+ async classifyText(state: TextState, signal?: AbortSignal): Promise<number> {
130
+ const q = state.role === "user" ? { durable: TEXT_QUESTIONS.durable_user } : { durable: TEXT_QUESTIONS.durable_agent };
131
+ const r = await this.client.systemOne({ state: { task: state.task, message: state.message }, questions: q, model: this.model }, { signal, timeout: 15000 });
132
+ return r.answers.durable.noul;
133
+ }
134
+ }
135
+
136
+ /** Deterministic stand-in for tests and dry runs. Never touches the network. */
137
+ export class MockClassifier implements Classifier {
138
+ constructor(private rule: (state: ItemState) => Probabilities = defaultMockRule) {}
139
+ async classifyToolResult(state: ItemState): Promise<Probabilities> {
140
+ return this.rule(state);
141
+ }
142
+ async classifyText(): Promise<number> {
143
+ return 0;
144
+ }
145
+ }
146
+
147
+ export function defaultMockRule(state: ItemState): Probabilities {
148
+ const big = state.item.total_chars > 2000;
149
+ const cmd = state.item.tool === "bash";
150
+ return { needed: big ? 0.1 : 0.9, outcomeOnly: cmd ? 0.9 : 0.1, durable: 0 };
151
+ }
package/src/config.ts ADDED
@@ -0,0 +1,188 @@
1
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
2
+ import { homedir } from "node:os";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath } from "node:url";
5
+
6
+ export type SealMode = "rolling" | "batch" | "budget";
7
+
8
+ export interface Config {
9
+ /** Disable all pruning (classification still runs and logs). */
10
+ enabled: boolean;
11
+ /**
12
+ * rolling: apply decisions at the next LLM call (smallest prompt, one cache rewrite per call while pruning).
13
+ * batch: apply only when the cache is cold or on compaction (best cache, prompt shrinks late).
14
+ * budget: like batch, but also apply when pending prunable tokens exceed a share of the prompt (one rewrite buys many calls).
15
+ */
16
+ mode: SealMode;
17
+ /** budget mode: apply pending decisions when they remove at least this fraction of the tail they would rewrite... */
18
+ budgetFraction: number;
19
+ /** ...and at least this many tokens. */
20
+ budgetMinTokens: number;
21
+ /** P(needed) below this → forget (stub). */
22
+ forgetBelow: number;
23
+ /** P(needed) below this and P(outcomeOnly) above trimAbove → trim to head+tail. */
24
+ trimBelow: number;
25
+ trimAbove: number;
26
+ /** P(durable) above this → written to the memory file. */
27
+ durableAbove: number;
28
+ /** Tool results smaller than this (estimated tokens) are never touched. */
29
+ minTokens: number;
30
+ /** Maximum wait for in-flight classifications at context, agent end and shutdown. */
31
+ classifyWaitMs: number;
32
+ /** Provider prompt-cache TTL; idle longer than this means the cache is cold. */
33
+ cacheTtlMs: number;
34
+ /** Lines kept at head/tail when trimming. */
35
+ trimHeadLines: number;
36
+ trimTailLines: number;
37
+ /** Max chars of tool output sent to jev (head + tail). */
38
+ stateHeadChars: number;
39
+ stateTailChars: number;
40
+ /** Pre-send compression of large tool results (jev picks a view before the output is ever sent). */
41
+ presend: boolean;
42
+ /** Only results at least this large (estimated tokens) are considered for pre-send compression. */
43
+ presendMinTokens: number;
44
+ /** Send full when P(needs full) is above this. */
45
+ presendNeedsFullAbove: number;
46
+ /** Send full when the "full" option itself gets more than this probability mass. */
47
+ presendFullMassAbove: number;
48
+ /** Separate needs-full threshold for command output (test runs), where "exact full text" is rarely what the agent needs. */
49
+ presendCommandNeedsFullAbove: number;
50
+ /** Code policy: "gate" (default) = jev's needs-full/full-mass gates decide between full and a view; "outline" = code is always sent as outline plus the blocks the second step expands (17 % edit-miss on 500 real trajectories, see STATUS). */
51
+ presendCodePolicy: "gate" | "outline";
52
+ /** Stricter needs-full threshold for source code, where a wrong view costs an edit (jev's answers vary run to run by ±0.2). */
53
+ presendCodeNeedsFullAbove: number;
54
+ /** Send full when the choice confidence is below this (0 = off; a spread over acceptable views is not a reason to send everything). */
55
+ presendMinConfidence: number;
56
+ /** Second step for code: expand the bodies of blocks jev says the agent will need (P above this). */
57
+ presendExpandAbove: number;
58
+ /** Command policy: "sections" (default) = when jev picks full for command output but needs-full is under the command threshold, send section headers and let the second step expand the sections it needs; "gate" = jev's view choice stands. */
59
+ presendCommandPolicy: "gate" | "sections";
60
+ /** Second step for command output: expand a section when P(needed) is above this. */
61
+ presendSectionExpandAbove: number;
62
+ /** Command output: when no section reaches this probability the step is uninformative and full is sent (0 = headers alone are allowed). */
63
+ presendSectionFloor: number;
64
+ model: string;
65
+ /** Optional variant file (JEV_LENS_VARIANT): { config, prompts, views } overrides, as produced by eval/bench/autoresearch.ts. */
66
+ variantFile: string | undefined;
67
+ /** Force the mock classifier even when a key is present (tests, dry runs). */
68
+ forceMock: boolean;
69
+ logFile: boolean;
70
+ apiKey: string | undefined;
71
+ }
72
+
73
+ const HERE = dirname(fileURLToPath(import.meta.url));
74
+
75
+ /** Where `/jev-lens key` stores the TypeSafe API key (overridable for tests). */
76
+ export function keyFilePath(): string {
77
+ return process.env.JEV_LENS_KEY_FILE || join(homedir(), ".pi", "agent", "jev-lens.json");
78
+ }
79
+
80
+ /** The stored key, if any. */
81
+ export function readStoredKey(): string | undefined {
82
+ try {
83
+ const p = keyFilePath();
84
+ if (!existsSync(p)) return undefined;
85
+ const key = (JSON.parse(readFileSync(p, "utf8")) as { apiKey?: unknown }).apiKey;
86
+ return typeof key === "string" && key.trim() ? key.trim() : undefined;
87
+ } catch {
88
+ return undefined;
89
+ }
90
+ }
91
+
92
+ /** Store the key in the user's pi directory, readable only by the user. */
93
+ export function storeKey(key: string): string {
94
+ const p = keyFilePath();
95
+ mkdirSync(dirname(p), { recursive: true });
96
+ writeFileSync(p, `${JSON.stringify({ apiKey: key.trim() }, null, 2)}\n`, { mode: 0o600 });
97
+ return p;
98
+ }
99
+
100
+ /** Key resolution: environment (or the package's .env, loaded into it), then the stored key. */
101
+ export function resolveApiKey(): string | undefined {
102
+ return process.env.TYPESAFE_API_KEY || readStoredKey();
103
+ }
104
+
105
+ /** Load KEY=VALUE lines from the extension's own .env (never from the target project). */
106
+ export function loadDotEnv(): void {
107
+ for (const dir of [join(HERE, ".."), HERE]) {
108
+ const p = join(dir, ".env");
109
+ if (!existsSync(p)) continue;
110
+ for (const line of readFileSync(p, "utf8").split("\n")) {
111
+ const m = line.match(/^\s*([A-Z0-9_]+)\s*=\s*(.*?)\s*$/);
112
+ if (!m || !m[2] || process.env[m[1]]) continue;
113
+ process.env[m[1]] = m[2].replace(/^["']|["']$/g, "");
114
+ }
115
+ }
116
+ }
117
+
118
+ function num(name: string, fallback: number): number {
119
+ const v = process.env[name];
120
+ if (v === undefined || v === "") return fallback;
121
+ const n = Number(v);
122
+ return Number.isFinite(n) ? n : fallback;
123
+ }
124
+
125
+ export function loadConfig(): Config {
126
+ loadDotEnv();
127
+ const envMode = process.env.JEV_LENS_MODE;
128
+ const mode: SealMode = envMode === "batch" || envMode === "budget" || envMode === "rolling" ? envMode : "budget";
129
+ return {
130
+ enabled: process.env.JEV_LENS_DISABLED !== "1",
131
+ mode,
132
+ budgetFraction: num("JEV_LENS_BUDGET_FRACTION", 0.5),
133
+ budgetMinTokens: num("JEV_LENS_BUDGET_MIN_TOKENS", 1000),
134
+ forgetBelow: num("JEV_LENS_FORGET_BELOW", 0.25),
135
+ trimBelow: num("JEV_LENS_TRIM_BELOW", 0.5),
136
+ trimAbove: num("JEV_LENS_TRIM_ABOVE", 0.6),
137
+ durableAbove: num("JEV_LENS_DURABLE_ABOVE", 0.7),
138
+ minTokens: num("JEV_LENS_MIN_TOKENS", 150),
139
+ classifyWaitMs: num("JEV_LENS_CLASSIFY_WAIT_MS", 2500),
140
+ cacheTtlMs: num("JEV_LENS_CACHE_TTL_MS", 5 * 60 * 1000),
141
+ trimHeadLines: num("JEV_LENS_TRIM_HEAD", 15),
142
+ trimTailLines: num("JEV_LENS_TRIM_TAIL", 15),
143
+ stateHeadChars: num("JEV_LENS_STATE_HEAD", 2500),
144
+ stateTailChars: num("JEV_LENS_STATE_TAIL", 800),
145
+ presend: process.env.JEV_LENS_PRESEND !== "0",
146
+ presendMinTokens: num("JEV_LENS_PRESEND_MIN_TOKENS", 1200),
147
+ presendNeedsFullAbove: num("JEV_LENS_PRESEND_NEEDS_FULL_ABOVE", 0.5),
148
+ presendFullMassAbove: num("JEV_LENS_PRESEND_FULL_MASS_ABOVE", 0.5),
149
+ presendCodeNeedsFullAbove: num("JEV_LENS_PRESEND_CODE_NEEDS_FULL_ABOVE", 0.5),
150
+ presendCommandNeedsFullAbove: num("JEV_LENS_PRESEND_COMMAND_NEEDS_FULL_ABOVE", 0.65),
151
+ presendCodePolicy: process.env.JEV_LENS_PRESEND_CODE_POLICY === "outline" ? "outline" : "gate",
152
+ presendMinConfidence: num("JEV_LENS_PRESEND_MIN_CONFIDENCE", 0),
153
+ presendExpandAbove: num("JEV_LENS_PRESEND_EXPAND_ABOVE", 0.5),
154
+ presendCommandPolicy: process.env.JEV_LENS_PRESEND_COMMAND_POLICY === "gate" ? "gate" : "sections",
155
+ presendSectionExpandAbove: num("JEV_LENS_PRESEND_SECTION_EXPAND_ABOVE", 0.5),
156
+ presendSectionFloor: num("JEV_LENS_PRESEND_SECTION_FLOOR", 0.3),
157
+ model: process.env.JEV_LENS_MODEL || "jev-latest",
158
+ variantFile: process.env.JEV_LENS_VARIANT || undefined,
159
+ forceMock: process.env.JEV_LENS_CLASSIFIER === "mock",
160
+ logFile: process.env.JEV_LENS_LOG !== "0",
161
+ apiKey: resolveApiKey(),
162
+ };
163
+ }
164
+
165
+ export interface VariantOverrides {
166
+ name?: string;
167
+ config?: Partial<Config>;
168
+ prompts?: Record<string, unknown>;
169
+ views?: Record<string, number>;
170
+ }
171
+
172
+ /** Read a variant file (either a bare variant or an autoresearch best.json with { variant }). */
173
+ export function loadVariant(path: string | undefined): VariantOverrides {
174
+ if (!path) return {};
175
+ try {
176
+ const raw = JSON.parse(readFileSync(path, "utf8")) as VariantOverrides & { variant?: VariantOverrides };
177
+ return raw.variant ?? raw;
178
+ } catch {
179
+ return {};
180
+ }
181
+ }
182
+
183
+ /** Config with a variant's config overrides applied. */
184
+ export function loadConfigWithVariant(): { cfg: Config; variant: VariantOverrides } {
185
+ const base = loadConfig();
186
+ const variant = loadVariant(base.variantFile);
187
+ return { cfg: { ...base, ...(variant.config ?? {}) }, variant };
188
+ }
package/src/ledger.ts ADDED
@@ -0,0 +1,25 @@
1
+ import type { Decision } from "./types.ts";
2
+
3
+ export const ENTRY_TYPE = "jev-lens";
4
+
5
+ export interface LedgerEntryData {
6
+ kind: "decision";
7
+ decision: Decision;
8
+ }
9
+
10
+ /** Rebuild the decision map from persisted custom entries (survives resume, reload, fork). */
11
+ export function rebuildLedger(entries: Iterable<unknown>): Map<string, Decision> {
12
+ const ledger = new Map<string, Decision>();
13
+ for (const e of entries) {
14
+ const entry = e as { type?: string; customType?: string; data?: LedgerEntryData };
15
+ if (entry.type !== "custom" || entry.customType !== ENTRY_TYPE || !entry.data) continue;
16
+ if (entry.data.kind === "decision") {
17
+ const d = entry.data.decision;
18
+ const prev = ledger.get(d.id);
19
+ // Later entries win; an applied entry never regresses to pending.
20
+ if (prev?.status === "applied" && d.status === "pending") continue;
21
+ ledger.set(d.id, { ...d });
22
+ }
23
+ }
24
+ return ledger;
25
+ }
@@ -0,0 +1,51 @@
1
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
2
+ import { dirname } from "node:path";
3
+ import type { DurableNote } from "./types.ts";
4
+
5
+ export const MEMORY_MAX_LINES = 150;
6
+ export const MEMORY_MAX_CHARS = 8000;
7
+
8
+ function hash(s: string): string {
9
+ let h = 0;
10
+ for (let i = 0; i < s.length; i++) h = (Math.imul(31, h) + s.charCodeAt(i)) | 0;
11
+ return (h >>> 0).toString(16);
12
+ }
13
+
14
+ export function readMemoryFile(path: string): string {
15
+ return existsSync(path) ? readFileSync(path, "utf8") : "";
16
+ }
17
+
18
+ /** Append durable notes as bullet lines, dedup by normalized text, cap size keeping the newest. */
19
+ export function appendNotes(path: string, notes: DurableNote[]): number {
20
+ if (notes.length === 0) return 0;
21
+ const existing = readMemoryFile(path);
22
+ const lines = existing.split("\n").filter((l) => l.startsWith("- "));
23
+ const seen = new Set(lines.map((l) => hash(l.replace(/^- \(\S+, \w+\) /, "").trim().toLowerCase())));
24
+ let added = 0;
25
+ for (const n of notes) {
26
+ const text = n.text.replace(/\s+/g, " ").trim();
27
+ if (!text) continue;
28
+ const key = hash(text.toLowerCase());
29
+ if (seen.has(key)) continue;
30
+ seen.add(key);
31
+ const date = new Date(n.at).toISOString().slice(0, 10);
32
+ lines.push(`- (${date}, ${n.source}) ${text.length > 400 ? `${text.slice(0, 400)}…` : text}`);
33
+ added++;
34
+ }
35
+ if (added === 0) return 0;
36
+ let kept = lines.slice(-MEMORY_MAX_LINES);
37
+ while (kept.join("\n").length > MEMORY_MAX_CHARS && kept.length > 1) kept = kept.slice(1);
38
+ const header = "# jev-lens\n\nDurable notes selected by jev from earlier sessions. Newest last.\n\n";
39
+ mkdirSync(dirname(path), { recursive: true });
40
+ writeFileSync(path, `${header}${kept.join("\n")}\n`, "utf8");
41
+ return added;
42
+ }
43
+
44
+ export function memoryPromptSection(snapshot: string): string {
45
+ const body = snapshot
46
+ .split("\n")
47
+ .filter((l) => l.startsWith("- "))
48
+ .join("\n");
49
+ if (!body) return "";
50
+ return `\n\n# Memory from earlier sessions (jev-lens)\nThese notes were kept from previous sessions in this project. Treat them as likely but verify before relying on details.\n${body}\n`;
51
+ }
@@ -0,0 +1,6 @@
1
+ import type { ContextEvent } from "@earendil-works/pi-coding-agent";
2
+
3
+ /** pi's AgentMessage union, taken from the context event so we track pi's own definition. */
4
+ export type AgentMessage = ContextEvent["messages"][number];
5
+ export type ToolResultMessage = Extract<AgentMessage, { role: "toolResult" }>;
6
+ export type AssistantMessage = Extract<AgentMessage, { role: "assistant" }>;
package/src/policy.ts ADDED
@@ -0,0 +1,145 @@
1
+ import type { AgentMessage, ToolResultMessage } from "./pi-types.ts";
2
+ import type { Config } from "./config.ts";
3
+ import type { Bucket, Decision } from "./types.ts";
4
+ import { contentText, estimateTokensOfText } from "./text.ts";
5
+
6
+ export const STUB_PREFIX = "[jev-lens pruned:";
7
+
8
+ export function stubText(summary: string): string {
9
+ return `${STUB_PREFIX} ${summary}. The output was judged no longer needed; call the tool again if you need it.]`;
10
+ }
11
+
12
+ export function trimText(text: string, headLines: number, tailLines: number): string {
13
+ const lines = text.split("\n");
14
+ if (lines.length <= headLines + tailLines + 2) return text;
15
+ const dropped = lines.length - headLines - tailLines;
16
+ return [
17
+ ...lines.slice(0, headLines),
18
+ `[jev-lens trimmed ${dropped} lines here; only the head and tail were judged useful. Call the tool again for the full output.]`,
19
+ ...lines.slice(-tailLines),
20
+ ].join("\n");
21
+ }
22
+
23
+ /** Deterministic transform of a tool result given a frozen decision. */
24
+ export function transformToolResult(message: AgentMessage, decision: Decision, cfg: Config): AgentMessage {
25
+ if (message.role !== "toolResult" || decision.bucket === "keep") return message;
26
+ // Classifiers only see text; never apply even a restored decision to unseen image content.
27
+ if (message.content.some((c) => c.type !== "text")) return message;
28
+ const original = contentText(message.content);
29
+ const text = decision.bucket === "forget" ? stubText(decision.summary) : trimText(original, cfg.trimHeadLines, cfg.trimTailLines);
30
+ return { ...message, content: [{ type: "text", text }] };
31
+ }
32
+
33
+ export function decideBucket(p: { needed: number; outcomeOnly: number }, cfg: Config): Bucket {
34
+ if (p.needed < cfg.forgetBelow) return "forget";
35
+ if (p.needed < cfg.trimBelow && p.outcomeOnly > cfg.trimAbove) return "trim";
36
+ return "keep";
37
+ }
38
+
39
+ /**
40
+ * Tokens that pending (not yet applied) decisions would remove from this message list, and the
41
+ * size of the tail that applying them would rewrite (everything from the first pending decision on).
42
+ */
43
+ export function pendingPrunable(messages: AgentMessage[], ledger: Map<string, Decision>, cfg: Config): { tokens: number; tailTokens: number } {
44
+ let tokens = 0;
45
+ let tailTokens = 0;
46
+ let inTail = false;
47
+ for (const m of messages) {
48
+ const size = estimateTokensOfText(contentText((m as { content?: unknown }).content));
49
+ if (m.role === "toolResult") {
50
+ const d = ledger.get(m.toolCallId);
51
+ if (d && d.status === "pending" && d.bucket !== "keep") {
52
+ const after = estimateTokensOfText(contentText((transformToolResult(m, d, cfg) as ToolResultMessage).content));
53
+ tokens += Math.max(0, size - after);
54
+ inTail = true;
55
+ }
56
+ }
57
+ if (inTail) tailTokens += size;
58
+ }
59
+ return { tokens, tailTokens };
60
+ }
61
+
62
+ /**
63
+ * Decide whether this call should apply pending decisions, given the mode and cache state.
64
+ * Budget mode reasons about the cache: applying rewrites the tail once (tailTokens at full price
65
+ * instead of the cached rate), and then saves `tokens` at the cached rate on every later call.
66
+ * With a 10x cache discount that pays back after ~9 × tailTokens / tokens calls, so we apply
67
+ * when the prunable share of the tail is at least `budgetFraction`.
68
+ */
69
+ export function shouldApplyPending(
70
+ mode: Config["mode"],
71
+ cfg: Config,
72
+ coldCache: boolean,
73
+ pending: { tokens: number; tailTokens: number },
74
+ ): { apply: boolean; reason: string } {
75
+ if (coldCache) return { apply: true, reason: "cold-cache" };
76
+ if (mode === "rolling") return { apply: true, reason: "rolling" };
77
+ if (mode === "budget" && pending.tokens >= cfg.budgetMinTokens && pending.tokens >= cfg.budgetFraction * pending.tailTokens) {
78
+ return { apply: true, reason: "budget" };
79
+ }
80
+ return { apply: false, reason: mode };
81
+ }
82
+
83
+ export interface ApplyResult {
84
+ messages: AgentMessage[];
85
+ tokensOriginal: number;
86
+ tokensSent: number;
87
+ appliedNow: Decision[];
88
+ frozen: number;
89
+ pendingHeld: number;
90
+ }
91
+
92
+ /**
93
+ * Apply the ledger to the outgoing message list.
94
+ * - Frozen (applied) decisions are always re-applied identically, so the prompt prefix stays stable.
95
+ * - Pending decisions are applied now only if `applyPending` is true; they then freeze.
96
+ * - Messages without a decision pass through untouched. Tool results are never removed,
97
+ * only rewritten, because every function_call needs a matching output.
98
+ */
99
+ export function applyLedger(
100
+ messages: AgentMessage[],
101
+ ledger: Map<string, Decision>,
102
+ cfg: Config,
103
+ applyPending: boolean,
104
+ callIndex: number,
105
+ reason: string,
106
+ ): ApplyResult {
107
+ const out: AgentMessage[] = [];
108
+ const appliedNow: Decision[] = [];
109
+ let frozen = 0;
110
+ let pendingHeld = 0;
111
+ let tokensOriginal = 0;
112
+ let tokensSent = 0;
113
+ for (const m of messages) {
114
+ if (m.role !== "toolResult") {
115
+ out.push(m);
116
+ continue;
117
+ }
118
+ const before = estimateTokensOfText(contentText(m.content));
119
+ tokensOriginal += before;
120
+ const d = ledger.get(m.toolCallId);
121
+ if (!d || d.bucket === "keep") {
122
+ out.push(m);
123
+ tokensSent += before;
124
+ continue;
125
+ }
126
+ if (d.status === "pending") {
127
+ if (!applyPending) {
128
+ pendingHeld++;
129
+ out.push(m);
130
+ tokensSent += before;
131
+ continue;
132
+ }
133
+ d.status = "applied";
134
+ d.appliedAtCall = callIndex;
135
+ d.appliedReason = reason;
136
+ appliedNow.push(d);
137
+ } else {
138
+ frozen++;
139
+ }
140
+ const t = transformToolResult(m, d, cfg) as ToolResultMessage;
141
+ out.push(t);
142
+ tokensSent += estimateTokensOfText(contentText(t.content));
143
+ }
144
+ return { messages: out, tokensOriginal, tokensSent, appliedNow, frozen, pendingHeld };
145
+ }