@gajae-code/agent-core 0.8.2 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -92,10 +92,23 @@ export declare class AppendOnlyLog {
92
92
  * ctx = mgr.build(context); // subsequent calls use cache
93
93
  * ```
94
94
  */
95
+ export interface AppendOnlyContextManagerOptions {
96
+ /**
97
+ * Invoked whenever the stable prefix fingerprint changes on `build()` (a
98
+ * provider prompt-cache prefix reset). Used for per-session diagnostics; must
99
+ * not throw. `from` is `<unbuilt>` on the first build.
100
+ */
101
+ readonly onPrefixChange?: (info: {
102
+ from: string;
103
+ to: string;
104
+ version: number;
105
+ }) => void;
106
+ }
95
107
  export declare class AppendOnlyContextManager {
96
108
  #private;
97
109
  readonly prefix: StablePrefix;
98
110
  readonly log: AppendOnlyLog;
111
+ constructor(options?: AppendOnlyContextManagerOptions);
99
112
  static forkFromSeed(args: {
100
113
  prefixSnapshot?: StablePrefixSnapshot;
101
114
  messages?: readonly Message[];
@@ -98,6 +98,11 @@ export declare const DEFAULT_EMERGENCY_COMPACTION_LIMITS: EmergencyCompactionLim
98
98
  */
99
99
  export declare function emergencyCompactionReason(sample: EmergencyCompactionSample, limits?: EmergencyCompactionLimits): CompactionTriggerReason | null;
100
100
  export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
101
+ /**
102
+ * Image content has no tokenizer representation; charge a fixed estimate
103
+ * matching what providers typically bill for inline images.
104
+ */
105
+ export declare const IMAGE_TOKEN_ESTIMATE = 1200;
101
106
  /**
102
107
  * Native-free chars/4 token estimate for a message. This is the only message
103
108
  * token estimator: provider usage (see {@link calculatePromptTokens}) anchors
@@ -243,8 +248,32 @@ export interface CompactionPreparation {
243
248
  fileOps: FileOperations;
244
249
  /** Compaction settions from settings.jsonl */
245
250
  settings: CompactionSettings;
251
+ /**
252
+ * Diagnostics for the keep-window token correction (Finding 7). `ratio` is the
253
+ * clamped heuristic→actual correction that was applied (1 when none supplied);
254
+ * `keepRecentTokensCorrected` is the heuristic budget findCutPoint actually used.
255
+ */
256
+ tokenCorrection: {
257
+ ratio: number;
258
+ keepRecentTokensCorrected: number;
259
+ };
260
+ }
261
+ /** Bounds for the keep-window token correction (Finding 7): never trust a ratio
262
+ * beyond 2x in either direction so a bad estimate cannot balloon or collapse the
263
+ * kept window. */
264
+ export declare const TOKEN_CORRECTION_MIN_RATIO = 0.5;
265
+ export declare const TOKEN_CORRECTION_MAX_RATIO = 2;
266
+ export interface PrepareCompactionOptions {
267
+ /**
268
+ * Observed heuristic→actual token correction for the post-boundary keep window
269
+ * (actualTokens / chars-4-heuristicTokens), supplied by the caller from per-turn
270
+ * Usage deltas or a stable-prefix-subtracted comparison. Clamped to
271
+ * [0.5, 2] and applied bidirectionally. When omitted, no correction is applied
272
+ * (the confounded raw promptTokens/estimatedTokens quotient is never used).
273
+ */
274
+ tokenCorrectionRatio?: number;
246
275
  }
247
- export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings): CompactionPreparation | undefined;
276
+ export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, options?: PrepareCompactionOptions): CompactionPreparation | undefined;
248
277
  /**
249
278
  * Generate summaries for compaction using prepared data.
250
279
  * Returns CompactionResult - SessionManager adds id/parentId when saving.
@@ -44,4 +44,26 @@ export interface AssistantArgumentPruneResult {
44
44
  prunedEntries: SessionMessageEntry[];
45
45
  }
46
46
  export declare function pruneAssistantToolArguments(entries: SessionEntry[], config?: PruneConfig): AssistantArgumentPruneResult;
47
+ /**
48
+ * Estimate the token savings {@link pruneToolOutputs} would achieve, without
49
+ * mutating any entry. Returns 0 savings when below the configured minimum so the
50
+ * caller sees the same gate the real prune enforces.
51
+ */
52
+ export declare function estimateToolOutputPruneSavings(entries: SessionEntry[], config?: PruneConfig): {
53
+ prunableCount: number;
54
+ tokensSaved: number;
55
+ };
56
+ /**
57
+ * Evidence gate for below-threshold maintenance pruning (Finding 13). Pruning
58
+ * forces a prompt-cache-epoch reset, so it only runs when opted in AND the
59
+ * estimated stale savings clear a high minimum AND exceed the one-time reset
60
+ * cost (so the reclaim pays the reset back). Default-off/blocked until live
61
+ * evidence justifies enabling.
62
+ */
63
+ export declare function shouldRunMaintenancePrune(args: {
64
+ enabled: boolean;
65
+ estimatedSavings: number;
66
+ minSavings: number;
67
+ cacheEpochResetCost: number;
68
+ }): boolean;
47
69
  export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
package/package.json CHANGED
@@ -1,13 +1,10 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/agent-core",
4
- "version": "0.8.2",
4
+ "version": "0.9.1",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://gajae-code.com",
7
- "author": "Yeachan-Heo",
8
- "contributors": [
9
- "Mario Zechner"
10
- ],
7
+ "author": "Yeachan-Heo and Gajae Code Contributors",
11
8
  "license": "MIT",
12
9
  "repository": {
13
10
  "type": "git",
@@ -35,9 +32,9 @@
35
32
  "fmt": "biome format --write ."
36
33
  },
37
34
  "dependencies": {
38
- "@gajae-code/ai": "0.8.2",
39
- "@gajae-code/natives": "0.8.2",
40
- "@gajae-code/utils": "0.8.2",
35
+ "@gajae-code/ai": "0.9.1",
36
+ "@gajae-code/natives": "0.9.1",
37
+ "@gajae-code/utils": "0.9.1",
41
38
  "@opentelemetry/api": "^1.9.0"
42
39
  },
43
40
  "devDependencies": {
package/src/agent-loop.ts CHANGED
@@ -342,9 +342,136 @@ export function normalizeMessagesForProvider(
342
342
  return changed ? normalized : messages;
343
343
  }
344
344
 
345
+ interface ConvertedContextCacheEntry {
346
+ messageHashes: string[];
347
+ modelKey: string;
348
+ toolKey: string;
349
+ intentTracing: boolean;
350
+ convertToLlm: AgentLoopConfig["convertToLlm"];
351
+ transformContext: AgentLoopConfig["transformContext"];
352
+ llmMessages: Context["messages"];
353
+ normalizedMessages: Context["messages"];
354
+ }
355
+
356
+ const convertedContextCache = new WeakMap<AgentLoopConfig, ConvertedContextCacheEntry>();
357
+
358
+ function stableCacheString(value: unknown): string | undefined {
359
+ try {
360
+ return JSON.stringify(value, (_key, item) =>
361
+ typeof item === "function" ? `[Function:${item.name || "anonymous"}]` : item,
362
+ );
363
+ } catch {
364
+ return undefined;
365
+ }
366
+ }
367
+
368
+ /**
369
+ * Hash a message by full content serialization.
370
+ *
371
+ * Deliberately NOT memoized by object identity: callers mutate messages in
372
+ * place (compaction rewrites, obfuscation, abort markers) and the cache's
373
+ * correctness contract requires detecting those mutations. The per-turn
374
+ * serialization cost is the price of that contract; the win is skipping
375
+ * convertToLlm + normalize on stable contexts, which dominates for
376
+ * image-heavy histories.
377
+ */
378
+ function hashMessageContent(message: AgentMessage): string | undefined {
379
+ return stableCacheString(message);
380
+ }
381
+
382
+ function buildConvertedContextCacheKeys(
383
+ messages: AgentMessage[],
384
+ context: AgentContext,
385
+ config: AgentLoopConfig,
386
+ ): Pick<ConvertedContextCacheEntry, "messageHashes" | "modelKey" | "toolKey" | "intentTracing"> | undefined {
387
+ const intentTracing = !!config.intentTracing;
388
+ const messageHashes = messages.map(hashMessageContent);
389
+ const modelKey = stableCacheString(config.model);
390
+ const toolKey = stableCacheString(normalizeTools(context.tools, intentTracing) ?? []);
391
+ if (messageHashes.some(hash => hash === undefined) || modelKey === undefined || toolKey === undefined) {
392
+ return undefined;
393
+ }
394
+ return {
395
+ messageHashes: messageHashes as string[],
396
+ modelKey,
397
+ toolKey,
398
+ intentTracing,
399
+ };
400
+ }
401
+
402
+ function findStablePrefixLength(previous: string[], next: string[]): number {
403
+ const max = Math.min(previous.length, next.length);
404
+ let index = 0;
405
+ while (index < max && previous[index] === next[index]) index++;
406
+ return index;
407
+ }
408
+
409
+ async function convertAndNormalizeMessages(
410
+ messages: AgentMessage[],
411
+ context: AgentContext,
412
+ config: AgentLoopConfig,
413
+ ): Promise<Context["messages"]> {
414
+ const keys = buildConvertedContextCacheKeys(messages, context, config);
415
+ if (!keys) {
416
+ return normalizeMessagesForProvider(await config.convertToLlm(messages), config.model);
417
+ }
418
+ const previous = convertedContextCache.get(config);
419
+ const canReuse =
420
+ previous &&
421
+ previous.convertToLlm === config.convertToLlm &&
422
+ previous.transformContext === config.transformContext &&
423
+ previous.modelKey === keys.modelKey &&
424
+ previous.toolKey === keys.toolKey &&
425
+ previous.intentTracing === keys.intentTracing;
426
+
427
+ if (canReuse) {
428
+ const stablePrefixLength = findStablePrefixLength(previous.messageHashes, keys.messageHashes);
429
+ if (stablePrefixLength === keys.messageHashes.length && stablePrefixLength === previous.messageHashes.length) {
430
+ return previous.normalizedMessages;
431
+ }
432
+ // Append-only fast path: convert only the new suffix and concatenate.
433
+ // CONTRACT: `convertToLlm` must be per-message (each output message
434
+ // derived solely from its input message). The bundled converters
435
+ // satisfy this — they map/filter message-by-message. A converter that
436
+ // merges adjacent messages or pairs across the suffix boundary would
437
+ // diverge from a full rebuild; such converters must not be combined
438
+ // with appendOnlyContext. Covered by the suffix-equivalence test in
439
+ // agent-loop-context-cache.test.ts.
440
+ if (
441
+ config.appendOnlyContext &&
442
+ stablePrefixLength === previous.messageHashes.length &&
443
+ keys.messageHashes.length > previous.messageHashes.length
444
+ ) {
445
+ const suffix = messages.slice(stablePrefixLength);
446
+ const convertedSuffix = await config.convertToLlm(suffix);
447
+ const llmMessages = [...previous.llmMessages, ...convertedSuffix];
448
+ const normalizedMessages = normalizeMessagesForProvider(llmMessages, config.model);
449
+ convertedContextCache.set(config, {
450
+ ...keys,
451
+ convertToLlm: config.convertToLlm,
452
+ transformContext: config.transformContext,
453
+ llmMessages,
454
+ normalizedMessages,
455
+ });
456
+ return normalizedMessages;
457
+ }
458
+ }
459
+
460
+ const llmMessages = await config.convertToLlm(messages);
461
+ const normalizedMessages = normalizeMessagesForProvider(llmMessages, config.model);
462
+ convertedContextCache.set(config, {
463
+ ...keys,
464
+ convertToLlm: config.convertToLlm,
465
+ transformContext: config.transformContext,
466
+ llmMessages,
467
+ normalizedMessages,
468
+ });
469
+ return normalizedMessages;
470
+ }
471
+
345
472
  export const INTENT_FIELD = "_i";
346
473
 
347
- function injectIntentIntoSchema(schema: unknown, mode: "require" | "optional" = "require"): unknown {
474
+ function injectIntentIntoSchema(schema: unknown, mode: "require" | "optional" = "optional"): unknown {
348
475
  if (!schema || typeof schema !== "object" || Array.isArray(schema)) return schema;
349
476
  const schemaRecord = schema as Record<string, unknown>;
350
477
  const propertiesValue = schemaRecord.properties;
@@ -400,7 +527,7 @@ export function normalizeTools(tools: AgentContext["tools"], injectIntent: boole
400
527
  function resolveIntentMode(intent: AgentTool["intent"]): "require" | "optional" | "omit" {
401
528
  if (typeof intent === "function") return "omit";
402
529
  if (intent === "optional" || intent === "omit") return intent;
403
- return "require";
530
+ return intent === "require" ? "require" : "optional";
404
531
  }
405
532
 
406
533
  function extractIntent(args: Record<string, unknown>): { intent?: string; strippedArgs: Record<string, unknown> } {
@@ -711,9 +838,9 @@ async function streamAssistantResponse(
711
838
  messages = await config.transformContext(messages, signal);
712
839
  }
713
840
 
714
- // Convert to LLM-compatible messages (AgentMessage[] → Message[])
715
- const llmMessages = await config.convertToLlm(messages);
716
- const normalizedMessages = normalizeMessagesForProvider(llmMessages, config.model);
841
+ // Convert to LLM-compatible messages (AgentMessage[] → Message[]) and normalize at the LLM boundary.
842
+ // Cache hits are keyed by provider-visible content hashes, never message object identity.
843
+ const normalizedMessages = await convertAndNormalizeMessages(messages, context, config);
717
844
 
718
845
  // Build LLM context — append-only mode caches system prompt + tools
719
846
  // AND keeps an append-only message log so prior-turn bytes are stable.
@@ -194,6 +194,15 @@ export class AppendOnlyLog {
194
194
  * ctx = mgr.build(context); // subsequent calls use cache
195
195
  * ```
196
196
  */
197
+ export interface AppendOnlyContextManagerOptions {
198
+ /**
199
+ * Invoked whenever the stable prefix fingerprint changes on `build()` (a
200
+ * provider prompt-cache prefix reset). Used for per-session diagnostics; must
201
+ * not throw. `from` is `<unbuilt>` on the first build.
202
+ */
203
+ readonly onPrefixChange?: (info: { from: string; to: string; version: number }) => void;
204
+ }
205
+
197
206
  export class AppendOnlyContextManager {
198
207
  readonly prefix = new StablePrefix();
199
208
  readonly log = new AppendOnlyLog();
@@ -203,6 +212,11 @@ export class AppendOnlyContextManager {
203
212
  #syncedHashes: (number | bigint)[] = [];
204
213
  /** Number of provider-normalized messages that were seeded before child-local messages. */
205
214
  #seededPrefixCount = 0;
215
+ readonly #onPrefixChange: AppendOnlyContextManagerOptions["onPrefixChange"];
216
+
217
+ constructor(options: AppendOnlyContextManagerOptions = {}) {
218
+ this.#onPrefixChange = options.onPrefixChange;
219
+ }
206
220
 
207
221
  static forkFromSeed(args: {
208
222
  prefixSnapshot?: StablePrefixSnapshot;
@@ -220,7 +234,15 @@ export class AppendOnlyContextManager {
220
234
  }
221
235
 
222
236
  build(context: AgentContext, options: BuildOptions): Context {
223
- this.prefix.build(context, options);
237
+ const previousFingerprint = this.prefix.fingerprint;
238
+ const changed = this.prefix.build(context, options);
239
+ if (changed && this.#onPrefixChange) {
240
+ this.#onPrefixChange({
241
+ from: previousFingerprint,
242
+ to: this.prefix.fingerprint,
243
+ version: this.prefix.version,
244
+ });
245
+ }
224
246
  const { systemPrompt, tools } = this.prefix.toContext();
225
247
  return { systemPrompt, messages: this.log.toMessages(), tools };
226
248
  }
@@ -315,7 +315,7 @@ export function resolveThresholdTokens(
315
315
  * Image content has no tokenizer representation; charge a fixed estimate
316
316
  * matching what providers typically bill for inline images.
317
317
  */
318
- const IMAGE_TOKEN_ESTIMATE = 1200;
318
+ export const IMAGE_TOKEN_ESTIMATE = 1200;
319
319
  /**
320
320
  * Estimate tokens for collected message fragments using the native-free
321
321
  * heuristic. Provider usage is the authoritative anchor for context-changing
@@ -977,11 +977,35 @@ export interface CompactionPreparation {
977
977
  fileOps: FileOperations;
978
978
  /** Compaction settions from settings.jsonl */
979
979
  settings: CompactionSettings;
980
+ /**
981
+ * Diagnostics for the keep-window token correction (Finding 7). `ratio` is the
982
+ * clamped heuristic→actual correction that was applied (1 when none supplied);
983
+ * `keepRecentTokensCorrected` is the heuristic budget findCutPoint actually used.
984
+ */
985
+ tokenCorrection: { ratio: number; keepRecentTokensCorrected: number };
986
+ }
987
+
988
+ /** Bounds for the keep-window token correction (Finding 7): never trust a ratio
989
+ * beyond 2x in either direction so a bad estimate cannot balloon or collapse the
990
+ * kept window. */
991
+ export const TOKEN_CORRECTION_MIN_RATIO = 0.5;
992
+ export const TOKEN_CORRECTION_MAX_RATIO = 2;
993
+
994
+ export interface PrepareCompactionOptions {
995
+ /**
996
+ * Observed heuristic→actual token correction for the post-boundary keep window
997
+ * (actualTokens / chars-4-heuristicTokens), supplied by the caller from per-turn
998
+ * Usage deltas or a stable-prefix-subtracted comparison. Clamped to
999
+ * [0.5, 2] and applied bidirectionally. When omitted, no correction is applied
1000
+ * (the confounded raw promptTokens/estimatedTokens quotient is never used).
1001
+ */
1002
+ tokenCorrectionRatio?: number;
980
1003
  }
981
1004
 
982
1005
  export function prepareCompaction(
983
1006
  pathEntries: SessionEntry[],
984
1007
  settings: CompactionSettings,
1008
+ options: PrepareCompactionOptions = {},
985
1009
  ): CompactionPreparation | undefined {
986
1010
  if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
987
1011
  return undefined;
@@ -999,17 +1023,22 @@ export function prepareCompaction(
999
1023
 
1000
1024
  const lastUsage = getLastAssistantUsage(pathEntries);
1001
1025
  const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
1002
- let keepRecentTokens = settings.keepRecentTokens;
1003
- if (lastUsage) {
1004
- const estimatedTokens = estimateEntriesTokens(pathEntries, boundaryStart, boundaryEnd);
1005
- const promptTokens = calculatePromptTokens(lastUsage);
1006
- const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
1007
- if (Number.isFinite(ratio) && ratio > 1) {
1008
- keepRecentTokens = Math.max(1, Math.floor(keepRecentTokens / ratio));
1009
- }
1010
- }
1011
1026
 
1012
- const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokens);
1027
+ // Correct the keep-window budget for the chars/4 heuristic error using the
1028
+ // caller-supplied observed ratio (actual/heuristic). The legacy raw
1029
+ // promptTokens/estimatedTokens quotient is intentionally NOT used: promptTokens
1030
+ // counts system+tools+full history while estimatedTokens counted only the
1031
+ // post-boundary slice, so it was confounded and only ever shrank the window.
1032
+ // Here the correction is bidirectional and clamped to [0.5, 2].
1033
+ const keepRecentTokens = settings.keepRecentTokens;
1034
+ const rawRatio = options.tokenCorrectionRatio;
1035
+ const appliedRatio =
1036
+ rawRatio !== undefined && Number.isFinite(rawRatio) && rawRatio > 0
1037
+ ? Math.min(TOKEN_CORRECTION_MAX_RATIO, Math.max(TOKEN_CORRECTION_MIN_RATIO, rawRatio))
1038
+ : 1;
1039
+ const keepRecentTokensCorrected = Math.max(1, Math.round(keepRecentTokens / appliedRatio));
1040
+
1041
+ const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokensCorrected);
1013
1042
 
1014
1043
  // Get ID of first kept entry
1015
1044
  const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
@@ -1077,6 +1106,7 @@ export function prepareCompaction(
1077
1106
  previousPreserveData,
1078
1107
  fileOps,
1079
1108
  settings,
1109
+ tokenCorrection: { ratio: appliedRatio, keepRecentTokensCorrected },
1080
1110
  };
1081
1111
  }
1082
1112
 
@@ -527,14 +527,29 @@ export function pruneAssistantToolArguments(
527
527
  return { argumentPrunedCount: candidates.length, argumentTokensSaved, prunedEntries };
528
528
  }
529
529
 
530
- export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
530
+ interface ToolOutputPruneCandidate {
531
+ entry: SessionMessageEntry;
532
+ tokens: number;
533
+ notice: string;
534
+ savings: number;
535
+ }
536
+
537
+ /**
538
+ * Read-only pass that collects the tool-result entries that {@link pruneToolOutputs}
539
+ * would prune, plus the total estimated token savings. Shared by the mutating
540
+ * prune and the non-mutating {@link estimateToolOutputPruneSavings} so the
541
+ * maintenance gate (Finding 13) can decide whether pruning is worth a cache-epoch
542
+ * reset without rewriting history.
543
+ */
544
+ function collectToolOutputPruneCandidates(
545
+ entries: SessionEntry[],
546
+ config: PruneConfig,
547
+ ): { candidates: ToolOutputPruneCandidate[]; tokensSaved: number } {
531
548
  let accumulatedTokens = 0;
532
- let tokensSaved = 0;
533
- let prunedCount = 0;
534
549
 
535
550
  const { staleResultIndices } = buildStalenessIndex(entries);
536
551
  const staleOverridable = new Set(config.staleOverridableTools ?? []);
537
- const candidates: Array<{ entry: SessionMessageEntry; tokens: number; notice: string; savings: number }> = [];
552
+ const candidates: ToolOutputPruneCandidate[] = [];
538
553
 
539
554
  for (let i = entries.length - 1; i >= 0; i--) {
540
555
  const entry = entries[i];
@@ -574,14 +589,56 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
574
589
  accumulatedTokens += tokens;
575
590
  }
576
591
 
592
+ let tokensSaved = 0;
577
593
  for (const candidate of candidates) {
578
594
  tokensSaved += candidate.savings;
579
595
  }
596
+ return { candidates, tokensSaved };
597
+ }
598
+
599
+ /**
600
+ * Estimate the token savings {@link pruneToolOutputs} would achieve, without
601
+ * mutating any entry. Returns 0 savings when below the configured minimum so the
602
+ * caller sees the same gate the real prune enforces.
603
+ */
604
+ export function estimateToolOutputPruneSavings(
605
+ entries: SessionEntry[],
606
+ config: PruneConfig = DEFAULT_PRUNE_CONFIG,
607
+ ): { prunableCount: number; tokensSaved: number } {
608
+ const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
609
+ if (tokensSaved < config.minimumSavings || candidates.length === 0) {
610
+ return { prunableCount: 0, tokensSaved: 0 };
611
+ }
612
+ return { prunableCount: candidates.length, tokensSaved };
613
+ }
614
+
615
+ /**
616
+ * Evidence gate for below-threshold maintenance pruning (Finding 13). Pruning
617
+ * forces a prompt-cache-epoch reset, so it only runs when opted in AND the
618
+ * estimated stale savings clear a high minimum AND exceed the one-time reset
619
+ * cost (so the reclaim pays the reset back). Default-off/blocked until live
620
+ * evidence justifies enabling.
621
+ */
622
+ export function shouldRunMaintenancePrune(args: {
623
+ enabled: boolean;
624
+ estimatedSavings: number;
625
+ minSavings: number;
626
+ cacheEpochResetCost: number;
627
+ }): boolean {
628
+ if (!args.enabled) return false;
629
+ if (args.estimatedSavings < args.minSavings) return false;
630
+ return args.estimatedSavings > args.cacheEpochResetCost;
631
+ }
632
+
633
+ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
634
+ const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
580
635
 
581
636
  if (tokensSaved < config.minimumSavings || candidates.length === 0) {
582
637
  return { prunedCount: 0, tokensSaved: 0, prunedEntries: [] };
583
638
  }
584
639
 
640
+ let prunedCount = 0;
641
+
585
642
  const prunedAt = Date.now();
586
643
  const prunedEntries: SessionMessageEntry[] = [];
587
644
  for (const candidate of candidates) {