@caupulican/pi-agent-core 0.81.7 → 0.81.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -118,7 +118,7 @@ export declare function findCutPoint(entries: SessionEntry[], startIndex: number
118
118
  * If previousSummary is provided, uses the update prompt to merge.
119
119
  */
120
120
  export declare function generateSummary(currentMessages: AgentMessage[], model: Model<any>, reserveTokens: number, apiKey: string | undefined, headers?: Record<string, string>, signal?: AbortSignal, customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, preDigest?: (conversationText: string, signal?: AbortSignal) => Promise<string>, factsBlock?: string, chunked?: boolean): Promise<string>;
121
- /** Hard ceiling for the facts-scaled summary budget; also the worst case selection must assume. */
121
+ /** Worst-case selection assumption for summary output when exact bounded facts are not available. */
122
122
  export declare const SUMMARY_BUDGET_MAX_TOKENS = 4000;
123
123
  /**
124
124
  * Whether a candidate summarizer can ingest a summarization input of the given size in ONE
@@ -1 +1 @@
1
- {"version":3,"file":"compaction.d.ts","sourceRoot":"","sources":["../../src/compaction/compaction.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EAA6B,KAAK,EAAuB,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAQtG,OAAO,EAA6C,KAAK,YAAY,EAAE,MAAM,+BAA+B,CAAC;AAC7G,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,aAAa,CAAC;AACzE,OAAO,EAAE,KAAK,eAAe,EAA4C,MAAM,iBAAiB,CAAC;AACjG,OAAO,EAIN,KAAK,cAAc,EAInB,MAAM,YAAY,CAAC;AACpB,OAAO,EAAoB,KAAK,kBAAkB,EAAiB,MAAM,mBAAmB,CAAC;AAM7F,kEAAkE;AAClE,MAAM,WAAW,iBAAiB;IACjC,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,aAAa,EAAE,MAAM,EAAE,CAAC;CACxB;AAkED,8EAA8E;AAC9E,MAAM,WAAW,gBAAgB,CAAC,CAAC,GAAG,OAAO;IAC5C,OAAO,EAAE,MAAM,CAAC;IAChB,gBAAgB,EAAE,MAAM,CAAC;IACzB,YAAY,EAAE,MAAM,CAAC;IACrB,+FAA+F;IAC/F,OAAO,CAAC,EAAE,CAAC,CAAC;IACZ,YAAY,CAAC,EAAE,kBAAkB,CAAC;CAClC;AAMD,MAAM,WAAW,kBAAkB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,aAAa,EAAE,MAAM,CAAC;IACtB,gBAAgB,EAAE,MAAM,CAAC;IACzB;;;;;;OAMG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CACxB;AAED,eAAO,MAAM,2BAA2B,EAAE,kBAKzC,CAAC;AAMF;;;GAGG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AAgBD;;GAEG;AACH,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,YAAY,EAAE,GAAG,KAAK,GAAG,SAAS,CAShF;AAED,MAAM,WAAW,oBAAoB;IACpC,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAUD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,YAAY,EAAE,GAAG,oBAAoB,CA4BpF;AAED;;;;;;GAMG;AACH,eAAO,MAAM,sBAAsB,OAAO,CAAC;AAE3C;;;;;;;;GAQG;AACH,wBAAgB,aAAa,CAC5B,aAAa,EAAE,MAAM,EACrB,aAAa,EAAE,MAAM,EACrB,QAAQ,EAAE,kBAAkB,EAC5B,aAAa,CAAC,EAAE,MAAM,GACpB,OAAO,CAmBT;AAwBD;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,OAAO,EAAE,YAAY,GAAG,MAAM,CAwC5D;AAiDD;;;;GAIG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,YAAY,EAAE,EAAE,UAAU,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,MAAM,CAe1G;AAED,MAAM,WAAW,cAAc;IAC9B,mCAAmC;IACnC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,qFAAqF;IACrF,cAAc,EAAE,MAAM,CAAC;IACvB,uEAAuE;IACvE,WAAW,EAAE,OAAO,CAAC;CACrB;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,YAAY,CAC3B,OAAO,EAAE,YAAY,EAAE,EACvB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,MAAM,EAChB,gBAAgB,EAAE,MAAM,GACtB,cAAc,CAyDhB;AAiED;;;GAGG;AACH,wBAAsB,eAAe,CACpC,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,SAAS,CAAC,EAAE,CAAC,gBAAgB,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,OAAO,CAAC,MAAM,CAAC,EAC/E,UAAU,SAAoC,EAC9C,OAAO,UAAQ,GACb,OAAO,CAAC,MAAM,CAAC,CAqEjB;AAOD,mGAAmG;AACnG,eAAO,MAAM,yBAAyB,OAAQ,CAAC;AA2B/C;;;;;;GAMG;AACH,wBAAgB,mBAAmB,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,oBAAoB,EAAE,MAAM,GAAG,OAAO,CAI5F;AAgBD,wBAAgB,gCAAgC,CAAC,UAAU,EAAE,MAAM,GAAG,MAAM,CAE3E;AAED,wBAAgB,6BAA6B,CAAC,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,CAEjG;AA0GD,wBAAgB,oBAAoB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEzD;AAMD,MAAM,WAAW,qBAAqB;IACrC,kCAAkC;IAClC,gBAAgB,EAAE,MAAM,CAAC;IACzB,qDAAqD;IACrD,mBAAmB,EAAE,YAAY,EAAE,CAAC;IACpC,2EAA2E;IAC3E,kBAAkB,EAAE,YAAY,EAAE,CAAC;IACnC,iEAAiE;IACjE,WAAW,EAAE,OAAO,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,yDAAyD;IACzD,OAAO,EAAE,cAAc,CAAC;IACxB,sEAAsE;IACtE,KAAK,CAAC,EAAE,eAAe,CAAC;IACxB,8CAA8C;IAC9C,QAAQ,EAAE,kBAAkB,CAAC;CAC7B;AAED,wBAAgB,iBAAiB,CAChC,WAAW,EAAE,YAAY,EAAE,EAC3B,QAAQ,EAAE,kBAAkB,EAC5B,OAAO,CAAC,EAAE;IAAE,iCAAiC,CAAC,EAAE,OAAO,CAAA;CAAE,GACvD,qBAAqB,GAAG,SAAS,CA+EnC;AAqBD;;;;;;GAMG;AACH,wBAAsB,OAAO,CAC5B,WAAW,EAAE,qBAAqB,EAClC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,kBAAkB,CAAC,EAAE,MAAM,EAC3B,MAAM,CAAC,EAAE,WAAW,EACpB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,SAAS,CAAC,EAAE,CAAC,gBAAgB,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,OAAO,CAAC,MAAM,CAAC,EAC/E,gBAAgB,CAAC,EAAE;IAAE,OAAO,CAAC,EAAE,OAAO,CAAA;CAAE,GACtC,OAAO,CAAC,gBAAgB,CAAC,CAkH3B;AAMD,wBAAgB,6BAA6B,CAAC,WAAW,EAAE,qBAAqB,GAAG,gBAAgB,CAwDlG","sourcesContent":["/**\n * Context compaction for long sessions.\n *\n * Pure functions for compaction logic. The session manager handles I/O,\n * and after compaction the session is reloaded.\n */\n\nimport type { AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from \"@caupulican/pi-ai\";\nimport { completeSimple } from \"@caupulican/pi-ai\";\nimport {\n\tconvertToLlm,\n\tcreateBranchSummaryMessage,\n\tcreateCompactionSummaryMessage,\n\tcreateCustomMessage,\n} from \"../messages.ts\";\nimport { buildSessionContext, type CompactionEntry, type SessionEntry } from \"../session/session-manager.ts\";\nimport type { AgentMessage, StreamFn, ThinkingLevel } from \"../types.ts\";\nimport { type CompactionFacts, extractCompactionFacts, renderFactsBlock } from \"./extraction.ts\";\nimport {\n\tcomputeFileLists,\n\tcreateFileOps,\n\textractFileOpsFromMessage,\n\ttype FileOperations,\n\tformatFileOperations,\n\tSUMMARIZATION_SYSTEM_PROMPT,\n\tserializeConversation,\n} from \"./utils.ts\";\nimport { buildRetryPrompt, type VerificationReport, verifySummary } from \"./verification.ts\";\n\n// ============================================================================\n// File Operation Tracking\n// ============================================================================\n\n/** Details stored in CompactionEntry.details for file tracking */\nexport interface CompactionDetails {\n\treadFiles: string[];\n\tmodifiedFiles: string[];\n}\n\n/**\n * Extract file operations from messages and previous compaction entries.\n */\nfunction extractFileOperations(\n\tmessages: AgentMessage[],\n\tentries: SessionEntry[],\n\tprevCompactionIndex: number,\n): FileOperations {\n\tconst fileOps = createFileOps();\n\n\t// Collect from previous compaction's details (if pi-generated)\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = entries[prevCompactionIndex] as CompactionEntry;\n\t\tif (!prevCompaction.fromHook && prevCompaction.details) {\n\t\t\t// fromHook field kept for session file compatibility\n\t\t\tconst details = prevCompaction.details as CompactionDetails;\n\t\t\tif (Array.isArray(details.readFiles)) {\n\t\t\t\tfor (const f of details.readFiles) fileOps.read.add(f);\n\t\t\t}\n\t\t\tif (Array.isArray(details.modifiedFiles)) {\n\t\t\t\tfor (const f of details.modifiedFiles) fileOps.edited.add(f);\n\t\t\t}\n\t\t}\n\t}\n\n\t// Extract from tool calls in messages\n\tfor (const msg of messages) {\n\t\textractFileOpsFromMessage(msg, fileOps);\n\t}\n\n\treturn fileOps;\n}\n\n// ============================================================================\n// Message Extraction\n// ============================================================================\n\n/**\n * Extract AgentMessage from an entry if it produces one.\n * Returns undefined for entries that don't contribute to LLM context.\n */\nfunction getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {\n\tif (entry.type === \"message\") {\n\t\treturn entry.message;\n\t}\n\tif (entry.type === \"custom_message\") {\n\t\treturn createCustomMessage(entry.customType, entry.content, entry.display, entry.details, entry.timestamp);\n\t}\n\tif (entry.type === \"branch_summary\") {\n\t\treturn createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);\n\t}\n\tif (entry.type === \"compaction\") {\n\t\treturn createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp);\n\t}\n\treturn undefined;\n}\n\nfunction getMessageFromEntryForCompaction(entry: SessionEntry): AgentMessage | undefined {\n\tif (entry.type === \"compaction\") {\n\t\treturn undefined;\n\t}\n\treturn getMessageFromEntry(entry);\n}\n\n/** Result from compact() - SessionManager adds uuid/parentUuid when saving */\nexport interface CompactionResult<T = unknown> {\n\tsummary: string;\n\tfirstKeptEntryId: string;\n\ttokensBefore: number;\n\t/** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */\n\tdetails?: T;\n\tverification?: VerificationReport;\n}\n\n// ============================================================================\n// Types\n// ============================================================================\n\nexport interface CompactionSettings {\n\tenabled: boolean;\n\treserveTokens: number;\n\tkeepRecentTokens: number;\n\t/**\n\t * Compaction also triggers once context exceeds this fraction of the model's window — not only when\n\t * it's nearly full (`contextWindow - reserveTokens`). On large-window models, waiting until nearly\n\t * full means every turn pays a huge input cost; a fractional cap keeps per-turn input bounded\n\t * (cost guard). The effective trigger is the LOWER of the two, so small-window models keep the\n\t * reserve-based behavior while large windows compact earlier. `0`/`1`+ disables the fractional cap.\n\t */\n\ttriggerPercent?: number;\n}\n\nexport const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {\n\tenabled: true,\n\treserveTokens: 16384,\n\tkeepRecentTokens: 20000,\n\ttriggerPercent: 0.7,\n};\n\n// ============================================================================\n// Token calculation\n// ============================================================================\n\n/**\n * Calculate total context tokens from usage.\n * Uses the native totalTokens field when available, falls back to computing from components.\n */\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\n/**\n * Get usage from an assistant message if available.\n * Skips aborted and error messages as they don't have valid usage data.\n */\nfunction getAssistantUsage(msg: AgentMessage): Usage | undefined {\n\tif (msg.role === \"assistant\" && \"usage\" in msg) {\n\t\tconst assistantMsg = msg as AssistantMessage;\n\t\tif (assistantMsg.stopReason !== \"aborted\" && assistantMsg.stopReason !== \"error\" && assistantMsg.usage) {\n\t\t\treturn assistantMsg.usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\n/**\n * Find the last non-aborted assistant message usage from session entries.\n */\nexport function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefined {\n\tfor (let i = entries.length - 1; i >= 0; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"message\") {\n\t\t\tconst usage = getAssistantUsage(entry.message);\n\t\t\tif (usage) return usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\nexport interface ContextUsageEstimate {\n\ttokens: number;\n\tusageTokens: number;\n\ttrailingTokens: number;\n\tlastUsageIndex: number | null;\n}\n\nfunction getLastAssistantUsageInfo(messages: AgentMessage[]): { usage: Usage; index: number } | undefined {\n\tfor (let i = messages.length - 1; i >= 0; i--) {\n\t\tconst usage = getAssistantUsage(messages[i]);\n\t\tif (usage) return { usage, index: i };\n\t}\n\treturn undefined;\n}\n\n/**\n * Estimate context tokens from messages, using the last assistant usage when available.\n * If there are messages after the last usage, estimate their tokens with estimateTokens.\n */\nexport function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\n\tif (!usageInfo) {\n\t\tlet estimated = 0;\n\t\tfor (const message of messages) {\n\t\t\testimated += estimateTokens(message);\n\t\t}\n\t\treturn {\n\t\t\ttokens: estimated,\n\t\t\tusageTokens: 0,\n\t\t\ttrailingTokens: estimated,\n\t\t\tlastUsageIndex: null,\n\t\t};\n\t}\n\n\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\tlet trailingTokens = 0;\n\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\ttrailingTokens += estimateTokens(messages[i]);\n\t}\n\n\treturn {\n\t\ttokens: usageTokens + trailingTokens,\n\t\tusageTokens,\n\t\ttrailingTokens,\n\t\tlastUsageIndex: usageInfo.index,\n\t};\n}\n\n/**\n * Minimum projected space saving for the EARLY (fractional) compaction trigger to fire. Anti-thrashing\n * (cost guard, #30): an early compaction whose summary would barely shrink the context (mostly recent,\n * protected content) just burns a summarization call for little gain — skip it and let the context grow\n * until either the saving is worthwhile or the hard (near-full) trigger forces it. Does NOT gate the\n * hard trigger, so overflow is always avoided.\n */\nexport const MIN_COMPACTION_SAVINGS = 0.12;\n\n/**\n * Check if compaction should trigger based on context usage.\n *\n * Two triggers:\n * - HARD: context exceeds `contextWindow - reserveTokens` (near-full) or an explicit `triggerTokens`\n * override — always compact (prevents overflow).\n * - EARLY (fractional, cost guard): context exceeds `contextWindow * triggerPercent` — compact only if\n * the summary would actually save enough (`MIN_COMPACTION_SAVINGS`), so we don't thrash for tiny gains.\n */\nexport function shouldCompact(\n\tcontextTokens: number,\n\tcontextWindow: number,\n\tsettings: CompactionSettings,\n\ttriggerTokens?: number,\n): boolean {\n\tif (!settings.enabled) return false;\n\n\t// Hard trigger: near-full, or a caller-supplied lower override. Always compacts (avoid overflow).\n\tconst reserveTrigger = contextWindow - settings.reserveTokens;\n\tconst hardTrigger = triggerTokens === undefined ? reserveTrigger : Math.min(reserveTrigger, triggerTokens);\n\tif (contextTokens > hardTrigger) return true;\n\n\t// Early fractional trigger: bounds per-turn input cost on large-window models, gated by anti-thrashing.\n\tconst pct = settings.triggerPercent ?? 0;\n\tif (pct > 0 && pct < 1) {\n\t\tconst fractionalTrigger = Math.floor(contextWindow * pct);\n\t\tif (contextTokens > fractionalTrigger) {\n\t\t\t// Projected saving ≈ the non-protected fraction (everything but the recent tail we keep).\n\t\t\tconst projectedSavings = contextTokens > 0 ? 1 - settings.keepRecentTokens / contextTokens : 0;\n\t\t\treturn projectedSavings >= MIN_COMPACTION_SAVINGS;\n\t\t}\n\t}\n\treturn false;\n}\n\n// ============================================================================\n// Cut point detection\n// ============================================================================\n\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nfunction estimateTextAndImageContentChars(content: string | Array<{ type: string; text?: string }>): number {\n\tif (typeof content === \"string\") {\n\t\treturn content.length;\n\t}\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\" && block.text) {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"image\") {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\n/**\n * Estimate token count for a message using chars/4 heuristic.\n * This is a rough planning heuristic; code and structured text can be denser than 4 chars/token,\n * so callers that must stay under a provider bound need additional headroom.\n */\nexport function estimateTokens(message: AgentMessage): number {\n\tlet chars = 0;\n\n\tswitch (message.role) {\n\t\tcase \"user\": {\n\t\t\tchars = estimateTextAndImageContentChars(\n\t\t\t\t(message as { content: string | Array<{ type: string; text?: string }> }).content,\n\t\t\t);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"assistant\": {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\tfor (const block of assistant.content) {\n\t\t\t\tif (block.type === \"text\") {\n\t\t\t\t\tchars += block.text.length;\n\t\t\t\t} else if (block.type === \"thinking\") {\n\t\t\t\t\tchars += block.thinking.length;\n\t\t\t\t} else if (block.type === \"toolCall\") {\n\t\t\t\t\tchars += block.name.length + JSON.stringify(block.arguments).length;\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"custom\":\n\t\tcase \"toolResult\": {\n\t\t\tchars = estimateTextAndImageContentChars(message.content);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"bashExecution\": {\n\t\t\tchars = message.command.length + message.output.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\": {\n\t\t\tchars = message.summary.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t}\n\n\treturn 0;\n}\n\n/**\n * Find valid cut points: indices of user, assistant, custom, or bashExecution messages.\n * Never cut at tool results (they must follow their tool call).\n * When we cut at an assistant message with tool calls, its tool results follow it\n * and will be kept.\n * BashExecutionMessage is treated like a user message (user-initiated context).\n */\nfunction findValidCutPoints(entries: SessionEntry[], startIndex: number, endIndex: number): number[] {\n\tconst cutPoints: number[] = [];\n\tfor (let i = startIndex; i < endIndex; i++) {\n\t\tconst entry = entries[i];\n\t\tswitch (entry.type) {\n\t\t\tcase \"message\": {\n\t\t\t\tconst role = entry.message.role;\n\t\t\t\tswitch (role) {\n\t\t\t\t\tcase \"bashExecution\":\n\t\t\t\t\tcase \"custom\":\n\t\t\t\t\tcase \"branchSummary\":\n\t\t\t\t\tcase \"compactionSummary\":\n\t\t\t\t\tcase \"user\":\n\t\t\t\t\tcase \"assistant\":\n\t\t\t\t\t\tcutPoints.push(i);\n\t\t\t\t\t\tbreak;\n\t\t\t\t\tcase \"toolResult\":\n\t\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t\tbreak;\n\t\t\t}\n\t\t\tcase \"thinking_level_change\":\n\t\t\tcase \"model_change\":\n\t\t\tcase \"compaction\":\n\t\t\tcase \"branch_summary\":\n\t\t\tcase \"custom\":\n\t\t\tcase \"custom_message\":\n\t\t\tcase \"label\":\n\t\t\tcase \"session_info\":\n\t\t\t\tbreak;\n\t\t}\n\n\t\t// branch_summary and custom_message are user-role messages, valid cut points\n\t\tif (entry.type === \"branch_summary\" || entry.type === \"custom_message\") {\n\t\t\tcutPoints.push(i);\n\t\t}\n\t}\n\treturn cutPoints;\n}\n\n/**\n * Find the user message (or bashExecution) that starts the turn containing the given entry index.\n * Returns -1 if no turn start found before the index.\n * BashExecutionMessage is treated like a user message for turn boundaries.\n */\nexport function findTurnStartIndex(entries: SessionEntry[], entryIndex: number, startIndex: number): number {\n\tfor (let i = entryIndex; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\t// branch_summary and custom_message are user-role messages, can start a turn\n\t\tif (entry.type === \"branch_summary\" || entry.type === \"custom_message\") {\n\t\t\treturn i;\n\t\t}\n\t\tif (entry.type === \"message\") {\n\t\t\tconst role = entry.message.role;\n\t\t\tif (role === \"user\" || role === \"bashExecution\") {\n\t\t\t\treturn i;\n\t\t\t}\n\t\t}\n\t}\n\treturn -1;\n}\n\nexport interface CutPointResult {\n\t/** Index of first entry to keep */\n\tfirstKeptEntryIndex: number;\n\t/** Index of user message that starts the turn being split, or -1 if not splitting */\n\tturnStartIndex: number;\n\t/** Whether this cut splits a turn (cut point is not a user message) */\n\tisSplitTurn: boolean;\n}\n\n/**\n * Find the cut point in session entries that keeps approximately `keepRecentTokens`.\n *\n * Algorithm: Walk backwards from newest, accumulating estimated message sizes.\n * Stop when we've accumulated >= keepRecentTokens. Cut at that point.\n *\n * Can cut at user OR assistant messages (never tool results). When cutting at an\n * assistant message with tool calls, its tool results come after and will be kept.\n *\n * Returns CutPointResult with:\n * - firstKeptEntryIndex: the entry index to start keeping from\n * - turnStartIndex: if cutting mid-turn, the user message that started that turn\n * - isSplitTurn: whether we're cutting in the middle of a turn\n *\n * Only considers entries between `startIndex` and `endIndex` (exclusive).\n */\nexport function findCutPoint(\n\tentries: SessionEntry[],\n\tstartIndex: number,\n\tendIndex: number,\n\tkeepRecentTokens: number,\n): CutPointResult {\n\tconst cutPoints = findValidCutPoints(entries, startIndex, endIndex);\n\n\tif (cutPoints.length === 0) {\n\t\treturn { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };\n\t}\n\n\t// Walk backwards from newest, accumulating estimated message sizes\n\tlet accumulatedTokens = 0;\n\tlet cutIndex = cutPoints[0]; // Default: keep from first message (not header)\n\n\tfor (let i = endIndex - 1; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type !== \"message\") continue;\n\n\t\t// Estimate this message's size\n\t\tconst messageTokens = estimateTokens(entry.message);\n\t\taccumulatedTokens += messageTokens;\n\n\t\t// Check if we've exceeded the budget\n\t\tif (accumulatedTokens >= keepRecentTokens) {\n\t\t\t// Find the closest valid cut point at or after this entry\n\t\t\tfor (let c = 0; c < cutPoints.length; c++) {\n\t\t\t\tif (cutPoints[c] >= i) {\n\t\t\t\t\tcutIndex = cutPoints[c];\n\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t}\n\t\t\tbreak;\n\t\t}\n\t}\n\n\t// Scan backwards from cutIndex to include any non-message entries (bash, settings, etc.)\n\twhile (cutIndex > startIndex) {\n\t\tconst prevEntry = entries[cutIndex - 1];\n\t\t// Stop at session header or compaction boundaries\n\t\tif (prevEntry.type === \"compaction\") {\n\t\t\tbreak;\n\t\t}\n\t\tif (prevEntry.type === \"message\") {\n\t\t\t// Stop if we hit any message\n\t\t\tbreak;\n\t\t}\n\t\t// Include this non-message entry (bash, settings change, etc.)\n\t\tcutIndex--;\n\t}\n\n\t// Determine if this is a split turn\n\tconst cutEntry = entries[cutIndex];\n\tconst isUserMessage = cutEntry.type === \"message\" && cutEntry.message.role === \"user\";\n\tconst turnStartIndex = isUserMessage ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);\n\n\treturn {\n\t\tfirstKeptEntryIndex: cutIndex,\n\t\tturnStartIndex,\n\t\tisSplitTurn: !isUserMessage && turnStartIndex !== -1,\n\t};\n}\n\n// ============================================================================\n// Summarization\n// ============================================================================\n\nconst SUMMARIZATION_PROMPT = `Checkpoint the conversation above. Format from your instructions, sections in this order:\n## Active Task\n### Mandatory Rules\n## Files\n## Done\n## Constraints & Preferences\n## Key Decisions\n## Blocked / Open\n## Critical Context\n\nPre-extracted facts (verified by tooling — include ALL of them, merged with what you read):\n<facts>\n{FACTS_BLOCK}\n</facts>\n\nBudget: ~{BUDGET} tokens. Concrete beats complete.`;\n\nconst UPDATE_SUMMARIZATION_PROMPT = `Update the checkpoint in <previous-summary> with the NEW turns above. RULES:\n- PRESERVE every existing ### Mandatory Rules bullet VERBATIM; append new ones.\n- Continue the ## Done numbering. Keep the 15 most recent numbered items verbatim; compress everything older into the single first line \"1. (earlier work compressed) <one line>\". The checkpoint must not grow without bound across updates.\n- Update ## Active Task to the newest unfulfilled user input; apply the cancellation rule.\n- Keep ## Files current (add new, keep still-relevant, drop obsolete).\n- Preserve exact paths, commands, errors.\n\nSame section order. Pre-extracted facts:\n<facts>\n{FACTS_BLOCK}\n</facts>\n\nBudget: ~{BUDGET} tokens.`;\n\nfunction createSummarizationOptions(\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n): SimpleStreamOptions {\n\tconst options: SimpleStreamOptions = { maxTokens, signal, apiKey, headers };\n\tif (model.reasoning && thinkingLevel && thinkingLevel !== \"off\") {\n\t\toptions.reasoning = thinkingLevel;\n\t}\n\treturn options;\n}\n\nasync function completeSummarization(\n\tmodel: Model<any>,\n\tcontext: Context,\n\toptions: SimpleStreamOptions,\n\tstreamFn?: StreamFn,\n): Promise<AssistantMessage> {\n\tif (!streamFn) {\n\t\treturn completeSimple(model, context, options);\n\t}\n\tconst stream = await streamFn(model, context, options);\n\treturn stream.result();\n}\n\n/**\n * Generate a summary of the conversation using the LLM.\n * If previousSummary is provided, uses the update prompt to merge.\n */\nexport async function generateSummary(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tpreDigest?: (conversationText: string, signal?: AbortSignal) => Promise<string>,\n\tfactsBlock = \"files:\\nactions:\\nprohibitions:\",\n\tchunked = false,\n): Promise<string> {\n\tconst summaryBudget = getSummaryBudget(reserveTokens, model, factsBlock);\n\tconst maxTokens = summaryBudget;\n\n\tlet promptSuffix = fillPromptTemplate(\n\t\tpreviousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT,\n\t\tfactsBlock,\n\t\tsummaryBudget,\n\t);\n\tif (customInstructions) {\n\t\tpromptSuffix = `${promptSuffix}\\n\\nAdditional focus: ${customInstructions}`;\n\t}\n\n\tconst llmMessages = convertToLlm(currentMessages);\n\tlet conversationText = serializeConversation(llmMessages);\n\tif (preDigest) {\n\t\ttry {\n\t\t\tconversationText = await preDigest(conversationText, signal);\n\t\t} catch {\n\t\t\t// Keep the verbatim conversation when an optional pre-digest fails.\n\t\t}\n\t}\n\n\tconst inputBound = getSummarizerInputBound(model, maxTokens);\n\tconst initialPromptText = buildSummarizationPrompt(conversationText, previousSummary, promptSuffix);\n\tif (estimateStringTokens(initialPromptText) > inputBound) {\n\t\tif (!chunked) {\n\t\t\tthrow new Error(\"input-overflow: summarization request exceeds summarizer window\");\n\t\t}\n\t\tconversationText = await summarizeChunks(\n\t\t\tconversationText,\n\t\t\tmodel,\n\t\t\tmaxTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tinputBound,\n\t\t);\n\t}\n\n\tconst promptText = buildSummarizationPrompt(conversationText, previousSummary, promptSuffix);\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\t{\n\t\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\t\tmessages: [\n\t\t\t\t{\n\t\t\t\t\trole: \"user\",\n\t\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t},\n\t\t\t],\n\t\t},\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\tstreamFn,\n\t);\n\n\tif (response.stopReason === \"error\") {\n\t\tthrow new Error(`Summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t}\n\t// A length-stopped checkpoint silently lost its tail sections — gating it as if complete\n\t// guarantees a verification failure. Fail loudly so the compaction ladder escalates instead.\n\tif (response.stopReason === \"length\") {\n\t\tthrow new Error(\"summary-length-stop: summarizer hit its output cap before completing the checkpoint\");\n\t}\n\n\treturn truncateSummaryToBudget(extractTextContent(response), summaryBudget);\n}\n\nfunction fillPromptTemplate(template: string, factsBlock: string, budget: number): string {\n\treturn template.replaceAll(\"{FACTS_BLOCK}\", factsBlock).replaceAll(\"{BUDGET}\", String(budget));\n}\n\nconst SUMMARY_BUDGET_BASE_TOKENS = 1_500;\n/** Hard ceiling for the facts-scaled summary budget; also the worst case selection must assume. */\nexport const SUMMARY_BUDGET_MAX_TOKENS = 4_000;\n/** Prompt-side margin beyond the raw conversation input (system prompt, tags, instructions). */\nconst SUMMARIZER_PROMPT_MARGIN_TOKENS = 2_000;\n\nfunction getSummaryBudget(reserveTokens: number, model: Model<any>, factsBlock?: string): number {\n\tconst modelMaxTokens = model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY;\n\t// The verification gate demands the summary restate every extracted fact (modified files,\n\t// actions, rules). A fixed budget makes large spans structurally fail: the model length-stops\n\t// and the gated sections are the casualties. Scale the budget with the demand, bounded.\n\tconst factsTokens = factsBlock ? estimateStringTokens(factsBlock) : 0;\n\tconst demandBudget = Math.min(SUMMARY_BUDGET_MAX_TOKENS, Math.max(SUMMARY_BUDGET_BASE_TOKENS, factsTokens + 500));\n\treturn Math.max(1, Math.min(demandBudget, Math.floor(0.8 * reserveTokens), modelMaxTokens));\n}\n\nfunction getEffectiveContextWindow(model: Model<any>): number {\n\tconst registered = model.contextWindow > 0 ? model.contextWindow : Number.POSITIVE_INFINITY;\n\tconst served = (model as { servedContextWindow?: unknown }).servedContextWindow;\n\treturn typeof served === \"number\" && served > 0 ? Math.min(registered, served) : registered;\n}\n\nfunction getSummarizerInputBound(model: Model<any>, maxTokens: number): number {\n\tconst contextWindow = getEffectiveContextWindow(model);\n\treturn contextWindow === Number.POSITIVE_INFINITY\n\t\t? contextWindow\n\t\t: Math.max(1, contextWindow - maxTokens - SUMMARIZER_PROMPT_MARGIN_TOKENS);\n}\n\n/**\n * Whether a candidate summarizer can ingest a summarization input of the given size in ONE\n * request (unchunked), using the same window arithmetic as {@link getSummarizerInputBound} with\n * the worst-case (facts-scaled) summary budget. Hosts use this at SELECTION time: a model that\n * fails this must not be handed the job — chunking cannot rescue recall-gated summarization, and\n * local servers silently truncate over-window prompts instead of erroring.\n */\nexport function summarizerCanIngest(model: Model<any>, estimatedInputTokens: number): boolean {\n\tconst contextWindow = getEffectiveContextWindow(model);\n\tif (contextWindow === Number.POSITIVE_INFINITY) return true;\n\treturn estimatedInputTokens <= contextWindow - SUMMARY_BUDGET_MAX_TOKENS - SUMMARIZER_PROMPT_MARGIN_TOKENS;\n}\n\nfunction buildSummarizationPrompt(\n\tconversationText: string,\n\tpreviousSummary: string | undefined,\n\tpromptSuffix: string,\n): string {\n\tlet promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n`;\n\tif (previousSummary) {\n\t\tpromptText += `<previous-summary>\\n${previousSummary}\\n</previous-summary>\\n\\n`;\n\t}\n\treturn promptText + promptSuffix;\n}\n\nconst CHUNK_SUMMARIZATION_HEADROOM_TOKENS = 1000;\n\nexport function getChunkSummarizationTokenBudget(inputBound: number): number {\n\treturn Math.max(1, inputBound - CHUNK_SUMMARIZATION_HEADROOM_TOKENS);\n}\n\nexport function buildChunkSummarizationPrompt(chunk: string, index: number, total: number): string {\n\treturn `<conversation-chunk index=\"${index}\" total=\"${total}\">\\n${chunk}\\n</conversation-chunk>\\n\\nSummarize this chunk for a later checkpoint merge. Preserve exact file paths, commands, errors, user prohibitions, and active work. Output concise notes only.`;\n}\n\nasync function summarizeChunks(\n\tconversationText: string,\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n\tstreamFn: StreamFn | undefined,\n\tinputBound: number,\n): Promise<string> {\n\tconst maxChunkTokens = getChunkSummarizationTokenBudget(inputBound);\n\tconst maxChunkChars = Math.max(1, maxChunkTokens * 4);\n\tconst chunks = splitText(conversationText, maxChunkChars);\n\tconst retainedChunks = chunks.slice(-4);\n\tconst omittedChunks = chunks.length - retainedChunks.length;\n\tconst summaries: string[] = [];\n\n\tfor (let i = 0; i < retainedChunks.length; i++) {\n\t\tconst promptText = buildChunkSummarizationPrompt(retainedChunks[i], i + 1, retainedChunks.length);\n\t\tconst response = await completeSummarization(\n\t\t\tmodel,\n\t\t\t{\n\t\t\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\t\t\tmessages: [\n\t\t\t\t\t{\n\t\t\t\t\t\trole: \"user\",\n\t\t\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t\t},\n\t\t\t\t],\n\t\t\t},\n\t\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\t\tstreamFn,\n\t\t);\n\t\tif (response.stopReason === \"error\") {\n\t\t\tthrow new Error(`Summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t\t}\n\t\tsummaries.push(extractTextContent(response));\n\t}\n\n\tconst omittedNote =\n\t\tomittedChunks > 0\n\t\t\t? `## Critical Context\\n${omittedChunks} older oversized chunks omitted; deterministic facts supplied separately.\\n\\n`\n\t\t\t: \"\";\n\treturn `${omittedNote}${summaries.join(\"\\n\\n\")}`;\n}\n\nfunction splitText(text: string, maxChars: number): string[] {\n\tconst chunks: string[] = [];\n\tfor (let start = 0; start < text.length; start += maxChars) {\n\t\tchunks.push(text.slice(start, start + maxChars));\n\t}\n\treturn chunks.length > 0 ? chunks : [\"\"];\n}\n\nfunction extractTextContent(message: AssistantMessage): string {\n\treturn message.content\n\t\t.filter((content): content is { type: \"text\"; text: string } => content.type === \"text\")\n\t\t.map((content) => content.text)\n\t\t.join(\"\\n\");\n}\n\nfunction truncateSummaryToBudget(summary: string, budget: number): string {\n\tconst maxTokens = Math.floor(budget * 1.3);\n\tif (estimateStringTokens(summary) <= maxTokens) {\n\t\treturn summary;\n\t}\n\n\tlet current = summary;\n\t// Never drop \"Files\" or \"Done\" here: the verification gate checks exactly those sections\n\t// (files-modified/read-recall, actions-overlap), so deleting them guarantees gate failure.\n\tfor (const heading of [\"Critical Context\", \"Blocked / Open\", \"Key Decisions\", \"Constraints & Preferences\"]) {\n\t\tconst next = removeSummarySection(current, heading);\n\t\tif (next === current) {\n\t\t\tcontinue;\n\t\t}\n\t\tcurrent = next;\n\t\tif (estimateStringTokens(current) <= maxTokens) {\n\t\t\treturn current;\n\t\t}\n\t}\n\treturn current;\n}\n\nfunction removeSummarySection(summary: string, heading: string): string {\n\tconst lines = summary.split(/\\r?\\n/);\n\tconst kept: string[] = [];\n\tlet skipping = false;\n\tfor (const line of lines) {\n\t\tconst match = /^(?:##|###)\\s+(.+?)\\s*$/.exec(line);\n\t\tif (match) {\n\t\t\tskipping = match[1].trim().toLowerCase() === heading.toLowerCase();\n\t\t\tif (skipping) {\n\t\t\t\tcontinue;\n\t\t\t}\n\t\t}\n\t\tif (!skipping) {\n\t\t\tkept.push(line);\n\t\t}\n\t}\n\treturn kept.join(\"\\n\").trim();\n}\n\nexport function estimateStringTokens(text: string): number {\n\treturn Math.ceil(text.length / 4);\n}\n\n// ============================================================================\n// Compaction Preparation (for extensions)\n// ============================================================================\n\nexport interface CompactionPreparation {\n\t/** UUID of first entry to keep */\n\tfirstKeptEntryId: string;\n\t/** Messages that will be summarized and discarded */\n\tmessagesToSummarize: AgentMessage[];\n\t/** Messages that will be turned into turn prefix summary (if splitting) */\n\tturnPrefixMessages: AgentMessage[];\n\t/** Whether this is a split turn (cut point in middle of turn) */\n\tisSplitTurn: boolean;\n\ttokensBefore: number;\n\t/** Summary from previous compaction, for iterative update */\n\tpreviousSummary?: string;\n\t/** File operations extracted from messagesToSummarize */\n\tfileOps: FileOperations;\n\t/** Facts extracted from the compacted span for verification gating */\n\tfacts?: CompactionFacts;\n\t/** Compaction settions from settings.jsonl\t*/\n\tsettings: CompactionSettings;\n}\n\nexport function prepareCompaction(\n\tpathEntries: SessionEntry[],\n\tsettings: CompactionSettings,\n\toptions?: { allowTrailingCompactionAsPrevious?: boolean },\n): CompactionPreparation | undefined {\n\tconst trailingEntry = pathEntries[pathEntries.length - 1];\n\tif (trailingEntry?.type === \"compaction\" && !options?.allowTrailingCompactionAsPrevious) {\n\t\treturn undefined;\n\t}\n\n\tlet prevCompactionIndex = -1;\n\tfor (let i = pathEntries.length - 1; i >= 0; i--) {\n\t\tif (pathEntries[i].type === \"compaction\") {\n\t\t\tprevCompactionIndex = i;\n\t\t\tbreak;\n\t\t}\n\t}\n\n\tlet previousSummary: string | undefined;\n\tlet boundaryStart = 0;\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;\n\t\tpreviousSummary = prevCompaction.summary;\n\t\tconst firstKeptEntryIndex = pathEntries.findIndex((entry) => entry.id === prevCompaction.firstKeptEntryId);\n\t\tboundaryStart = firstKeptEntryIndex >= 0 ? firstKeptEntryIndex : prevCompactionIndex + 1;\n\t}\n\tconst boundaryEnd =\n\t\toptions?.allowTrailingCompactionAsPrevious && pathEntries[pathEntries.length - 1]?.type === \"compaction\"\n\t\t\t? pathEntries.length - 1\n\t\t\t: pathEntries.length;\n\n\tconst tokensBefore = estimateContextTokens(buildSessionContext(pathEntries).messages).tokens;\n\n\tconst cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, settings.keepRecentTokens);\n\n\t// Get UUID of first kept entry\n\tconst firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];\n\tif (!firstKeptEntry?.id) {\n\t\treturn undefined; // Session needs migration\n\t}\n\tconst firstKeptEntryId = firstKeptEntry.id;\n\n\tconst historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;\n\n\t// Messages to summarize (will be discarded after summary)\n\tconst messagesToSummarize: AgentMessage[] = [];\n\tfor (let i = boundaryStart; i < historyEnd; i++) {\n\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\tif (msg) messagesToSummarize.push(msg);\n\t}\n\n\t// Messages for turn prefix summary (if splitting a turn)\n\tconst turnPrefixMessages: AgentMessage[] = [];\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {\n\t\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\t\tif (msg) turnPrefixMessages.push(msg);\n\t\t}\n\t}\n\n\t// Extract file operations from messages and previous compaction\n\tconst fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);\n\n\t// Also extract file ops from turn prefix if splitting\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (const msg of turnPrefixMessages) {\n\t\t\textractFileOpsFromMessage(msg, fileOps);\n\t\t}\n\t}\n\n\tconst facts = extractCompactionFacts(pathEntries, boundaryStart, historyEnd);\n\n\treturn {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn: cutPoint.isSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tfacts,\n\t\tsettings,\n\t};\n}\n\n// ============================================================================\n// Main compaction function\n// ============================================================================\n\nconst TURN_PREFIX_SUMMARIZATION_PROMPT = `This is the PREFIX of a turn that was too large to keep. The SUFFIX (recent work) is retained.\n\nSummarize the prefix to provide context for the retained suffix:\n\n## Original Request\n[What did the user ask for in this turn?]\n\n## Early Progress\n- [Key decisions and work done in the prefix]\n\n## Context for Suffix\n- [Information needed to understand the retained recent work]\n\nBe concise. Focus on what's needed to understand the kept suffix.`;\n\n/**\n * Generate summaries for compaction using prepared data.\n * Returns CompactionResult - SessionManager adds uuid/parentUuid when saving.\n *\n * @param preparation - Pre-calculated preparation from prepareCompaction()\n * @param customInstructions - Optional custom focus for the summary\n */\nexport async function compact(\n\tpreparation: CompactionPreparation,\n\tmodel: Model<any>,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tcustomInstructions?: string,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tpreDigest?: (conversationText: string, signal?: AbortSignal) => Promise<string>,\n\texecutionOptions?: { chunked?: boolean },\n): Promise<CompactionResult> {\n\tconst {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t\tfacts: factsFromPreparation,\n\t} = preparation;\n\n\tconst facts = factsFromPreparation ?? {\n\t\tfiles: [],\n\t\tactions: [],\n\t\tprohibitions: [],\n\t\tcancelledText: \"\",\n\t\tactiveTaskSource: \"\",\n\t};\n\tconst factsBlock = renderFactsBlock(facts);\n\tlet verification: VerificationReport | undefined;\n\tlet summary = \"\";\n\n\tif (isSplitTurn && messagesToSummarize.length > 0) {\n\t\tlet historySummary = \"No prior history.\";\n\t\tlet historyInstructions = customInstructions;\n\t\tfor (let attempt = 0; attempt < 2; attempt++) {\n\t\t\thistorySummary = await generateSummary(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\thistoryInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tpreDigest,\n\t\t\t\tfactsBlock,\n\t\t\t\texecutionOptions?.chunked ?? false,\n\t\t\t);\n\n\t\t\tverification = verifySummary(historySummary, facts);\n\t\t\tif (verification.ok) {\n\t\t\t\tbreak;\n\t\t\t}\n\n\t\t\tif (attempt >= 1) {\n\t\t\t\tthrow new Error(`gate-failed: ${formatVerificationFailures(verification)}`);\n\t\t\t}\n\n\t\t\thistoryInstructions = buildRetryPrompt(verification, historySummary);\n\t\t}\n\n\t\tconst turnPrefixSummary = await generateTurnPrefixSummary(\n\t\t\tturnPrefixMessages,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t);\n\t\tsummary = `${historySummary}\\n\\n---\\n\\n**Turn Context (split turn):**\\n\\n${turnPrefixSummary}`;\n\t} else {\n\t\tlet customSummaryInstructions = customInstructions;\n\t\tfor (let attempt = 0; attempt < 2; attempt++) {\n\t\t\tsummary = await generateSummary(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\tcustomSummaryInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tpreDigest,\n\t\t\t\tfactsBlock,\n\t\t\t\texecutionOptions?.chunked ?? false,\n\t\t\t);\n\n\t\t\tverification = verifySummary(summary, facts);\n\t\t\tif (verification.ok) {\n\t\t\t\tbreak;\n\t\t\t}\n\n\t\t\tif (attempt >= 1) {\n\t\t\t\tthrow new Error(`gate-failed: ${formatVerificationFailures(verification)}`);\n\t\t\t}\n\n\t\t\tcustomSummaryInstructions = buildRetryPrompt(verification, summary);\n\t\t}\n\t}\n\n\t// Compute file lists and append to summary\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\tsummary += formatFileOperations(readFiles, modifiedFiles);\n\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t\tverification,\n\t};\n}\n\nfunction formatVerificationFailures(verification: VerificationReport): string {\n\treturn verification.failures.map((failure) => `${failure.check}: ${failure.detail}`).join(\", \");\n}\n\nexport function createDeterministicCompaction(preparation: CompactionPreparation): CompactionResult {\n\tconst { firstKeptEntryId, tokensBefore, fileOps, facts } = preparation;\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\tconst factsText = renderFactsBlock(\n\t\tfacts ?? {\n\t\t\tfiles: [],\n\t\t\tactions: [],\n\t\t\tprohibitions: [],\n\t\t\tcancelledText: \"\",\n\t\t\tactiveTaskSource: \"\",\n\t\t},\n\t);\n\tconst fileLines = facts?.files.length\n\t\t? facts.files.map((file) => `- ${file.path} — ${file.note || file.kind} (${file.kind})`)\n\t\t: [`- read: ${readFiles.length}`, `- modified: ${modifiedFiles.length}`];\n\tconst mandatoryRuleLines = facts?.prohibitions.length ? facts.prohibitions.map((rule) => `- ${rule}`) : [\"(none)\"];\n\tconst doneLines = facts?.actions.length\n\t\t? facts.actions.map((action, index) => `${index + 1}. ${action}`)\n\t\t: [\"1. CHECKPOINT deterministic fallback — repeated compaction retries exhausted\"];\n\tconst summary = [\n\t\t\"## Active Task\",\n\t\tfacts?.activeTaskSource ? `User: ${facts.activeTaskSource}` : \"Continue from the deterministic compact snapshot.\",\n\t\t\"\",\n\t\t\"### Mandatory Rules\",\n\t\t...mandatoryRuleLines,\n\t\t\"\",\n\t\t\"## Files\",\n\t\t...fileLines,\n\t\t\"\",\n\t\t\"## Done\",\n\t\t...doneLines,\n\t\t\"\",\n\t\t\"## Constraints & Preferences\",\n\t\t\"Preserve exact file paths, commands, line numbers, and error strings.\",\n\t\t\"\",\n\t\t\"## Key Decisions\",\n\t\t\"- Deterministic checkpoint used after repeated compaction retries.\",\n\t\t\"\",\n\t\t\"## Blocked / Open\",\n\t\t\"(none)\",\n\t\t\"\",\n\t\t\"## Critical Context\",\n\t\t\"- Deterministic facts-only checkpoint; no LLM summary was accepted.\",\n\t\tfactsText,\n\t].join(\"\\n\");\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t};\n}\n\n/**\n * Generate a summary for a turn prefix (when splitting a turn).\n */\nasync function generateTurnPrefixSummary(\n\tmessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n): Promise<string> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.5 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t); // Smaller budget for turn prefix\n\tconst llmMessages = convertToLlm(messages);\n\tconst conversationText = serializeConversation(llmMessages);\n\tconst promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;\n\tconst summarizationMessages = [\n\t\t{\n\t\t\trole: \"user\" as const,\n\t\t\tcontent: [{ type: \"text\" as const, text: promptText }],\n\t\t\ttimestamp: Date.now(),\n\t\t},\n\t];\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\t{ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages },\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\tstreamFn,\n\t);\n\n\tif (response.stopReason === \"error\") {\n\t\tthrow new Error(`Turn prefix summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t}\n\n\treturn response.content\n\t\t.filter((c): c is { type: \"text\"; text: string } => c.type === \"text\")\n\t\t.map((c) => c.text)\n\t\t.join(\"\\n\");\n}\n"]}
1
+ {"version":3,"file":"compaction.d.ts","sourceRoot":"","sources":["../../src/compaction/compaction.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EAA6B,KAAK,EAAuB,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAQtG,OAAO,EAA6C,KAAK,YAAY,EAAE,MAAM,+BAA+B,CAAC;AAC7G,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,aAAa,CAAC;AACzE,OAAO,EAAE,KAAK,eAAe,EAA4C,MAAM,iBAAiB,CAAC;AACjG,OAAO,EAIN,KAAK,cAAc,EAGnB,MAAM,YAAY,CAAC;AACpB,OAAO,EAAoB,KAAK,kBAAkB,EAAiB,MAAM,mBAAmB,CAAC;AAM7F,kEAAkE;AAClE,MAAM,WAAW,iBAAiB;IACjC,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,aAAa,EAAE,MAAM,EAAE,CAAC;CACxB;AAkED,8EAA8E;AAC9E,MAAM,WAAW,gBAAgB,CAAC,CAAC,GAAG,OAAO;IAC5C,OAAO,EAAE,MAAM,CAAC;IAChB,gBAAgB,EAAE,MAAM,CAAC;IACzB,YAAY,EAAE,MAAM,CAAC;IACrB,+FAA+F;IAC/F,OAAO,CAAC,EAAE,CAAC,CAAC;IACZ,YAAY,CAAC,EAAE,kBAAkB,CAAC;CAClC;AAMD,MAAM,WAAW,kBAAkB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,aAAa,EAAE,MAAM,CAAC;IACtB,gBAAgB,EAAE,MAAM,CAAC;IACzB;;;;;;OAMG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CACxB;AAED,eAAO,MAAM,2BAA2B,EAAE,kBAKzC,CAAC;AAMF;;;GAGG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AAgBD;;GAEG;AACH,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,YAAY,EAAE,GAAG,KAAK,GAAG,SAAS,CAShF;AAED,MAAM,WAAW,oBAAoB;IACpC,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAUD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,YAAY,EAAE,GAAG,oBAAoB,CA4BpF;AAED;;;;;;GAMG;AACH,eAAO,MAAM,sBAAsB,OAAO,CAAC;AAE3C;;;;;;;;GAQG;AACH,wBAAgB,aAAa,CAC5B,aAAa,EAAE,MAAM,EACrB,aAAa,EAAE,MAAM,EACrB,QAAQ,EAAE,kBAAkB,EAC5B,aAAa,CAAC,EAAE,MAAM,GACpB,OAAO,CAmBT;AAwBD;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,OAAO,EAAE,YAAY,GAAG,MAAM,CAwC5D;AAiDD;;;;GAIG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,YAAY,EAAE,EAAE,UAAU,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,MAAM,CAe1G;AAED,MAAM,WAAW,cAAc;IAC9B,mCAAmC;IACnC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,qFAAqF;IACrF,cAAc,EAAE,MAAM,CAAC;IACvB,uEAAuE;IACvE,WAAW,EAAE,OAAO,CAAC;CACrB;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,YAAY,CAC3B,OAAO,EAAE,YAAY,EAAE,EACvB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,MAAM,EAChB,gBAAgB,EAAE,MAAM,GACtB,cAAc,CAyDhB;AAuED;;;GAGG;AACH,wBAAsB,eAAe,CACpC,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,SAAS,CAAC,EAAE,CAAC,gBAAgB,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,OAAO,CAAC,MAAM,CAAC,EAC/E,UAAU,SAAoC,EAC9C,OAAO,UAAQ,GACb,OAAO,CAAC,MAAM,CAAC,CA0EjB;AAOD,qGAAqG;AACrG,eAAO,MAAM,yBAAyB,OAAQ,CAAC;AAkC/C;;;;;;GAMG;AACH,wBAAgB,mBAAmB,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,oBAAoB,EAAE,MAAM,GAAG,OAAO,CAI5F;AAgBD,wBAAgB,gCAAgC,CAAC,UAAU,EAAE,MAAM,GAAG,MAAM,CAE3E;AAED,wBAAgB,6BAA6B,CAAC,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,CAEjG;AAsID,wBAAgB,oBAAoB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEzD;AAMD,MAAM,WAAW,qBAAqB;IACrC,kCAAkC;IAClC,gBAAgB,EAAE,MAAM,CAAC;IACzB,qDAAqD;IACrD,mBAAmB,EAAE,YAAY,EAAE,CAAC;IACpC,2EAA2E;IAC3E,kBAAkB,EAAE,YAAY,EAAE,CAAC;IACnC,iEAAiE;IACjE,WAAW,EAAE,OAAO,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,yDAAyD;IACzD,OAAO,EAAE,cAAc,CAAC;IACxB,sEAAsE;IACtE,KAAK,CAAC,EAAE,eAAe,CAAC;IACxB,8CAA8C;IAC9C,QAAQ,EAAE,kBAAkB,CAAC;CAC7B;AAED,wBAAgB,iBAAiB,CAChC,WAAW,EAAE,YAAY,EAAE,EAC3B,QAAQ,EAAE,kBAAkB,EAC5B,OAAO,CAAC,EAAE;IAAE,iCAAiC,CAAC,EAAE,OAAO,CAAA;CAAE,GACvD,qBAAqB,GAAG,SAAS,CA+EnC;AAqBD;;;;;;GAMG;AACH,wBAAsB,OAAO,CAC5B,WAAW,EAAE,qBAAqB,EAClC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,kBAAkB,CAAC,EAAE,MAAM,EAC3B,MAAM,CAAC,EAAE,WAAW,EACpB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,SAAS,CAAC,EAAE,CAAC,gBAAgB,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,OAAO,CAAC,MAAM,CAAC,EAC/E,gBAAgB,CAAC,EAAE;IAAE,OAAO,CAAC,EAAE,OAAO,CAAA;CAAE,GACtC,OAAO,CAAC,gBAAgB,CAAC,CAkH3B;AAMD,wBAAgB,6BAA6B,CAAC,WAAW,EAAE,qBAAqB,GAAG,gBAAgB,CAmElG","sourcesContent":["/**\n * Context compaction for long sessions.\n *\n * Pure functions for compaction logic. The session manager handles I/O,\n * and after compaction the session is reloaded.\n */\n\nimport type { AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from \"@caupulican/pi-ai\";\nimport { completeSimple } from \"@caupulican/pi-ai\";\nimport {\n\tconvertToLlm,\n\tcreateBranchSummaryMessage,\n\tcreateCompactionSummaryMessage,\n\tcreateCustomMessage,\n} from \"../messages.ts\";\nimport { buildSessionContext, type CompactionEntry, type SessionEntry } from \"../session/session-manager.ts\";\nimport type { AgentMessage, StreamFn, ThinkingLevel } from \"../types.ts\";\nimport { type CompactionFacts, extractCompactionFacts, renderFactsBlock } from \"./extraction.ts\";\nimport {\n\tcomputeFileLists,\n\tcreateFileOps,\n\textractFileOpsFromMessage,\n\ttype FileOperations,\n\tSUMMARIZATION_SYSTEM_PROMPT,\n\tserializeConversation,\n} from \"./utils.ts\";\nimport { buildRetryPrompt, type VerificationReport, verifySummary } from \"./verification.ts\";\n\n// ============================================================================\n// File Operation Tracking\n// ============================================================================\n\n/** Details stored in CompactionEntry.details for file tracking */\nexport interface CompactionDetails {\n\treadFiles: string[];\n\tmodifiedFiles: string[];\n}\n\n/**\n * Extract file operations from messages and previous compaction entries.\n */\nfunction extractFileOperations(\n\tmessages: AgentMessage[],\n\tentries: SessionEntry[],\n\tprevCompactionIndex: number,\n): FileOperations {\n\tconst fileOps = createFileOps();\n\n\t// Collect from previous compaction's details (if pi-generated)\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = entries[prevCompactionIndex] as CompactionEntry;\n\t\tif (!prevCompaction.fromHook && prevCompaction.details) {\n\t\t\t// fromHook field kept for session file compatibility\n\t\t\tconst details = prevCompaction.details as CompactionDetails;\n\t\t\tif (Array.isArray(details.readFiles)) {\n\t\t\t\tfor (const f of details.readFiles) fileOps.read.add(f);\n\t\t\t}\n\t\t\tif (Array.isArray(details.modifiedFiles)) {\n\t\t\t\tfor (const f of details.modifiedFiles) fileOps.edited.add(f);\n\t\t\t}\n\t\t}\n\t}\n\n\t// Extract from tool calls in messages\n\tfor (const msg of messages) {\n\t\textractFileOpsFromMessage(msg, fileOps);\n\t}\n\n\treturn fileOps;\n}\n\n// ============================================================================\n// Message Extraction\n// ============================================================================\n\n/**\n * Extract AgentMessage from an entry if it produces one.\n * Returns undefined for entries that don't contribute to LLM context.\n */\nfunction getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {\n\tif (entry.type === \"message\") {\n\t\treturn entry.message;\n\t}\n\tif (entry.type === \"custom_message\") {\n\t\treturn createCustomMessage(entry.customType, entry.content, entry.display, entry.details, entry.timestamp);\n\t}\n\tif (entry.type === \"branch_summary\") {\n\t\treturn createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);\n\t}\n\tif (entry.type === \"compaction\") {\n\t\treturn createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp);\n\t}\n\treturn undefined;\n}\n\nfunction getMessageFromEntryForCompaction(entry: SessionEntry): AgentMessage | undefined {\n\tif (entry.type === \"compaction\") {\n\t\treturn undefined;\n\t}\n\treturn getMessageFromEntry(entry);\n}\n\n/** Result from compact() - SessionManager adds uuid/parentUuid when saving */\nexport interface CompactionResult<T = unknown> {\n\tsummary: string;\n\tfirstKeptEntryId: string;\n\ttokensBefore: number;\n\t/** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */\n\tdetails?: T;\n\tverification?: VerificationReport;\n}\n\n// ============================================================================\n// Types\n// ============================================================================\n\nexport interface CompactionSettings {\n\tenabled: boolean;\n\treserveTokens: number;\n\tkeepRecentTokens: number;\n\t/**\n\t * Compaction also triggers once context exceeds this fraction of the model's window — not only when\n\t * it's nearly full (`contextWindow - reserveTokens`). On large-window models, waiting until nearly\n\t * full means every turn pays a huge input cost; a fractional cap keeps per-turn input bounded\n\t * (cost guard). The effective trigger is the LOWER of the two, so small-window models keep the\n\t * reserve-based behavior while large windows compact earlier. `0`/`1`+ disables the fractional cap.\n\t */\n\ttriggerPercent?: number;\n}\n\nexport const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {\n\tenabled: true,\n\treserveTokens: 16384,\n\tkeepRecentTokens: 20000,\n\ttriggerPercent: 0.7,\n};\n\n// ============================================================================\n// Token calculation\n// ============================================================================\n\n/**\n * Calculate total context tokens from usage.\n * Uses the native totalTokens field when available, falls back to computing from components.\n */\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\n/**\n * Get usage from an assistant message if available.\n * Skips aborted and error messages as they don't have valid usage data.\n */\nfunction getAssistantUsage(msg: AgentMessage): Usage | undefined {\n\tif (msg.role === \"assistant\" && \"usage\" in msg) {\n\t\tconst assistantMsg = msg as AssistantMessage;\n\t\tif (assistantMsg.stopReason !== \"aborted\" && assistantMsg.stopReason !== \"error\" && assistantMsg.usage) {\n\t\t\treturn assistantMsg.usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\n/**\n * Find the last non-aborted assistant message usage from session entries.\n */\nexport function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefined {\n\tfor (let i = entries.length - 1; i >= 0; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"message\") {\n\t\t\tconst usage = getAssistantUsage(entry.message);\n\t\t\tif (usage) return usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\nexport interface ContextUsageEstimate {\n\ttokens: number;\n\tusageTokens: number;\n\ttrailingTokens: number;\n\tlastUsageIndex: number | null;\n}\n\nfunction getLastAssistantUsageInfo(messages: AgentMessage[]): { usage: Usage; index: number } | undefined {\n\tfor (let i = messages.length - 1; i >= 0; i--) {\n\t\tconst usage = getAssistantUsage(messages[i]);\n\t\tif (usage) return { usage, index: i };\n\t}\n\treturn undefined;\n}\n\n/**\n * Estimate context tokens from messages, using the last assistant usage when available.\n * If there are messages after the last usage, estimate their tokens with estimateTokens.\n */\nexport function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\n\tif (!usageInfo) {\n\t\tlet estimated = 0;\n\t\tfor (const message of messages) {\n\t\t\testimated += estimateTokens(message);\n\t\t}\n\t\treturn {\n\t\t\ttokens: estimated,\n\t\t\tusageTokens: 0,\n\t\t\ttrailingTokens: estimated,\n\t\t\tlastUsageIndex: null,\n\t\t};\n\t}\n\n\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\tlet trailingTokens = 0;\n\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\ttrailingTokens += estimateTokens(messages[i]);\n\t}\n\n\treturn {\n\t\ttokens: usageTokens + trailingTokens,\n\t\tusageTokens,\n\t\ttrailingTokens,\n\t\tlastUsageIndex: usageInfo.index,\n\t};\n}\n\n/**\n * Minimum projected space saving for the EARLY (fractional) compaction trigger to fire. Anti-thrashing\n * (cost guard, #30): an early compaction whose summary would barely shrink the context (mostly recent,\n * protected content) just burns a summarization call for little gain — skip it and let the context grow\n * until either the saving is worthwhile or the hard (near-full) trigger forces it. Does NOT gate the\n * hard trigger, so overflow is always avoided.\n */\nexport const MIN_COMPACTION_SAVINGS = 0.12;\n\n/**\n * Check if compaction should trigger based on context usage.\n *\n * Two triggers:\n * - HARD: context exceeds `contextWindow - reserveTokens` (near-full) or an explicit `triggerTokens`\n * override — always compact (prevents overflow).\n * - EARLY (fractional, cost guard): context exceeds `contextWindow * triggerPercent` — compact only if\n * the summary would actually save enough (`MIN_COMPACTION_SAVINGS`), so we don't thrash for tiny gains.\n */\nexport function shouldCompact(\n\tcontextTokens: number,\n\tcontextWindow: number,\n\tsettings: CompactionSettings,\n\ttriggerTokens?: number,\n): boolean {\n\tif (!settings.enabled) return false;\n\n\t// Hard trigger: near-full, or a caller-supplied lower override. Always compacts (avoid overflow).\n\tconst reserveTrigger = contextWindow - settings.reserveTokens;\n\tconst hardTrigger = triggerTokens === undefined ? reserveTrigger : Math.min(reserveTrigger, triggerTokens);\n\tif (contextTokens > hardTrigger) return true;\n\n\t// Early fractional trigger: bounds per-turn input cost on large-window models, gated by anti-thrashing.\n\tconst pct = settings.triggerPercent ?? 0;\n\tif (pct > 0 && pct < 1) {\n\t\tconst fractionalTrigger = Math.floor(contextWindow * pct);\n\t\tif (contextTokens > fractionalTrigger) {\n\t\t\t// Projected saving ≈ the non-protected fraction (everything but the recent tail we keep).\n\t\t\tconst projectedSavings = contextTokens > 0 ? 1 - settings.keepRecentTokens / contextTokens : 0;\n\t\t\treturn projectedSavings >= MIN_COMPACTION_SAVINGS;\n\t\t}\n\t}\n\treturn false;\n}\n\n// ============================================================================\n// Cut point detection\n// ============================================================================\n\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nfunction estimateTextAndImageContentChars(content: string | Array<{ type: string; text?: string }>): number {\n\tif (typeof content === \"string\") {\n\t\treturn content.length;\n\t}\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\" && block.text) {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"image\") {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\n/**\n * Estimate token count for a message using chars/4 heuristic.\n * This is a rough planning heuristic; code and structured text can be denser than 4 chars/token,\n * so callers that must stay under a provider bound need additional headroom.\n */\nexport function estimateTokens(message: AgentMessage): number {\n\tlet chars = 0;\n\n\tswitch (message.role) {\n\t\tcase \"user\": {\n\t\t\tchars = estimateTextAndImageContentChars(\n\t\t\t\t(message as { content: string | Array<{ type: string; text?: string }> }).content,\n\t\t\t);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"assistant\": {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\tfor (const block of assistant.content) {\n\t\t\t\tif (block.type === \"text\") {\n\t\t\t\t\tchars += block.text.length;\n\t\t\t\t} else if (block.type === \"thinking\") {\n\t\t\t\t\tchars += block.thinking.length;\n\t\t\t\t} else if (block.type === \"toolCall\") {\n\t\t\t\t\tchars += block.name.length + JSON.stringify(block.arguments).length;\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"custom\":\n\t\tcase \"toolResult\": {\n\t\t\tchars = estimateTextAndImageContentChars(message.content);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"bashExecution\": {\n\t\t\tchars = message.command.length + message.output.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\": {\n\t\t\tchars = message.summary.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t}\n\n\treturn 0;\n}\n\n/**\n * Find valid cut points: indices of user, assistant, custom, or bashExecution messages.\n * Never cut at tool results (they must follow their tool call).\n * When we cut at an assistant message with tool calls, its tool results follow it\n * and will be kept.\n * BashExecutionMessage is treated like a user message (user-initiated context).\n */\nfunction findValidCutPoints(entries: SessionEntry[], startIndex: number, endIndex: number): number[] {\n\tconst cutPoints: number[] = [];\n\tfor (let i = startIndex; i < endIndex; i++) {\n\t\tconst entry = entries[i];\n\t\tswitch (entry.type) {\n\t\t\tcase \"message\": {\n\t\t\t\tconst role = entry.message.role;\n\t\t\t\tswitch (role) {\n\t\t\t\t\tcase \"bashExecution\":\n\t\t\t\t\tcase \"custom\":\n\t\t\t\t\tcase \"branchSummary\":\n\t\t\t\t\tcase \"compactionSummary\":\n\t\t\t\t\tcase \"user\":\n\t\t\t\t\tcase \"assistant\":\n\t\t\t\t\t\tcutPoints.push(i);\n\t\t\t\t\t\tbreak;\n\t\t\t\t\tcase \"toolResult\":\n\t\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t\tbreak;\n\t\t\t}\n\t\t\tcase \"thinking_level_change\":\n\t\t\tcase \"model_change\":\n\t\t\tcase \"compaction\":\n\t\t\tcase \"branch_summary\":\n\t\t\tcase \"custom\":\n\t\t\tcase \"custom_message\":\n\t\t\tcase \"label\":\n\t\t\tcase \"session_info\":\n\t\t\t\tbreak;\n\t\t}\n\n\t\t// branch_summary and custom_message are user-role messages, valid cut points\n\t\tif (entry.type === \"branch_summary\" || entry.type === \"custom_message\") {\n\t\t\tcutPoints.push(i);\n\t\t}\n\t}\n\treturn cutPoints;\n}\n\n/**\n * Find the user message (or bashExecution) that starts the turn containing the given entry index.\n * Returns -1 if no turn start found before the index.\n * BashExecutionMessage is treated like a user message for turn boundaries.\n */\nexport function findTurnStartIndex(entries: SessionEntry[], entryIndex: number, startIndex: number): number {\n\tfor (let i = entryIndex; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\t// branch_summary and custom_message are user-role messages, can start a turn\n\t\tif (entry.type === \"branch_summary\" || entry.type === \"custom_message\") {\n\t\t\treturn i;\n\t\t}\n\t\tif (entry.type === \"message\") {\n\t\t\tconst role = entry.message.role;\n\t\t\tif (role === \"user\" || role === \"bashExecution\") {\n\t\t\t\treturn i;\n\t\t\t}\n\t\t}\n\t}\n\treturn -1;\n}\n\nexport interface CutPointResult {\n\t/** Index of first entry to keep */\n\tfirstKeptEntryIndex: number;\n\t/** Index of user message that starts the turn being split, or -1 if not splitting */\n\tturnStartIndex: number;\n\t/** Whether this cut splits a turn (cut point is not a user message) */\n\tisSplitTurn: boolean;\n}\n\n/**\n * Find the cut point in session entries that keeps approximately `keepRecentTokens`.\n *\n * Algorithm: Walk backwards from newest, accumulating estimated message sizes.\n * Stop when we've accumulated >= keepRecentTokens. Cut at that point.\n *\n * Can cut at user OR assistant messages (never tool results). When cutting at an\n * assistant message with tool calls, its tool results come after and will be kept.\n *\n * Returns CutPointResult with:\n * - firstKeptEntryIndex: the entry index to start keeping from\n * - turnStartIndex: if cutting mid-turn, the user message that started that turn\n * - isSplitTurn: whether we're cutting in the middle of a turn\n *\n * Only considers entries between `startIndex` and `endIndex` (exclusive).\n */\nexport function findCutPoint(\n\tentries: SessionEntry[],\n\tstartIndex: number,\n\tendIndex: number,\n\tkeepRecentTokens: number,\n): CutPointResult {\n\tconst cutPoints = findValidCutPoints(entries, startIndex, endIndex);\n\n\tif (cutPoints.length === 0) {\n\t\treturn { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };\n\t}\n\n\t// Walk backwards from newest, accumulating estimated message sizes\n\tlet accumulatedTokens = 0;\n\tlet cutIndex = cutPoints[0]; // Default: keep from first message (not header)\n\n\tfor (let i = endIndex - 1; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type !== \"message\") continue;\n\n\t\t// Estimate this message's size\n\t\tconst messageTokens = estimateTokens(entry.message);\n\t\taccumulatedTokens += messageTokens;\n\n\t\t// Check if we've exceeded the budget\n\t\tif (accumulatedTokens >= keepRecentTokens) {\n\t\t\t// Find the closest valid cut point at or after this entry\n\t\t\tfor (let c = 0; c < cutPoints.length; c++) {\n\t\t\t\tif (cutPoints[c] >= i) {\n\t\t\t\t\tcutIndex = cutPoints[c];\n\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t}\n\t\t\tbreak;\n\t\t}\n\t}\n\n\t// Scan backwards from cutIndex to include any non-message entries (bash, settings, etc.)\n\twhile (cutIndex > startIndex) {\n\t\tconst prevEntry = entries[cutIndex - 1];\n\t\t// Stop at session header or compaction boundaries\n\t\tif (prevEntry.type === \"compaction\") {\n\t\t\tbreak;\n\t\t}\n\t\tif (prevEntry.type === \"message\") {\n\t\t\t// Stop if we hit any message\n\t\t\tbreak;\n\t\t}\n\t\t// Include this non-message entry (bash, settings change, etc.)\n\t\tcutIndex--;\n\t}\n\n\t// Determine if this is a split turn\n\tconst cutEntry = entries[cutIndex];\n\tconst isUserMessage = cutEntry.type === \"message\" && cutEntry.message.role === \"user\";\n\tconst turnStartIndex = isUserMessage ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);\n\n\treturn {\n\t\tfirstKeptEntryIndex: cutIndex,\n\t\tturnStartIndex,\n\t\tisSplitTurn: !isUserMessage && turnStartIndex !== -1,\n\t};\n}\n\n// ============================================================================\n// Summarization\n// ============================================================================\n\nconst SUMMARIZATION_PROMPT = `Checkpoint the conversation above. Format from your instructions, sections in this order:\n## Active Task\n### Mandatory Rules\n## Working Set\n## Files\n## Open Problems\n## Done\n## Key Decisions\n## Constraints & Preferences\n## Critical Context\n\nDo NOT carry resolved/transient errors, superseded approaches, or file contents. Record paths and intent, never bodies.\n\nPre-extracted facts (verified by tooling — include ALL of them, merged with what you read):\n<facts>\n{FACTS_BLOCK}\n</facts>\n\nBudget: ~{BUDGET} tokens. Concrete beats complete.`;\n\nconst UPDATE_SUMMARIZATION_PROMPT = `Update the checkpoint in <previous-summary> with the NEW turns above. RULES:\n- PRESERVE every existing ### Mandatory Rules bullet VERBATIM; append new ones.\n- Continue the ## Done numbering. Keep the 15 most recent numbered items verbatim; compress everything older into the single first line \"1. (earlier work compressed) <one line>\". The checkpoint must not grow without bound across updates.\n- Update ## Active Task to the newest unfulfilled user input; apply the cancellation rule.\n- Keep ## Files current (add new, keep still-relevant, drop obsolete).\n- Drop previous ## Open Problems resolved by the new turns.\n- Drop ## Working Set files untouched since the previous checkpoint unless the active task references them.\n- Preserve exact paths, commands, errors.\n- Do NOT carry resolved/transient errors, superseded approaches, or file contents. Record paths and intent, never bodies.\n\nSame section order. Pre-extracted facts:\n<facts>\n{FACTS_BLOCK}\n</facts>\n\nBudget: ~{BUDGET} tokens.`;\n\nfunction createSummarizationOptions(\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n): SimpleStreamOptions {\n\tconst options: SimpleStreamOptions = { maxTokens, signal, apiKey, headers };\n\tif (model.reasoning && thinkingLevel && thinkingLevel !== \"off\") {\n\t\toptions.reasoning = thinkingLevel;\n\t}\n\treturn options;\n}\n\nasync function completeSummarization(\n\tmodel: Model<any>,\n\tcontext: Context,\n\toptions: SimpleStreamOptions,\n\tstreamFn?: StreamFn,\n): Promise<AssistantMessage> {\n\tif (!streamFn) {\n\t\treturn completeSimple(model, context, options);\n\t}\n\tconst stream = await streamFn(model, context, options);\n\treturn stream.result();\n}\n\n/**\n * Generate a summary of the conversation using the LLM.\n * If previousSummary is provided, uses the update prompt to merge.\n */\nexport async function generateSummary(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tpreDigest?: (conversationText: string, signal?: AbortSignal) => Promise<string>,\n\tfactsBlock = \"files:\\nactions:\\nprohibitions:\",\n\tchunked = false,\n): Promise<string> {\n\tconst summaryBudget = getSummaryBudget(reserveTokens, model, factsBlock);\n\tconst maxTokens = summaryBudget;\n\n\tlet promptSuffix = fillPromptTemplate(\n\t\tpreviousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT,\n\t\tfactsBlock,\n\t\tsummaryBudget,\n\t);\n\tif (customInstructions) {\n\t\tpromptSuffix = `${promptSuffix}\\n\\nAdditional focus: ${customInstructions}`;\n\t}\n\n\tconst llmMessages = convertToLlm(currentMessages);\n\tlet conversationText = serializeConversation(llmMessages);\n\tif (preDigest) {\n\t\ttry {\n\t\t\tconversationText = await preDigest(conversationText, signal);\n\t\t} catch {\n\t\t\t// Keep the verbatim conversation when an optional pre-digest fails.\n\t\t}\n\t}\n\n\tconst inputBound = getSummarizerInputBound(model, maxTokens);\n\tconst initialPromptText = buildSummarizationPrompt(conversationText, previousSummary, promptSuffix);\n\tif (estimateStringTokens(initialPromptText) > inputBound) {\n\t\tif (!chunked) {\n\t\t\tthrow new Error(\"input-overflow: summarization request exceeds summarizer window\");\n\t\t}\n\t\tconversationText = await summarizeChunks(\n\t\t\tconversationText,\n\t\t\tmodel,\n\t\t\tmaxTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tinputBound,\n\t\t\tpreviousSummary,\n\t\t\tpromptSuffix,\n\t\t);\n\t}\n\n\tconst promptText = buildSummarizationPrompt(conversationText, previousSummary, promptSuffix);\n\tif (estimateStringTokens(promptText) > inputBound) {\n\t\tthrow new Error(\"input-overflow: chunked summarization merge still exceeds summarizer window\");\n\t}\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\t{\n\t\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\t\tmessages: [\n\t\t\t\t{\n\t\t\t\t\trole: \"user\",\n\t\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t},\n\t\t\t],\n\t\t},\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\tstreamFn,\n\t);\n\n\tif (response.stopReason === \"error\") {\n\t\tthrow new Error(`Summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t}\n\t// A length-stopped checkpoint silently lost its tail sections — gating it as if complete\n\t// guarantees a verification failure. Fail loudly so the compaction ladder escalates instead.\n\tif (response.stopReason === \"length\") {\n\t\tthrow new Error(\"summary-length-stop: summarizer hit its output cap before completing the checkpoint\");\n\t}\n\n\treturn truncateSummaryToBudget(extractTextContent(response), summaryBudget);\n}\n\nfunction fillPromptTemplate(template: string, factsBlock: string, budget: number): string {\n\treturn template.replaceAll(\"{FACTS_BLOCK}\", factsBlock).replaceAll(\"{BUDGET}\", String(budget));\n}\n\nconst SUMMARY_BUDGET_BASE_TOKENS = 1_500;\n/** Worst-case selection assumption for summary output when exact bounded facts are not available. */\nexport const SUMMARY_BUDGET_MAX_TOKENS = 4_000;\n/** Prompt-side margin beyond the raw conversation input (system prompt, tags, instructions). */\nconst SUMMARIZER_PROMPT_MARGIN_TOKENS = 2_000;\n\nfunction getSummaryBudget(reserveTokens: number, model: Model<any>, factsBlock?: string): number {\n\tconst modelMaxTokens = model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY;\n\t// Verification demand is bounded at extraction time, so the summary budget can be derived from\n\t// the actual gate demand instead of a blind hard cap. If the demanded facts cannot fit inside the\n\t// caller's reserve budget, deterministic compaction is the only honest path.\n\tconst factsTokens = factsBlock ? estimateStringTokens(factsBlock) : 0;\n\tconst gateDemandBudget = factsTokens + 500;\n\tconst demandBudget = Math.max(SUMMARY_BUDGET_BASE_TOKENS, gateDemandBudget);\n\tconst reserveBudget = Math.floor(0.8 * reserveTokens);\n\tif (factsTokens > reserveBudget || gateDemandBudget > modelMaxTokens) {\n\t\tthrow new Error(\n\t\t\t`summary-demand-exceeds-reserve: required ${factsTokens} fact tokens, reserve budget ${reserveBudget}, model max ${modelMaxTokens}`,\n\t\t);\n\t}\n\treturn Math.max(1, Math.min(demandBudget, modelMaxTokens));\n}\n\nfunction getEffectiveContextWindow(model: Model<any>): number {\n\tconst registered = model.contextWindow > 0 ? model.contextWindow : Number.POSITIVE_INFINITY;\n\tconst served = (model as { servedContextWindow?: unknown }).servedContextWindow;\n\treturn typeof served === \"number\" && served > 0 ? Math.min(registered, served) : registered;\n}\n\nfunction getSummarizerInputBound(model: Model<any>, maxTokens: number): number {\n\tconst contextWindow = getEffectiveContextWindow(model);\n\treturn contextWindow === Number.POSITIVE_INFINITY\n\t\t? contextWindow\n\t\t: Math.max(1, contextWindow - maxTokens - SUMMARIZER_PROMPT_MARGIN_TOKENS);\n}\n\n/**\n * Whether a candidate summarizer can ingest a summarization input of the given size in ONE\n * request (unchunked), using the same window arithmetic as {@link getSummarizerInputBound} with\n * the worst-case (facts-scaled) summary budget. Hosts use this at SELECTION time: a model that\n * fails this must not be handed the job — chunking cannot rescue recall-gated summarization, and\n * local servers silently truncate over-window prompts instead of erroring.\n */\nexport function summarizerCanIngest(model: Model<any>, estimatedInputTokens: number): boolean {\n\tconst contextWindow = getEffectiveContextWindow(model);\n\tif (contextWindow === Number.POSITIVE_INFINITY) return true;\n\treturn estimatedInputTokens <= contextWindow - SUMMARY_BUDGET_MAX_TOKENS - SUMMARIZER_PROMPT_MARGIN_TOKENS;\n}\n\nfunction buildSummarizationPrompt(\n\tconversationText: string,\n\tpreviousSummary: string | undefined,\n\tpromptSuffix: string,\n): string {\n\tlet promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n`;\n\tif (previousSummary) {\n\t\tpromptText += `<previous-summary>\\n${previousSummary}\\n</previous-summary>\\n\\n`;\n\t}\n\treturn promptText + promptSuffix;\n}\n\nconst CHUNK_SUMMARIZATION_HEADROOM_TOKENS = 1000;\n\nexport function getChunkSummarizationTokenBudget(inputBound: number): number {\n\treturn Math.max(1, inputBound - CHUNK_SUMMARIZATION_HEADROOM_TOKENS);\n}\n\nexport function buildChunkSummarizationPrompt(chunk: string, index: number, total: number): string {\n\treturn `<conversation-chunk index=\"${index}\" total=\"${total}\">\\n${chunk}\\n</conversation-chunk>\\n\\nSummarize this chunk for a later checkpoint merge. Preserve exact file paths, commands, errors, user prohibitions, and active work. Output concise notes only.`;\n}\n\nasync function summarizeChunks(\n\tconversationText: string,\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n\tstreamFn: StreamFn | undefined,\n\tinputBound: number,\n\tpreviousSummary: string | undefined,\n\tpromptSuffix: string,\n): Promise<string> {\n\tlet reducedText = conversationText;\n\tfor (let pass = 0; pass < 3; pass++) {\n\t\tconst summary = await summarizeChunkPass(\n\t\t\treducedText,\n\t\t\tmodel,\n\t\t\tmaxTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tinputBound,\n\t\t);\n\t\tif (estimateStringTokens(buildSummarizationPrompt(summary, previousSummary, promptSuffix)) <= inputBound) {\n\t\t\treturn summary;\n\t\t}\n\t\treducedText = summary;\n\t}\n\tthrow new Error(\"input-overflow: chunked summarization merge still exceeds summarizer window\");\n}\n\nasync function summarizeChunkPass(\n\tconversationText: string,\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n\tstreamFn: StreamFn | undefined,\n\tinputBound: number,\n): Promise<string> {\n\tconst maxChunkTokens = getChunkSummarizationTokenBudget(inputBound);\n\tconst maxChunkChars = Math.max(1, maxChunkTokens * 4);\n\tconst chunks = splitText(conversationText, maxChunkChars);\n\tconst summaries: string[] = [];\n\n\tfor (let i = 0; i < chunks.length; i++) {\n\t\tconst promptText = buildChunkSummarizationPrompt(chunks[i], i + 1, chunks.length);\n\t\tconst response = await completeSummarization(\n\t\t\tmodel,\n\t\t\t{\n\t\t\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\t\t\tmessages: [\n\t\t\t\t\t{\n\t\t\t\t\t\trole: \"user\",\n\t\t\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t\t},\n\t\t\t\t],\n\t\t\t},\n\t\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\t\tstreamFn,\n\t\t);\n\t\tif (response.stopReason === \"error\") {\n\t\t\tthrow new Error(`Summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t\t}\n\t\tsummaries.push(extractTextContent(response));\n\t}\n\n\treturn summaries.join(\"\\n\\n\");\n}\n\nfunction splitText(text: string, maxChars: number): string[] {\n\tconst chunks: string[] = [];\n\tfor (let start = 0; start < text.length; start += maxChars) {\n\t\tchunks.push(text.slice(start, start + maxChars));\n\t}\n\treturn chunks.length > 0 ? chunks : [\"\"];\n}\n\nfunction extractTextContent(message: AssistantMessage): string {\n\treturn message.content\n\t\t.filter((content): content is { type: \"text\"; text: string } => content.type === \"text\")\n\t\t.map((content) => content.text)\n\t\t.join(\"\\n\");\n}\n\nfunction truncateSummaryToBudget(summary: string, budget: number): string {\n\tconst maxTokens = Math.floor(budget * 1.3);\n\tif (estimateStringTokens(summary) <= maxTokens) {\n\t\treturn summary;\n\t}\n\n\tlet current = summary;\n\t// Never drop \"Files\" or \"Done\" here: the verification gate checks exactly those sections\n\t// (files-modified/read-recall, actions-overlap), so deleting them guarantees gate failure.\n\tfor (const heading of [\"Critical Context\", \"Blocked / Open\", \"Key Decisions\", \"Constraints & Preferences\"]) {\n\t\tconst next = removeSummarySection(current, heading);\n\t\tif (next === current) {\n\t\t\tcontinue;\n\t\t}\n\t\tcurrent = next;\n\t\tif (estimateStringTokens(current) <= maxTokens) {\n\t\t\treturn current;\n\t\t}\n\t}\n\treturn current;\n}\n\nfunction removeSummarySection(summary: string, heading: string): string {\n\tconst lines = summary.split(/\\r?\\n/);\n\tconst kept: string[] = [];\n\tlet skipping = false;\n\tfor (const line of lines) {\n\t\tconst match = /^(?:##|###)\\s+(.+?)\\s*$/.exec(line);\n\t\tif (match) {\n\t\t\tskipping = match[1].trim().toLowerCase() === heading.toLowerCase();\n\t\t\tif (skipping) {\n\t\t\t\tcontinue;\n\t\t\t}\n\t\t}\n\t\tif (!skipping) {\n\t\t\tkept.push(line);\n\t\t}\n\t}\n\treturn kept.join(\"\\n\").trim();\n}\n\nexport function estimateStringTokens(text: string): number {\n\treturn Math.ceil(text.length / 4);\n}\n\n// ============================================================================\n// Compaction Preparation (for extensions)\n// ============================================================================\n\nexport interface CompactionPreparation {\n\t/** UUID of first entry to keep */\n\tfirstKeptEntryId: string;\n\t/** Messages that will be summarized and discarded */\n\tmessagesToSummarize: AgentMessage[];\n\t/** Messages that will be turned into turn prefix summary (if splitting) */\n\tturnPrefixMessages: AgentMessage[];\n\t/** Whether this is a split turn (cut point in middle of turn) */\n\tisSplitTurn: boolean;\n\ttokensBefore: number;\n\t/** Summary from previous compaction, for iterative update */\n\tpreviousSummary?: string;\n\t/** File operations extracted from messagesToSummarize */\n\tfileOps: FileOperations;\n\t/** Facts extracted from the compacted span for verification gating */\n\tfacts?: CompactionFacts;\n\t/** Compaction settions from settings.jsonl\t*/\n\tsettings: CompactionSettings;\n}\n\nexport function prepareCompaction(\n\tpathEntries: SessionEntry[],\n\tsettings: CompactionSettings,\n\toptions?: { allowTrailingCompactionAsPrevious?: boolean },\n): CompactionPreparation | undefined {\n\tconst trailingEntry = pathEntries[pathEntries.length - 1];\n\tif (trailingEntry?.type === \"compaction\" && !options?.allowTrailingCompactionAsPrevious) {\n\t\treturn undefined;\n\t}\n\n\tlet prevCompactionIndex = -1;\n\tfor (let i = pathEntries.length - 1; i >= 0; i--) {\n\t\tif (pathEntries[i].type === \"compaction\") {\n\t\t\tprevCompactionIndex = i;\n\t\t\tbreak;\n\t\t}\n\t}\n\n\tlet previousSummary: string | undefined;\n\tlet boundaryStart = 0;\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;\n\t\tpreviousSummary = prevCompaction.summary;\n\t\tconst firstKeptEntryIndex = pathEntries.findIndex((entry) => entry.id === prevCompaction.firstKeptEntryId);\n\t\tboundaryStart = firstKeptEntryIndex >= 0 ? firstKeptEntryIndex : prevCompactionIndex + 1;\n\t}\n\tconst boundaryEnd =\n\t\toptions?.allowTrailingCompactionAsPrevious && pathEntries[pathEntries.length - 1]?.type === \"compaction\"\n\t\t\t? pathEntries.length - 1\n\t\t\t: pathEntries.length;\n\n\tconst tokensBefore = estimateContextTokens(buildSessionContext(pathEntries).messages).tokens;\n\n\tconst cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, settings.keepRecentTokens);\n\n\t// Get UUID of first kept entry\n\tconst firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];\n\tif (!firstKeptEntry?.id) {\n\t\treturn undefined; // Session needs migration\n\t}\n\tconst firstKeptEntryId = firstKeptEntry.id;\n\n\tconst historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;\n\n\t// Messages to summarize (will be discarded after summary)\n\tconst messagesToSummarize: AgentMessage[] = [];\n\tfor (let i = boundaryStart; i < historyEnd; i++) {\n\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\tif (msg) messagesToSummarize.push(msg);\n\t}\n\n\t// Messages for turn prefix summary (if splitting a turn)\n\tconst turnPrefixMessages: AgentMessage[] = [];\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {\n\t\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\t\tif (msg) turnPrefixMessages.push(msg);\n\t\t}\n\t}\n\n\t// Extract file operations from messages and previous compaction\n\tconst fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);\n\n\t// Also extract file ops from turn prefix if splitting\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (const msg of turnPrefixMessages) {\n\t\t\textractFileOpsFromMessage(msg, fileOps);\n\t\t}\n\t}\n\n\tconst facts = extractCompactionFacts(pathEntries, boundaryStart, historyEnd);\n\n\treturn {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn: cutPoint.isSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tfacts,\n\t\tsettings,\n\t};\n}\n\n// ============================================================================\n// Main compaction function\n// ============================================================================\n\nconst TURN_PREFIX_SUMMARIZATION_PROMPT = `This is the PREFIX of a turn that was too large to keep. The SUFFIX (recent work) is retained.\n\nSummarize the prefix to provide context for the retained suffix:\n\n## Original Request\n[What did the user ask for in this turn?]\n\n## Early Progress\n- [Key decisions and work done in the prefix]\n\n## Context for Suffix\n- [Information needed to understand the retained recent work]\n\nBe concise. Focus on what's needed to understand the kept suffix.`;\n\n/**\n * Generate summaries for compaction using prepared data.\n * Returns CompactionResult - SessionManager adds uuid/parentUuid when saving.\n *\n * @param preparation - Pre-calculated preparation from prepareCompaction()\n * @param customInstructions - Optional custom focus for the summary\n */\nexport async function compact(\n\tpreparation: CompactionPreparation,\n\tmodel: Model<any>,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tcustomInstructions?: string,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tpreDigest?: (conversationText: string, signal?: AbortSignal) => Promise<string>,\n\texecutionOptions?: { chunked?: boolean },\n): Promise<CompactionResult> {\n\tconst {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t\tfacts: factsFromPreparation,\n\t} = preparation;\n\n\tconst facts = factsFromPreparation ?? {\n\t\tfiles: [],\n\t\tworkingSet: [],\n\t\tactions: [],\n\t\terrorFacts: [],\n\t\tprohibitions: [],\n\t\tcancelledText: \"\",\n\t\tactiveTaskSource: \"\",\n\t};\n\tconst factsBlock = renderFactsBlock(facts);\n\tlet verification: VerificationReport | undefined;\n\tlet summary = \"\";\n\n\tif (isSplitTurn && messagesToSummarize.length > 0) {\n\t\tlet historySummary = \"No prior history.\";\n\t\tlet historyInstructions = customInstructions;\n\t\tfor (let attempt = 0; attempt < 2; attempt++) {\n\t\t\thistorySummary = await generateSummary(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\thistoryInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tpreDigest,\n\t\t\t\tfactsBlock,\n\t\t\t\texecutionOptions?.chunked ?? false,\n\t\t\t);\n\n\t\t\tverification = verifySummary(historySummary, facts);\n\t\t\tif (verification.ok) {\n\t\t\t\tbreak;\n\t\t\t}\n\n\t\t\tif (attempt >= 1) {\n\t\t\t\tthrow new Error(`gate-failed: ${formatVerificationFailures(verification)}`);\n\t\t\t}\n\n\t\t\thistoryInstructions = buildRetryPrompt(verification, historySummary);\n\t\t}\n\n\t\tconst turnPrefixSummary = await generateTurnPrefixSummary(\n\t\t\tturnPrefixMessages,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t);\n\t\tsummary = `${historySummary}\\n\\n---\\n\\n**Turn Context (split turn):**\\n\\n${turnPrefixSummary}`;\n\t} else {\n\t\tlet customSummaryInstructions = customInstructions;\n\t\tfor (let attempt = 0; attempt < 2; attempt++) {\n\t\t\tsummary = await generateSummary(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\tcustomSummaryInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tpreDigest,\n\t\t\t\tfactsBlock,\n\t\t\t\texecutionOptions?.chunked ?? false,\n\t\t\t);\n\n\t\t\tverification = verifySummary(summary, facts);\n\t\t\tif (verification.ok) {\n\t\t\t\tbreak;\n\t\t\t}\n\n\t\t\tif (attempt >= 1) {\n\t\t\t\tthrow new Error(`gate-failed: ${formatVerificationFailures(verification)}`);\n\t\t\t}\n\n\t\t\tcustomSummaryInstructions = buildRetryPrompt(verification, summary);\n\t\t}\n\t}\n\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t\tverification,\n\t};\n}\n\nfunction formatVerificationFailures(verification: VerificationReport): string {\n\treturn verification.failures.map((failure) => `${failure.check}: ${failure.detail}`).join(\", \");\n}\n\nexport function createDeterministicCompaction(preparation: CompactionPreparation): CompactionResult {\n\tconst { firstKeptEntryId, tokensBefore, fileOps, facts } = preparation;\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\tconst factsText = renderFactsBlock(\n\t\tfacts ?? {\n\t\t\tfiles: [],\n\t\t\tworkingSet: [],\n\t\t\tactions: [],\n\t\t\terrorFacts: [],\n\t\t\tprohibitions: [],\n\t\t\tcancelledText: \"\",\n\t\t\tactiveTaskSource: \"\",\n\t\t},\n\t);\n\tconst workingSetLines = facts?.workingSet.length\n\t\t? facts.workingSet.map((file) => `- ${file.path} — ${file.note || file.kind}`)\n\t\t: [\"(none)\"];\n\tconst fileLines = facts?.files.length\n\t\t? facts.files.map((file) => `- ${file.path}`)\n\t\t: [`- read: ${readFiles.length}`, `- modified: ${modifiedFiles.length}`];\n\tconst openProblemLines = facts?.errorFacts.length\n\t\t? facts.errorFacts.map((error) => `- ${error.operation}: ${error.error}`)\n\t\t: [\"(none)\"];\n\tconst mandatoryRuleLines = facts?.prohibitions.length ? facts.prohibitions.map((rule) => `- ${rule}`) : [\"(none)\"];\n\tconst doneLines = facts?.actions.length\n\t\t? facts.actions.map((action, index) => `${index + 1}. ${action}`)\n\t\t: [\"1. CHECKPOINT deterministic fallback — repeated compaction retries exhausted\"];\n\tconst summary = [\n\t\t\"## Active Task\",\n\t\tfacts?.activeTaskSource ? `User: ${facts.activeTaskSource}` : \"Continue from the deterministic compact snapshot.\",\n\t\t\"\",\n\t\t\"### Mandatory Rules\",\n\t\t...mandatoryRuleLines,\n\t\t\"\",\n\t\t\"## Working Set\",\n\t\t...workingSetLines,\n\t\t\"\",\n\t\t\"## Files\",\n\t\t...fileLines,\n\t\t\"\",\n\t\t\"## Open Problems\",\n\t\t...openProblemLines,\n\t\t\"\",\n\t\t\"## Done\",\n\t\t...doneLines,\n\t\t\"\",\n\t\t\"## Key Decisions\",\n\t\t\"- Deterministic checkpoint used after repeated compaction retries.\",\n\t\t\"\",\n\t\t\"## Constraints & Preferences\",\n\t\t\"Preserve exact file paths, commands, line numbers, and error strings.\",\n\t\t\"\",\n\t\t\"## Critical Context\",\n\t\t\"- Deterministic facts-only checkpoint; no LLM summary was accepted.\",\n\t\tfactsText,\n\t].join(\"\\n\");\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t};\n}\n\n/**\n * Generate a summary for a turn prefix (when splitting a turn).\n */\nasync function generateTurnPrefixSummary(\n\tmessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n): Promise<string> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.5 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t); // Smaller budget for turn prefix\n\tconst llmMessages = convertToLlm(messages);\n\tconst conversationText = serializeConversation(llmMessages);\n\tconst promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;\n\tconst summarizationMessages = [\n\t\t{\n\t\t\trole: \"user\" as const,\n\t\t\tcontent: [{ type: \"text\" as const, text: promptText }],\n\t\t\ttimestamp: Date.now(),\n\t\t},\n\t];\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\t{ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, messages: summarizationMessages },\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, signal, thinkingLevel),\n\t\tstreamFn,\n\t);\n\n\tif (response.stopReason === \"error\") {\n\t\tthrow new Error(`Turn prefix summarization failed: ${response.errorMessage || \"Unknown error\"}`);\n\t}\n\n\treturn response.content\n\t\t.filter((c): c is { type: \"text\"; text: string } => c.type === \"text\")\n\t\t.map((c) => c.text)\n\t\t.join(\"\\n\");\n}\n"]}
@@ -8,7 +8,7 @@ import { completeSimple } from "@caupulican/pi-ai";
8
8
  import { convertToLlm, createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage, } from "../messages.js";
9
9
  import { buildSessionContext } from "../session/session-manager.js";
10
10
  import { extractCompactionFacts, renderFactsBlock } from "./extraction.js";
11
- import { computeFileLists, createFileOps, extractFileOpsFromMessage, formatFileOperations, SUMMARIZATION_SYSTEM_PROMPT, serializeConversation, } from "./utils.js";
11
+ import { computeFileLists, createFileOps, extractFileOpsFromMessage, SUMMARIZATION_SYSTEM_PROMPT, serializeConversation, } from "./utils.js";
12
12
  import { buildRetryPrompt, verifySummary } from "./verification.js";
13
13
  /**
14
14
  * Extract file operations from messages and previous compaction entries.
@@ -385,13 +385,16 @@ export function findCutPoint(entries, startIndex, endIndex, keepRecentTokens) {
385
385
  const SUMMARIZATION_PROMPT = `Checkpoint the conversation above. Format from your instructions, sections in this order:
386
386
  ## Active Task
387
387
  ### Mandatory Rules
388
+ ## Working Set
388
389
  ## Files
390
+ ## Open Problems
389
391
  ## Done
390
- ## Constraints & Preferences
391
392
  ## Key Decisions
392
- ## Blocked / Open
393
+ ## Constraints & Preferences
393
394
  ## Critical Context
394
395
 
396
+ Do NOT carry resolved/transient errors, superseded approaches, or file contents. Record paths and intent, never bodies.
397
+
395
398
  Pre-extracted facts (verified by tooling — include ALL of them, merged with what you read):
396
399
  <facts>
397
400
  {FACTS_BLOCK}
@@ -403,7 +406,10 @@ const UPDATE_SUMMARIZATION_PROMPT = `Update the checkpoint in <previous-summary>
403
406
  - Continue the ## Done numbering. Keep the 15 most recent numbered items verbatim; compress everything older into the single first line "1. (earlier work compressed) <one line>". The checkpoint must not grow without bound across updates.
404
407
  - Update ## Active Task to the newest unfulfilled user input; apply the cancellation rule.
405
408
  - Keep ## Files current (add new, keep still-relevant, drop obsolete).
409
+ - Drop previous ## Open Problems resolved by the new turns.
410
+ - Drop ## Working Set files untouched since the previous checkpoint unless the active task references them.
406
411
  - Preserve exact paths, commands, errors.
412
+ - Do NOT carry resolved/transient errors, superseded approaches, or file contents. Record paths and intent, never bodies.
407
413
 
408
414
  Same section order. Pre-extracted facts:
409
415
  <facts>
@@ -452,9 +458,12 @@ export async function generateSummary(currentMessages, model, reserveTokens, api
452
458
  if (!chunked) {
453
459
  throw new Error("input-overflow: summarization request exceeds summarizer window");
454
460
  }
455
- conversationText = await summarizeChunks(conversationText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound);
461
+ conversationText = await summarizeChunks(conversationText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound, previousSummary, promptSuffix);
456
462
  }
457
463
  const promptText = buildSummarizationPrompt(conversationText, previousSummary, promptSuffix);
464
+ if (estimateStringTokens(promptText) > inputBound) {
465
+ throw new Error("input-overflow: chunked summarization merge still exceeds summarizer window");
466
+ }
458
467
  const response = await completeSummarization(model, {
459
468
  systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
460
469
  messages: [
@@ -479,18 +488,23 @@ function fillPromptTemplate(template, factsBlock, budget) {
479
488
  return template.replaceAll("{FACTS_BLOCK}", factsBlock).replaceAll("{BUDGET}", String(budget));
480
489
  }
481
490
  const SUMMARY_BUDGET_BASE_TOKENS = 1_500;
482
- /** Hard ceiling for the facts-scaled summary budget; also the worst case selection must assume. */
491
+ /** Worst-case selection assumption for summary output when exact bounded facts are not available. */
483
492
  export const SUMMARY_BUDGET_MAX_TOKENS = 4_000;
484
493
  /** Prompt-side margin beyond the raw conversation input (system prompt, tags, instructions). */
485
494
  const SUMMARIZER_PROMPT_MARGIN_TOKENS = 2_000;
486
495
  function getSummaryBudget(reserveTokens, model, factsBlock) {
487
496
  const modelMaxTokens = model.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY;
488
- // The verification gate demands the summary restate every extracted fact (modified files,
489
- // actions, rules). A fixed budget makes large spans structurally fail: the model length-stops
490
- // and the gated sections are the casualties. Scale the budget with the demand, bounded.
497
+ // Verification demand is bounded at extraction time, so the summary budget can be derived from
498
+ // the actual gate demand instead of a blind hard cap. If the demanded facts cannot fit inside the
499
+ // caller's reserve budget, deterministic compaction is the only honest path.
491
500
  const factsTokens = factsBlock ? estimateStringTokens(factsBlock) : 0;
492
- const demandBudget = Math.min(SUMMARY_BUDGET_MAX_TOKENS, Math.max(SUMMARY_BUDGET_BASE_TOKENS, factsTokens + 500));
493
- return Math.max(1, Math.min(demandBudget, Math.floor(0.8 * reserveTokens), modelMaxTokens));
501
+ const gateDemandBudget = factsTokens + 500;
502
+ const demandBudget = Math.max(SUMMARY_BUDGET_BASE_TOKENS, gateDemandBudget);
503
+ const reserveBudget = Math.floor(0.8 * reserveTokens);
504
+ if (factsTokens > reserveBudget || gateDemandBudget > modelMaxTokens) {
505
+ throw new Error(`summary-demand-exceeds-reserve: required ${factsTokens} fact tokens, reserve budget ${reserveBudget}, model max ${modelMaxTokens}`);
506
+ }
507
+ return Math.max(1, Math.min(demandBudget, modelMaxTokens));
494
508
  }
495
509
  function getEffectiveContextWindow(model) {
496
510
  const registered = model.contextWindow > 0 ? model.contextWindow : Number.POSITIVE_INFINITY;
@@ -530,15 +544,24 @@ export function getChunkSummarizationTokenBudget(inputBound) {
530
544
  export function buildChunkSummarizationPrompt(chunk, index, total) {
531
545
  return `<conversation-chunk index="${index}" total="${total}">\n${chunk}\n</conversation-chunk>\n\nSummarize this chunk for a later checkpoint merge. Preserve exact file paths, commands, errors, user prohibitions, and active work. Output concise notes only.`;
532
546
  }
533
- async function summarizeChunks(conversationText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound) {
547
+ async function summarizeChunks(conversationText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound, previousSummary, promptSuffix) {
548
+ let reducedText = conversationText;
549
+ for (let pass = 0; pass < 3; pass++) {
550
+ const summary = await summarizeChunkPass(reducedText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound);
551
+ if (estimateStringTokens(buildSummarizationPrompt(summary, previousSummary, promptSuffix)) <= inputBound) {
552
+ return summary;
553
+ }
554
+ reducedText = summary;
555
+ }
556
+ throw new Error("input-overflow: chunked summarization merge still exceeds summarizer window");
557
+ }
558
+ async function summarizeChunkPass(conversationText, model, maxTokens, apiKey, headers, signal, thinkingLevel, streamFn, inputBound) {
534
559
  const maxChunkTokens = getChunkSummarizationTokenBudget(inputBound);
535
560
  const maxChunkChars = Math.max(1, maxChunkTokens * 4);
536
561
  const chunks = splitText(conversationText, maxChunkChars);
537
- const retainedChunks = chunks.slice(-4);
538
- const omittedChunks = chunks.length - retainedChunks.length;
539
562
  const summaries = [];
540
- for (let i = 0; i < retainedChunks.length; i++) {
541
- const promptText = buildChunkSummarizationPrompt(retainedChunks[i], i + 1, retainedChunks.length);
563
+ for (let i = 0; i < chunks.length; i++) {
564
+ const promptText = buildChunkSummarizationPrompt(chunks[i], i + 1, chunks.length);
542
565
  const response = await completeSummarization(model, {
543
566
  systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
544
567
  messages: [
@@ -554,10 +577,7 @@ async function summarizeChunks(conversationText, model, maxTokens, apiKey, heade
554
577
  }
555
578
  summaries.push(extractTextContent(response));
556
579
  }
557
- const omittedNote = omittedChunks > 0
558
- ? `## Critical Context\n${omittedChunks} older oversized chunks omitted; deterministic facts supplied separately.\n\n`
559
- : "";
560
- return `${omittedNote}${summaries.join("\n\n")}`;
580
+ return summaries.join("\n\n");
561
581
  }
562
582
  function splitText(text, maxChars) {
563
583
  const chunks = [];
@@ -710,7 +730,9 @@ export async function compact(preparation, model, apiKey, headers, customInstruc
710
730
  const { firstKeptEntryId, messagesToSummarize, turnPrefixMessages, isSplitTurn, tokensBefore, previousSummary, fileOps, settings, facts: factsFromPreparation, } = preparation;
711
731
  const facts = factsFromPreparation ?? {
712
732
  files: [],
733
+ workingSet: [],
713
734
  actions: [],
735
+ errorFacts: [],
714
736
  prohibitions: [],
715
737
  cancelledText: "",
716
738
  activeTaskSource: "",
@@ -749,9 +771,7 @@ export async function compact(preparation, model, apiKey, headers, customInstruc
749
771
  customSummaryInstructions = buildRetryPrompt(verification, summary);
750
772
  }
751
773
  }
752
- // Compute file lists and append to summary
753
774
  const { readFiles, modifiedFiles } = computeFileLists(fileOps);
754
- summary += formatFileOperations(readFiles, modifiedFiles);
755
775
  if (!firstKeptEntryId) {
756
776
  throw new Error("First kept entry has no UUID - session may need migration");
757
777
  }
@@ -774,14 +794,22 @@ export function createDeterministicCompaction(preparation) {
774
794
  const { readFiles, modifiedFiles } = computeFileLists(fileOps);
775
795
  const factsText = renderFactsBlock(facts ?? {
776
796
  files: [],
797
+ workingSet: [],
777
798
  actions: [],
799
+ errorFacts: [],
778
800
  prohibitions: [],
779
801
  cancelledText: "",
780
802
  activeTaskSource: "",
781
803
  });
804
+ const workingSetLines = facts?.workingSet.length
805
+ ? facts.workingSet.map((file) => `- ${file.path} — ${file.note || file.kind}`)
806
+ : ["(none)"];
782
807
  const fileLines = facts?.files.length
783
- ? facts.files.map((file) => `- ${file.path} — ${file.note || file.kind} (${file.kind})`)
808
+ ? facts.files.map((file) => `- ${file.path}`)
784
809
  : [`- read: ${readFiles.length}`, `- modified: ${modifiedFiles.length}`];
810
+ const openProblemLines = facts?.errorFacts.length
811
+ ? facts.errorFacts.map((error) => `- ${error.operation}: ${error.error}`)
812
+ : ["(none)"];
785
813
  const mandatoryRuleLines = facts?.prohibitions.length ? facts.prohibitions.map((rule) => `- ${rule}`) : ["(none)"];
786
814
  const doneLines = facts?.actions.length
787
815
  ? facts.actions.map((action, index) => `${index + 1}. ${action}`)
@@ -793,20 +821,23 @@ export function createDeterministicCompaction(preparation) {
793
821
  "### Mandatory Rules",
794
822
  ...mandatoryRuleLines,
795
823
  "",
824
+ "## Working Set",
825
+ ...workingSetLines,
826
+ "",
796
827
  "## Files",
797
828
  ...fileLines,
798
829
  "",
830
+ "## Open Problems",
831
+ ...openProblemLines,
832
+ "",
799
833
  "## Done",
800
834
  ...doneLines,
801
835
  "",
802
- "## Constraints & Preferences",
803
- "Preserve exact file paths, commands, line numbers, and error strings.",
804
- "",
805
836
  "## Key Decisions",
806
837
  "- Deterministic checkpoint used after repeated compaction retries.",
807
838
  "",
808
- "## Blocked / Open",
809
- "(none)",
839
+ "## Constraints & Preferences",
840
+ "Preserve exact file paths, commands, line numbers, and error strings.",
810
841
  "",
811
842
  "## Critical Context",
812
843
  "- Deterministic facts-only checkpoint; no LLM summary was accepted.",