@earendil-works/pi-coding-agent 0.86.1 → 0.87.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (160) hide show
  1. package/CHANGELOG.md +62 -0
  2. package/README.md +25 -675
  3. package/dist/bundle/chunks/{anthropic-messages-MYU5ZMRF.js → anthropic-messages-J5WXPPPC.js} +1 -1
  4. package/dist/bundle/chunks/chunk-65HAU2C5.js +2 -0
  5. package/dist/bundle/chunks/{chunk-CMRUVXTE.js → chunk-OJP47DM6.js} +48 -42
  6. package/dist/bundle/chunks/github-copilot.js +1 -1
  7. package/dist/bundle/chunks/{openai-completions-CYGM3XXP.js → openai-completions-OBX42CLD.js} +2 -2
  8. package/dist/bundle/chunks/{virtual-modules-MGTKWDID.js → virtual-modules-VHMJYYWQ.js} +1 -1
  9. package/dist/bundle/cli-runtime.js +1 -1
  10. package/dist/bundle/index.js +1 -1
  11. package/dist/bundle/rpc-entry.js +1 -1
  12. package/dist/cli/args.d.ts.map +1 -1
  13. package/dist/cli/args.js +14 -4
  14. package/dist/cli/args.js.map +1 -1
  15. package/dist/cli/file-processor.d.ts +1 -1
  16. package/dist/cli/file-processor.d.ts.map +1 -1
  17. package/dist/cli/file-processor.js.map +1 -1
  18. package/dist/core/agent-session-runtime.d.ts.map +1 -1
  19. package/dist/core/agent-session-runtime.js +1 -1
  20. package/dist/core/agent-session-runtime.js.map +1 -1
  21. package/dist/core/agent-session.d.ts +27 -3
  22. package/dist/core/agent-session.d.ts.map +1 -1
  23. package/dist/core/agent-session.js +389 -119
  24. package/dist/core/agent-session.js.map +1 -1
  25. package/dist/core/cache-warmer.d.ts +1 -0
  26. package/dist/core/cache-warmer.d.ts.map +1 -1
  27. package/dist/core/cache-warmer.js +15 -1
  28. package/dist/core/cache-warmer.js.map +1 -1
  29. package/dist/core/compaction/compaction.d.ts +3 -1
  30. package/dist/core/compaction/compaction.d.ts.map +1 -1
  31. package/dist/core/compaction/compaction.js +155 -57
  32. package/dist/core/compaction/compaction.js.map +1 -1
  33. package/dist/core/crash-log.d.ts +5 -0
  34. package/dist/core/crash-log.d.ts.map +1 -1
  35. package/dist/core/crash-log.js +68 -0
  36. package/dist/core/crash-log.js.map +1 -1
  37. package/dist/core/export-html/template.js +6 -1
  38. package/dist/core/extensions/index.d.ts +1 -1
  39. package/dist/core/extensions/index.d.ts.map +1 -1
  40. package/dist/core/extensions/index.js.map +1 -1
  41. package/dist/core/extensions/runner.d.ts +16 -3
  42. package/dist/core/extensions/runner.d.ts.map +1 -1
  43. package/dist/core/extensions/runner.js +110 -5
  44. package/dist/core/extensions/runner.js.map +1 -1
  45. package/dist/core/extensions/types.d.ts +77 -6
  46. package/dist/core/extensions/types.d.ts.map +1 -1
  47. package/dist/core/extensions/types.js.map +1 -1
  48. package/dist/core/index.d.ts +1 -1
  49. package/dist/core/index.d.ts.map +1 -1
  50. package/dist/core/index.js.map +1 -1
  51. package/dist/core/model-config.d.ts +52 -0
  52. package/dist/core/model-config.d.ts.map +1 -1
  53. package/dist/core/model-config.js +16 -0
  54. package/dist/core/model-config.js.map +1 -1
  55. package/dist/core/model-resolver.d.ts.map +1 -1
  56. package/dist/core/model-resolver.js +1 -1
  57. package/dist/core/model-resolver.js.map +1 -1
  58. package/dist/core/prompt-templates.d.ts +6 -1
  59. package/dist/core/prompt-templates.d.ts.map +1 -1
  60. package/dist/core/prompt-templates.js +61 -35
  61. package/dist/core/prompt-templates.js.map +1 -1
  62. package/dist/core/provider-composer.d.ts +1 -0
  63. package/dist/core/provider-composer.d.ts.map +1 -1
  64. package/dist/core/provider-composer.js +19 -0
  65. package/dist/core/provider-composer.js.map +1 -1
  66. package/dist/core/resource-loader.d.ts.map +1 -1
  67. package/dist/core/resource-loader.js +6 -2
  68. package/dist/core/resource-loader.js.map +1 -1
  69. package/dist/core/sdk.d.ts.map +1 -1
  70. package/dist/core/sdk.js +3 -4
  71. package/dist/core/sdk.js.map +1 -1
  72. package/dist/core/session-manager.d.ts +36 -9
  73. package/dist/core/session-manager.d.ts.map +1 -1
  74. package/dist/core/session-manager.js +97 -7
  75. package/dist/core/session-manager.js.map +1 -1
  76. package/dist/core/tools/read.d.ts +4 -1
  77. package/dist/core/tools/read.d.ts.map +1 -1
  78. package/dist/core/tools/read.js +5 -1
  79. package/dist/core/tools/read.js.map +1 -1
  80. package/dist/index.d.ts +2 -2
  81. package/dist/index.d.ts.map +1 -1
  82. package/dist/index.js +1 -1
  83. package/dist/index.js.map +1 -1
  84. package/dist/main.d.ts.map +1 -1
  85. package/dist/main.js +4 -3
  86. package/dist/main.js.map +1 -1
  87. package/dist/modes/interactive/bug-report.d.ts.map +1 -1
  88. package/dist/modes/interactive/bug-report.js +4 -0
  89. package/dist/modes/interactive/bug-report.js.map +1 -1
  90. package/dist/modes/interactive/components/tree-selector.d.ts.map +1 -1
  91. package/dist/modes/interactive/components/tree-selector.js +7 -0
  92. package/dist/modes/interactive/components/tree-selector.js.map +1 -1
  93. package/dist/modes/interactive/interactive-mode.d.ts +3 -0
  94. package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
  95. package/dist/modes/interactive/interactive-mode.js +64 -2
  96. package/dist/modes/interactive/interactive-mode.js.map +1 -1
  97. package/dist/utils/mime.d.ts.map +1 -1
  98. package/dist/utils/mime.js +1 -1
  99. package/dist/utils/mime.js.map +1 -1
  100. package/dist/utils/tool-result-images.d.ts +3 -1
  101. package/dist/utils/tool-result-images.d.ts.map +1 -1
  102. package/dist/utils/tool-result-images.js +4 -1
  103. package/dist/utils/tool-result-images.js.map +1 -1
  104. package/docs/cli-integration.md +106 -0
  105. package/docs/cli.md +268 -0
  106. package/docs/compaction.md +45 -26
  107. package/docs/configuration.md +45 -0
  108. package/docs/containerization.md +109 -82
  109. package/docs/custom-provider.md +132 -784
  110. package/docs/docs.json +139 -99
  111. package/docs/environment-variables.md +3 -5
  112. package/docs/extensions.md +134 -2956
  113. package/docs/how-pi-works.md +49 -0
  114. package/docs/images/interactive-mode.png +0 -0
  115. package/docs/index.md +24 -69
  116. package/docs/json.md +193 -65
  117. package/docs/keybindings.md +57 -102
  118. package/docs/llama-cpp.md +3 -3
  119. package/docs/message-types.md +261 -0
  120. package/docs/models.md +64 -546
  121. package/docs/packages.md +66 -167
  122. package/docs/prompt-templates.md +31 -68
  123. package/docs/providers.md +102 -240
  124. package/docs/quickstart.md +61 -106
  125. package/docs/rpc-commands.md +854 -0
  126. package/docs/rpc-extension-ui.md +200 -0
  127. package/docs/rpc.md +129 -1556
  128. package/docs/sdk.md +76 -1160
  129. package/docs/security.md +70 -32
  130. package/docs/session-format.md +25 -216
  131. package/docs/sessions.md +35 -141
  132. package/docs/settings.md +109 -387
  133. package/docs/shell-aliases.md +85 -5
  134. package/docs/skills.md +51 -190
  135. package/docs/slash-commands.md +60 -0
  136. package/docs/terminal-setup.md +105 -78
  137. package/docs/termux.md +74 -83
  138. package/docs/themes.md +68 -280
  139. package/docs/tmux.md +31 -39
  140. package/docs/tui.md +69 -923
  141. package/docs/usage.md +54 -272
  142. package/docs/windows.md +43 -17
  143. package/examples/README.md +13 -2
  144. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  145. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  146. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  147. package/examples/extensions/gondolin/package-lock.json +2 -2
  148. package/examples/extensions/gondolin/package.json +1 -1
  149. package/examples/extensions/sandbox/package-lock.json +2 -2
  150. package/examples/extensions/sandbox/package.json +1 -1
  151. package/examples/extensions/with-deps/package-lock.json +2 -2
  152. package/examples/extensions/with-deps/package.json +1 -1
  153. package/examples/plugins/pi-example-plugin/src/session.ts +3 -2
  154. package/examples/rpc-client.ts +35 -0
  155. package/examples/rpc-extension-ui.ts +25 -5
  156. package/examples/sdk/README.md +1 -1
  157. package/npm-shrinkwrap.json +20 -20
  158. package/package.json +8 -8
  159. package/dist/bundle/chunks/chunk-HTEQD2HM.js +0 -2
  160. package/docs/development.md +0 -90
@@ -91,6 +91,7 @@ export declare class CacheWarmer {
91
91
  private stop;
92
92
  private schedule;
93
93
  private refresh;
94
+ private refreshDeadlineMissed;
94
95
  private validateRun;
95
96
  private getModeStopReason;
96
97
  private evaluate;
@@ -1 +1 @@
1
- {"version":3,"file":"cache-warmer.d.ts","sourceRoot":"","sources":["../../src/core/cache-warmer.ts"],"names":[],"mappings":"AAAA,OAAO,EACN,KAAK,GAAG,EACR,KAAK,OAAO,EAEZ,KAAK,KAAK,EACV,KAAK,yBAAyB,EAC9B,KAAK,mBAAmB,EAExB,MAAM,uBAAuB,CAAC;AAE/B,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AACvD,OAAO,KAAK,EAAgB,cAAc,EAAE,UAAU,EAAE,MAAM,sBAAsB,CAAC;AACrF,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,uBAAuB,CAAC;AAe9D,iFAAiF;AACjF,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAGxE;AAED;;;;GAIG;AACH,wBAAgB,mBAAmB,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,OAAO,EAAE,mBAAmB,GAAG,SAAS,GAAG,MAAM,GAAG,SAAS,CAOnH;AAED;;;;;;GAMG;AACH,wBAAgB,YAAY,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,OAAO,EAAE,mBAAmB,GAAG,SAAS,GAAG,OAAO,CAGjG;AA8BD,MAAM,MAAM,kBAAkB,GAAG,MAAM,GAAG,MAAM,CAAC;AAEjD,+EAA+E;AAC/E,MAAM,WAAW,oBAAoB;IACpC,6EAA6E;IAC7E,KAAK,EAAE,WAAW,GAAG,MAAM,CAAC;IAC5B,+EAA+E;IAC/E,QAAQ,EAAE,MAAM,CAAC;IACjB,uEAAuE;IACvE,QAAQ,EAAE,MAAM,CAAC;IACjB,6EAA6E;IAC7E,uBAAuB,EAAE,MAAM,CAAC;IAChC,uDAAuD;IACvD,eAAe,EAAE,MAAM,CAAC;IACxB,oEAAoE;IACpE,kBAAkB,EAAE,OAAO,CAAC;IAC5B,sEAAsE;IACtE,MAAM,EAAE,kBAAkB,CAAC;CAC3B;AAED;;;GAGG;AACH,MAAM,WAAW,yBAChB,SAAQ,IAAI,CAAC,oBAAoB,EAAE,UAAU,GAAG,UAAU,GAAG,yBAAyB,GAAG,QAAQ,CAAC;IAClG,IAAI,EAAE,wBAAwB,CAAC;CAC/B;AAED,MAAM,WAAW,+BAA+B;IAC/C,8FAA8F;IAC9F,MAAM,CAAC,EAAE,kBAAkB,CAAC;CAC5B;AAED,MAAM,WAAW,kBAAkB;IAClC,wFAAwF;IACxF,KAAK,EAAE,UAAU,GAAG,WAAW,GAAG,YAAY,CAAC;IAC/C,gCAAgC;IAChC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,oBAAoB,CAAC;IAChC,wDAAwD;IACxD,iBAAiB,CAAC,EAAE,OAAO,CAAC;CAC5B;AAED,wFAAwF;AACxF,MAAM,WAAW,gBAAgB;IAChC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IAClB,OAAO,EAAE,OAAO,CAAC;IACjB,OAAO,EAAE,yBAAyB,CAAC;CACnC;AAeD;;;;GAIG;AACH,qBAAa,WAAW;IACvB,OAAO,CAAC,GAAG,CAAC,CAAY;IACxB,OAAO,CAAC,QAAQ,CAAqB;IACrC,OAAO,CAAC,QAAQ,CAAC,MAAM,CAAqC;IAC5D,OAAO,CAAC,QAAQ,CAAC,cAAc,CAAoD;IACnF,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAyB;IACjD,oFAAoF;IACpF,OAAO,CAAC,QAAQ,CAAC,MAAM,CAAoE;IAC3F,2EAA2E;IAC3E,QAAQ,CAAC,EAAE,CAAC,KAAK,EAAE,UAAU,KAAK,IAAI,CAAC;IAEvC,YACC,MAAM,EAAE,IAAI,CAAC,YAAY,EAAE,cAAc,CAAC,EAC1C,cAAc,EAAE,IAAI,CAAC,cAAc,EAAE,aAAa,GAAG,WAAW,CAAC,EACjE,OAAO,EAAE,MAAM,gBAAgB,EAC/B,MAAM,GAAE,CAAC,KAAK,EAAE,yBAAyB,KAAK,OAAO,CAAC,kBAAkB,CAAiC,EAOzG;IAED,IAAI,MAAM,IAAI,kBAAkB,CAgB/B;IAED,qFAAqF;IACrF,KAAK,CAAC,OAAO,EAAE,gBAAgB,EAAE,SAAS,EAAE,MAAM,OAAO,GAAG,IAAI,CAoC/D;IAED,cAAc,IAAI,IAAI,CAYrB;IAED,wEAAwE;IACxE,aAAa,IAAI,IAAI,CAKpB;IAED,MAAM,IAAI,IAAI,CAEb;IAED,OAAO,CAAC,QAAQ;IAQhB,OAAO,CAAC,IAAI;IAKZ,OAAO,CAAC,QAAQ;YAYF,OAAO;IAwDrB,OAAO,CAAC,WAAW;IAQnB,OAAO,CAAC,iBAAiB;IAOzB,OAAO,CAAC,QAAQ;CAuBhB;AA+BD,sCAAsC;AACtC,wBAAgB,wBAAwB,CAAC,MAAM,EAAE,kBAAkB,EAAE,GAAG,SAAa,GAAG,MAAM,CAa7F;AAED,kEAAkE;AAClE,wBAAgB,uBAAuB,CAAC,KAAK,EAAE,UAAU,GAAG,MAAM,CAIjE","sourcesContent":["import {\n\ttype Api,\n\ttype Context,\n\tcalculateCost,\n\ttype Model,\n\ttype ModelsSimpleStreamOptions,\n\ttype SimpleStreamOptions,\n\ttype Usage,\n} from \"@earendil-works/pi-ai\";\nimport { getProviderEnvValue } from \"@earendil-works/pi-ai/utils/provider-env\";\nimport type { ModelRuntime } from \"./model-runtime.ts\";\nimport type { SessionEntry, SessionManager, UsageEntry } from \"./session-manager.ts\";\nimport type { CacheWarmingMode } from \"./settings-manager.ts\";\n\n/** Streaming warming never continues past this long after the real request that started it. */\nconst MAX_WARMING_AGE_MS = 60 * 60_000;\n/** Idle warming uses a shorter horizon because continuation estimates become less reliable with age. */\nconst MAX_IDLE_WARMING_AGE_MS = 30 * 60_000;\n/** A refresh is sent only when it is expected to save at least this many dollars. */\nconst CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS = 0.05;\n/**\n * Chance that a real request arrives before the cache entry expires while the\n * agent sits idle. Measured from our own usage; per-session estimates were not\n * better than this constant.\n */\nconst IDLE_CONTINUATION_PROBABILITY = 0.15;\n\n/** Refresh at 90% of the TTL while preserving at least ten seconds of margin. */\nexport function getCacheWarmingDelayMs(ttlMs: number): number | undefined {\n\tif (ttlMs <= 10_000) return undefined;\n\treturn Math.max(1, Math.floor(Math.min(ttlMs * 0.9, ttlMs - 10_000)));\n}\n\n/**\n * Lifetime of the prompt cache entry a request writes, from the model's\n * `promptCache` tier for the retention the request used. Undefined when the\n * model has no lifetime for that tier or caching is off.\n */\nexport function getPromptCacheTtlMs(model: Model<Api>, options: SimpleStreamOptions | undefined): number | undefined {\n\tconst retention =\n\t\toptions?.cacheRetention ??\n\t\t(getProviderEnvValue(\"PI_CACHE_RETENTION\", options?.env) === \"long\" ? \"long\" : \"short\");\n\tif (retention === \"none\") return undefined;\n\tconst seconds = model.promptCache?.[retention];\n\treturn seconds === undefined ? undefined : seconds * 1000;\n}\n\n/**\n * Whether replaying the request with a one-token output cap leaves its cache\n * entry untouched. Anthropic's budget-based thinking (Claude models without\n * adaptive thinking) derives `budget_tokens` from `max_tokens`; the replay\n * would get a different budget, which Anthropic keys the message cache on,\n * and the model could still think for thousands of tokens.\n */\nexport function isReplayable(model: Model<Api>, options: SimpleStreamOptions | undefined): boolean {\n\tif (!options?.reasoning || model.api !== \"anthropic-messages\") return true;\n\treturn (model as Model<\"anthropic-messages\">).compat?.forceAdaptiveThinking === true;\n}\n\n/** Prompt size of the most recent real request on the branch, as reported by the provider. */\nfunction lastPromptTokens(entries: SessionEntry[]): number {\n\tfor (let index = entries.length - 1; index >= 0; index--) {\n\t\tconst entry = entries[index];\n\t\tif (entry.type === \"message\" && entry.message.role === \"assistant\") {\n\t\t\tconst usage = entry.message.usage;\n\t\t\treturn usage.input + usage.cacheRead + usage.cacheWrite;\n\t\t}\n\t}\n\treturn 0;\n}\n\nfunction price(\n\tmodel: Model<Api>,\n\ttokens: Partial<Pick<Usage, \"input\" | \"output\" | \"cacheRead\" | \"cacheWrite\">>,\n): number {\n\tconst usage: Usage = {\n\t\tinput: 0,\n\t\toutput: 0,\n\t\tcacheRead: 0,\n\t\tcacheWrite: 0,\n\t\ttotalTokens: 0,\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t...tokens,\n\t};\n\treturn calculateCost(model, usage).total;\n}\n\nexport type CacheWarmingAction = \"warm\" | \"stop\";\n\n/** Inputs and outcome of one warm-or-stop decision, as shown by `/session`. */\nexport interface CacheWarmingDecision {\n\t/** \"streaming\" while the agent run that sent the request is still active. */\n\tphase: \"streaming\" | \"idle\";\n\t/** Price of this refresh: a cache read of the prompt plus one output token. */\n\twarmCost: number;\n\t/** Extra price of the next real request if the cache entry is lost. */\n\tmissCost: number;\n\t/** Estimated chance that a real request arrives before the entry expires. */\n\tcontinuationProbability: number;\n\t/** `continuationProbability * missCost - warmCost`. */\n\texpectedSavings: number;\n\t/** False when the prompt size or the model's prices are unknown. */\n\teconomicsAvailable: boolean;\n\t/** Pi's decision: \"warm\" when `expectedSavings` is at least $0.05. */\n\taction: CacheWarmingAction;\n}\n\n/**\n * Fired before each refresh with pi's decision filled in. Everything else an\n * extension might want (model, idle state, context size) is on the context.\n */\nexport interface CacheWarmingDecisionEvent\n\textends Pick<CacheWarmingDecision, \"warmCost\" | \"missCost\" | \"continuationProbability\" | \"action\"> {\n\ttype: \"cache_warming_decision\";\n}\n\nexport interface CacheWarmingDecisionEventResult {\n\t/** Override whether this refresh is sent. \"stop\" ends warming until the next real request. */\n\taction?: CacheWarmingAction;\n}\n\nexport interface CacheWarmingStatus {\n\t/** \"scheduled\": a refresh timer is armed; \"refreshing\": a warm request is in flight. */\n\tstate: \"inactive\" | \"scheduled\" | \"refreshing\";\n\t/** Why nothing is scheduled. */\n\treason?: string;\n\tnextWarmAt?: number;\n\t/** The pending decision, or the decision that stopped warming. */\n\tdecision?: CacheWarmingDecision;\n\t/** True when an extension changed `decision.action`. */\n\textensionOverride?: boolean;\n}\n\n/** The request whose prompt cache entry should be kept warm, exactly as it was sent. */\nexport interface CacheWarmRequest {\n\tmodel: Model<Api>;\n\tcontext: Context;\n\toptions: ModelsSimpleStreamOptions;\n}\n\ninterface ActiveRun extends CacheWarmRequest {\n\t/** False once the session's model or messages no longer match the request. */\n\tisCurrent: () => boolean;\n\tdelayMs: number;\n\tstartedAt: number;\n\tcontroller: AbortController;\n\tphase: \"streaming\" | \"idle\";\n\tnextWarmAt: number;\n\t/** Set while a refresh that an extension forced is in flight. */\n\textensionOverride: boolean;\n\ttimer?: ReturnType<typeof setTimeout>;\n}\n\n/**\n * Keeps one prompt cache entry alive by re-sending its request with a\n * one-token output cap before the entry expires. `start` replaces any\n * previous run; warm requests never extend the fixed safety windows.\n */\nexport class CacheWarmer {\n\tprivate run?: ActiveRun;\n\tprivate inactive: CacheWarmingStatus;\n\tprivate readonly models: Pick<ModelRuntime, \"streamSimple\">;\n\tprivate readonly sessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">;\n\tprivate readonly getMode: () => CacheWarmingMode;\n\t/** Lets extensions override `event.action`; failures fall back to pi's decision. */\n\tprivate readonly decide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction>;\n\t/** Called with the persisted usage entry after each successful refresh. */\n\tonWarmed?: (entry: UsageEntry) => void;\n\n\tconstructor(\n\t\tmodels: Pick<ModelRuntime, \"streamSimple\">,\n\t\tsessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">,\n\t\tgetMode: () => CacheWarmingMode,\n\t\tdecide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction> = async (event) => event.action,\n\t) {\n\t\tthis.models = models;\n\t\tthis.sessionManager = sessionManager;\n\t\tthis.getMode = getMode;\n\t\tthis.decide = decide;\n\t\tthis.inactive = { state: \"inactive\", reason: \"waiting for first request\" };\n\t}\n\n\tget status(): CacheWarmingStatus {\n\t\tif (this.getMode() === \"off\") return { state: \"inactive\", reason: \"cache warming disabled\" };\n\t\tconst run = this.run;\n\t\tif (!run) return this.inactive;\n\t\tif (!run.isCurrent()) return { state: \"inactive\", reason: \"conversation context changed\" };\n\t\tconst decision = this.evaluate(run);\n\t\tconst refreshing = run.timer === undefined;\n\t\tif (!decision.economicsAvailable && !refreshing) {\n\t\t\treturn { state: \"inactive\", reason: \"cache economics unavailable\" };\n\t\t}\n\t\treturn {\n\t\t\tstate: refreshing ? \"refreshing\" : \"scheduled\",\n\t\t\tnextWarmAt: run.nextWarmAt,\n\t\t\tdecision,\n\t\t\textensionOverride: run.extensionOverride,\n\t\t};\n\t}\n\n\t/** Keep the prompt cache entry written by `request` warm while `isCurrent` holds. */\n\tstart(request: CacheWarmRequest, isCurrent: () => boolean): void {\n\t\tthis.clearRun();\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") {\n\t\t\tthis.stop(\"cache warming disabled\");\n\t\t\treturn;\n\t\t}\n\t\tif (!isReplayable(request.model, request.options)) {\n\t\t\tthis.stop(\"request cannot be replayed safely\");\n\t\t\treturn;\n\t\t}\n\t\tconst ttlMs = getPromptCacheTtlMs(request.model, request.options);\n\t\tif (ttlMs === undefined) {\n\t\t\tthis.stop(\n\t\t\t\trequest.options.cacheRetention === \"none\"\n\t\t\t\t\t? \"request disabled prompt caching\"\n\t\t\t\t\t: \"cache lifetime unavailable\",\n\t\t\t);\n\t\t\treturn;\n\t\t}\n\t\tconst delayMs = getCacheWarmingDelayMs(ttlMs);\n\t\tif (delayMs === undefined) {\n\t\t\tthis.stop(\"cache lifetime unavailable\");\n\t\t\treturn;\n\t\t}\n\t\tthis.run = {\n\t\t\t...request,\n\t\t\tisCurrent,\n\t\t\tdelayMs,\n\t\t\tstartedAt: Date.now(),\n\t\t\tcontroller: new AbortController(),\n\t\t\tphase: \"streaming\",\n\t\t\tnextWarmAt: 0,\n\t\t\textensionOverride: false,\n\t\t};\n\t\tthis.schedule(this.run);\n\t}\n\n\tonAgentSettled(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tif (this.getMode() === \"streaming\") {\n\t\t\tthis.stop(\"agent run settled\");\n\t\t\treturn;\n\t\t}\n\t\trun.phase = \"idle\";\n\t\tconst deadline = run.startedAt + MAX_IDLE_WARMING_AGE_MS;\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(\"30-minute idle safety limit reached\");\n\t\t}\n\t}\n\n\t/** Reconcile an active run after the persisted warming mode changes. */\n\tonModeChanged(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tconst reason = this.getModeStopReason(run);\n\t\tif (reason) this.stop(reason);\n\t}\n\n\tcancel(): void {\n\t\tthis.stop(\"inactive\");\n\t}\n\n\tprivate clearRun(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tthis.run = undefined;\n\t\tif (run.timer) clearTimeout(run.timer);\n\t\trun.controller.abort();\n\t}\n\n\tprivate stop(reason: string, stopped?: Pick<CacheWarmingStatus, \"decision\" | \"extensionOverride\">): void {\n\t\tthis.clearRun();\n\t\tthis.inactive = { state: \"inactive\", reason, ...stopped };\n\t}\n\n\tprivate schedule(run: ActiveRun): void {\n\t\trun.extensionOverride = false;\n\t\trun.nextWarmAt = Date.now() + run.delayMs;\n\t\tconst deadline = run.startedAt + (run.phase === \"idle\" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(run.phase === \"idle\" ? \"30-minute idle safety limit reached\" : \"one-hour safety limit reached\");\n\t\t\treturn;\n\t\t}\n\t\trun.timer = setTimeout(() => void this.refresh(run), Math.max(0, run.nextWarmAt - Date.now()));\n\t\trun.timer.unref?.();\n\t}\n\n\tprivate async refresh(run: ActiveRun): Promise<void> {\n\t\trun.timer = undefined;\n\t\tif (!this.validateRun(run)) return;\n\t\tconst decision = this.evaluate(run);\n\t\tconst { warmCost, missCost, continuationProbability } = decision;\n\t\tlet action = decision.action;\n\t\ttry {\n\t\t\taction = await this.decide({\n\t\t\t\ttype: \"cache_warming_decision\",\n\t\t\t\twarmCost,\n\t\t\t\tmissCost,\n\t\t\t\tcontinuationProbability,\n\t\t\t\taction,\n\t\t\t});\n\t\t} catch {\n\t\t\t// Extension failures fall back to pi's own decision.\n\t\t}\n\t\tif (!this.validateRun(run)) return;\n\t\tconst extensionOverride = action !== decision.action;\n\t\tif (action === \"stop\") {\n\t\t\tconst reason = extensionOverride\n\t\t\t\t? \"stopped by extension\"\n\t\t\t\t: decision.economicsAvailable\n\t\t\t\t\t? \"expected savings below threshold\"\n\t\t\t\t\t: \"cache economics unavailable\";\n\t\t\tthis.stop(reason, { decision, extensionOverride });\n\t\t\treturn;\n\t\t}\n\n\t\trun.extensionOverride = extensionOverride;\n\t\ttry {\n\t\t\tconst message = await this.models\n\t\t\t\t.streamSimple(run.model, run.context, {\n\t\t\t\t\t...run.options,\n\t\t\t\t\tmaxTokens: 1,\n\t\t\t\t\tmaxRetries: 0,\n\t\t\t\t\tsignal: run.controller.signal,\n\t\t\t\t})\n\t\t\t\t.result();\n\t\t\tif (!this.validateRun(run)) return;\n\t\t\tif (message.stopReason !== \"error\" && message.stopReason !== \"aborted\") {\n\t\t\t\tconst entry = this.sessionManager.appendUsage(\n\t\t\t\t\t\"cache_warm\",\n\t\t\t\t\tmessage.provider,\n\t\t\t\t\tmessage.responseModel ?? message.model,\n\t\t\t\t\tmessage.usage,\n\t\t\t\t\textensionOverride ? \"extension override\" : undefined,\n\t\t\t\t);\n\t\t\t\tthis.onWarmed?.(entry);\n\t\t\t}\n\t\t} catch {\n\t\t\t// Cache warming is best-effort and must not affect the active agent run.\n\t\t}\n\t\tif (this.run === run) this.schedule(run);\n\t}\n\n\tprivate validateRun(run: ActiveRun): boolean {\n\t\tif (this.run !== run) return false;\n\t\tconst reason = this.getModeStopReason(run) ?? (!run.isCurrent() ? \"conversation context changed\" : undefined);\n\t\tif (!reason) return true;\n\t\tthis.stop(reason);\n\t\treturn false;\n\t}\n\n\tprivate getModeStopReason(run: ActiveRun): string | undefined {\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") return \"cache warming disabled\";\n\t\tif (mode === \"streaming\" && run.phase === \"idle\") return \"agent run settled\";\n\t\treturn undefined;\n\t}\n\n\tprivate evaluate(run: ActiveRun): CacheWarmingDecision {\n\t\tconst model = run.model;\n\t\tconst promptTokens = lastPromptTokens(this.sessionManager.getBranch());\n\t\tconst cacheHitCost = price(model, { cacheRead: promptTokens });\n\t\tconst cacheMissCost = price(\n\t\t\tmodel,\n\t\t\tmodel.cost.cacheWrite > 0 ? { cacheWrite: promptTokens } : { input: promptTokens },\n\t\t);\n\t\tconst warmCost = price(model, { cacheRead: promptTokens, output: 1 });\n\t\tconst missCost = Math.max(0, cacheMissCost - cacheHitCost);\n\t\tconst continuationProbability = run.phase === \"idle\" ? IDLE_CONTINUATION_PROBABILITY : 1;\n\t\tconst economicsAvailable = promptTokens > 0 && (cacheHitCost > 0 || cacheMissCost > 0);\n\t\tconst expectedSavings = continuationProbability * missCost - warmCost;\n\t\treturn {\n\t\t\tphase: run.phase,\n\t\t\twarmCost,\n\t\t\tmissCost,\n\t\t\tcontinuationProbability,\n\t\t\texpectedSavings,\n\t\t\teconomicsAvailable,\n\t\t\taction: expectedSavings >= CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS ? \"warm\" : \"stop\",\n\t\t};\n\t}\n}\n\nfunction formatDollars(value: number): string {\n\treturn value < 0 ? `-$${Math.abs(value).toFixed(3)}` : `$${value.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingEconomics(decision: CacheWarmingDecision): string {\n\tif (!decision.economicsAvailable) return \"cache economics unavailable\";\n\tconst probability = Math.round(decision.continuationProbability * 100);\n\tconst probabilityText =\n\t\tdecision.phase === \"streaming\"\n\t\t\t? `${probability}% continuation probability while agent is running`\n\t\t\t: `${probability}% continuation probability`;\n\tconst comparison = decision.action === \"warm\" ? \">=\" : \"<\";\n\treturn `${probabilityText}, expected savings ${formatDollars(decision.expectedSavings)} ${comparison} $${CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingDecisionTime(nextWarmAt: number | undefined, now: number): string {\n\tif (nextWarmAt === undefined || nextWarmAt <= now) return \"Decision now\";\n\tlet remainingSeconds = Math.ceil((nextWarmAt - now) / 1000);\n\tconst hours = Math.floor(remainingSeconds / 3600);\n\tremainingSeconds %= 3600;\n\tconst minutes = Math.floor(remainingSeconds / 60);\n\tconst seconds = remainingSeconds % 60;\n\tconst parts: string[] = [];\n\tif (hours > 0) parts.push(`${hours}h`);\n\tif (minutes > 0) parts.push(`${minutes}m`);\n\tif (seconds > 0 || parts.length === 0) parts.push(`${seconds}s`);\n\treturn `Decision in ${parts.join(\" \")}`;\n}\n\n/** One-line status for `/session`. */\nexport function formatCacheWarmingStatus(status: CacheWarmingStatus, now = Date.now()): string {\n\tconst decision = status.decision;\n\t// A decision is attached once pi (or an extension) acted on it; \"inactive\"\n\t// without one never got that far.\n\tif (!decision || (status.state === \"inactive\" && !decision.economicsAvailable && !status.extensionOverride)) {\n\t\treturn `Inactive (${status.reason ?? \"unknown reason\"})`;\n\t}\n\tconst details = status.extensionOverride\n\t\t? `extension override, ${formatCacheWarmingEconomics(decision)}`\n\t\t: `${formatCacheWarmingEconomics(decision)} -> ${decision.action}`;\n\tif (status.state === \"inactive\") return `Stopped (${details})`;\n\tif (status.state === \"refreshing\") return `Warming cache (${details})`;\n\treturn `${formatCacheWarmingDecisionTime(status.nextWarmAt, now)} (${details})`;\n}\n\n/** One-line transcript text for persisted cache-warming usage. */\nexport function formatCacheWarmingUsage(entry: UsageEntry): string {\n\tconst note = entry.note ? ` (${entry.note})` : \"\";\n\tconst cost = entry.usage.cost.total.toFixed(6).replace(/(\\.\\d{3}\\d*?)0+$/, \"$1\");\n\treturn `Cache warmed${note}: $${cost}`;\n}\n"]}
1
+ {"version":3,"file":"cache-warmer.d.ts","sourceRoot":"","sources":["../../src/core/cache-warmer.ts"],"names":[],"mappings":"AAAA,OAAO,EACN,KAAK,GAAG,EACR,KAAK,OAAO,EAEZ,KAAK,KAAK,EACV,KAAK,yBAAyB,EAC9B,KAAK,mBAAmB,EAExB,MAAM,uBAAuB,CAAC;AAE/B,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AACvD,OAAO,KAAK,EAAgB,cAAc,EAAE,UAAU,EAAE,MAAM,sBAAsB,CAAC;AACrF,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,uBAAuB,CAAC;AAe9D,iFAAiF;AACjF,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAGxE;AAED;;;;GAIG;AACH,wBAAgB,mBAAmB,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,OAAO,EAAE,mBAAmB,GAAG,SAAS,GAAG,MAAM,GAAG,SAAS,CAOnH;AAED;;;;;;GAMG;AACH,wBAAgB,YAAY,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EAAE,OAAO,EAAE,mBAAmB,GAAG,SAAS,GAAG,OAAO,CAGjG;AA8BD,MAAM,MAAM,kBAAkB,GAAG,MAAM,GAAG,MAAM,CAAC;AAEjD,+EAA+E;AAC/E,MAAM,WAAW,oBAAoB;IACpC,6EAA6E;IAC7E,KAAK,EAAE,WAAW,GAAG,MAAM,CAAC;IAC5B,+EAA+E;IAC/E,QAAQ,EAAE,MAAM,CAAC;IACjB,uEAAuE;IACvE,QAAQ,EAAE,MAAM,CAAC;IACjB,6EAA6E;IAC7E,uBAAuB,EAAE,MAAM,CAAC;IAChC,uDAAuD;IACvD,eAAe,EAAE,MAAM,CAAC;IACxB,oEAAoE;IACpE,kBAAkB,EAAE,OAAO,CAAC;IAC5B,sEAAsE;IACtE,MAAM,EAAE,kBAAkB,CAAC;CAC3B;AAED;;;GAGG;AACH,MAAM,WAAW,yBAChB,SAAQ,IAAI,CAAC,oBAAoB,EAAE,UAAU,GAAG,UAAU,GAAG,yBAAyB,GAAG,QAAQ,CAAC;IAClG,IAAI,EAAE,wBAAwB,CAAC;CAC/B;AAED,MAAM,WAAW,+BAA+B;IAC/C,8FAA8F;IAC9F,MAAM,CAAC,EAAE,kBAAkB,CAAC;CAC5B;AAED,MAAM,WAAW,kBAAkB;IAClC,wFAAwF;IACxF,KAAK,EAAE,UAAU,GAAG,WAAW,GAAG,YAAY,CAAC;IAC/C,gCAAgC;IAChC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,oBAAoB,CAAC;IAChC,wDAAwD;IACxD,iBAAiB,CAAC,EAAE,OAAO,CAAC;CAC5B;AAED,wFAAwF;AACxF,MAAM,WAAW,gBAAgB;IAChC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IAClB,OAAO,EAAE,OAAO,CAAC;IACjB,OAAO,EAAE,yBAAyB,CAAC;CACnC;AAkBD;;;;GAIG;AACH,qBAAa,WAAW;IACvB,OAAO,CAAC,GAAG,CAAC,CAAY;IACxB,OAAO,CAAC,QAAQ,CAAqB;IACrC,OAAO,CAAC,QAAQ,CAAC,MAAM,CAAqC;IAC5D,OAAO,CAAC,QAAQ,CAAC,cAAc,CAAoD;IACnF,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAyB;IACjD,oFAAoF;IACpF,OAAO,CAAC,QAAQ,CAAC,MAAM,CAAoE;IAC3F,2EAA2E;IAC3E,QAAQ,CAAC,EAAE,CAAC,KAAK,EAAE,UAAU,KAAK,IAAI,CAAC;IAEvC,YACC,MAAM,EAAE,IAAI,CAAC,YAAY,EAAE,cAAc,CAAC,EAC1C,cAAc,EAAE,IAAI,CAAC,cAAc,EAAE,aAAa,GAAG,WAAW,CAAC,EACjE,OAAO,EAAE,MAAM,gBAAgB,EAC/B,MAAM,GAAE,CAAC,KAAK,EAAE,yBAAyB,KAAK,OAAO,CAAC,kBAAkB,CAAiC,EAOzG;IAED,IAAI,MAAM,IAAI,kBAAkB,CAgB/B;IAED,qFAAqF;IACrF,KAAK,CAAC,OAAO,EAAE,gBAAgB,EAAE,SAAS,EAAE,MAAM,OAAO,GAAG,IAAI,CAsC/D;IAED,cAAc,IAAI,IAAI,CAYrB;IAED,wEAAwE;IACxE,aAAa,IAAI,IAAI,CAKpB;IAED,MAAM,IAAI,IAAI,CAEb;IAED,OAAO,CAAC,QAAQ;IAQhB,OAAO,CAAC,IAAI;IAKZ,OAAO,CAAC,QAAQ;YAgBF,OAAO;IAyDrB,OAAO,CAAC,qBAAqB;IAM7B,OAAO,CAAC,WAAW;IAQnB,OAAO,CAAC,iBAAiB;IAOzB,OAAO,CAAC,QAAQ;CAuBhB;AA+BD,sCAAsC;AACtC,wBAAgB,wBAAwB,CAAC,MAAM,EAAE,kBAAkB,EAAE,GAAG,SAAa,GAAG,MAAM,CAa7F;AAED,kEAAkE;AAClE,wBAAgB,uBAAuB,CAAC,KAAK,EAAE,UAAU,GAAG,MAAM,CAIjE","sourcesContent":["import {\n\ttype Api,\n\ttype Context,\n\tcalculateCost,\n\ttype Model,\n\ttype ModelsSimpleStreamOptions,\n\ttype SimpleStreamOptions,\n\ttype Usage,\n} from \"@earendil-works/pi-ai\";\nimport { getProviderEnvValue } from \"@earendil-works/pi-ai/utils/provider-env\";\nimport type { ModelRuntime } from \"./model-runtime.ts\";\nimport type { SessionEntry, SessionManager, UsageEntry } from \"./session-manager.ts\";\nimport type { CacheWarmingMode } from \"./settings-manager.ts\";\n\n/** Streaming warming never continues past this long after the real request that started it. */\nconst MAX_WARMING_AGE_MS = 60 * 60_000;\n/** Idle warming uses a shorter horizon because continuation estimates become less reliable with age. */\nconst MAX_IDLE_WARMING_AGE_MS = 30 * 60_000;\n/** A refresh is sent only when it is expected to save at least this many dollars. */\nconst CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS = 0.05;\n/**\n * Chance that a real request arrives before the cache entry expires while the\n * agent sits idle. Measured from our own usage; per-session estimates were not\n * better than this constant.\n */\nconst IDLE_CONTINUATION_PROBABILITY = 0.15;\n\n/** Refresh at 90% of the TTL while preserving at least ten seconds of margin. */\nexport function getCacheWarmingDelayMs(ttlMs: number): number | undefined {\n\tif (ttlMs <= 10_000) return undefined;\n\treturn Math.max(1, Math.floor(Math.min(ttlMs * 0.9, ttlMs - 10_000)));\n}\n\n/**\n * Lifetime of the prompt cache entry a request writes, from the model's\n * `promptCache` tier for the retention the request used. Undefined when the\n * model has no lifetime for that tier or caching is off.\n */\nexport function getPromptCacheTtlMs(model: Model<Api>, options: SimpleStreamOptions | undefined): number | undefined {\n\tconst retention =\n\t\toptions?.cacheRetention ??\n\t\t(getProviderEnvValue(\"PI_CACHE_RETENTION\", options?.env) === \"long\" ? \"long\" : \"short\");\n\tif (retention === \"none\") return undefined;\n\tconst seconds = model.promptCache?.[retention];\n\treturn seconds === undefined ? undefined : seconds * 1000;\n}\n\n/**\n * Whether replaying the request with a one-token output cap leaves its cache\n * entry untouched. Anthropic's budget-based thinking (Claude models without\n * adaptive thinking) derives `budget_tokens` from `max_tokens`; the replay\n * would get a different budget, which Anthropic keys the message cache on,\n * and the model could still think for thousands of tokens.\n */\nexport function isReplayable(model: Model<Api>, options: SimpleStreamOptions | undefined): boolean {\n\tif (!options?.reasoning || model.api !== \"anthropic-messages\") return true;\n\treturn (model as Model<\"anthropic-messages\">).compat?.forceAdaptiveThinking === true;\n}\n\n/** Prompt size of the most recent real request on the branch, as reported by the provider. */\nfunction lastPromptTokens(entries: SessionEntry[]): number {\n\tfor (let index = entries.length - 1; index >= 0; index--) {\n\t\tconst entry = entries[index];\n\t\tif (entry.type === \"message\" && entry.message.role === \"assistant\") {\n\t\t\tconst usage = entry.message.usage;\n\t\t\treturn usage.input + usage.cacheRead + usage.cacheWrite;\n\t\t}\n\t}\n\treturn 0;\n}\n\nfunction price(\n\tmodel: Model<Api>,\n\ttokens: Partial<Pick<Usage, \"input\" | \"output\" | \"cacheRead\" | \"cacheWrite\">>,\n): number {\n\tconst usage: Usage = {\n\t\tinput: 0,\n\t\toutput: 0,\n\t\tcacheRead: 0,\n\t\tcacheWrite: 0,\n\t\ttotalTokens: 0,\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t...tokens,\n\t};\n\treturn calculateCost(model, usage).total;\n}\n\nexport type CacheWarmingAction = \"warm\" | \"stop\";\n\n/** Inputs and outcome of one warm-or-stop decision, as shown by `/session`. */\nexport interface CacheWarmingDecision {\n\t/** \"streaming\" while the agent run that sent the request is still active. */\n\tphase: \"streaming\" | \"idle\";\n\t/** Price of this refresh: a cache read of the prompt plus one output token. */\n\twarmCost: number;\n\t/** Extra price of the next real request if the cache entry is lost. */\n\tmissCost: number;\n\t/** Estimated chance that a real request arrives before the entry expires. */\n\tcontinuationProbability: number;\n\t/** `continuationProbability * missCost - warmCost`. */\n\texpectedSavings: number;\n\t/** False when the prompt size or the model's prices are unknown. */\n\teconomicsAvailable: boolean;\n\t/** Pi's decision: \"warm\" when `expectedSavings` is at least $0.05. */\n\taction: CacheWarmingAction;\n}\n\n/**\n * Fired before each refresh with pi's decision filled in. Everything else an\n * extension might want (model, idle state, context size) is on the context.\n */\nexport interface CacheWarmingDecisionEvent\n\textends Pick<CacheWarmingDecision, \"warmCost\" | \"missCost\" | \"continuationProbability\" | \"action\"> {\n\ttype: \"cache_warming_decision\";\n}\n\nexport interface CacheWarmingDecisionEventResult {\n\t/** Override whether this refresh is sent. \"stop\" ends warming until the next real request. */\n\taction?: CacheWarmingAction;\n}\n\nexport interface CacheWarmingStatus {\n\t/** \"scheduled\": a refresh timer is armed; \"refreshing\": a warm request is in flight. */\n\tstate: \"inactive\" | \"scheduled\" | \"refreshing\";\n\t/** Why nothing is scheduled. */\n\treason?: string;\n\tnextWarmAt?: number;\n\t/** The pending decision, or the decision that stopped warming. */\n\tdecision?: CacheWarmingDecision;\n\t/** True when an extension changed `decision.action`. */\n\textensionOverride?: boolean;\n}\n\n/** The request whose prompt cache entry should be kept warm, exactly as it was sent. */\nexport interface CacheWarmRequest {\n\tmodel: Model<Api>;\n\tcontext: Context;\n\toptions: ModelsSimpleStreamOptions;\n}\n\ninterface ActiveRun extends CacheWarmRequest {\n\t/** False once the session's model or messages no longer match the request. */\n\tisCurrent: () => boolean;\n\tttlMs: number;\n\tdelayMs: number;\n\t/** Latest safe time to send this refresh, leaving half the original expiry margin. */\n\trefreshDeadlineAt: number;\n\tstartedAt: number;\n\tcontroller: AbortController;\n\tphase: \"streaming\" | \"idle\";\n\tnextWarmAt: number;\n\t/** Set while a refresh that an extension forced is in flight. */\n\textensionOverride: boolean;\n\ttimer?: ReturnType<typeof setTimeout>;\n}\n\n/**\n * Keeps one prompt cache entry alive by re-sending its request with a\n * one-token output cap before the entry expires. `start` replaces any\n * previous run; warm requests never extend the fixed safety windows.\n */\nexport class CacheWarmer {\n\tprivate run?: ActiveRun;\n\tprivate inactive: CacheWarmingStatus;\n\tprivate readonly models: Pick<ModelRuntime, \"streamSimple\">;\n\tprivate readonly sessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">;\n\tprivate readonly getMode: () => CacheWarmingMode;\n\t/** Lets extensions override `event.action`; failures fall back to pi's decision. */\n\tprivate readonly decide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction>;\n\t/** Called with the persisted usage entry after each successful refresh. */\n\tonWarmed?: (entry: UsageEntry) => void;\n\n\tconstructor(\n\t\tmodels: Pick<ModelRuntime, \"streamSimple\">,\n\t\tsessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">,\n\t\tgetMode: () => CacheWarmingMode,\n\t\tdecide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction> = async (event) => event.action,\n\t) {\n\t\tthis.models = models;\n\t\tthis.sessionManager = sessionManager;\n\t\tthis.getMode = getMode;\n\t\tthis.decide = decide;\n\t\tthis.inactive = { state: \"inactive\", reason: \"waiting for first request\" };\n\t}\n\n\tget status(): CacheWarmingStatus {\n\t\tif (this.getMode() === \"off\") return { state: \"inactive\", reason: \"cache warming disabled\" };\n\t\tconst run = this.run;\n\t\tif (!run) return this.inactive;\n\t\tif (!run.isCurrent()) return { state: \"inactive\", reason: \"conversation context changed\" };\n\t\tconst decision = this.evaluate(run);\n\t\tconst refreshing = run.timer === undefined;\n\t\tif (!decision.economicsAvailable && !refreshing) {\n\t\t\treturn { state: \"inactive\", reason: \"cache economics unavailable\" };\n\t\t}\n\t\treturn {\n\t\t\tstate: refreshing ? \"refreshing\" : \"scheduled\",\n\t\t\tnextWarmAt: run.nextWarmAt,\n\t\t\tdecision,\n\t\t\textensionOverride: run.extensionOverride,\n\t\t};\n\t}\n\n\t/** Keep the prompt cache entry written by `request` warm while `isCurrent` holds. */\n\tstart(request: CacheWarmRequest, isCurrent: () => boolean): void {\n\t\tthis.clearRun();\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") {\n\t\t\tthis.stop(\"cache warming disabled\");\n\t\t\treturn;\n\t\t}\n\t\tif (!isReplayable(request.model, request.options)) {\n\t\t\tthis.stop(\"request cannot be replayed safely\");\n\t\t\treturn;\n\t\t}\n\t\tconst ttlMs = getPromptCacheTtlMs(request.model, request.options);\n\t\tif (ttlMs === undefined) {\n\t\t\tthis.stop(\n\t\t\t\trequest.options.cacheRetention === \"none\"\n\t\t\t\t\t? \"request disabled prompt caching\"\n\t\t\t\t\t: \"cache lifetime unavailable\",\n\t\t\t);\n\t\t\treturn;\n\t\t}\n\t\tconst delayMs = getCacheWarmingDelayMs(ttlMs);\n\t\tif (delayMs === undefined) {\n\t\t\tthis.stop(\"cache lifetime unavailable\");\n\t\t\treturn;\n\t\t}\n\t\tthis.run = {\n\t\t\t...request,\n\t\t\tisCurrent,\n\t\t\tttlMs,\n\t\t\tdelayMs,\n\t\t\trefreshDeadlineAt: 0,\n\t\t\tstartedAt: Date.now(),\n\t\t\tcontroller: new AbortController(),\n\t\t\tphase: \"streaming\",\n\t\t\tnextWarmAt: 0,\n\t\t\textensionOverride: false,\n\t\t};\n\t\tthis.schedule(this.run);\n\t}\n\n\tonAgentSettled(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tif (this.getMode() === \"streaming\") {\n\t\t\tthis.stop(\"agent run settled\");\n\t\t\treturn;\n\t\t}\n\t\trun.phase = \"idle\";\n\t\tconst deadline = run.startedAt + MAX_IDLE_WARMING_AGE_MS;\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(\"30-minute idle safety limit reached\");\n\t\t}\n\t}\n\n\t/** Reconcile an active run after the persisted warming mode changes. */\n\tonModeChanged(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tconst reason = this.getModeStopReason(run);\n\t\tif (reason) this.stop(reason);\n\t}\n\n\tcancel(): void {\n\t\tthis.stop(\"inactive\");\n\t}\n\n\tprivate clearRun(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tthis.run = undefined;\n\t\tif (run.timer) clearTimeout(run.timer);\n\t\trun.controller.abort();\n\t}\n\n\tprivate stop(reason: string, stopped?: Pick<CacheWarmingStatus, \"decision\" | \"extensionOverride\">): void {\n\t\tthis.clearRun();\n\t\tthis.inactive = { state: \"inactive\", reason, ...stopped };\n\t}\n\n\tprivate schedule(run: ActiveRun): void {\n\t\trun.extensionOverride = false;\n\t\trun.nextWarmAt = Date.now() + run.delayMs;\n\t\t// A timer can run late after sleep or event-loop blockage. Keep half of\n\t\t// the planned pre-expiry margin for that delay and request dispatch; a\n\t\t// late refresh is likely a full-price cache write, not a cache warm.\n\t\trun.refreshDeadlineAt = run.nextWarmAt + Math.floor((run.ttlMs - run.delayMs) / 2);\n\t\tconst deadline = run.startedAt + (run.phase === \"idle\" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(run.phase === \"idle\" ? \"30-minute idle safety limit reached\" : \"one-hour safety limit reached\");\n\t\t\treturn;\n\t\t}\n\t\trun.timer = setTimeout(() => void this.refresh(run), Math.max(0, run.nextWarmAt - Date.now()));\n\t\trun.timer.unref?.();\n\t}\n\n\tprivate async refresh(run: ActiveRun): Promise<void> {\n\t\trun.timer = undefined;\n\t\tif (!this.validateRun(run)) return;\n\t\tif (this.refreshDeadlineMissed(run)) return;\n\t\tconst decision = this.evaluate(run);\n\t\tconst { warmCost, missCost, continuationProbability } = decision;\n\t\tlet action = decision.action;\n\t\ttry {\n\t\t\taction = await this.decide({\n\t\t\t\ttype: \"cache_warming_decision\",\n\t\t\t\twarmCost,\n\t\t\t\tmissCost,\n\t\t\t\tcontinuationProbability,\n\t\t\t\taction,\n\t\t\t});\n\t\t} catch {\n\t\t\t// Extension failures fall back to pi's own decision.\n\t\t}\n\t\tif (!this.validateRun(run) || this.refreshDeadlineMissed(run)) return;\n\t\tconst extensionOverride = action !== decision.action;\n\t\tif (action === \"stop\") {\n\t\t\tconst reason = extensionOverride\n\t\t\t\t? \"stopped by extension\"\n\t\t\t\t: decision.economicsAvailable\n\t\t\t\t\t? \"expected savings below threshold\"\n\t\t\t\t\t: \"cache economics unavailable\";\n\t\t\tthis.stop(reason, { decision, extensionOverride });\n\t\t\treturn;\n\t\t}\n\n\t\trun.extensionOverride = extensionOverride;\n\t\ttry {\n\t\t\tconst message = await this.models\n\t\t\t\t.streamSimple(run.model, run.context, {\n\t\t\t\t\t...run.options,\n\t\t\t\t\tmaxTokens: 1,\n\t\t\t\t\tmaxRetries: 0,\n\t\t\t\t\tsignal: run.controller.signal,\n\t\t\t\t})\n\t\t\t\t.result();\n\t\t\tif (!this.validateRun(run)) return;\n\t\t\tif (message.stopReason !== \"error\" && message.stopReason !== \"aborted\") {\n\t\t\t\tconst entry = this.sessionManager.appendUsage(\n\t\t\t\t\t\"cache_warm\",\n\t\t\t\t\tmessage.provider,\n\t\t\t\t\tmessage.responseModel ?? message.model,\n\t\t\t\t\tmessage.usage,\n\t\t\t\t\textensionOverride ? \"extension override\" : undefined,\n\t\t\t\t);\n\t\t\t\tthis.onWarmed?.(entry);\n\t\t\t}\n\t\t} catch {\n\t\t\t// Cache warming is best-effort and must not affect the active agent run.\n\t\t}\n\t\tif (this.run === run) this.schedule(run);\n\t}\n\n\tprivate refreshDeadlineMissed(run: ActiveRun): boolean {\n\t\tif (Date.now() <= run.refreshDeadlineAt) return false;\n\t\tthis.stop(\"cache refresh deadline missed\");\n\t\treturn true;\n\t}\n\n\tprivate validateRun(run: ActiveRun): boolean {\n\t\tif (this.run !== run) return false;\n\t\tconst reason = this.getModeStopReason(run) ?? (!run.isCurrent() ? \"conversation context changed\" : undefined);\n\t\tif (!reason) return true;\n\t\tthis.stop(reason);\n\t\treturn false;\n\t}\n\n\tprivate getModeStopReason(run: ActiveRun): string | undefined {\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") return \"cache warming disabled\";\n\t\tif (mode === \"streaming\" && run.phase === \"idle\") return \"agent run settled\";\n\t\treturn undefined;\n\t}\n\n\tprivate evaluate(run: ActiveRun): CacheWarmingDecision {\n\t\tconst model = run.model;\n\t\tconst promptTokens = lastPromptTokens(this.sessionManager.getBranch());\n\t\tconst cacheHitCost = price(model, { cacheRead: promptTokens });\n\t\tconst cacheMissCost = price(\n\t\t\tmodel,\n\t\t\tmodel.cost.cacheWrite > 0 ? { cacheWrite: promptTokens } : { input: promptTokens },\n\t\t);\n\t\tconst warmCost = price(model, { cacheRead: promptTokens, output: 1 });\n\t\tconst missCost = Math.max(0, cacheMissCost - cacheHitCost);\n\t\tconst continuationProbability = run.phase === \"idle\" ? IDLE_CONTINUATION_PROBABILITY : 1;\n\t\tconst economicsAvailable = promptTokens > 0 && (cacheHitCost > 0 || cacheMissCost > 0);\n\t\tconst expectedSavings = continuationProbability * missCost - warmCost;\n\t\treturn {\n\t\t\tphase: run.phase,\n\t\t\twarmCost,\n\t\t\tmissCost,\n\t\t\tcontinuationProbability,\n\t\t\texpectedSavings,\n\t\t\teconomicsAvailable,\n\t\t\taction: expectedSavings >= CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS ? \"warm\" : \"stop\",\n\t\t};\n\t}\n}\n\nfunction formatDollars(value: number): string {\n\treturn value < 0 ? `-$${Math.abs(value).toFixed(3)}` : `$${value.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingEconomics(decision: CacheWarmingDecision): string {\n\tif (!decision.economicsAvailable) return \"cache economics unavailable\";\n\tconst probability = Math.round(decision.continuationProbability * 100);\n\tconst probabilityText =\n\t\tdecision.phase === \"streaming\"\n\t\t\t? `${probability}% continuation probability while agent is running`\n\t\t\t: `${probability}% continuation probability`;\n\tconst comparison = decision.action === \"warm\" ? \">=\" : \"<\";\n\treturn `${probabilityText}, expected savings ${formatDollars(decision.expectedSavings)} ${comparison} $${CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingDecisionTime(nextWarmAt: number | undefined, now: number): string {\n\tif (nextWarmAt === undefined || nextWarmAt <= now) return \"Decision now\";\n\tlet remainingSeconds = Math.ceil((nextWarmAt - now) / 1000);\n\tconst hours = Math.floor(remainingSeconds / 3600);\n\tremainingSeconds %= 3600;\n\tconst minutes = Math.floor(remainingSeconds / 60);\n\tconst seconds = remainingSeconds % 60;\n\tconst parts: string[] = [];\n\tif (hours > 0) parts.push(`${hours}h`);\n\tif (minutes > 0) parts.push(`${minutes}m`);\n\tif (seconds > 0 || parts.length === 0) parts.push(`${seconds}s`);\n\treturn `Decision in ${parts.join(\" \")}`;\n}\n\n/** One-line status for `/session`. */\nexport function formatCacheWarmingStatus(status: CacheWarmingStatus, now = Date.now()): string {\n\tconst decision = status.decision;\n\t// A decision is attached once pi (or an extension) acted on it; \"inactive\"\n\t// without one never got that far.\n\tif (!decision || (status.state === \"inactive\" && !decision.economicsAvailable && !status.extensionOverride)) {\n\t\treturn `Inactive (${status.reason ?? \"unknown reason\"})`;\n\t}\n\tconst details = status.extensionOverride\n\t\t? `extension override, ${formatCacheWarmingEconomics(decision)}`\n\t\t: `${formatCacheWarmingEconomics(decision)} -> ${decision.action}`;\n\tif (status.state === \"inactive\") return `Stopped (${details})`;\n\tif (status.state === \"refreshing\") return `Warming cache (${details})`;\n\treturn `${formatCacheWarmingDecisionTime(status.nextWarmAt, now)} (${details})`;\n}\n\n/** One-line transcript text for persisted cache-warming usage. */\nexport function formatCacheWarmingUsage(entry: UsageEntry): string {\n\tconst note = entry.note ? ` (${entry.note})` : \"\";\n\tconst cost = entry.usage.cost.total.toFixed(6).replace(/(\\.\\d{3}\\d*?)0+$/, \"$1\");\n\treturn `Cache warmed${note}: $${cost}`;\n}\n"]}
@@ -135,7 +135,9 @@ export class CacheWarmer {
135
135
  this.run = {
136
136
  ...request,
137
137
  isCurrent,
138
+ ttlMs,
138
139
  delayMs,
140
+ refreshDeadlineAt: 0,
139
141
  startedAt: Date.now(),
140
142
  controller: new AbortController(),
141
143
  phase: "streaming",
@@ -186,6 +188,10 @@ export class CacheWarmer {
186
188
  schedule(run) {
187
189
  run.extensionOverride = false;
188
190
  run.nextWarmAt = Date.now() + run.delayMs;
191
+ // A timer can run late after sleep or event-loop blockage. Keep half of
192
+ // the planned pre-expiry margin for that delay and request dispatch; a
193
+ // late refresh is likely a full-price cache write, not a cache warm.
194
+ run.refreshDeadlineAt = run.nextWarmAt + Math.floor((run.ttlMs - run.delayMs) / 2);
189
195
  const deadline = run.startedAt + (run.phase === "idle" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);
190
196
  if (run.nextWarmAt > deadline || Date.now() >= deadline) {
191
197
  this.stop(run.phase === "idle" ? "30-minute idle safety limit reached" : "one-hour safety limit reached");
@@ -198,6 +204,8 @@ export class CacheWarmer {
198
204
  run.timer = undefined;
199
205
  if (!this.validateRun(run))
200
206
  return;
207
+ if (this.refreshDeadlineMissed(run))
208
+ return;
201
209
  const decision = this.evaluate(run);
202
210
  const { warmCost, missCost, continuationProbability } = decision;
203
211
  let action = decision.action;
@@ -213,7 +221,7 @@ export class CacheWarmer {
213
221
  catch {
214
222
  // Extension failures fall back to pi's own decision.
215
223
  }
216
- if (!this.validateRun(run))
224
+ if (!this.validateRun(run) || this.refreshDeadlineMissed(run))
217
225
  return;
218
226
  const extensionOverride = action !== decision.action;
219
227
  if (action === "stop") {
@@ -248,6 +256,12 @@ export class CacheWarmer {
248
256
  if (this.run === run)
249
257
  this.schedule(run);
250
258
  }
259
+ refreshDeadlineMissed(run) {
260
+ if (Date.now() <= run.refreshDeadlineAt)
261
+ return false;
262
+ this.stop("cache refresh deadline missed");
263
+ return true;
264
+ }
251
265
  validateRun(run) {
252
266
  if (this.run !== run)
253
267
  return false;
@@ -1 +1 @@
1
- {"version":3,"file":"cache-warmer.js","sourceRoot":"","sources":["../../src/core/cache-warmer.ts"],"names":[],"mappings":"AAAA,OAAO,EAGN,aAAa,GAKb,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EAAE,mBAAmB,EAAE,MAAM,0CAA0C,CAAC;AAK/E,+FAA+F;AAC/F,MAAM,kBAAkB,GAAG,EAAE,GAAG,MAAM,CAAC;AACvC,wGAAwG;AACxG,MAAM,uBAAuB,GAAG,EAAE,GAAG,MAAM,CAAC;AAC5C,qFAAqF;AACrF,MAAM,sCAAsC,GAAG,IAAI,CAAC;AACpD;;;;GAIG;AACH,MAAM,6BAA6B,GAAG,IAAI,CAAC;AAE3C,iFAAiF;AACjF,MAAM,UAAU,sBAAsB,CAAC,KAAa,EAAsB;IACzE,IAAI,KAAK,IAAI,MAAM;QAAE,OAAO,SAAS,CAAC;IACtC,OAAO,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,GAAG,EAAE,KAAK,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;AAAA,CACtE;AAED;;;;GAIG;AACH,MAAM,UAAU,mBAAmB,CAAC,KAAiB,EAAE,OAAwC,EAAsB;IACpH,MAAM,SAAS,GACd,OAAO,EAAE,cAAc;QACvB,CAAC,mBAAmB,CAAC,oBAAoB,EAAE,OAAO,EAAE,GAAG,CAAC,KAAK,MAAM,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC;IACzF,IAAI,SAAS,KAAK,MAAM;QAAE,OAAO,SAAS,CAAC;IAC3C,MAAM,OAAO,GAAG,KAAK,CAAC,WAAW,EAAE,CAAC,SAAS,CAAC,CAAC;IAC/C,OAAO,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,OAAO,GAAG,IAAI,CAAC;AAAA,CAC1D;AAED;;;;;;GAMG;AACH,MAAM,UAAU,YAAY,CAAC,KAAiB,EAAE,OAAwC,EAAW;IAClG,IAAI,CAAC,OAAO,EAAE,SAAS,IAAI,KAAK,CAAC,GAAG,KAAK,oBAAoB;QAAE,OAAO,IAAI,CAAC;IAC3E,OAAQ,KAAqC,CAAC,MAAM,EAAE,qBAAqB,KAAK,IAAI,CAAC;AAAA,CACrF;AAED,8FAA8F;AAC9F,SAAS,gBAAgB,CAAC,OAAuB,EAAU;IAC1D,KAAK,IAAI,KAAK,GAAG,OAAO,CAAC,MAAM,GAAG,CAAC,EAAE,KAAK,IAAI,CAAC,EAAE,KAAK,EAAE,EAAE,CAAC;QAC1D,MAAM,KAAK,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC;QAC7B,IAAI,KAAK,CAAC,IAAI,KAAK,SAAS,IAAI,KAAK,CAAC,OAAO,CAAC,IAAI,KAAK,WAAW,EAAE,CAAC;YACpE,MAAM,KAAK,GAAG,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YAClC,OAAO,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,SAAS,GAAG,KAAK,CAAC,UAAU,CAAC;QACzD,CAAC;IACF,CAAC;IACD,OAAO,CAAC,CAAC;AAAA,CACT;AAED,SAAS,KAAK,CACb,KAAiB,EACjB,MAA6E,EACpE;IACT,MAAM,KAAK,GAAU;QACpB,KAAK,EAAE,CAAC;QACR,MAAM,EAAE,CAAC;QACT,SAAS,EAAE,CAAC;QACZ,UAAU,EAAE,CAAC;QACb,WAAW,EAAE,CAAC;QACd,IAAI,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE;QACpE,GAAG,MAAM;KACT,CAAC;IACF,OAAO,aAAa,CAAC,KAAK,EAAE,KAAK,CAAC,CAAC,KAAK,CAAC;AAAA,CACzC;AAoED;;;;GAIG;AACH,MAAM,OAAO,WAAW;IACf,GAAG,CAAa;IAChB,QAAQ,CAAqB;IACpB,MAAM,CAAqC;IAC3C,cAAc,CAAoD;IAClE,OAAO,CAAyB;IACjD,oFAAoF;IACnE,MAAM,CAAoE;IAC3F,2EAA2E;IAC3E,QAAQ,CAA+B;IAEvC,YACC,MAA0C,EAC1C,cAAiE,EACjE,OAA+B,EAC/B,MAAM,GAAsE,KAAK,EAAE,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,MAAM,EACxG;QACD,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC;QACrB,IAAI,CAAC,cAAc,GAAG,cAAc,CAAC;QACrC,IAAI,CAAC,OAAO,GAAG,OAAO,CAAC;QACvB,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC;QACrB,IAAI,CAAC,QAAQ,GAAG,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,2BAA2B,EAAE,CAAC;IAAA,CAC3E;IAED,IAAI,MAAM,GAAuB;QAChC,IAAI,IAAI,CAAC,OAAO,EAAE,KAAK,KAAK;YAAE,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,wBAAwB,EAAE,CAAC;QAC7F,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO,IAAI,CAAC,QAAQ,CAAC;QAC/B,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE;YAAE,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,8BAA8B,EAAE,CAAC;QAC3F,MAAM,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;QACpC,MAAM,UAAU,GAAG,GAAG,CAAC,KAAK,KAAK,SAAS,CAAC;QAC3C,IAAI,CAAC,QAAQ,CAAC,kBAAkB,IAAI,CAAC,UAAU,EAAE,CAAC;YACjD,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,6BAA6B,EAAE,CAAC;QACrE,CAAC;QACD,OAAO;YACN,KAAK,EAAE,UAAU,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,WAAW;YAC9C,UAAU,EAAE,GAAG,CAAC,UAAU;YAC1B,QAAQ;YACR,iBAAiB,EAAE,GAAG,CAAC,iBAAiB;SACxC,CAAC;IAAA,CACF;IAED,qFAAqF;IACrF,KAAK,CAAC,OAAyB,EAAE,SAAwB,EAAQ;QAChE,IAAI,CAAC,QAAQ,EAAE,CAAC;QAChB,MAAM,IAAI,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC;QAC5B,IAAI,IAAI,KAAK,KAAK,EAAE,CAAC;YACpB,IAAI,CAAC,IAAI,CAAC,wBAAwB,CAAC,CAAC;YACpC,OAAO;QACR,CAAC;QACD,IAAI,CAAC,YAAY,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,EAAE,CAAC;YACnD,IAAI,CAAC,IAAI,CAAC,mCAAmC,CAAC,CAAC;YAC/C,OAAO;QACR,CAAC;QACD,MAAM,KAAK,GAAG,mBAAmB,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,CAAC;QAClE,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;YACzB,IAAI,CAAC,IAAI,CACR,OAAO,CAAC,OAAO,CAAC,cAAc,KAAK,MAAM;gBACxC,CAAC,CAAC,iCAAiC;gBACnC,CAAC,CAAC,4BAA4B,CAC/B,CAAC;YACF,OAAO;QACR,CAAC;QACD,MAAM,OAAO,GAAG,sBAAsB,CAAC,KAAK,CAAC,CAAC;QAC9C,IAAI,OAAO,KAAK,SAAS,EAAE,CAAC;YAC3B,IAAI,CAAC,IAAI,CAAC,4BAA4B,CAAC,CAAC;YACxC,OAAO;QACR,CAAC;QACD,IAAI,CAAC,GAAG,GAAG;YACV,GAAG,OAAO;YACV,SAAS;YACT,OAAO;YACP,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE;YACrB,UAAU,EAAE,IAAI,eAAe,EAAE;YACjC,KAAK,EAAE,WAAW;YAClB,UAAU,EAAE,CAAC;YACb,iBAAiB,EAAE,KAAK;SACxB,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;IAAA,CACxB;IAED,cAAc,GAAS;QACtB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,IAAI,IAAI,CAAC,OAAO,EAAE,KAAK,WAAW,EAAE,CAAC;YACpC,IAAI,CAAC,IAAI,CAAC,mBAAmB,CAAC,CAAC;YAC/B,OAAO;QACR,CAAC;QACD,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC;QACnB,MAAM,QAAQ,GAAG,GAAG,CAAC,SAAS,GAAG,uBAAuB,CAAC;QACzD,IAAI,GAAG,CAAC,UAAU,GAAG,QAAQ,IAAI,IAAI,CAAC,GAAG,EAAE,IAAI,QAAQ,EAAE,CAAC;YACzD,IAAI,CAAC,IAAI,CAAC,qCAAqC,CAAC,CAAC;QAClD,CAAC;IAAA,CACD;IAED,wEAAwE;IACxE,aAAa,GAAS;QACrB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,MAAM,MAAM,GAAG,IAAI,CAAC,iBAAiB,CAAC,GAAG,CAAC,CAAC;QAC3C,IAAI,MAAM;YAAE,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;IAAA,CAC9B;IAED,MAAM,GAAS;QACd,IAAI,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC;IAAA,CACtB;IAEO,QAAQ,GAAS;QACxB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,IAAI,CAAC,GAAG,GAAG,SAAS,CAAC;QACrB,IAAI,GAAG,CAAC,KAAK;YAAE,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;QACvC,GAAG,CAAC,UAAU,CAAC,KAAK,EAAE,CAAC;IAAA,CACvB;IAEO,IAAI,CAAC,MAAc,EAAE,OAAoE,EAAQ;QACxG,IAAI,CAAC,QAAQ,EAAE,CAAC;QAChB,IAAI,CAAC,QAAQ,GAAG,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,GAAG,OAAO,EAAE,CAAC;IAAA,CAC1D;IAEO,QAAQ,CAAC,GAAc,EAAQ;QACtC,GAAG,CAAC,iBAAiB,GAAG,KAAK,CAAC;QAC9B,GAAG,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,GAAG,CAAC,OAAO,CAAC;QAC1C,MAAM,QAAQ,GAAG,GAAG,CAAC,SAAS,GAAG,CAAC,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,uBAAuB,CAAC,CAAC,CAAC,kBAAkB,CAAC,CAAC;QACvG,IAAI,GAAG,CAAC,UAAU,GAAG,QAAQ,IAAI,IAAI,CAAC,GAAG,EAAE,IAAI,QAAQ,EAAE,CAAC;YACzD,IAAI,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,qCAAqC,CAAC,CAAC,CAAC,+BAA+B,CAAC,CAAC;YAC1G,OAAO;QACR,CAAC;QACD,GAAG,CAAC,KAAK,GAAG,UAAU,CAAC,GAAG,EAAE,CAAC,KAAK,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,GAAG,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC;QAC/F,GAAG,CAAC,KAAK,CAAC,KAAK,EAAE,EAAE,CAAC;IAAA,CACpB;IAEO,KAAK,CAAC,OAAO,CAAC,GAAc,EAAiB;QACpD,GAAG,CAAC,KAAK,GAAG,SAAS,CAAC;QACtB,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC;YAAE,OAAO;QACnC,MAAM,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;QACpC,MAAM,EAAE,QAAQ,EAAE,QAAQ,EAAE,uBAAuB,EAAE,GAAG,QAAQ,CAAC;QACjE,IAAI,MAAM,GAAG,QAAQ,CAAC,MAAM,CAAC;QAC7B,IAAI,CAAC;YACJ,MAAM,GAAG,MAAM,IAAI,CAAC,MAAM,CAAC;gBAC1B,IAAI,EAAE,wBAAwB;gBAC9B,QAAQ;gBACR,QAAQ;gBACR,uBAAuB;gBACvB,MAAM;aACN,CAAC,CAAC;QACJ,CAAC;QAAC,MAAM,CAAC;YACR,qDAAqD;QACtD,CAAC;QACD,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC;YAAE,OAAO;QACnC,MAAM,iBAAiB,GAAG,MAAM,KAAK,QAAQ,CAAC,MAAM,CAAC;QACrD,IAAI,MAAM,KAAK,MAAM,EAAE,CAAC;YACvB,MAAM,MAAM,GAAG,iBAAiB;gBAC/B,CAAC,CAAC,sBAAsB;gBACxB,CAAC,CAAC,QAAQ,CAAC,kBAAkB;oBAC5B,CAAC,CAAC,kCAAkC;oBACpC,CAAC,CAAC,6BAA6B,CAAC;YAClC,IAAI,CAAC,IAAI,CAAC,MAAM,EAAE,EAAE,QAAQ,EAAE,iBAAiB,EAAE,CAAC,CAAC;YACnD,OAAO;QACR,CAAC;QAED,GAAG,CAAC,iBAAiB,GAAG,iBAAiB,CAAC;QAC1C,IAAI,CAAC;YACJ,MAAM,OAAO,GAAG,MAAM,IAAI,CAAC,MAAM;iBAC/B,YAAY,CAAC,GAAG,CAAC,KAAK,EAAE,GAAG,CAAC,OAAO,EAAE;gBACrC,GAAG,GAAG,CAAC,OAAO;gBACd,SAAS,EAAE,CAAC;gBACZ,UAAU,EAAE,CAAC;gBACb,MAAM,EAAE,GAAG,CAAC,UAAU,CAAC,MAAM;aAC7B,CAAC;iBACD,MAAM,EAAE,CAAC;YACX,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC;gBAAE,OAAO;YACnC,IAAI,OAAO,CAAC,UAAU,KAAK,OAAO,IAAI,OAAO,CAAC,UAAU,KAAK,SAAS,EAAE,CAAC;gBACxE,MAAM,KAAK,GAAG,IAAI,CAAC,cAAc,CAAC,WAAW,CAC5C,YAAY,EACZ,OAAO,CAAC,QAAQ,EAChB,OAAO,CAAC,aAAa,IAAI,OAAO,CAAC,KAAK,EACtC,OAAO,CAAC,KAAK,EACb,iBAAiB,CAAC,CAAC,CAAC,oBAAoB,CAAC,CAAC,CAAC,SAAS,CACpD,CAAC;gBACF,IAAI,CAAC,QAAQ,EAAE,CAAC,KAAK,CAAC,CAAC;YACxB,CAAC;QACF,CAAC;QAAC,MAAM,CAAC;YACR,yEAAyE;QAC1E,CAAC;QACD,IAAI,IAAI,CAAC,GAAG,KAAK,GAAG;YAAE,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;IAAA,CACzC;IAEO,WAAW,CAAC,GAAc,EAAW;QAC5C,IAAI,IAAI,CAAC,GAAG,KAAK,GAAG;YAAE,OAAO,KAAK,CAAC;QACnC,MAAM,MAAM,GAAG,IAAI,CAAC,iBAAiB,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,SAAS,EAAE,CAAC,CAAC,CAAC,8BAA8B,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC;QAC9G,IAAI,CAAC,MAAM;YAAE,OAAO,IAAI,CAAC;QACzB,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;QAClB,OAAO,KAAK,CAAC;IAAA,CACb;IAEO,iBAAiB,CAAC,GAAc,EAAsB;QAC7D,MAAM,IAAI,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC;QAC5B,IAAI,IAAI,KAAK,KAAK;YAAE,OAAO,wBAAwB,CAAC;QACpD,IAAI,IAAI,KAAK,WAAW,IAAI,GAAG,CAAC,KAAK,KAAK,MAAM;YAAE,OAAO,mBAAmB,CAAC;QAC7E,OAAO,SAAS,CAAC;IAAA,CACjB;IAEO,QAAQ,CAAC,GAAc,EAAwB;QACtD,MAAM,KAAK,GAAG,GAAG,CAAC,KAAK,CAAC;QACxB,MAAM,YAAY,GAAG,gBAAgB,CAAC,IAAI,CAAC,cAAc,CAAC,SAAS,EAAE,CAAC,CAAC;QACvE,MAAM,YAAY,GAAG,KAAK,CAAC,KAAK,EAAE,EAAE,SAAS,EAAE,YAAY,EAAE,CAAC,CAAC;QAC/D,MAAM,aAAa,GAAG,KAAK,CAC1B,KAAK,EACL,KAAK,CAAC,IAAI,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,EAAE,UAAU,EAAE,YAAY,EAAE,CAAC,CAAC,CAAC,EAAE,KAAK,EAAE,YAAY,EAAE,CAClF,CAAC;QACF,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAK,EAAE,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,CAAC,EAAE,CAAC,CAAC;QACtE,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,aAAa,GAAG,YAAY,CAAC,CAAC;QAC3D,MAAM,uBAAuB,GAAG,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,6BAA6B,CAAC,CAAC,CAAC,CAAC,CAAC;QACzF,MAAM,kBAAkB,GAAG,YAAY,GAAG,CAAC,IAAI,CAAC,YAAY,GAAG,CAAC,IAAI,aAAa,GAAG,CAAC,CAAC,CAAC;QACvF,MAAM,eAAe,GAAG,uBAAuB,GAAG,QAAQ,GAAG,QAAQ,CAAC;QACtE,OAAO;YACN,KAAK,EAAE,GAAG,CAAC,KAAK;YAChB,QAAQ;YACR,QAAQ;YACR,uBAAuB;YACvB,eAAe;YACf,kBAAkB;YAClB,MAAM,EAAE,eAAe,IAAI,sCAAsC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,MAAM;SACnF,CAAC;IAAA,CACF;CACD;AAED,SAAS,aAAa,CAAC,KAAa,EAAU;IAC7C,OAAO,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC;AAAA,CAC9E;AAED,SAAS,2BAA2B,CAAC,QAA8B,EAAU;IAC5E,IAAI,CAAC,QAAQ,CAAC,kBAAkB;QAAE,OAAO,6BAA6B,CAAC;IACvE,MAAM,WAAW,GAAG,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,uBAAuB,GAAG,GAAG,CAAC,CAAC;IACvE,MAAM,eAAe,GACpB,QAAQ,CAAC,KAAK,KAAK,WAAW;QAC7B,CAAC,CAAC,GAAG,WAAW,mDAAmD;QACnE,CAAC,CAAC,GAAG,WAAW,4BAA4B,CAAC;IAC/C,MAAM,UAAU,GAAG,QAAQ,CAAC,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,GAAG,CAAC;IAC3D,OAAO,GAAG,eAAe,sBAAsB,aAAa,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,UAAU,KAAK,sCAAsC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC;AAAA,CAC7J;AAED,SAAS,8BAA8B,CAAC,UAA8B,EAAE,GAAW,EAAU;IAC5F,IAAI,UAAU,KAAK,SAAS,IAAI,UAAU,IAAI,GAAG;QAAE,OAAO,cAAc,CAAC;IACzE,IAAI,gBAAgB,GAAG,IAAI,CAAC,IAAI,CAAC,CAAC,UAAU,GAAG,GAAG,CAAC,GAAG,IAAI,CAAC,CAAC;IAC5D,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,gBAAgB,GAAG,IAAI,CAAC,CAAC;IAClD,gBAAgB,IAAI,IAAI,CAAC;IACzB,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,gBAAgB,GAAG,EAAE,CAAC,CAAC;IAClD,MAAM,OAAO,GAAG,gBAAgB,GAAG,EAAE,CAAC;IACtC,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,KAAK,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,KAAK,GAAG,CAAC,CAAC;IACvC,IAAI,OAAO,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,OAAO,GAAG,CAAC,CAAC;IAC3C,IAAI,OAAO,GAAG,CAAC,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,OAAO,GAAG,CAAC,CAAC;IACjE,OAAO,eAAe,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC;AAAA,CACxC;AAED,sCAAsC;AACtC,MAAM,UAAU,wBAAwB,CAAC,MAA0B,EAAE,GAAG,GAAG,IAAI,CAAC,GAAG,EAAE,EAAU;IAC9F,MAAM,QAAQ,GAAG,MAAM,CAAC,QAAQ,CAAC;IACjC,2EAA2E;IAC3E,kCAAkC;IAClC,IAAI,CAAC,QAAQ,IAAI,CAAC,MAAM,CAAC,KAAK,KAAK,UAAU,IAAI,CAAC,QAAQ,CAAC,kBAAkB,IAAI,CAAC,MAAM,CAAC,iBAAiB,CAAC,EAAE,CAAC;QAC7G,OAAO,aAAa,MAAM,CAAC,MAAM,IAAI,gBAAgB,GAAG,CAAC;IAC1D,CAAC;IACD,MAAM,OAAO,GAAG,MAAM,CAAC,iBAAiB;QACvC,CAAC,CAAC,uBAAuB,2BAA2B,CAAC,QAAQ,CAAC,EAAE;QAChE,CAAC,CAAC,GAAG,2BAA2B,CAAC,QAAQ,CAAC,OAAO,QAAQ,CAAC,MAAM,EAAE,CAAC;IACpE,IAAI,MAAM,CAAC,KAAK,KAAK,UAAU;QAAE,OAAO,YAAY,OAAO,GAAG,CAAC;IAC/D,IAAI,MAAM,CAAC,KAAK,KAAK,YAAY;QAAE,OAAO,kBAAkB,OAAO,GAAG,CAAC;IACvE,OAAO,GAAG,8BAA8B,CAAC,MAAM,CAAC,UAAU,EAAE,GAAG,CAAC,KAAK,OAAO,GAAG,CAAC;AAAA,CAChF;AAED,kEAAkE;AAClE,MAAM,UAAU,uBAAuB,CAAC,KAAiB,EAAU;IAClE,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,KAAK,KAAK,CAAC,IAAI,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;IAClD,MAAM,IAAI,GAAG,KAAK,CAAC,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,kBAAkB,EAAE,IAAI,CAAC,CAAC;IACjF,OAAO,eAAe,IAAI,MAAM,IAAI,EAAE,CAAC;AAAA,CACvC","sourcesContent":["import {\n\ttype Api,\n\ttype Context,\n\tcalculateCost,\n\ttype Model,\n\ttype ModelsSimpleStreamOptions,\n\ttype SimpleStreamOptions,\n\ttype Usage,\n} from \"@earendil-works/pi-ai\";\nimport { getProviderEnvValue } from \"@earendil-works/pi-ai/utils/provider-env\";\nimport type { ModelRuntime } from \"./model-runtime.ts\";\nimport type { SessionEntry, SessionManager, UsageEntry } from \"./session-manager.ts\";\nimport type { CacheWarmingMode } from \"./settings-manager.ts\";\n\n/** Streaming warming never continues past this long after the real request that started it. */\nconst MAX_WARMING_AGE_MS = 60 * 60_000;\n/** Idle warming uses a shorter horizon because continuation estimates become less reliable with age. */\nconst MAX_IDLE_WARMING_AGE_MS = 30 * 60_000;\n/** A refresh is sent only when it is expected to save at least this many dollars. */\nconst CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS = 0.05;\n/**\n * Chance that a real request arrives before the cache entry expires while the\n * agent sits idle. Measured from our own usage; per-session estimates were not\n * better than this constant.\n */\nconst IDLE_CONTINUATION_PROBABILITY = 0.15;\n\n/** Refresh at 90% of the TTL while preserving at least ten seconds of margin. */\nexport function getCacheWarmingDelayMs(ttlMs: number): number | undefined {\n\tif (ttlMs <= 10_000) return undefined;\n\treturn Math.max(1, Math.floor(Math.min(ttlMs * 0.9, ttlMs - 10_000)));\n}\n\n/**\n * Lifetime of the prompt cache entry a request writes, from the model's\n * `promptCache` tier for the retention the request used. Undefined when the\n * model has no lifetime for that tier or caching is off.\n */\nexport function getPromptCacheTtlMs(model: Model<Api>, options: SimpleStreamOptions | undefined): number | undefined {\n\tconst retention =\n\t\toptions?.cacheRetention ??\n\t\t(getProviderEnvValue(\"PI_CACHE_RETENTION\", options?.env) === \"long\" ? \"long\" : \"short\");\n\tif (retention === \"none\") return undefined;\n\tconst seconds = model.promptCache?.[retention];\n\treturn seconds === undefined ? undefined : seconds * 1000;\n}\n\n/**\n * Whether replaying the request with a one-token output cap leaves its cache\n * entry untouched. Anthropic's budget-based thinking (Claude models without\n * adaptive thinking) derives `budget_tokens` from `max_tokens`; the replay\n * would get a different budget, which Anthropic keys the message cache on,\n * and the model could still think for thousands of tokens.\n */\nexport function isReplayable(model: Model<Api>, options: SimpleStreamOptions | undefined): boolean {\n\tif (!options?.reasoning || model.api !== \"anthropic-messages\") return true;\n\treturn (model as Model<\"anthropic-messages\">).compat?.forceAdaptiveThinking === true;\n}\n\n/** Prompt size of the most recent real request on the branch, as reported by the provider. */\nfunction lastPromptTokens(entries: SessionEntry[]): number {\n\tfor (let index = entries.length - 1; index >= 0; index--) {\n\t\tconst entry = entries[index];\n\t\tif (entry.type === \"message\" && entry.message.role === \"assistant\") {\n\t\t\tconst usage = entry.message.usage;\n\t\t\treturn usage.input + usage.cacheRead + usage.cacheWrite;\n\t\t}\n\t}\n\treturn 0;\n}\n\nfunction price(\n\tmodel: Model<Api>,\n\ttokens: Partial<Pick<Usage, \"input\" | \"output\" | \"cacheRead\" | \"cacheWrite\">>,\n): number {\n\tconst usage: Usage = {\n\t\tinput: 0,\n\t\toutput: 0,\n\t\tcacheRead: 0,\n\t\tcacheWrite: 0,\n\t\ttotalTokens: 0,\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t...tokens,\n\t};\n\treturn calculateCost(model, usage).total;\n}\n\nexport type CacheWarmingAction = \"warm\" | \"stop\";\n\n/** Inputs and outcome of one warm-or-stop decision, as shown by `/session`. */\nexport interface CacheWarmingDecision {\n\t/** \"streaming\" while the agent run that sent the request is still active. */\n\tphase: \"streaming\" | \"idle\";\n\t/** Price of this refresh: a cache read of the prompt plus one output token. */\n\twarmCost: number;\n\t/** Extra price of the next real request if the cache entry is lost. */\n\tmissCost: number;\n\t/** Estimated chance that a real request arrives before the entry expires. */\n\tcontinuationProbability: number;\n\t/** `continuationProbability * missCost - warmCost`. */\n\texpectedSavings: number;\n\t/** False when the prompt size or the model's prices are unknown. */\n\teconomicsAvailable: boolean;\n\t/** Pi's decision: \"warm\" when `expectedSavings` is at least $0.05. */\n\taction: CacheWarmingAction;\n}\n\n/**\n * Fired before each refresh with pi's decision filled in. Everything else an\n * extension might want (model, idle state, context size) is on the context.\n */\nexport interface CacheWarmingDecisionEvent\n\textends Pick<CacheWarmingDecision, \"warmCost\" | \"missCost\" | \"continuationProbability\" | \"action\"> {\n\ttype: \"cache_warming_decision\";\n}\n\nexport interface CacheWarmingDecisionEventResult {\n\t/** Override whether this refresh is sent. \"stop\" ends warming until the next real request. */\n\taction?: CacheWarmingAction;\n}\n\nexport interface CacheWarmingStatus {\n\t/** \"scheduled\": a refresh timer is armed; \"refreshing\": a warm request is in flight. */\n\tstate: \"inactive\" | \"scheduled\" | \"refreshing\";\n\t/** Why nothing is scheduled. */\n\treason?: string;\n\tnextWarmAt?: number;\n\t/** The pending decision, or the decision that stopped warming. */\n\tdecision?: CacheWarmingDecision;\n\t/** True when an extension changed `decision.action`. */\n\textensionOverride?: boolean;\n}\n\n/** The request whose prompt cache entry should be kept warm, exactly as it was sent. */\nexport interface CacheWarmRequest {\n\tmodel: Model<Api>;\n\tcontext: Context;\n\toptions: ModelsSimpleStreamOptions;\n}\n\ninterface ActiveRun extends CacheWarmRequest {\n\t/** False once the session's model or messages no longer match the request. */\n\tisCurrent: () => boolean;\n\tdelayMs: number;\n\tstartedAt: number;\n\tcontroller: AbortController;\n\tphase: \"streaming\" | \"idle\";\n\tnextWarmAt: number;\n\t/** Set while a refresh that an extension forced is in flight. */\n\textensionOverride: boolean;\n\ttimer?: ReturnType<typeof setTimeout>;\n}\n\n/**\n * Keeps one prompt cache entry alive by re-sending its request with a\n * one-token output cap before the entry expires. `start` replaces any\n * previous run; warm requests never extend the fixed safety windows.\n */\nexport class CacheWarmer {\n\tprivate run?: ActiveRun;\n\tprivate inactive: CacheWarmingStatus;\n\tprivate readonly models: Pick<ModelRuntime, \"streamSimple\">;\n\tprivate readonly sessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">;\n\tprivate readonly getMode: () => CacheWarmingMode;\n\t/** Lets extensions override `event.action`; failures fall back to pi's decision. */\n\tprivate readonly decide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction>;\n\t/** Called with the persisted usage entry after each successful refresh. */\n\tonWarmed?: (entry: UsageEntry) => void;\n\n\tconstructor(\n\t\tmodels: Pick<ModelRuntime, \"streamSimple\">,\n\t\tsessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">,\n\t\tgetMode: () => CacheWarmingMode,\n\t\tdecide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction> = async (event) => event.action,\n\t) {\n\t\tthis.models = models;\n\t\tthis.sessionManager = sessionManager;\n\t\tthis.getMode = getMode;\n\t\tthis.decide = decide;\n\t\tthis.inactive = { state: \"inactive\", reason: \"waiting for first request\" };\n\t}\n\n\tget status(): CacheWarmingStatus {\n\t\tif (this.getMode() === \"off\") return { state: \"inactive\", reason: \"cache warming disabled\" };\n\t\tconst run = this.run;\n\t\tif (!run) return this.inactive;\n\t\tif (!run.isCurrent()) return { state: \"inactive\", reason: \"conversation context changed\" };\n\t\tconst decision = this.evaluate(run);\n\t\tconst refreshing = run.timer === undefined;\n\t\tif (!decision.economicsAvailable && !refreshing) {\n\t\t\treturn { state: \"inactive\", reason: \"cache economics unavailable\" };\n\t\t}\n\t\treturn {\n\t\t\tstate: refreshing ? \"refreshing\" : \"scheduled\",\n\t\t\tnextWarmAt: run.nextWarmAt,\n\t\t\tdecision,\n\t\t\textensionOverride: run.extensionOverride,\n\t\t};\n\t}\n\n\t/** Keep the prompt cache entry written by `request` warm while `isCurrent` holds. */\n\tstart(request: CacheWarmRequest, isCurrent: () => boolean): void {\n\t\tthis.clearRun();\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") {\n\t\t\tthis.stop(\"cache warming disabled\");\n\t\t\treturn;\n\t\t}\n\t\tif (!isReplayable(request.model, request.options)) {\n\t\t\tthis.stop(\"request cannot be replayed safely\");\n\t\t\treturn;\n\t\t}\n\t\tconst ttlMs = getPromptCacheTtlMs(request.model, request.options);\n\t\tif (ttlMs === undefined) {\n\t\t\tthis.stop(\n\t\t\t\trequest.options.cacheRetention === \"none\"\n\t\t\t\t\t? \"request disabled prompt caching\"\n\t\t\t\t\t: \"cache lifetime unavailable\",\n\t\t\t);\n\t\t\treturn;\n\t\t}\n\t\tconst delayMs = getCacheWarmingDelayMs(ttlMs);\n\t\tif (delayMs === undefined) {\n\t\t\tthis.stop(\"cache lifetime unavailable\");\n\t\t\treturn;\n\t\t}\n\t\tthis.run = {\n\t\t\t...request,\n\t\t\tisCurrent,\n\t\t\tdelayMs,\n\t\t\tstartedAt: Date.now(),\n\t\t\tcontroller: new AbortController(),\n\t\t\tphase: \"streaming\",\n\t\t\tnextWarmAt: 0,\n\t\t\textensionOverride: false,\n\t\t};\n\t\tthis.schedule(this.run);\n\t}\n\n\tonAgentSettled(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tif (this.getMode() === \"streaming\") {\n\t\t\tthis.stop(\"agent run settled\");\n\t\t\treturn;\n\t\t}\n\t\trun.phase = \"idle\";\n\t\tconst deadline = run.startedAt + MAX_IDLE_WARMING_AGE_MS;\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(\"30-minute idle safety limit reached\");\n\t\t}\n\t}\n\n\t/** Reconcile an active run after the persisted warming mode changes. */\n\tonModeChanged(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tconst reason = this.getModeStopReason(run);\n\t\tif (reason) this.stop(reason);\n\t}\n\n\tcancel(): void {\n\t\tthis.stop(\"inactive\");\n\t}\n\n\tprivate clearRun(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tthis.run = undefined;\n\t\tif (run.timer) clearTimeout(run.timer);\n\t\trun.controller.abort();\n\t}\n\n\tprivate stop(reason: string, stopped?: Pick<CacheWarmingStatus, \"decision\" | \"extensionOverride\">): void {\n\t\tthis.clearRun();\n\t\tthis.inactive = { state: \"inactive\", reason, ...stopped };\n\t}\n\n\tprivate schedule(run: ActiveRun): void {\n\t\trun.extensionOverride = false;\n\t\trun.nextWarmAt = Date.now() + run.delayMs;\n\t\tconst deadline = run.startedAt + (run.phase === \"idle\" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(run.phase === \"idle\" ? \"30-minute idle safety limit reached\" : \"one-hour safety limit reached\");\n\t\t\treturn;\n\t\t}\n\t\trun.timer = setTimeout(() => void this.refresh(run), Math.max(0, run.nextWarmAt - Date.now()));\n\t\trun.timer.unref?.();\n\t}\n\n\tprivate async refresh(run: ActiveRun): Promise<void> {\n\t\trun.timer = undefined;\n\t\tif (!this.validateRun(run)) return;\n\t\tconst decision = this.evaluate(run);\n\t\tconst { warmCost, missCost, continuationProbability } = decision;\n\t\tlet action = decision.action;\n\t\ttry {\n\t\t\taction = await this.decide({\n\t\t\t\ttype: \"cache_warming_decision\",\n\t\t\t\twarmCost,\n\t\t\t\tmissCost,\n\t\t\t\tcontinuationProbability,\n\t\t\t\taction,\n\t\t\t});\n\t\t} catch {\n\t\t\t// Extension failures fall back to pi's own decision.\n\t\t}\n\t\tif (!this.validateRun(run)) return;\n\t\tconst extensionOverride = action !== decision.action;\n\t\tif (action === \"stop\") {\n\t\t\tconst reason = extensionOverride\n\t\t\t\t? \"stopped by extension\"\n\t\t\t\t: decision.economicsAvailable\n\t\t\t\t\t? \"expected savings below threshold\"\n\t\t\t\t\t: \"cache economics unavailable\";\n\t\t\tthis.stop(reason, { decision, extensionOverride });\n\t\t\treturn;\n\t\t}\n\n\t\trun.extensionOverride = extensionOverride;\n\t\ttry {\n\t\t\tconst message = await this.models\n\t\t\t\t.streamSimple(run.model, run.context, {\n\t\t\t\t\t...run.options,\n\t\t\t\t\tmaxTokens: 1,\n\t\t\t\t\tmaxRetries: 0,\n\t\t\t\t\tsignal: run.controller.signal,\n\t\t\t\t})\n\t\t\t\t.result();\n\t\t\tif (!this.validateRun(run)) return;\n\t\t\tif (message.stopReason !== \"error\" && message.stopReason !== \"aborted\") {\n\t\t\t\tconst entry = this.sessionManager.appendUsage(\n\t\t\t\t\t\"cache_warm\",\n\t\t\t\t\tmessage.provider,\n\t\t\t\t\tmessage.responseModel ?? message.model,\n\t\t\t\t\tmessage.usage,\n\t\t\t\t\textensionOverride ? \"extension override\" : undefined,\n\t\t\t\t);\n\t\t\t\tthis.onWarmed?.(entry);\n\t\t\t}\n\t\t} catch {\n\t\t\t// Cache warming is best-effort and must not affect the active agent run.\n\t\t}\n\t\tif (this.run === run) this.schedule(run);\n\t}\n\n\tprivate validateRun(run: ActiveRun): boolean {\n\t\tif (this.run !== run) return false;\n\t\tconst reason = this.getModeStopReason(run) ?? (!run.isCurrent() ? \"conversation context changed\" : undefined);\n\t\tif (!reason) return true;\n\t\tthis.stop(reason);\n\t\treturn false;\n\t}\n\n\tprivate getModeStopReason(run: ActiveRun): string | undefined {\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") return \"cache warming disabled\";\n\t\tif (mode === \"streaming\" && run.phase === \"idle\") return \"agent run settled\";\n\t\treturn undefined;\n\t}\n\n\tprivate evaluate(run: ActiveRun): CacheWarmingDecision {\n\t\tconst model = run.model;\n\t\tconst promptTokens = lastPromptTokens(this.sessionManager.getBranch());\n\t\tconst cacheHitCost = price(model, { cacheRead: promptTokens });\n\t\tconst cacheMissCost = price(\n\t\t\tmodel,\n\t\t\tmodel.cost.cacheWrite > 0 ? { cacheWrite: promptTokens } : { input: promptTokens },\n\t\t);\n\t\tconst warmCost = price(model, { cacheRead: promptTokens, output: 1 });\n\t\tconst missCost = Math.max(0, cacheMissCost - cacheHitCost);\n\t\tconst continuationProbability = run.phase === \"idle\" ? IDLE_CONTINUATION_PROBABILITY : 1;\n\t\tconst economicsAvailable = promptTokens > 0 && (cacheHitCost > 0 || cacheMissCost > 0);\n\t\tconst expectedSavings = continuationProbability * missCost - warmCost;\n\t\treturn {\n\t\t\tphase: run.phase,\n\t\t\twarmCost,\n\t\t\tmissCost,\n\t\t\tcontinuationProbability,\n\t\t\texpectedSavings,\n\t\t\teconomicsAvailable,\n\t\t\taction: expectedSavings >= CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS ? \"warm\" : \"stop\",\n\t\t};\n\t}\n}\n\nfunction formatDollars(value: number): string {\n\treturn value < 0 ? `-$${Math.abs(value).toFixed(3)}` : `$${value.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingEconomics(decision: CacheWarmingDecision): string {\n\tif (!decision.economicsAvailable) return \"cache economics unavailable\";\n\tconst probability = Math.round(decision.continuationProbability * 100);\n\tconst probabilityText =\n\t\tdecision.phase === \"streaming\"\n\t\t\t? `${probability}% continuation probability while agent is running`\n\t\t\t: `${probability}% continuation probability`;\n\tconst comparison = decision.action === \"warm\" ? \">=\" : \"<\";\n\treturn `${probabilityText}, expected savings ${formatDollars(decision.expectedSavings)} ${comparison} $${CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingDecisionTime(nextWarmAt: number | undefined, now: number): string {\n\tif (nextWarmAt === undefined || nextWarmAt <= now) return \"Decision now\";\n\tlet remainingSeconds = Math.ceil((nextWarmAt - now) / 1000);\n\tconst hours = Math.floor(remainingSeconds / 3600);\n\tremainingSeconds %= 3600;\n\tconst minutes = Math.floor(remainingSeconds / 60);\n\tconst seconds = remainingSeconds % 60;\n\tconst parts: string[] = [];\n\tif (hours > 0) parts.push(`${hours}h`);\n\tif (minutes > 0) parts.push(`${minutes}m`);\n\tif (seconds > 0 || parts.length === 0) parts.push(`${seconds}s`);\n\treturn `Decision in ${parts.join(\" \")}`;\n}\n\n/** One-line status for `/session`. */\nexport function formatCacheWarmingStatus(status: CacheWarmingStatus, now = Date.now()): string {\n\tconst decision = status.decision;\n\t// A decision is attached once pi (or an extension) acted on it; \"inactive\"\n\t// without one never got that far.\n\tif (!decision || (status.state === \"inactive\" && !decision.economicsAvailable && !status.extensionOverride)) {\n\t\treturn `Inactive (${status.reason ?? \"unknown reason\"})`;\n\t}\n\tconst details = status.extensionOverride\n\t\t? `extension override, ${formatCacheWarmingEconomics(decision)}`\n\t\t: `${formatCacheWarmingEconomics(decision)} -> ${decision.action}`;\n\tif (status.state === \"inactive\") return `Stopped (${details})`;\n\tif (status.state === \"refreshing\") return `Warming cache (${details})`;\n\treturn `${formatCacheWarmingDecisionTime(status.nextWarmAt, now)} (${details})`;\n}\n\n/** One-line transcript text for persisted cache-warming usage. */\nexport function formatCacheWarmingUsage(entry: UsageEntry): string {\n\tconst note = entry.note ? ` (${entry.note})` : \"\";\n\tconst cost = entry.usage.cost.total.toFixed(6).replace(/(\\.\\d{3}\\d*?)0+$/, \"$1\");\n\treturn `Cache warmed${note}: $${cost}`;\n}\n"]}
1
+ {"version":3,"file":"cache-warmer.js","sourceRoot":"","sources":["../../src/core/cache-warmer.ts"],"names":[],"mappings":"AAAA,OAAO,EAGN,aAAa,GAKb,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EAAE,mBAAmB,EAAE,MAAM,0CAA0C,CAAC;AAK/E,+FAA+F;AAC/F,MAAM,kBAAkB,GAAG,EAAE,GAAG,MAAM,CAAC;AACvC,wGAAwG;AACxG,MAAM,uBAAuB,GAAG,EAAE,GAAG,MAAM,CAAC;AAC5C,qFAAqF;AACrF,MAAM,sCAAsC,GAAG,IAAI,CAAC;AACpD;;;;GAIG;AACH,MAAM,6BAA6B,GAAG,IAAI,CAAC;AAE3C,iFAAiF;AACjF,MAAM,UAAU,sBAAsB,CAAC,KAAa,EAAsB;IACzE,IAAI,KAAK,IAAI,MAAM;QAAE,OAAO,SAAS,CAAC;IACtC,OAAO,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,GAAG,EAAE,KAAK,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;AAAA,CACtE;AAED;;;;GAIG;AACH,MAAM,UAAU,mBAAmB,CAAC,KAAiB,EAAE,OAAwC,EAAsB;IACpH,MAAM,SAAS,GACd,OAAO,EAAE,cAAc;QACvB,CAAC,mBAAmB,CAAC,oBAAoB,EAAE,OAAO,EAAE,GAAG,CAAC,KAAK,MAAM,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC;IACzF,IAAI,SAAS,KAAK,MAAM;QAAE,OAAO,SAAS,CAAC;IAC3C,MAAM,OAAO,GAAG,KAAK,CAAC,WAAW,EAAE,CAAC,SAAS,CAAC,CAAC;IAC/C,OAAO,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,OAAO,GAAG,IAAI,CAAC;AAAA,CAC1D;AAED;;;;;;GAMG;AACH,MAAM,UAAU,YAAY,CAAC,KAAiB,EAAE,OAAwC,EAAW;IAClG,IAAI,CAAC,OAAO,EAAE,SAAS,IAAI,KAAK,CAAC,GAAG,KAAK,oBAAoB;QAAE,OAAO,IAAI,CAAC;IAC3E,OAAQ,KAAqC,CAAC,MAAM,EAAE,qBAAqB,KAAK,IAAI,CAAC;AAAA,CACrF;AAED,8FAA8F;AAC9F,SAAS,gBAAgB,CAAC,OAAuB,EAAU;IAC1D,KAAK,IAAI,KAAK,GAAG,OAAO,CAAC,MAAM,GAAG,CAAC,EAAE,KAAK,IAAI,CAAC,EAAE,KAAK,EAAE,EAAE,CAAC;QAC1D,MAAM,KAAK,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC;QAC7B,IAAI,KAAK,CAAC,IAAI,KAAK,SAAS,IAAI,KAAK,CAAC,OAAO,CAAC,IAAI,KAAK,WAAW,EAAE,CAAC;YACpE,MAAM,KAAK,GAAG,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YAClC,OAAO,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,SAAS,GAAG,KAAK,CAAC,UAAU,CAAC;QACzD,CAAC;IACF,CAAC;IACD,OAAO,CAAC,CAAC;AAAA,CACT;AAED,SAAS,KAAK,CACb,KAAiB,EACjB,MAA6E,EACpE;IACT,MAAM,KAAK,GAAU;QACpB,KAAK,EAAE,CAAC;QACR,MAAM,EAAE,CAAC;QACT,SAAS,EAAE,CAAC;QACZ,UAAU,EAAE,CAAC;QACb,WAAW,EAAE,CAAC;QACd,IAAI,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE;QACpE,GAAG,MAAM;KACT,CAAC;IACF,OAAO,aAAa,CAAC,KAAK,EAAE,KAAK,CAAC,CAAC,KAAK,CAAC;AAAA,CACzC;AAuED;;;;GAIG;AACH,MAAM,OAAO,WAAW;IACf,GAAG,CAAa;IAChB,QAAQ,CAAqB;IACpB,MAAM,CAAqC;IAC3C,cAAc,CAAoD;IAClE,OAAO,CAAyB;IACjD,oFAAoF;IACnE,MAAM,CAAoE;IAC3F,2EAA2E;IAC3E,QAAQ,CAA+B;IAEvC,YACC,MAA0C,EAC1C,cAAiE,EACjE,OAA+B,EAC/B,MAAM,GAAsE,KAAK,EAAE,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,MAAM,EACxG;QACD,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC;QACrB,IAAI,CAAC,cAAc,GAAG,cAAc,CAAC;QACrC,IAAI,CAAC,OAAO,GAAG,OAAO,CAAC;QACvB,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC;QACrB,IAAI,CAAC,QAAQ,GAAG,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,2BAA2B,EAAE,CAAC;IAAA,CAC3E;IAED,IAAI,MAAM,GAAuB;QAChC,IAAI,IAAI,CAAC,OAAO,EAAE,KAAK,KAAK;YAAE,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,wBAAwB,EAAE,CAAC;QAC7F,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO,IAAI,CAAC,QAAQ,CAAC;QAC/B,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE;YAAE,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,8BAA8B,EAAE,CAAC;QAC3F,MAAM,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;QACpC,MAAM,UAAU,GAAG,GAAG,CAAC,KAAK,KAAK,SAAS,CAAC;QAC3C,IAAI,CAAC,QAAQ,CAAC,kBAAkB,IAAI,CAAC,UAAU,EAAE,CAAC;YACjD,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,6BAA6B,EAAE,CAAC;QACrE,CAAC;QACD,OAAO;YACN,KAAK,EAAE,UAAU,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,WAAW;YAC9C,UAAU,EAAE,GAAG,CAAC,UAAU;YAC1B,QAAQ;YACR,iBAAiB,EAAE,GAAG,CAAC,iBAAiB;SACxC,CAAC;IAAA,CACF;IAED,qFAAqF;IACrF,KAAK,CAAC,OAAyB,EAAE,SAAwB,EAAQ;QAChE,IAAI,CAAC,QAAQ,EAAE,CAAC;QAChB,MAAM,IAAI,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC;QAC5B,IAAI,IAAI,KAAK,KAAK,EAAE,CAAC;YACpB,IAAI,CAAC,IAAI,CAAC,wBAAwB,CAAC,CAAC;YACpC,OAAO;QACR,CAAC;QACD,IAAI,CAAC,YAAY,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,EAAE,CAAC;YACnD,IAAI,CAAC,IAAI,CAAC,mCAAmC,CAAC,CAAC;YAC/C,OAAO;QACR,CAAC;QACD,MAAM,KAAK,GAAG,mBAAmB,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,CAAC;QAClE,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;YACzB,IAAI,CAAC,IAAI,CACR,OAAO,CAAC,OAAO,CAAC,cAAc,KAAK,MAAM;gBACxC,CAAC,CAAC,iCAAiC;gBACnC,CAAC,CAAC,4BAA4B,CAC/B,CAAC;YACF,OAAO;QACR,CAAC;QACD,MAAM,OAAO,GAAG,sBAAsB,CAAC,KAAK,CAAC,CAAC;QAC9C,IAAI,OAAO,KAAK,SAAS,EAAE,CAAC;YAC3B,IAAI,CAAC,IAAI,CAAC,4BAA4B,CAAC,CAAC;YACxC,OAAO;QACR,CAAC;QACD,IAAI,CAAC,GAAG,GAAG;YACV,GAAG,OAAO;YACV,SAAS;YACT,KAAK;YACL,OAAO;YACP,iBAAiB,EAAE,CAAC;YACpB,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE;YACrB,UAAU,EAAE,IAAI,eAAe,EAAE;YACjC,KAAK,EAAE,WAAW;YAClB,UAAU,EAAE,CAAC;YACb,iBAAiB,EAAE,KAAK;SACxB,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;IAAA,CACxB;IAED,cAAc,GAAS;QACtB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,IAAI,IAAI,CAAC,OAAO,EAAE,KAAK,WAAW,EAAE,CAAC;YACpC,IAAI,CAAC,IAAI,CAAC,mBAAmB,CAAC,CAAC;YAC/B,OAAO;QACR,CAAC;QACD,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC;QACnB,MAAM,QAAQ,GAAG,GAAG,CAAC,SAAS,GAAG,uBAAuB,CAAC;QACzD,IAAI,GAAG,CAAC,UAAU,GAAG,QAAQ,IAAI,IAAI,CAAC,GAAG,EAAE,IAAI,QAAQ,EAAE,CAAC;YACzD,IAAI,CAAC,IAAI,CAAC,qCAAqC,CAAC,CAAC;QAClD,CAAC;IAAA,CACD;IAED,wEAAwE;IACxE,aAAa,GAAS;QACrB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,MAAM,MAAM,GAAG,IAAI,CAAC,iBAAiB,CAAC,GAAG,CAAC,CAAC;QAC3C,IAAI,MAAM;YAAE,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;IAAA,CAC9B;IAED,MAAM,GAAS;QACd,IAAI,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC;IAAA,CACtB;IAEO,QAAQ,GAAS;QACxB,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC;QACrB,IAAI,CAAC,GAAG;YAAE,OAAO;QACjB,IAAI,CAAC,GAAG,GAAG,SAAS,CAAC;QACrB,IAAI,GAAG,CAAC,KAAK;YAAE,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;QACvC,GAAG,CAAC,UAAU,CAAC,KAAK,EAAE,CAAC;IAAA,CACvB;IAEO,IAAI,CAAC,MAAc,EAAE,OAAoE,EAAQ;QACxG,IAAI,CAAC,QAAQ,EAAE,CAAC;QAChB,IAAI,CAAC,QAAQ,GAAG,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,GAAG,OAAO,EAAE,CAAC;IAAA,CAC1D;IAEO,QAAQ,CAAC,GAAc,EAAQ;QACtC,GAAG,CAAC,iBAAiB,GAAG,KAAK,CAAC;QAC9B,GAAG,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,GAAG,CAAC,OAAO,CAAC;QAC1C,wEAAwE;QACxE,uEAAuE;QACvE,qEAAqE;QACrE,GAAG,CAAC,iBAAiB,GAAG,GAAG,CAAC,UAAU,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,CAAC,KAAK,GAAG,GAAG,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC;QACnF,MAAM,QAAQ,GAAG,GAAG,CAAC,SAAS,GAAG,CAAC,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,uBAAuB,CAAC,CAAC,CAAC,kBAAkB,CAAC,CAAC;QACvG,IAAI,GAAG,CAAC,UAAU,GAAG,QAAQ,IAAI,IAAI,CAAC,GAAG,EAAE,IAAI,QAAQ,EAAE,CAAC;YACzD,IAAI,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,qCAAqC,CAAC,CAAC,CAAC,+BAA+B,CAAC,CAAC;YAC1G,OAAO;QACR,CAAC;QACD,GAAG,CAAC,KAAK,GAAG,UAAU,CAAC,GAAG,EAAE,CAAC,KAAK,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,GAAG,CAAC,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC;QAC/F,GAAG,CAAC,KAAK,CAAC,KAAK,EAAE,EAAE,CAAC;IAAA,CACpB;IAEO,KAAK,CAAC,OAAO,CAAC,GAAc,EAAiB;QACpD,GAAG,CAAC,KAAK,GAAG,SAAS,CAAC;QACtB,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC;YAAE,OAAO;QACnC,IAAI,IAAI,CAAC,qBAAqB,CAAC,GAAG,CAAC;YAAE,OAAO;QAC5C,MAAM,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;QACpC,MAAM,EAAE,QAAQ,EAAE,QAAQ,EAAE,uBAAuB,EAAE,GAAG,QAAQ,CAAC;QACjE,IAAI,MAAM,GAAG,QAAQ,CAAC,MAAM,CAAC;QAC7B,IAAI,CAAC;YACJ,MAAM,GAAG,MAAM,IAAI,CAAC,MAAM,CAAC;gBAC1B,IAAI,EAAE,wBAAwB;gBAC9B,QAAQ;gBACR,QAAQ;gBACR,uBAAuB;gBACvB,MAAM;aACN,CAAC,CAAC;QACJ,CAAC;QAAC,MAAM,CAAC;YACR,qDAAqD;QACtD,CAAC;QACD,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,IAAI,IAAI,CAAC,qBAAqB,CAAC,GAAG,CAAC;YAAE,OAAO;QACtE,MAAM,iBAAiB,GAAG,MAAM,KAAK,QAAQ,CAAC,MAAM,CAAC;QACrD,IAAI,MAAM,KAAK,MAAM,EAAE,CAAC;YACvB,MAAM,MAAM,GAAG,iBAAiB;gBAC/B,CAAC,CAAC,sBAAsB;gBACxB,CAAC,CAAC,QAAQ,CAAC,kBAAkB;oBAC5B,CAAC,CAAC,kCAAkC;oBACpC,CAAC,CAAC,6BAA6B,CAAC;YAClC,IAAI,CAAC,IAAI,CAAC,MAAM,EAAE,EAAE,QAAQ,EAAE,iBAAiB,EAAE,CAAC,CAAC;YACnD,OAAO;QACR,CAAC;QAED,GAAG,CAAC,iBAAiB,GAAG,iBAAiB,CAAC;QAC1C,IAAI,CAAC;YACJ,MAAM,OAAO,GAAG,MAAM,IAAI,CAAC,MAAM;iBAC/B,YAAY,CAAC,GAAG,CAAC,KAAK,EAAE,GAAG,CAAC,OAAO,EAAE;gBACrC,GAAG,GAAG,CAAC,OAAO;gBACd,SAAS,EAAE,CAAC;gBACZ,UAAU,EAAE,CAAC;gBACb,MAAM,EAAE,GAAG,CAAC,UAAU,CAAC,MAAM;aAC7B,CAAC;iBACD,MAAM,EAAE,CAAC;YACX,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC;gBAAE,OAAO;YACnC,IAAI,OAAO,CAAC,UAAU,KAAK,OAAO,IAAI,OAAO,CAAC,UAAU,KAAK,SAAS,EAAE,CAAC;gBACxE,MAAM,KAAK,GAAG,IAAI,CAAC,cAAc,CAAC,WAAW,CAC5C,YAAY,EACZ,OAAO,CAAC,QAAQ,EAChB,OAAO,CAAC,aAAa,IAAI,OAAO,CAAC,KAAK,EACtC,OAAO,CAAC,KAAK,EACb,iBAAiB,CAAC,CAAC,CAAC,oBAAoB,CAAC,CAAC,CAAC,SAAS,CACpD,CAAC;gBACF,IAAI,CAAC,QAAQ,EAAE,CAAC,KAAK,CAAC,CAAC;YACxB,CAAC;QACF,CAAC;QAAC,MAAM,CAAC;YACR,yEAAyE;QAC1E,CAAC;QACD,IAAI,IAAI,CAAC,GAAG,KAAK,GAAG;YAAE,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC;IAAA,CACzC;IAEO,qBAAqB,CAAC,GAAc,EAAW;QACtD,IAAI,IAAI,CAAC,GAAG,EAAE,IAAI,GAAG,CAAC,iBAAiB;YAAE,OAAO,KAAK,CAAC;QACtD,IAAI,CAAC,IAAI,CAAC,+BAA+B,CAAC,CAAC;QAC3C,OAAO,IAAI,CAAC;IAAA,CACZ;IAEO,WAAW,CAAC,GAAc,EAAW;QAC5C,IAAI,IAAI,CAAC,GAAG,KAAK,GAAG;YAAE,OAAO,KAAK,CAAC;QACnC,MAAM,MAAM,GAAG,IAAI,CAAC,iBAAiB,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,SAAS,EAAE,CAAC,CAAC,CAAC,8BAA8B,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC;QAC9G,IAAI,CAAC,MAAM;YAAE,OAAO,IAAI,CAAC;QACzB,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;QAClB,OAAO,KAAK,CAAC;IAAA,CACb;IAEO,iBAAiB,CAAC,GAAc,EAAsB;QAC7D,MAAM,IAAI,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC;QAC5B,IAAI,IAAI,KAAK,KAAK;YAAE,OAAO,wBAAwB,CAAC;QACpD,IAAI,IAAI,KAAK,WAAW,IAAI,GAAG,CAAC,KAAK,KAAK,MAAM;YAAE,OAAO,mBAAmB,CAAC;QAC7E,OAAO,SAAS,CAAC;IAAA,CACjB;IAEO,QAAQ,CAAC,GAAc,EAAwB;QACtD,MAAM,KAAK,GAAG,GAAG,CAAC,KAAK,CAAC;QACxB,MAAM,YAAY,GAAG,gBAAgB,CAAC,IAAI,CAAC,cAAc,CAAC,SAAS,EAAE,CAAC,CAAC;QACvE,MAAM,YAAY,GAAG,KAAK,CAAC,KAAK,EAAE,EAAE,SAAS,EAAE,YAAY,EAAE,CAAC,CAAC;QAC/D,MAAM,aAAa,GAAG,KAAK,CAC1B,KAAK,EACL,KAAK,CAAC,IAAI,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,EAAE,UAAU,EAAE,YAAY,EAAE,CAAC,CAAC,CAAC,EAAE,KAAK,EAAE,YAAY,EAAE,CAClF,CAAC;QACF,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAK,EAAE,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,CAAC,EAAE,CAAC,CAAC;QACtE,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,aAAa,GAAG,YAAY,CAAC,CAAC;QAC3D,MAAM,uBAAuB,GAAG,GAAG,CAAC,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,6BAA6B,CAAC,CAAC,CAAC,CAAC,CAAC;QACzF,MAAM,kBAAkB,GAAG,YAAY,GAAG,CAAC,IAAI,CAAC,YAAY,GAAG,CAAC,IAAI,aAAa,GAAG,CAAC,CAAC,CAAC;QACvF,MAAM,eAAe,GAAG,uBAAuB,GAAG,QAAQ,GAAG,QAAQ,CAAC;QACtE,OAAO;YACN,KAAK,EAAE,GAAG,CAAC,KAAK;YAChB,QAAQ;YACR,QAAQ;YACR,uBAAuB;YACvB,eAAe;YACf,kBAAkB;YAClB,MAAM,EAAE,eAAe,IAAI,sCAAsC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,MAAM;SACnF,CAAC;IAAA,CACF;CACD;AAED,SAAS,aAAa,CAAC,KAAa,EAAU;IAC7C,OAAO,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC;AAAA,CAC9E;AAED,SAAS,2BAA2B,CAAC,QAA8B,EAAU;IAC5E,IAAI,CAAC,QAAQ,CAAC,kBAAkB;QAAE,OAAO,6BAA6B,CAAC;IACvE,MAAM,WAAW,GAAG,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,uBAAuB,GAAG,GAAG,CAAC,CAAC;IACvE,MAAM,eAAe,GACpB,QAAQ,CAAC,KAAK,KAAK,WAAW;QAC7B,CAAC,CAAC,GAAG,WAAW,mDAAmD;QACnE,CAAC,CAAC,GAAG,WAAW,4BAA4B,CAAC;IAC/C,MAAM,UAAU,GAAG,QAAQ,CAAC,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,GAAG,CAAC;IAC3D,OAAO,GAAG,eAAe,sBAAsB,aAAa,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,UAAU,KAAK,sCAAsC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC;AAAA,CAC7J;AAED,SAAS,8BAA8B,CAAC,UAA8B,EAAE,GAAW,EAAU;IAC5F,IAAI,UAAU,KAAK,SAAS,IAAI,UAAU,IAAI,GAAG;QAAE,OAAO,cAAc,CAAC;IACzE,IAAI,gBAAgB,GAAG,IAAI,CAAC,IAAI,CAAC,CAAC,UAAU,GAAG,GAAG,CAAC,GAAG,IAAI,CAAC,CAAC;IAC5D,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,gBAAgB,GAAG,IAAI,CAAC,CAAC;IAClD,gBAAgB,IAAI,IAAI,CAAC;IACzB,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,gBAAgB,GAAG,EAAE,CAAC,CAAC;IAClD,MAAM,OAAO,GAAG,gBAAgB,GAAG,EAAE,CAAC;IACtC,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,KAAK,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,KAAK,GAAG,CAAC,CAAC;IACvC,IAAI,OAAO,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,OAAO,GAAG,CAAC,CAAC;IAC3C,IAAI,OAAO,GAAG,CAAC,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,OAAO,GAAG,CAAC,CAAC;IACjE,OAAO,eAAe,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC;AAAA,CACxC;AAED,sCAAsC;AACtC,MAAM,UAAU,wBAAwB,CAAC,MAA0B,EAAE,GAAG,GAAG,IAAI,CAAC,GAAG,EAAE,EAAU;IAC9F,MAAM,QAAQ,GAAG,MAAM,CAAC,QAAQ,CAAC;IACjC,2EAA2E;IAC3E,kCAAkC;IAClC,IAAI,CAAC,QAAQ,IAAI,CAAC,MAAM,CAAC,KAAK,KAAK,UAAU,IAAI,CAAC,QAAQ,CAAC,kBAAkB,IAAI,CAAC,MAAM,CAAC,iBAAiB,CAAC,EAAE,CAAC;QAC7G,OAAO,aAAa,MAAM,CAAC,MAAM,IAAI,gBAAgB,GAAG,CAAC;IAC1D,CAAC;IACD,MAAM,OAAO,GAAG,MAAM,CAAC,iBAAiB;QACvC,CAAC,CAAC,uBAAuB,2BAA2B,CAAC,QAAQ,CAAC,EAAE;QAChE,CAAC,CAAC,GAAG,2BAA2B,CAAC,QAAQ,CAAC,OAAO,QAAQ,CAAC,MAAM,EAAE,CAAC;IACpE,IAAI,MAAM,CAAC,KAAK,KAAK,UAAU;QAAE,OAAO,YAAY,OAAO,GAAG,CAAC;IAC/D,IAAI,MAAM,CAAC,KAAK,KAAK,YAAY;QAAE,OAAO,kBAAkB,OAAO,GAAG,CAAC;IACvE,OAAO,GAAG,8BAA8B,CAAC,MAAM,CAAC,UAAU,EAAE,GAAG,CAAC,KAAK,OAAO,GAAG,CAAC;AAAA,CAChF;AAED,kEAAkE;AAClE,MAAM,UAAU,uBAAuB,CAAC,KAAiB,EAAU;IAClE,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,KAAK,KAAK,CAAC,IAAI,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;IAClD,MAAM,IAAI,GAAG,KAAK,CAAC,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,kBAAkB,EAAE,IAAI,CAAC,CAAC;IACjF,OAAO,eAAe,IAAI,MAAM,IAAI,EAAE,CAAC;AAAA,CACvC","sourcesContent":["import {\n\ttype Api,\n\ttype Context,\n\tcalculateCost,\n\ttype Model,\n\ttype ModelsSimpleStreamOptions,\n\ttype SimpleStreamOptions,\n\ttype Usage,\n} from \"@earendil-works/pi-ai\";\nimport { getProviderEnvValue } from \"@earendil-works/pi-ai/utils/provider-env\";\nimport type { ModelRuntime } from \"./model-runtime.ts\";\nimport type { SessionEntry, SessionManager, UsageEntry } from \"./session-manager.ts\";\nimport type { CacheWarmingMode } from \"./settings-manager.ts\";\n\n/** Streaming warming never continues past this long after the real request that started it. */\nconst MAX_WARMING_AGE_MS = 60 * 60_000;\n/** Idle warming uses a shorter horizon because continuation estimates become less reliable with age. */\nconst MAX_IDLE_WARMING_AGE_MS = 30 * 60_000;\n/** A refresh is sent only when it is expected to save at least this many dollars. */\nconst CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS = 0.05;\n/**\n * Chance that a real request arrives before the cache entry expires while the\n * agent sits idle. Measured from our own usage; per-session estimates were not\n * better than this constant.\n */\nconst IDLE_CONTINUATION_PROBABILITY = 0.15;\n\n/** Refresh at 90% of the TTL while preserving at least ten seconds of margin. */\nexport function getCacheWarmingDelayMs(ttlMs: number): number | undefined {\n\tif (ttlMs <= 10_000) return undefined;\n\treturn Math.max(1, Math.floor(Math.min(ttlMs * 0.9, ttlMs - 10_000)));\n}\n\n/**\n * Lifetime of the prompt cache entry a request writes, from the model's\n * `promptCache` tier for the retention the request used. Undefined when the\n * model has no lifetime for that tier or caching is off.\n */\nexport function getPromptCacheTtlMs(model: Model<Api>, options: SimpleStreamOptions | undefined): number | undefined {\n\tconst retention =\n\t\toptions?.cacheRetention ??\n\t\t(getProviderEnvValue(\"PI_CACHE_RETENTION\", options?.env) === \"long\" ? \"long\" : \"short\");\n\tif (retention === \"none\") return undefined;\n\tconst seconds = model.promptCache?.[retention];\n\treturn seconds === undefined ? undefined : seconds * 1000;\n}\n\n/**\n * Whether replaying the request with a one-token output cap leaves its cache\n * entry untouched. Anthropic's budget-based thinking (Claude models without\n * adaptive thinking) derives `budget_tokens` from `max_tokens`; the replay\n * would get a different budget, which Anthropic keys the message cache on,\n * and the model could still think for thousands of tokens.\n */\nexport function isReplayable(model: Model<Api>, options: SimpleStreamOptions | undefined): boolean {\n\tif (!options?.reasoning || model.api !== \"anthropic-messages\") return true;\n\treturn (model as Model<\"anthropic-messages\">).compat?.forceAdaptiveThinking === true;\n}\n\n/** Prompt size of the most recent real request on the branch, as reported by the provider. */\nfunction lastPromptTokens(entries: SessionEntry[]): number {\n\tfor (let index = entries.length - 1; index >= 0; index--) {\n\t\tconst entry = entries[index];\n\t\tif (entry.type === \"message\" && entry.message.role === \"assistant\") {\n\t\t\tconst usage = entry.message.usage;\n\t\t\treturn usage.input + usage.cacheRead + usage.cacheWrite;\n\t\t}\n\t}\n\treturn 0;\n}\n\nfunction price(\n\tmodel: Model<Api>,\n\ttokens: Partial<Pick<Usage, \"input\" | \"output\" | \"cacheRead\" | \"cacheWrite\">>,\n): number {\n\tconst usage: Usage = {\n\t\tinput: 0,\n\t\toutput: 0,\n\t\tcacheRead: 0,\n\t\tcacheWrite: 0,\n\t\ttotalTokens: 0,\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t...tokens,\n\t};\n\treturn calculateCost(model, usage).total;\n}\n\nexport type CacheWarmingAction = \"warm\" | \"stop\";\n\n/** Inputs and outcome of one warm-or-stop decision, as shown by `/session`. */\nexport interface CacheWarmingDecision {\n\t/** \"streaming\" while the agent run that sent the request is still active. */\n\tphase: \"streaming\" | \"idle\";\n\t/** Price of this refresh: a cache read of the prompt plus one output token. */\n\twarmCost: number;\n\t/** Extra price of the next real request if the cache entry is lost. */\n\tmissCost: number;\n\t/** Estimated chance that a real request arrives before the entry expires. */\n\tcontinuationProbability: number;\n\t/** `continuationProbability * missCost - warmCost`. */\n\texpectedSavings: number;\n\t/** False when the prompt size or the model's prices are unknown. */\n\teconomicsAvailable: boolean;\n\t/** Pi's decision: \"warm\" when `expectedSavings` is at least $0.05. */\n\taction: CacheWarmingAction;\n}\n\n/**\n * Fired before each refresh with pi's decision filled in. Everything else an\n * extension might want (model, idle state, context size) is on the context.\n */\nexport interface CacheWarmingDecisionEvent\n\textends Pick<CacheWarmingDecision, \"warmCost\" | \"missCost\" | \"continuationProbability\" | \"action\"> {\n\ttype: \"cache_warming_decision\";\n}\n\nexport interface CacheWarmingDecisionEventResult {\n\t/** Override whether this refresh is sent. \"stop\" ends warming until the next real request. */\n\taction?: CacheWarmingAction;\n}\n\nexport interface CacheWarmingStatus {\n\t/** \"scheduled\": a refresh timer is armed; \"refreshing\": a warm request is in flight. */\n\tstate: \"inactive\" | \"scheduled\" | \"refreshing\";\n\t/** Why nothing is scheduled. */\n\treason?: string;\n\tnextWarmAt?: number;\n\t/** The pending decision, or the decision that stopped warming. */\n\tdecision?: CacheWarmingDecision;\n\t/** True when an extension changed `decision.action`. */\n\textensionOverride?: boolean;\n}\n\n/** The request whose prompt cache entry should be kept warm, exactly as it was sent. */\nexport interface CacheWarmRequest {\n\tmodel: Model<Api>;\n\tcontext: Context;\n\toptions: ModelsSimpleStreamOptions;\n}\n\ninterface ActiveRun extends CacheWarmRequest {\n\t/** False once the session's model or messages no longer match the request. */\n\tisCurrent: () => boolean;\n\tttlMs: number;\n\tdelayMs: number;\n\t/** Latest safe time to send this refresh, leaving half the original expiry margin. */\n\trefreshDeadlineAt: number;\n\tstartedAt: number;\n\tcontroller: AbortController;\n\tphase: \"streaming\" | \"idle\";\n\tnextWarmAt: number;\n\t/** Set while a refresh that an extension forced is in flight. */\n\textensionOverride: boolean;\n\ttimer?: ReturnType<typeof setTimeout>;\n}\n\n/**\n * Keeps one prompt cache entry alive by re-sending its request with a\n * one-token output cap before the entry expires. `start` replaces any\n * previous run; warm requests never extend the fixed safety windows.\n */\nexport class CacheWarmer {\n\tprivate run?: ActiveRun;\n\tprivate inactive: CacheWarmingStatus;\n\tprivate readonly models: Pick<ModelRuntime, \"streamSimple\">;\n\tprivate readonly sessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">;\n\tprivate readonly getMode: () => CacheWarmingMode;\n\t/** Lets extensions override `event.action`; failures fall back to pi's decision. */\n\tprivate readonly decide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction>;\n\t/** Called with the persisted usage entry after each successful refresh. */\n\tonWarmed?: (entry: UsageEntry) => void;\n\n\tconstructor(\n\t\tmodels: Pick<ModelRuntime, \"streamSimple\">,\n\t\tsessionManager: Pick<SessionManager, \"appendUsage\" | \"getBranch\">,\n\t\tgetMode: () => CacheWarmingMode,\n\t\tdecide: (event: CacheWarmingDecisionEvent) => Promise<CacheWarmingAction> = async (event) => event.action,\n\t) {\n\t\tthis.models = models;\n\t\tthis.sessionManager = sessionManager;\n\t\tthis.getMode = getMode;\n\t\tthis.decide = decide;\n\t\tthis.inactive = { state: \"inactive\", reason: \"waiting for first request\" };\n\t}\n\n\tget status(): CacheWarmingStatus {\n\t\tif (this.getMode() === \"off\") return { state: \"inactive\", reason: \"cache warming disabled\" };\n\t\tconst run = this.run;\n\t\tif (!run) return this.inactive;\n\t\tif (!run.isCurrent()) return { state: \"inactive\", reason: \"conversation context changed\" };\n\t\tconst decision = this.evaluate(run);\n\t\tconst refreshing = run.timer === undefined;\n\t\tif (!decision.economicsAvailable && !refreshing) {\n\t\t\treturn { state: \"inactive\", reason: \"cache economics unavailable\" };\n\t\t}\n\t\treturn {\n\t\t\tstate: refreshing ? \"refreshing\" : \"scheduled\",\n\t\t\tnextWarmAt: run.nextWarmAt,\n\t\t\tdecision,\n\t\t\textensionOverride: run.extensionOverride,\n\t\t};\n\t}\n\n\t/** Keep the prompt cache entry written by `request` warm while `isCurrent` holds. */\n\tstart(request: CacheWarmRequest, isCurrent: () => boolean): void {\n\t\tthis.clearRun();\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") {\n\t\t\tthis.stop(\"cache warming disabled\");\n\t\t\treturn;\n\t\t}\n\t\tif (!isReplayable(request.model, request.options)) {\n\t\t\tthis.stop(\"request cannot be replayed safely\");\n\t\t\treturn;\n\t\t}\n\t\tconst ttlMs = getPromptCacheTtlMs(request.model, request.options);\n\t\tif (ttlMs === undefined) {\n\t\t\tthis.stop(\n\t\t\t\trequest.options.cacheRetention === \"none\"\n\t\t\t\t\t? \"request disabled prompt caching\"\n\t\t\t\t\t: \"cache lifetime unavailable\",\n\t\t\t);\n\t\t\treturn;\n\t\t}\n\t\tconst delayMs = getCacheWarmingDelayMs(ttlMs);\n\t\tif (delayMs === undefined) {\n\t\t\tthis.stop(\"cache lifetime unavailable\");\n\t\t\treturn;\n\t\t}\n\t\tthis.run = {\n\t\t\t...request,\n\t\t\tisCurrent,\n\t\t\tttlMs,\n\t\t\tdelayMs,\n\t\t\trefreshDeadlineAt: 0,\n\t\t\tstartedAt: Date.now(),\n\t\t\tcontroller: new AbortController(),\n\t\t\tphase: \"streaming\",\n\t\t\tnextWarmAt: 0,\n\t\t\textensionOverride: false,\n\t\t};\n\t\tthis.schedule(this.run);\n\t}\n\n\tonAgentSettled(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tif (this.getMode() === \"streaming\") {\n\t\t\tthis.stop(\"agent run settled\");\n\t\t\treturn;\n\t\t}\n\t\trun.phase = \"idle\";\n\t\tconst deadline = run.startedAt + MAX_IDLE_WARMING_AGE_MS;\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(\"30-minute idle safety limit reached\");\n\t\t}\n\t}\n\n\t/** Reconcile an active run after the persisted warming mode changes. */\n\tonModeChanged(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tconst reason = this.getModeStopReason(run);\n\t\tif (reason) this.stop(reason);\n\t}\n\n\tcancel(): void {\n\t\tthis.stop(\"inactive\");\n\t}\n\n\tprivate clearRun(): void {\n\t\tconst run = this.run;\n\t\tif (!run) return;\n\t\tthis.run = undefined;\n\t\tif (run.timer) clearTimeout(run.timer);\n\t\trun.controller.abort();\n\t}\n\n\tprivate stop(reason: string, stopped?: Pick<CacheWarmingStatus, \"decision\" | \"extensionOverride\">): void {\n\t\tthis.clearRun();\n\t\tthis.inactive = { state: \"inactive\", reason, ...stopped };\n\t}\n\n\tprivate schedule(run: ActiveRun): void {\n\t\trun.extensionOverride = false;\n\t\trun.nextWarmAt = Date.now() + run.delayMs;\n\t\t// A timer can run late after sleep or event-loop blockage. Keep half of\n\t\t// the planned pre-expiry margin for that delay and request dispatch; a\n\t\t// late refresh is likely a full-price cache write, not a cache warm.\n\t\trun.refreshDeadlineAt = run.nextWarmAt + Math.floor((run.ttlMs - run.delayMs) / 2);\n\t\tconst deadline = run.startedAt + (run.phase === \"idle\" ? MAX_IDLE_WARMING_AGE_MS : MAX_WARMING_AGE_MS);\n\t\tif (run.nextWarmAt > deadline || Date.now() >= deadline) {\n\t\t\tthis.stop(run.phase === \"idle\" ? \"30-minute idle safety limit reached\" : \"one-hour safety limit reached\");\n\t\t\treturn;\n\t\t}\n\t\trun.timer = setTimeout(() => void this.refresh(run), Math.max(0, run.nextWarmAt - Date.now()));\n\t\trun.timer.unref?.();\n\t}\n\n\tprivate async refresh(run: ActiveRun): Promise<void> {\n\t\trun.timer = undefined;\n\t\tif (!this.validateRun(run)) return;\n\t\tif (this.refreshDeadlineMissed(run)) return;\n\t\tconst decision = this.evaluate(run);\n\t\tconst { warmCost, missCost, continuationProbability } = decision;\n\t\tlet action = decision.action;\n\t\ttry {\n\t\t\taction = await this.decide({\n\t\t\t\ttype: \"cache_warming_decision\",\n\t\t\t\twarmCost,\n\t\t\t\tmissCost,\n\t\t\t\tcontinuationProbability,\n\t\t\t\taction,\n\t\t\t});\n\t\t} catch {\n\t\t\t// Extension failures fall back to pi's own decision.\n\t\t}\n\t\tif (!this.validateRun(run) || this.refreshDeadlineMissed(run)) return;\n\t\tconst extensionOverride = action !== decision.action;\n\t\tif (action === \"stop\") {\n\t\t\tconst reason = extensionOverride\n\t\t\t\t? \"stopped by extension\"\n\t\t\t\t: decision.economicsAvailable\n\t\t\t\t\t? \"expected savings below threshold\"\n\t\t\t\t\t: \"cache economics unavailable\";\n\t\t\tthis.stop(reason, { decision, extensionOverride });\n\t\t\treturn;\n\t\t}\n\n\t\trun.extensionOverride = extensionOverride;\n\t\ttry {\n\t\t\tconst message = await this.models\n\t\t\t\t.streamSimple(run.model, run.context, {\n\t\t\t\t\t...run.options,\n\t\t\t\t\tmaxTokens: 1,\n\t\t\t\t\tmaxRetries: 0,\n\t\t\t\t\tsignal: run.controller.signal,\n\t\t\t\t})\n\t\t\t\t.result();\n\t\t\tif (!this.validateRun(run)) return;\n\t\t\tif (message.stopReason !== \"error\" && message.stopReason !== \"aborted\") {\n\t\t\t\tconst entry = this.sessionManager.appendUsage(\n\t\t\t\t\t\"cache_warm\",\n\t\t\t\t\tmessage.provider,\n\t\t\t\t\tmessage.responseModel ?? message.model,\n\t\t\t\t\tmessage.usage,\n\t\t\t\t\textensionOverride ? \"extension override\" : undefined,\n\t\t\t\t);\n\t\t\t\tthis.onWarmed?.(entry);\n\t\t\t}\n\t\t} catch {\n\t\t\t// Cache warming is best-effort and must not affect the active agent run.\n\t\t}\n\t\tif (this.run === run) this.schedule(run);\n\t}\n\n\tprivate refreshDeadlineMissed(run: ActiveRun): boolean {\n\t\tif (Date.now() <= run.refreshDeadlineAt) return false;\n\t\tthis.stop(\"cache refresh deadline missed\");\n\t\treturn true;\n\t}\n\n\tprivate validateRun(run: ActiveRun): boolean {\n\t\tif (this.run !== run) return false;\n\t\tconst reason = this.getModeStopReason(run) ?? (!run.isCurrent() ? \"conversation context changed\" : undefined);\n\t\tif (!reason) return true;\n\t\tthis.stop(reason);\n\t\treturn false;\n\t}\n\n\tprivate getModeStopReason(run: ActiveRun): string | undefined {\n\t\tconst mode = this.getMode();\n\t\tif (mode === \"off\") return \"cache warming disabled\";\n\t\tif (mode === \"streaming\" && run.phase === \"idle\") return \"agent run settled\";\n\t\treturn undefined;\n\t}\n\n\tprivate evaluate(run: ActiveRun): CacheWarmingDecision {\n\t\tconst model = run.model;\n\t\tconst promptTokens = lastPromptTokens(this.sessionManager.getBranch());\n\t\tconst cacheHitCost = price(model, { cacheRead: promptTokens });\n\t\tconst cacheMissCost = price(\n\t\t\tmodel,\n\t\t\tmodel.cost.cacheWrite > 0 ? { cacheWrite: promptTokens } : { input: promptTokens },\n\t\t);\n\t\tconst warmCost = price(model, { cacheRead: promptTokens, output: 1 });\n\t\tconst missCost = Math.max(0, cacheMissCost - cacheHitCost);\n\t\tconst continuationProbability = run.phase === \"idle\" ? IDLE_CONTINUATION_PROBABILITY : 1;\n\t\tconst economicsAvailable = promptTokens > 0 && (cacheHitCost > 0 || cacheMissCost > 0);\n\t\tconst expectedSavings = continuationProbability * missCost - warmCost;\n\t\treturn {\n\t\t\tphase: run.phase,\n\t\t\twarmCost,\n\t\t\tmissCost,\n\t\t\tcontinuationProbability,\n\t\t\texpectedSavings,\n\t\t\teconomicsAvailable,\n\t\t\taction: expectedSavings >= CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS ? \"warm\" : \"stop\",\n\t\t};\n\t}\n}\n\nfunction formatDollars(value: number): string {\n\treturn value < 0 ? `-$${Math.abs(value).toFixed(3)}` : `$${value.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingEconomics(decision: CacheWarmingDecision): string {\n\tif (!decision.economicsAvailable) return \"cache economics unavailable\";\n\tconst probability = Math.round(decision.continuationProbability * 100);\n\tconst probabilityText =\n\t\tdecision.phase === \"streaming\"\n\t\t\t? `${probability}% continuation probability while agent is running`\n\t\t\t: `${probability}% continuation probability`;\n\tconst comparison = decision.action === \"warm\" ? \">=\" : \"<\";\n\treturn `${probabilityText}, expected savings ${formatDollars(decision.expectedSavings)} ${comparison} $${CACHE_WARMING_MINIMUM_EXPECTED_SAVINGS.toFixed(3)}`;\n}\n\nfunction formatCacheWarmingDecisionTime(nextWarmAt: number | undefined, now: number): string {\n\tif (nextWarmAt === undefined || nextWarmAt <= now) return \"Decision now\";\n\tlet remainingSeconds = Math.ceil((nextWarmAt - now) / 1000);\n\tconst hours = Math.floor(remainingSeconds / 3600);\n\tremainingSeconds %= 3600;\n\tconst minutes = Math.floor(remainingSeconds / 60);\n\tconst seconds = remainingSeconds % 60;\n\tconst parts: string[] = [];\n\tif (hours > 0) parts.push(`${hours}h`);\n\tif (minutes > 0) parts.push(`${minutes}m`);\n\tif (seconds > 0 || parts.length === 0) parts.push(`${seconds}s`);\n\treturn `Decision in ${parts.join(\" \")}`;\n}\n\n/** One-line status for `/session`. */\nexport function formatCacheWarmingStatus(status: CacheWarmingStatus, now = Date.now()): string {\n\tconst decision = status.decision;\n\t// A decision is attached once pi (or an extension) acted on it; \"inactive\"\n\t// without one never got that far.\n\tif (!decision || (status.state === \"inactive\" && !decision.economicsAvailable && !status.extensionOverride)) {\n\t\treturn `Inactive (${status.reason ?? \"unknown reason\"})`;\n\t}\n\tconst details = status.extensionOverride\n\t\t? `extension override, ${formatCacheWarmingEconomics(decision)}`\n\t\t: `${formatCacheWarmingEconomics(decision)} -> ${decision.action}`;\n\tif (status.state === \"inactive\") return `Stopped (${details})`;\n\tif (status.state === \"refreshing\") return `Warming cache (${details})`;\n\treturn `${formatCacheWarmingDecisionTime(status.nextWarmAt, now)} (${details})`;\n}\n\n/** One-line transcript text for persisted cache-warming usage. */\nexport function formatCacheWarmingUsage(entry: UsageEntry): string {\n\tconst note = entry.note ? ` (${entry.note})` : \"\";\n\tconst cost = entry.usage.cost.total.toFixed(6).replace(/(\\.\\d{3}\\d*?)0+$/, \"$1\");\n\treturn `Cache warmed${note}: $${cost}`;\n}\n"]}
@@ -7,7 +7,7 @@
7
7
  import type { AgentMessage, StreamFn, ThinkingLevel } from "@earendil-works/pi-agent-core";
8
8
  import { type RetryCallbacks, type RetryPolicy } from "@earendil-works/pi-ai";
9
9
  import type { AssistantMessage, Model, SimpleStreamOptions, TranscriptContext, Usage } from "@earendil-works/pi-ai/compat";
10
- import { type SessionEntry } from "../session-manager.ts";
10
+ import { type SessionEntry, type SessionProjection } from "../session-manager.ts";
11
11
  import { type FileOperations } from "./utils.ts";
12
12
  /** Details stored in CompactionEntry.details for file tracking */
13
13
  export interface CompactionDetails {
@@ -51,6 +51,8 @@ export interface ContextUsageEstimate {
51
51
  * If there are messages after the last usage, estimate their tokens with estimateTokens.
52
52
  */
53
53
  export declare function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate;
54
+ /** Estimate projected context without trusting usage captured before a later edit or compaction. */
55
+ export declare function estimateProjectedContextTokens(projection: SessionProjection, branchEntries: SessionEntry[]): ContextUsageEstimate;
54
56
  /**
55
57
  * Check if compaction should trigger based on context usage.
56
58
  */
@@ -1 +1 @@
1
- {"version":3,"file":"compaction.d.ts","sourceRoot":"","sources":["../../../src/core/compaction/compaction.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,+BAA+B,CAAC;AAC3F,OAAO,EAGN,KAAK,cAAc,EACnB,KAAK,WAAW,EAGhB,MAAM,uBAAuB,CAAC;AAC/B,OAAO,KAAK,EACX,gBAAgB,EAChB,KAAK,EACL,mBAAmB,EACnB,iBAAiB,EACjB,KAAK,EACL,MAAM,8BAA8B,CAAC;AAGtC,OAAO,EAGN,KAAK,YAAY,EAEjB,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EAIN,KAAK,cAAc,EAInB,MAAM,YAAY,CAAC;AAMpB,kEAAkE;AAClE,MAAM,WAAW,iBAAiB;IACjC,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,aAAa,EAAE,MAAM,EAAE,CAAC;CACxB;AAoDD,8EAA8E;AAC9E,MAAM,WAAW,gBAAgB,CAAC,CAAC,GAAG,OAAO;IAC5C,OAAO,EAAE,MAAM,CAAC;IAChB,gBAAgB,EAAE,MAAM,CAAC;IACzB,YAAY,EAAE,MAAM,CAAC;IACrB,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,2EAA2E;IAC3E,KAAK,CAAC,EAAE,KAAK,CAAC;IACd,+FAA+F;IAC/F,OAAO,CAAC,EAAE,CAAC,CAAC;CACZ;AA6BD,MAAM,WAAW,kBAAkB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,aAAa,EAAE,MAAM,CAAC;IACtB,gBAAgB,EAAE,MAAM,CAAC;CACzB;AAED,eAAO,MAAM,2BAA2B,EAAE,kBAIzC,CAAC;AAMF;;;GAGG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AAqBD;;GAEG;AACH,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,YAAY,EAAE,GAAG,KAAK,GAAG,SAAS,CAShF;AAED,MAAM,WAAW,oBAAoB;IACpC,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAUD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,YAAY,EAAE,GAAG,oBAAoB,CA4BpF;AAED;;GAEG;AACH,wBAAgB,aAAa,CAAC,aAAa,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,EAAE,QAAQ,EAAE,kBAAkB,GAAG,OAAO,CAGjH;AAwBD;;;GAGG;AACH,wBAAgB,cAAc,CAAC,OAAO,EAAE,YAAY,GAAG,MAAM,CAwC5D;AA2DD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,YAAY,EAAE,EAAE,UAAU,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,MAAM,CAO1G;AAED,MAAM,WAAW,cAAc;IAC9B,mCAAmC;IACnC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,qFAAqF;IACrF,cAAc,EAAE,MAAM,CAAC;IACvB,uEAAuE;IACvE,WAAW,EAAE,OAAO,CAAC;CACrB;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,YAAY,CAC3B,OAAO,EAAE,YAAY,EAAE,EACvB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,MAAM,EAChB,gBAAgB,EAAE,MAAM,GACtB,cAAc,CAkDhB;AAgFD;;;GAGG;AACH,wBAAgB,uBAAuB,CAAC,QAAQ,EAAE,gBAAgB,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAQrG;AAmBD;;;;;;GAMG;AACH,wBAAsB,qBAAqB,CAC1C,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,OAAO,EAAE,iBAAiB,EAC1B,OAAO,EAAE,mBAAmB,EAC5B,QAAQ,CAAC,EAAE,QAAQ,EACnB,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,GACxB,OAAO,CAAC,gBAAgB,CAAC,CAa3B;AAED;;;GAGG;AACH,wBAAsB,eAAe,CACpC,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC,MAAM,CAAC,CAmBjB;AAgBD,+EAA+E;AAC/E,wBAAsB,wBAAwB,CAC7C,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,KAAK,CAAA;CAAE,CAAC,CAuDzC;AAMD,MAAM,WAAW,qBAAqB;IACrC,kCAAkC;IAClC,gBAAgB,EAAE,MAAM,CAAC;IACzB,qDAAqD;IACrD,mBAAmB,EAAE,YAAY,EAAE,CAAC;IACpC,2EAA2E;IAC3E,kBAAkB,EAAE,YAAY,EAAE,CAAC;IACnC,iEAAiE;IACjE,WAAW,EAAE,OAAO,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,yDAAyD;IACzD,OAAO,EAAE,cAAc,CAAC;IACxB,8CAA8C;IAC9C,QAAQ,EAAE,kBAAkB,CAAC;CAC7B;AAED,wBAAgB,iBAAiB,CAChC,WAAW,EAAE,YAAY,EAAE,EAC3B,QAAQ,EAAE,kBAAkB,GAC1B,qBAAqB,GAAG,SAAS,CA4EnC;AAqBD;;;;;;;GAOG;AACH,wBAAsB,OAAO,CAC5B,WAAW,EAAE,qBAAqB,EAClC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,kBAAkB,CAAC,EAAE,MAAM,EAC3B,MAAM,CAAC,EAAE,WAAW,EACpB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC,gBAAgB,CAAC,CA6F3B","sourcesContent":["/**\n * Context compaction for long sessions.\n *\n * Pure functions for compaction logic. The session manager handles I/O,\n * and after compaction the session is reloaded.\n */\n\nimport type { AgentMessage, StreamFn, ThinkingLevel } from \"@earendil-works/pi-agent-core\";\nimport {\n\tcontentText,\n\tnormalizeContext,\n\ttype RetryCallbacks,\n\ttype RetryPolicy,\n\tretryAssistantCall,\n\tuuidv7,\n} from \"@earendil-works/pi-ai\";\nimport type {\n\tAssistantMessage,\n\tModel,\n\tSimpleStreamOptions,\n\tTranscriptContext,\n\tUsage,\n} from \"@earendil-works/pi-ai/compat\";\nimport { completeSimple } from \"@earendil-works/pi-ai/compat\";\nimport { convertToLlm } from \"../messages.ts\";\nimport {\n\tbuildSessionContext,\n\ttype CompactionEntry,\n\ttype SessionEntry,\n\tsessionEntryToContextMessages,\n} from \"../session-manager.ts\";\nimport {\n\tcomputeFileLists,\n\tcreateFileOps,\n\textractFileOpsFromMessage,\n\ttype FileOperations,\n\tformatFileOperations,\n\tSUMMARIZATION_SYSTEM_PROMPT,\n\tserializeConversation,\n} from \"./utils.ts\";\n\n// ============================================================================\n// File Operation Tracking\n// ============================================================================\n\n/** Details stored in CompactionEntry.details for file tracking */\nexport interface CompactionDetails {\n\treadFiles: string[];\n\tmodifiedFiles: string[];\n}\n\n/**\n * Extract file operations from messages and previous compaction entries.\n */\nfunction extractFileOperations(\n\tmessages: AgentMessage[],\n\tentries: SessionEntry[],\n\tprevCompactionIndex: number,\n): FileOperations {\n\tconst fileOps = createFileOps();\n\n\t// Collect from previous compaction's details (if pi-generated)\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = entries[prevCompactionIndex] as CompactionEntry;\n\t\tif (!prevCompaction.fromHook && prevCompaction.details) {\n\t\t\t// fromHook field kept for session file compatibility\n\t\t\tconst details = prevCompaction.details as CompactionDetails;\n\t\t\tif (Array.isArray(details.readFiles)) {\n\t\t\t\tfor (const f of details.readFiles) fileOps.read.add(f);\n\t\t\t}\n\t\t\tif (Array.isArray(details.modifiedFiles)) {\n\t\t\t\tfor (const f of details.modifiedFiles) fileOps.edited.add(f);\n\t\t\t}\n\t\t}\n\t}\n\n\t// Extract from tool calls in messages\n\tfor (const msg of messages) {\n\t\textractFileOpsFromMessage(msg, fileOps);\n\t}\n\n\treturn fileOps;\n}\n\n// ============================================================================\n// Message Extraction\n// ============================================================================\n\n/**\n * Extract AgentMessage from an entry if it produces one.\n * Returns undefined for entries that don't contribute to LLM context.\n */\nfunction getMessageFromEntryForCompaction(entry: SessionEntry): AgentMessage | undefined {\n\tif (entry.type === \"compaction\") {\n\t\treturn undefined;\n\t}\n\t// System messages are prompt state, not conversation; the compaction entry carries their replay.\n\tconst message = sessionEntryToContextMessages(entry)[0];\n\treturn message?.role === \"system\" ? undefined : message;\n}\n\n/** Result from compact() - SessionManager adds uuid/parentUuid when saving */\nexport interface CompactionResult<T = unknown> {\n\tsummary: string;\n\tfirstKeptEntryId: string;\n\ttokensBefore: number;\n\testimatedTokensAfter?: number;\n\t/** Usage from the LLM call(s) that generated this summary, if available */\n\tusage?: Usage;\n\t/** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */\n\tdetails?: T;\n}\n\nfunction combineUsage(first: Usage, second: Usage): Usage {\n\treturn {\n\t\tinput: first.input + second.input,\n\t\toutput: first.output + second.output,\n\t\tcacheRead: first.cacheRead + second.cacheRead,\n\t\tcacheWrite: first.cacheWrite + second.cacheWrite,\n\t\t...(first.cacheWrite1h !== undefined || second.cacheWrite1h !== undefined\n\t\t\t? { cacheWrite1h: (first.cacheWrite1h ?? 0) + (second.cacheWrite1h ?? 0) }\n\t\t\t: {}),\n\t\t...(first.reasoning !== undefined || second.reasoning !== undefined\n\t\t\t? { reasoning: (first.reasoning ?? 0) + (second.reasoning ?? 0) }\n\t\t\t: {}),\n\t\ttotalTokens: first.totalTokens + second.totalTokens,\n\t\tcost: {\n\t\t\tinput: first.cost.input + second.cost.input,\n\t\t\toutput: first.cost.output + second.cost.output,\n\t\t\tcacheRead: first.cost.cacheRead + second.cost.cacheRead,\n\t\t\tcacheWrite: first.cost.cacheWrite + second.cost.cacheWrite,\n\t\t\ttotal: first.cost.total + second.cost.total,\n\t\t},\n\t};\n}\n\n// ============================================================================\n// Types\n// ============================================================================\n\nexport interface CompactionSettings {\n\tenabled: boolean;\n\treserveTokens: number;\n\tkeepRecentTokens: number;\n}\n\nexport const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {\n\tenabled: true,\n\treserveTokens: 16384,\n\tkeepRecentTokens: 20000,\n};\n\n// ============================================================================\n// Token calculation\n// ============================================================================\n\n/**\n * Calculate total context tokens from usage.\n * Uses the native totalTokens field when available, falls back to computing from components.\n */\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\n/**\n * Get usage from an assistant message if available.\n * Skips aborted, error, and all-zero usage messages as they don't have valid usage data.\n */\nfunction getAssistantUsage(msg: AgentMessage): Usage | undefined {\n\tif (msg.role === \"assistant\" && \"usage\" in msg) {\n\t\tconst assistantMsg = msg as AssistantMessage;\n\t\tif (\n\t\t\tassistantMsg.stopReason !== \"aborted\" &&\n\t\t\tassistantMsg.stopReason !== \"error\" &&\n\t\t\tassistantMsg.usage &&\n\t\t\tcalculateContextTokens(assistantMsg.usage) > 0\n\t\t) {\n\t\t\treturn assistantMsg.usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\n/**\n * Find the last valid assistant message usage from session entries.\n */\nexport function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefined {\n\tfor (let i = entries.length - 1; i >= 0; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"message\") {\n\t\t\tconst usage = getAssistantUsage(entry.message);\n\t\t\tif (usage) return usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\nexport interface ContextUsageEstimate {\n\ttokens: number;\n\tusageTokens: number;\n\ttrailingTokens: number;\n\tlastUsageIndex: number | null;\n}\n\nfunction getLastAssistantUsageInfo(messages: AgentMessage[]): { usage: Usage; index: number } | undefined {\n\tfor (let i = messages.length - 1; i >= 0; i--) {\n\t\tconst usage = getAssistantUsage(messages[i]);\n\t\tif (usage) return { usage, index: i };\n\t}\n\treturn undefined;\n}\n\n/**\n * Estimate context tokens from messages, using the last assistant usage when available.\n * If there are messages after the last usage, estimate their tokens with estimateTokens.\n */\nexport function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\n\tif (!usageInfo) {\n\t\tlet estimated = 0;\n\t\tfor (const message of messages) {\n\t\t\testimated += estimateTokens(message);\n\t\t}\n\t\treturn {\n\t\t\ttokens: estimated,\n\t\t\tusageTokens: 0,\n\t\t\ttrailingTokens: estimated,\n\t\t\tlastUsageIndex: null,\n\t\t};\n\t}\n\n\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\tlet trailingTokens = 0;\n\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\ttrailingTokens += estimateTokens(messages[i]);\n\t}\n\n\treturn {\n\t\ttokens: usageTokens + trailingTokens,\n\t\tusageTokens,\n\t\ttrailingTokens,\n\t\tlastUsageIndex: usageInfo.index,\n\t};\n}\n\n/**\n * Check if compaction should trigger based on context usage.\n */\nexport function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings): boolean {\n\tif (!settings.enabled) return false;\n\treturn contextTokens > contextWindow - settings.reserveTokens;\n}\n\n// ============================================================================\n// Cut point detection\n// ============================================================================\n\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nfunction estimateTextAndImageContentChars(content: string | Array<{ type: string; text?: string }>): number {\n\tif (typeof content === \"string\") {\n\t\treturn content.length;\n\t}\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\" && block.text) {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"image\") {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\n/**\n * Estimate token count for a message using chars/4 heuristic.\n * This is conservative (overestimates tokens).\n */\nexport function estimateTokens(message: AgentMessage): number {\n\tlet chars = 0;\n\n\tswitch (message.role) {\n\t\tcase \"user\": {\n\t\t\tchars = estimateTextAndImageContentChars(\n\t\t\t\t(message as { content: string | Array<{ type: string; text?: string }> }).content,\n\t\t\t);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"assistant\": {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\tfor (const block of assistant.content) {\n\t\t\t\tif (block.type === \"text\") {\n\t\t\t\t\tchars += block.text.length;\n\t\t\t\t} else if (block.type === \"thinking\") {\n\t\t\t\t\tchars += block.thinking.length;\n\t\t\t\t} else if (block.type === \"toolCall\") {\n\t\t\t\t\tchars += block.name.length + JSON.stringify(block.arguments).length;\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"custom\":\n\t\tcase \"toolResult\": {\n\t\t\tchars = estimateTextAndImageContentChars(message.content);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"bashExecution\": {\n\t\t\tchars = message.command.length + message.output.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\": {\n\t\t\tchars = message.summary.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t}\n\n\treturn 0;\n}\n\nfunction isCutPointMessage(message: AgentMessage): boolean {\n\tswitch (message.role) {\n\t\tcase \"user\":\n\t\tcase \"assistant\":\n\t\tcase \"bashExecution\":\n\t\tcase \"custom\":\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\":\n\t\t\treturn true;\n\t\tcase \"toolResult\":\n\t\t\treturn false;\n\t}\n\treturn false;\n}\n\nfunction isTurnStartMessage(message: AgentMessage): boolean {\n\tswitch (message.role) {\n\t\tcase \"user\":\n\t\tcase \"bashExecution\":\n\t\tcase \"custom\":\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\":\n\t\t\treturn true;\n\t\tcase \"assistant\":\n\t\tcase \"toolResult\":\n\t\t\treturn false;\n\t}\n\treturn false;\n}\n\nfunction isTurnStartEntry(entry: SessionEntry): boolean {\n\tif (entry.type === \"compaction\") {\n\t\treturn false;\n\t}\n\treturn sessionEntryToContextMessages(entry).some(isTurnStartMessage);\n}\n\n/**\n * Find valid cut points: indices of context-visible user-like or assistant messages.\n * Never cut at tool results (they must follow their tool call).\n * When we cut at an assistant message with tool calls, its tool results follow it\n * and will be kept.\n */\nfunction findValidCutPoints(entries: SessionEntry[], startIndex: number, endIndex: number): number[] {\n\tconst cutPoints: number[] = [];\n\tfor (let i = startIndex; i < endIndex; i++) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"compaction\") {\n\t\t\tcontinue;\n\t\t}\n\t\tif (sessionEntryToContextMessages(entry).some(isCutPointMessage)) {\n\t\t\tcutPoints.push(i);\n\t\t}\n\t}\n\treturn cutPoints;\n}\n\n/**\n * Find the context-visible user-role message that starts the turn containing the given entry index.\n * Returns -1 if no turn start found before the index.\n */\nexport function findTurnStartIndex(entries: SessionEntry[], entryIndex: number, startIndex: number): number {\n\tfor (let i = entryIndex; i >= startIndex; i--) {\n\t\tif (isTurnStartEntry(entries[i])) {\n\t\t\treturn i;\n\t\t}\n\t}\n\treturn -1;\n}\n\nexport interface CutPointResult {\n\t/** Index of first entry to keep */\n\tfirstKeptEntryIndex: number;\n\t/** Index of user message that starts the turn being split, or -1 if not splitting */\n\tturnStartIndex: number;\n\t/** Whether this cut splits a turn (cut point is not a user message) */\n\tisSplitTurn: boolean;\n}\n\n/**\n * Find the cut point in session entries that keeps approximately `keepRecentTokens`.\n *\n * Algorithm: Walk backwards from newest, accumulating estimated message sizes.\n * Stop when we've accumulated >= keepRecentTokens. Cut at that point.\n *\n * Can cut at user OR assistant messages (never tool results). When cutting at an\n * assistant message with tool calls, its tool results come after and will be kept.\n *\n * Returns CutPointResult with:\n * - firstKeptEntryIndex: the entry index to start keeping from\n * - turnStartIndex: if cutting mid-turn, the user message that started that turn\n * - isSplitTurn: whether we're cutting in the middle of a turn\n *\n * Only considers entries between `startIndex` and `endIndex` (exclusive).\n */\nexport function findCutPoint(\n\tentries: SessionEntry[],\n\tstartIndex: number,\n\tendIndex: number,\n\tkeepRecentTokens: number,\n): CutPointResult {\n\tconst cutPoints = findValidCutPoints(entries, startIndex, endIndex);\n\n\tif (cutPoints.length === 0) {\n\t\treturn { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };\n\t}\n\n\t// Walk backwards from newest, accumulating estimated message sizes\n\tlet accumulatedTokens = 0;\n\tlet cutIndex = cutPoints[0]; // Default: keep from first message (not header)\n\n\tfor (let i = endIndex - 1; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\tconst messageTokens = sessionEntryToContextMessages(entry).reduce(\n\t\t\t(sum, message) => sum + estimateTokens(message),\n\t\t\t0,\n\t\t);\n\t\tif (messageTokens === 0) continue;\n\t\taccumulatedTokens += messageTokens;\n\n\t\t// Check if we've exceeded the budget\n\t\tif (accumulatedTokens >= keepRecentTokens) {\n\t\t\t// Prefer the closest valid cut point at or after this entry. If trailing\n\t\t\t// tool results exceed the budget by themselves, keep their preceding\n\t\t\t// assistant tool call instead of falling back to the first message.\n\t\t\tcutIndex = cutPoints.find((candidate) => candidate >= i) ?? cutPoints[cutPoints.length - 1];\n\t\t\tbreak;\n\t\t}\n\t}\n\n\t// Scan backwards from cutIndex to include adjacent metadata entries that do not affect context.\n\twhile (cutIndex > startIndex) {\n\t\tconst prevEntry = entries[cutIndex - 1];\n\t\t// Stop at compaction boundaries or context-visible entries.\n\t\tif (prevEntry.type === \"compaction\" || sessionEntryToContextMessages(prevEntry).length > 0) {\n\t\t\tbreak;\n\t\t}\n\t\tcutIndex--;\n\t}\n\n\t// Determine if this is a split turn\n\tconst cutEntry = entries[cutIndex];\n\tconst startsTurn = isTurnStartEntry(cutEntry);\n\tconst turnStartIndex = startsTurn ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);\n\n\treturn {\n\t\tfirstKeptEntryIndex: cutIndex,\n\t\tturnStartIndex,\n\t\tisSplitTurn: !startsTurn && turnStartIndex !== -1,\n\t};\n}\n\n// ============================================================================\n// Summarization\n// ============================================================================\n\nconst SUMMARIZATION_PROMPT = `The messages above are a conversation to summarize. Create a structured context checkpoint summary that another LLM will use to continue the work.\n\nUse this EXACT format:\n\n## Goal\n[What is the user trying to accomplish? Can be multiple items if the session covers different tasks.]\n\n## Constraints & Preferences\n- [Any constraints, preferences, or requirements mentioned by user]\n- [Or \"(none)\" if none were mentioned]\n\n## Progress\n### Done\n- [x] [Completed tasks/changes]\n\n### In Progress\n- [ ] [Current work]\n\n### Blocked\n- [Issues preventing progress, if any]\n\n## Key Decisions\n- **[Decision]**: [Brief rationale]\n\n## Next Steps\n1. [Ordered list of what should happen next]\n\n## Critical Context\n- [Any data, examples, or references needed to continue]\n- [Or \"(none)\" if not applicable]\n\nKeep each section concise. Preserve exact file paths, function names, and error messages.`;\n\nconst UPDATE_SUMMARIZATION_INSTRUCTIONS = `Update the existing structured summary with new information. RULES:\n- PRESERVE all existing information from the previous summary\n- ADD new progress, decisions, and context from the new messages\n- UPDATE the Progress section: move items from \"In Progress\" to \"Done\" when completed\n- UPDATE \"Next Steps\" based on what was accomplished\n- PRESERVE exact file paths, function names, and error messages\n- If something is no longer relevant, you may remove it\n\nUse this EXACT format:\n\n## Goal\n[Preserve existing goals, add new ones if the task expanded]\n\n## Constraints & Preferences\n- [Preserve existing, add new ones discovered]\n\n## Progress\n### Done\n- [x] [Include previously done items AND newly completed items]\n\n### In Progress\n- [ ] [Current work - update based on progress]\n\n### Blocked\n- [Current blockers - remove if resolved]\n\n## Key Decisions\n- **[Decision]**: [Brief rationale] (preserve all previous, add new)\n\n## Next Steps\n1. [Update based on current state]\n\n## Critical Context\n- [Preserve important context, add new if needed]\n\nKeep each section concise. Preserve exact file paths, function names, and error messages.`;\n\nconst UPDATE_SUMMARIZATION_PROMPT = `The messages above are NEW conversation messages to incorporate into the existing summary provided in <previous-summary> tags.\n\n${UPDATE_SUMMARIZATION_INSTRUCTIONS}`;\n\n/**\n * Returns an error message when a summarization response cannot safely be persisted.\n * A length stop contains partial text and must not become a session checkpoint.\n */\nexport function getSummarizationFailure(response: AssistantMessage, label: string): string | undefined {\n\tif (response.stopReason === \"error\") {\n\t\treturn `${label} failed: ${response.errorMessage || \"Unknown error\"}`;\n\t}\n\tif (response.stopReason === \"length\") {\n\t\treturn `${label} failed: generation hit the token cap and the summary is incomplete`;\n\t}\n\treturn undefined;\n}\n\nfunction createSummarizationOptions(\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tenv: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n\tsessionId: string | undefined,\n): SimpleStreamOptions {\n\tconst options: SimpleStreamOptions = { maxTokens, signal, apiKey, headers, env, sessionId };\n\tif (model.reasoning && thinkingLevel && thinkingLevel !== \"off\") {\n\t\toptions.reasoning = thinkingLevel;\n\t}\n\treturn options;\n}\n\n/**\n * Shared choke point for every compaction/branch-summary summarization call. Wraps the\n * single LLM call in {@link retryAssistantCall} so transient stream drops (e.g.\n * `terminated`, socket close) honor the configured retry policy instead of failing\n * the whole compaction on the first attempt. Deterministic errors and aborts return\n * immediately (see {@link retryAssistantCall}).\n */\nexport async function completeSummarization(\n\tmodel: Model<any>,\n\tcontext: TranscriptContext,\n\toptions: SimpleStreamOptions,\n\tstreamFn?: StreamFn,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n): Promise<AssistantMessage> {\n\t// Avoid cache writes for one-off summaries. Reuse caller-supplied routing when available;\n\t// callers without a session ID, including branch summaries, receive a fresh routing ID.\n\tconst requestOptions: SimpleStreamOptions = {\n\t\t...options,\n\t\tcacheRetention: \"none\",\n\t\tsessionId: options.sessionId ?? uuidv7(),\n\t};\n\tconst produce = async (): Promise<AssistantMessage> =>\n\t\tstreamFn\n\t\t\t? (await streamFn(model, context, requestOptions)).result()\n\t\t\t: completeSimple(model, context, requestOptions);\n\treturn retryAssistantCall(produce, retry, requestOptions.signal, callbacks);\n}\n\n/**\n * Generate a summary of the conversation using the LLM.\n * If previousSummary is provided, uses the update prompt to merge.\n */\nexport async function generateSummary(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<string> {\n\treturn (\n\t\tawait generateSummaryWithUsage(\n\t\t\tcurrentMessages,\n\t\t\tmodel,\n\t\t\treserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tcustomInstructions,\n\t\t\tpreviousSummary,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tenv,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t)\n\t).text;\n}\n\n/** Build the provider context for a standalone summary request. */\nfunction buildSummarizationContext(promptText: string): TranscriptContext {\n\treturn normalizeContext({\n\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\tmessages: [\n\t\t\t{\n\t\t\t\trole: \"user\",\n\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\ttimestamp: Date.now(),\n\t\t\t},\n\t\t],\n\t});\n}\n\n/** Generate or update a conversation summary and return its provider usage. */\nexport async function generateSummaryWithUsage(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<{ text: string; usage: Usage }> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.8 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t);\n\n\t// Use update prompt if we have a previous summary, otherwise initial prompt\n\tlet basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;\n\tif (customInstructions) {\n\t\tbasePrompt = `${basePrompt}\\n\\nAdditional focus: ${customInstructions}`;\n\t}\n\n\t// Serialize conversation to text so model doesn't try to continue it\n\t// Convert to LLM messages first (handles custom types like bashExecution, custom, etc.)\n\tconst llmMessages = convertToLlm(currentMessages);\n\tconst conversationText = serializeConversation(llmMessages);\n\n\t// Build the prompt with conversation wrapped in tags\n\tlet promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n`;\n\tif (previousSummary) {\n\t\tpromptText += `<previous-summary>\\n${previousSummary}\\n</previous-summary>\\n\\n`;\n\t}\n\tpromptText += basePrompt;\n\n\tconst completionOptions = createSummarizationOptions(\n\t\tmodel,\n\t\tmaxTokens,\n\t\tapiKey,\n\t\theaders,\n\t\tenv,\n\t\tsignal,\n\t\tthinkingLevel,\n\t\tsessionId,\n\t);\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\tbuildSummarizationContext(promptText),\n\t\tcompletionOptions,\n\t\tstreamFn,\n\t\tretry,\n\t\tcallbacks,\n\t);\n\n\tconst failure = getSummarizationFailure(response, \"Summarization\");\n\tif (failure) {\n\t\tthrow new Error(failure);\n\t}\n\tif (response.content.some((block) => block.type === \"toolCall\")) {\n\t\tthrow new Error(\"Summarization attempted to call a tool\");\n\t}\n\n\tconst textContent = contentText(response.content);\n\n\treturn { text: textContent, usage: response.usage };\n}\n\n// ============================================================================\n// Compaction Preparation (for extensions)\n// ============================================================================\n\nexport interface CompactionPreparation {\n\t/** UUID of first entry to keep */\n\tfirstKeptEntryId: string;\n\t/** Messages that will be summarized and discarded */\n\tmessagesToSummarize: AgentMessage[];\n\t/** Messages that will be turned into turn prefix summary (if splitting) */\n\tturnPrefixMessages: AgentMessage[];\n\t/** Whether this is a split turn (cut point in middle of turn) */\n\tisSplitTurn: boolean;\n\ttokensBefore: number;\n\t/** Summary from previous compaction, for iterative update */\n\tpreviousSummary?: string;\n\t/** File operations extracted from messagesToSummarize */\n\tfileOps: FileOperations;\n\t/** Compaction settions from settings.jsonl\t*/\n\tsettings: CompactionSettings;\n}\n\nexport function prepareCompaction(\n\tpathEntries: SessionEntry[],\n\tsettings: CompactionSettings,\n): CompactionPreparation | undefined {\n\tif (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === \"compaction\") {\n\t\treturn undefined;\n\t}\n\n\tlet prevCompactionIndex = -1;\n\tfor (let i = pathEntries.length - 1; i >= 0; i--) {\n\t\tif (pathEntries[i].type === \"compaction\") {\n\t\t\tprevCompactionIndex = i;\n\t\t\tbreak;\n\t\t}\n\t}\n\n\tlet previousSummary: string | undefined;\n\tlet boundaryStart = 0;\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;\n\t\tpreviousSummary = prevCompaction.summary;\n\t\tconst firstKeptEntryIndex = pathEntries.findIndex((entry) => entry.id === prevCompaction.firstKeptEntryId);\n\t\tboundaryStart = firstKeptEntryIndex >= 0 ? firstKeptEntryIndex : prevCompactionIndex + 1;\n\t}\n\tconst boundaryEnd = pathEntries.length;\n\n\tconst tokensBefore = estimateContextTokens(buildSessionContext(pathEntries).messages).tokens;\n\n\tconst cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, settings.keepRecentTokens);\n\n\t// Get UUID of first kept entry\n\tconst firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];\n\tif (!firstKeptEntry?.id) {\n\t\treturn undefined; // Session needs migration\n\t}\n\tconst firstKeptEntryId = firstKeptEntry.id;\n\n\tconst historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;\n\n\t// Messages to summarize (will be discarded after summary)\n\tconst messagesToSummarize: AgentMessage[] = [];\n\tfor (let i = boundaryStart; i < historyEnd; i++) {\n\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\tif (msg) messagesToSummarize.push(msg);\n\t}\n\n\t// Messages for turn prefix summary (if splitting a turn)\n\tconst turnPrefixMessages: AgentMessage[] = [];\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {\n\t\t\tconst msg = getMessageFromEntryForCompaction(pathEntries[i]);\n\t\t\tif (msg) turnPrefixMessages.push(msg);\n\t\t}\n\t}\n\n\tif (messagesToSummarize.length === 0 && turnPrefixMessages.length === 0) {\n\t\treturn undefined;\n\t}\n\n\t// Extract file operations from messages and previous compaction\n\tconst fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);\n\n\t// Also extract file ops from turn prefix if splitting\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (const msg of turnPrefixMessages) {\n\t\t\textractFileOpsFromMessage(msg, fileOps);\n\t\t}\n\t}\n\n\treturn {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn: cutPoint.isSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t};\n}\n\n// ============================================================================\n// Main compaction function\n// ============================================================================\n\nconst TURN_PREFIX_SUMMARIZATION_PROMPT = `This is the PREFIX of a turn that was too large to keep. The SUFFIX (recent work) is retained.\n\nSummarize the prefix to provide context for the retained suffix:\n\n## Original Request\n[What did the user ask for in this turn?]\n\n## Early Progress\n- [Key decisions and work done in the prefix]\n\n## Context for Suffix\n- [Information needed to understand the retained recent work]\n\nBe concise. Focus on what's needed to understand the kept suffix.`;\n\n/**\n * Generate summaries for compaction using prepared data.\n * Returns CompactionResult - SessionManager adds uuid/parentUuid when saving.\n *\n * @param preparation - Pre-calculated preparation from prepareCompaction()\n * @param customInstructions - Optional custom focus for the summary\n * @param sessionId - Optional routing session ID forwarded without enabling prompt caching\n */\nexport async function compact(\n\tpreparation: CompactionPreparation,\n\tmodel: Model<any>,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tcustomInstructions?: string,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<CompactionResult> {\n\tconst {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t} = preparation;\n\n\t// Generate summaries and merge into one\n\tlet summary: string;\n\tlet summaryUsage: Usage;\n\n\tif (isSplitTurn && turnPrefixMessages.length > 0) {\n\t\tlet historyText = previousSummary ?? \"No prior history.\";\n\t\tlet historyUsage: Usage | undefined;\n\t\tif (messagesToSummarize.length > 0) {\n\t\t\tconst historyResult = await generateSummaryWithUsage(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\tcustomInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tenv,\n\t\t\t\tretry,\n\t\t\t\tcallbacks,\n\t\t\t\tsessionId,\n\t\t\t);\n\t\t\thistoryText = historyResult.text;\n\t\t\thistoryUsage = historyResult.usage;\n\t\t}\n\t\tconst turnPrefixResult = await generateTurnPrefixSummary(\n\t\t\tturnPrefixMessages,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tenv,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t);\n\t\t// Merge into single summary\n\t\tsummary = `${historyText}\\n\\n---\\n\\n**Turn Context (split turn):**\\n\\n${turnPrefixResult.text}`;\n\t\tsummaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage;\n\t} else {\n\t\t// Just generate history summary\n\t\tconst result = await generateSummaryWithUsage(\n\t\t\tmessagesToSummarize,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tcustomInstructions,\n\t\t\tpreviousSummary,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tenv,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t);\n\t\tsummary = result.text;\n\t\tsummaryUsage = result.usage;\n\t}\n\n\t// Compute file lists and append to summary\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\tsummary += formatFileOperations(readFiles, modifiedFiles);\n\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tusage: summaryUsage,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t};\n}\n\n/**\n * Generate a summary for a turn prefix (when splitting a turn).\n */\nasync function generateTurnPrefixSummary(\n\tmessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tenv?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<{ text: string; usage: Usage }> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.5 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t); // Smaller budget for turn prefix\n\tconst llmMessages = convertToLlm(messages);\n\tconst conversationText = serializeConversation(llmMessages);\n\tconst promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\tbuildSummarizationContext(promptText),\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel, sessionId),\n\t\tstreamFn,\n\t\tretry,\n\t\tcallbacks,\n\t);\n\n\tconst failure = getSummarizationFailure(response, \"Turn prefix summarization\");\n\tif (failure) {\n\t\tthrow new Error(failure);\n\t}\n\tif (response.content.some((block) => block.type === \"toolCall\")) {\n\t\tthrow new Error(\"Turn prefix summarization attempted to call a tool\");\n\t}\n\n\treturn {\n\t\ttext: contentText(response.content),\n\t\tusage: response.usage,\n\t};\n}\n"]}
1
+ {"version":3,"file":"compaction.d.ts","sourceRoot":"","sources":["../../../src/core/compaction/compaction.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,+BAA+B,CAAC;AAC3F,OAAO,EAIN,KAAK,cAAc,EACnB,KAAK,WAAW,EAGhB,MAAM,uBAAuB,CAAC;AAC/B,OAAO,KAAK,EACX,gBAAgB,EAChB,KAAK,EACL,mBAAmB,EAEnB,iBAAiB,EACjB,KAAK,EACL,MAAM,8BAA8B,CAAC;AAGtC,OAAO,EAIN,KAAK,YAAY,EACjB,KAAK,iBAAiB,EAEtB,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EAIN,KAAK,cAAc,EAInB,MAAM,YAAY,CAAC;AAMpB,kEAAkE;AAClE,MAAM,WAAW,iBAAiB;IACjC,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,aAAa,EAAE,MAAM,EAAE,CAAC;CACxB;AAiDD,8EAA8E;AAC9E,MAAM,WAAW,gBAAgB,CAAC,CAAC,GAAG,OAAO;IAC5C,OAAO,EAAE,MAAM,CAAC;IAChB,gBAAgB,EAAE,MAAM,CAAC;IACzB,YAAY,EAAE,MAAM,CAAC;IACrB,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,2EAA2E;IAC3E,KAAK,CAAC,EAAE,KAAK,CAAC;IACd,+FAA+F;IAC/F,OAAO,CAAC,EAAE,CAAC,CAAC;CACZ;AA6BD,MAAM,WAAW,kBAAkB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,aAAa,EAAE,MAAM,CAAC;IACtB,gBAAgB,EAAE,MAAM,CAAC;CACzB;AAED,eAAO,MAAM,2BAA2B,EAAE,kBAIzC,CAAC;AAMF;;;GAGG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AAqBD;;GAEG;AACH,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,YAAY,EAAE,GAAG,KAAK,GAAG,SAAS,CAShF;AAED,MAAM,WAAW,oBAAoB;IACpC,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAUD;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,YAAY,EAAE,GAAG,oBAAoB,CA4BpF;AAED,oGAAoG;AACpG,wBAAgB,8BAA8B,CAC7C,UAAU,EAAE,iBAAiB,EAC7B,aAAa,EAAE,YAAY,EAAE,GAC3B,oBAAoB,CAgCtB;AAED;;GAEG;AACH,wBAAgB,aAAa,CAAC,aAAa,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,EAAE,QAAQ,EAAE,kBAAkB,GAAG,OAAO,CAGjH;AAwBD;;;GAGG;AACH,wBAAgB,cAAc,CAAC,OAAO,EAAE,YAAY,GAAG,MAAM,CAmD5D;AA2DD;;;GAGG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,YAAY,EAAE,EAAE,UAAU,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,MAAM,CAO1G;AAED,MAAM,WAAW,cAAc;IAC9B,mCAAmC;IACnC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,qFAAqF;IACrF,cAAc,EAAE,MAAM,CAAC;IACvB,uEAAuE;IACvE,WAAW,EAAE,OAAO,CAAC;CACrB;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,YAAY,CAC3B,OAAO,EAAE,YAAY,EAAE,EACvB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,MAAM,EAChB,gBAAgB,EAAE,MAAM,GACtB,cAAc,CAkDhB;AAgFD;;;GAGG;AACH,wBAAgB,uBAAuB,CAAC,QAAQ,EAAE,gBAAgB,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAQrG;AAmBD;;;;;;GAMG;AACH,wBAAsB,qBAAqB,CAC1C,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,OAAO,EAAE,iBAAiB,EAC1B,OAAO,EAAE,mBAAmB,EAC5B,QAAQ,CAAC,EAAE,QAAQ,EACnB,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,GACxB,OAAO,CAAC,gBAAgB,CAAC,CAa3B;AAED;;;GAGG;AACH,wBAAsB,eAAe,CACpC,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC,MAAM,CAAC,CAmBjB;AAgBD,+EAA+E;AAC/E,wBAAsB,wBAAwB,CAC7C,eAAe,EAAE,YAAY,EAAE,EAC/B,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,MAAM,CAAC,EAAE,WAAW,EACpB,kBAAkB,CAAC,EAAE,MAAM,EAC3B,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,KAAK,CAAA;CAAE,CAAC,CAuDzC;AAMD,MAAM,WAAW,qBAAqB;IACrC,kCAAkC;IAClC,gBAAgB,EAAE,MAAM,CAAC;IACzB,qDAAqD;IACrD,mBAAmB,EAAE,YAAY,EAAE,CAAC;IACpC,2EAA2E;IAC3E,kBAAkB,EAAE,YAAY,EAAE,CAAC;IACnC,iEAAiE;IACjE,WAAW,EAAE,OAAO,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,yDAAyD;IACzD,OAAO,EAAE,cAAc,CAAC;IACxB,8CAA8C;IAC9C,QAAQ,EAAE,kBAAkB,CAAC;CAC7B;AAoFD,wBAAgB,iBAAiB,CAChC,WAAW,EAAE,YAAY,EAAE,EAC3B,QAAQ,EAAE,kBAAkB,GAC1B,qBAAqB,GAAG,SAAS,CA6DnC;AAqBD;;;;;;;GAOG;AACH,wBAAsB,OAAO,CAC5B,WAAW,EAAE,qBAAqB,EAClC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAChC,kBAAkB,CAAC,EAAE,MAAM,EAC3B,MAAM,CAAC,EAAE,WAAW,EACpB,aAAa,CAAC,EAAE,aAAa,EAC7B,QAAQ,CAAC,EAAE,QAAQ,EACnB,GAAG,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAC5B,KAAK,CAAC,EAAE,WAAW,EACnB,SAAS,CAAC,EAAE,cAAc,EAC1B,SAAS,CAAC,EAAE,MAAM,GAChB,OAAO,CAAC,gBAAgB,CAAC,CA6F3B","sourcesContent":["/**\n * Context compaction for long sessions.\n *\n * Pure functions for compaction logic. The session manager handles I/O,\n * and after compaction the session is reloaded.\n */\n\nimport type { AgentMessage, StreamFn, ThinkingLevel } from \"@earendil-works/pi-agent-core\";\nimport {\n\tcontentText,\n\tgetCurrentSystemMessage,\n\tnormalizeContext,\n\ttype RetryCallbacks,\n\ttype RetryPolicy,\n\tretryAssistantCall,\n\tuuidv7,\n} from \"@earendil-works/pi-ai\";\nimport type {\n\tAssistantMessage,\n\tModel,\n\tSimpleStreamOptions,\n\tSystemMessage,\n\tTranscriptContext,\n\tUsage,\n} from \"@earendil-works/pi-ai/compat\";\nimport { completeSimple } from \"@earendil-works/pi-ai/compat\";\nimport { convertToLlm } from \"../messages.ts\";\nimport {\n\tbuildSessionProjection,\n\ttype CompactionEntry,\n\ttype ProjectedSessionEntry,\n\ttype SessionEntry,\n\ttype SessionProjection,\n\tsessionEntryToContextMessages,\n} from \"../session-manager.ts\";\nimport {\n\tcomputeFileLists,\n\tcreateFileOps,\n\textractFileOpsFromMessage,\n\ttype FileOperations,\n\tformatFileOperations,\n\tSUMMARIZATION_SYSTEM_PROMPT,\n\tserializeConversation,\n} from \"./utils.ts\";\n\n// ============================================================================\n// File Operation Tracking\n// ============================================================================\n\n/** Details stored in CompactionEntry.details for file tracking */\nexport interface CompactionDetails {\n\treadFiles: string[];\n\tmodifiedFiles: string[];\n}\n\n/**\n * Extract file operations from messages and previous compaction entries.\n */\nfunction extractFileOperations(\n\tmessages: AgentMessage[],\n\tentries: SessionEntry[],\n\tprevCompactionIndex: number,\n): FileOperations {\n\tconst fileOps = createFileOps();\n\n\t// Collect from previous compaction's details (if pi-generated)\n\tif (prevCompactionIndex >= 0) {\n\t\tconst prevCompaction = entries[prevCompactionIndex] as CompactionEntry;\n\t\tif (!prevCompaction.fromHook && prevCompaction.details) {\n\t\t\t// fromHook field kept for session file compatibility\n\t\t\tconst details = prevCompaction.details as CompactionDetails;\n\t\t\tif (Array.isArray(details.readFiles)) {\n\t\t\t\tfor (const f of details.readFiles) fileOps.read.add(f);\n\t\t\t}\n\t\t\tif (Array.isArray(details.modifiedFiles)) {\n\t\t\t\tfor (const f of details.modifiedFiles) fileOps.edited.add(f);\n\t\t\t}\n\t\t}\n\t}\n\n\t// Extract from tool calls in messages\n\tfor (const msg of messages) {\n\t\textractFileOpsFromMessage(msg, fileOps);\n\t}\n\n\treturn fileOps;\n}\n\n// ============================================================================\n// Message Extraction\n// ============================================================================\n\n/**\n * Extract AgentMessage from an entry if it produces one.\n * Returns undefined for entries that don't contribute to LLM context.\n */\nfunction getMessagesFromProjectedEntryForCompaction(entry: ProjectedSessionEntry): AgentMessage[] {\n\tif (entry.sourceEntry.type === \"compaction\") return [];\n\t// System messages are prompt state, not conversation; the compaction entry carries their replay.\n\treturn entry.messages.filter((message) => message.role !== \"system\");\n}\n\n/** Result from compact() - SessionManager adds uuid/parentUuid when saving */\nexport interface CompactionResult<T = unknown> {\n\tsummary: string;\n\tfirstKeptEntryId: string;\n\ttokensBefore: number;\n\testimatedTokensAfter?: number;\n\t/** Usage from the LLM call(s) that generated this summary, if available */\n\tusage?: Usage;\n\t/** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */\n\tdetails?: T;\n}\n\nfunction combineUsage(first: Usage, second: Usage): Usage {\n\treturn {\n\t\tinput: first.input + second.input,\n\t\toutput: first.output + second.output,\n\t\tcacheRead: first.cacheRead + second.cacheRead,\n\t\tcacheWrite: first.cacheWrite + second.cacheWrite,\n\t\t...(first.cacheWrite1h !== undefined || second.cacheWrite1h !== undefined\n\t\t\t? { cacheWrite1h: (first.cacheWrite1h ?? 0) + (second.cacheWrite1h ?? 0) }\n\t\t\t: {}),\n\t\t...(first.reasoning !== undefined || second.reasoning !== undefined\n\t\t\t? { reasoning: (first.reasoning ?? 0) + (second.reasoning ?? 0) }\n\t\t\t: {}),\n\t\ttotalTokens: first.totalTokens + second.totalTokens,\n\t\tcost: {\n\t\t\tinput: first.cost.input + second.cost.input,\n\t\t\toutput: first.cost.output + second.cost.output,\n\t\t\tcacheRead: first.cost.cacheRead + second.cost.cacheRead,\n\t\t\tcacheWrite: first.cost.cacheWrite + second.cost.cacheWrite,\n\t\t\ttotal: first.cost.total + second.cost.total,\n\t\t},\n\t};\n}\n\n// ============================================================================\n// Types\n// ============================================================================\n\nexport interface CompactionSettings {\n\tenabled: boolean;\n\treserveTokens: number;\n\tkeepRecentTokens: number;\n}\n\nexport const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {\n\tenabled: true,\n\treserveTokens: 16384,\n\tkeepRecentTokens: 20000,\n};\n\n// ============================================================================\n// Token calculation\n// ============================================================================\n\n/**\n * Calculate total context tokens from usage.\n * Uses the native totalTokens field when available, falls back to computing from components.\n */\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\n/**\n * Get usage from an assistant message if available.\n * Skips aborted, error, and all-zero usage messages as they don't have valid usage data.\n */\nfunction getAssistantUsage(msg: AgentMessage): Usage | undefined {\n\tif (msg.role === \"assistant\" && \"usage\" in msg) {\n\t\tconst assistantMsg = msg as AssistantMessage;\n\t\tif (\n\t\t\tassistantMsg.stopReason !== \"aborted\" &&\n\t\t\tassistantMsg.stopReason !== \"error\" &&\n\t\t\tassistantMsg.usage &&\n\t\t\tcalculateContextTokens(assistantMsg.usage) > 0\n\t\t) {\n\t\t\treturn assistantMsg.usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\n/**\n * Find the last valid assistant message usage from session entries.\n */\nexport function getLastAssistantUsage(entries: SessionEntry[]): Usage | undefined {\n\tfor (let i = entries.length - 1; i >= 0; i--) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"message\") {\n\t\t\tconst usage = getAssistantUsage(entry.message);\n\t\t\tif (usage) return usage;\n\t\t}\n\t}\n\treturn undefined;\n}\n\nexport interface ContextUsageEstimate {\n\ttokens: number;\n\tusageTokens: number;\n\ttrailingTokens: number;\n\tlastUsageIndex: number | null;\n}\n\nfunction getLastAssistantUsageInfo(messages: AgentMessage[]): { usage: Usage; index: number } | undefined {\n\tfor (let i = messages.length - 1; i >= 0; i--) {\n\t\tconst usage = getAssistantUsage(messages[i]);\n\t\tif (usage) return { usage, index: i };\n\t}\n\treturn undefined;\n}\n\n/**\n * Estimate context tokens from messages, using the last assistant usage when available.\n * If there are messages after the last usage, estimate their tokens with estimateTokens.\n */\nexport function estimateContextTokens(messages: AgentMessage[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\n\tif (!usageInfo) {\n\t\tlet estimated = 0;\n\t\tfor (const message of messages) {\n\t\t\testimated += estimateTokens(message);\n\t\t}\n\t\treturn {\n\t\t\ttokens: estimated,\n\t\t\tusageTokens: 0,\n\t\t\ttrailingTokens: estimated,\n\t\t\tlastUsageIndex: null,\n\t\t};\n\t}\n\n\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\tlet trailingTokens = 0;\n\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\ttrailingTokens += estimateTokens(messages[i]);\n\t}\n\n\treturn {\n\t\ttokens: usageTokens + trailingTokens,\n\t\tusageTokens,\n\t\ttrailingTokens,\n\t\tlastUsageIndex: usageInfo.index,\n\t};\n}\n\n/** Estimate projected context without trusting usage captured before a later edit or compaction. */\nexport function estimateProjectedContextTokens(\n\tprojection: SessionProjection,\n\tbranchEntries: SessionEntry[],\n): ContextUsageEstimate {\n\tconst estimate = estimateContextTokens(projection.messages);\n\tif (estimate.lastUsageIndex !== null) {\n\t\tlet projectedMessageIndex = 0;\n\t\tlet usageEntryId: string | undefined;\n\t\tfor (const entry of projection.entries) {\n\t\t\tconst nextMessageIndex = projectedMessageIndex + entry.messages.length;\n\t\t\tif (estimate.lastUsageIndex < nextMessageIndex) {\n\t\t\t\tusageEntryId = entry.sourceEntry.id;\n\t\t\t\tbreak;\n\t\t\t}\n\t\t\tprojectedMessageIndex = nextMessageIndex;\n\t\t}\n\n\t\tconst usageEntryIndex = usageEntryId ? branchEntries.findIndex((entry) => entry.id === usageEntryId) : -1;\n\t\tlet latestInvalidatingEntryIndex = -1;\n\t\tfor (let i = branchEntries.length - 1; i >= 0; i--) {\n\t\t\tconst entry = branchEntries[i];\n\t\t\tif (entry.type === \"context_edit\" || entry.type === \"compaction\") {\n\t\t\t\tlatestInvalidatingEntryIndex = i;\n\t\t\t\tbreak;\n\t\t\t}\n\t\t}\n\t\tif (usageEntryIndex > latestInvalidatingEntryIndex) return estimate;\n\t}\n\n\tconst currentSystem = getCurrentSystemMessage(projection.messages);\n\tlet tokens = currentSystem ? estimateTokens(currentSystem) : 0;\n\tfor (const message of projection.messages) {\n\t\tif (message.role !== \"system\") tokens += estimateTokens(message);\n\t}\n\treturn { tokens, usageTokens: 0, trailingTokens: tokens, lastUsageIndex: null };\n}\n\n/**\n * Check if compaction should trigger based on context usage.\n */\nexport function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings): boolean {\n\tif (!settings.enabled) return false;\n\treturn contextTokens > contextWindow - settings.reserveTokens;\n}\n\n// ============================================================================\n// Cut point detection\n// ============================================================================\n\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nfunction estimateTextAndImageContentChars(content: string | Array<{ type: string; text?: string }>): number {\n\tif (typeof content === \"string\") {\n\t\treturn content.length;\n\t}\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\" && block.text) {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"image\") {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\n/**\n * Estimate token count for a message using chars/4 heuristic.\n * This is conservative (overestimates tokens).\n */\nexport function estimateTokens(message: AgentMessage): number {\n\tlet chars = 0;\n\n\tswitch (message.role) {\n\t\tcase \"system\": {\n\t\t\tconst system = message as SystemMessage;\n\t\t\tchars = estimateTextAndImageContentChars(system.content);\n\t\t\tif (system.sections) {\n\t\t\t\tfor (const section of Object.values(system.sections)) {\n\t\t\t\t\tif (section) chars += section.length;\n\t\t\t\t}\n\t\t\t}\n\t\t\tif (system.toolsAdded) chars += JSON.stringify(system.toolsAdded).length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"user\": {\n\t\t\tchars = estimateTextAndImageContentChars(\n\t\t\t\t(message as { content: string | Array<{ type: string; text?: string }> }).content,\n\t\t\t);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"assistant\": {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\tfor (const block of assistant.content) {\n\t\t\t\tif (block.type === \"text\") {\n\t\t\t\t\tchars += block.text.length;\n\t\t\t\t} else if (block.type === \"thinking\") {\n\t\t\t\t\tchars += block.thinking.length;\n\t\t\t\t} else if (block.type === \"toolCall\") {\n\t\t\t\t\tchars += block.name.length + JSON.stringify(block.arguments).length;\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"custom\":\n\t\tcase \"toolResult\": {\n\t\t\tchars = estimateTextAndImageContentChars(message.content);\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"bashExecution\": {\n\t\t\tchars = message.command.length + message.output.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\": {\n\t\t\tchars = message.summary.length;\n\t\t\treturn Math.ceil(chars / 4);\n\t\t}\n\t}\n\n\treturn 0;\n}\n\nfunction isCutPointMessage(message: AgentMessage): boolean {\n\tswitch (message.role) {\n\t\tcase \"user\":\n\t\tcase \"assistant\":\n\t\tcase \"bashExecution\":\n\t\tcase \"custom\":\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\":\n\t\t\treturn true;\n\t\tcase \"toolResult\":\n\t\t\treturn false;\n\t}\n\treturn false;\n}\n\nfunction isTurnStartMessage(message: AgentMessage): boolean {\n\tswitch (message.role) {\n\t\tcase \"user\":\n\t\tcase \"bashExecution\":\n\t\tcase \"custom\":\n\t\tcase \"branchSummary\":\n\t\tcase \"compactionSummary\":\n\t\t\treturn true;\n\t\tcase \"assistant\":\n\t\tcase \"toolResult\":\n\t\t\treturn false;\n\t}\n\treturn false;\n}\n\nfunction isTurnStartEntry(entry: SessionEntry): boolean {\n\tif (entry.type === \"compaction\") {\n\t\treturn false;\n\t}\n\treturn sessionEntryToContextMessages(entry).some(isTurnStartMessage);\n}\n\n/**\n * Find valid cut points: indices of context-visible user-like or assistant messages.\n * Never cut at tool results (they must follow their tool call).\n * When we cut at an assistant message with tool calls, its tool results follow it\n * and will be kept.\n */\nfunction findValidCutPoints(entries: SessionEntry[], startIndex: number, endIndex: number): number[] {\n\tconst cutPoints: number[] = [];\n\tfor (let i = startIndex; i < endIndex; i++) {\n\t\tconst entry = entries[i];\n\t\tif (entry.type === \"compaction\") {\n\t\t\tcontinue;\n\t\t}\n\t\tif (sessionEntryToContextMessages(entry).some(isCutPointMessage)) {\n\t\t\tcutPoints.push(i);\n\t\t}\n\t}\n\treturn cutPoints;\n}\n\n/**\n * Find the context-visible user-role message that starts the turn containing the given entry index.\n * Returns -1 if no turn start found before the index.\n */\nexport function findTurnStartIndex(entries: SessionEntry[], entryIndex: number, startIndex: number): number {\n\tfor (let i = entryIndex; i >= startIndex; i--) {\n\t\tif (isTurnStartEntry(entries[i])) {\n\t\t\treturn i;\n\t\t}\n\t}\n\treturn -1;\n}\n\nexport interface CutPointResult {\n\t/** Index of first entry to keep */\n\tfirstKeptEntryIndex: number;\n\t/** Index of user message that starts the turn being split, or -1 if not splitting */\n\tturnStartIndex: number;\n\t/** Whether this cut splits a turn (cut point is not a user message) */\n\tisSplitTurn: boolean;\n}\n\n/**\n * Find the cut point in session entries that keeps approximately `keepRecentTokens`.\n *\n * Algorithm: Walk backwards from newest, accumulating estimated message sizes.\n * Stop when we've accumulated >= keepRecentTokens. Cut at that point.\n *\n * Can cut at user OR assistant messages (never tool results). When cutting at an\n * assistant message with tool calls, its tool results come after and will be kept.\n *\n * Returns CutPointResult with:\n * - firstKeptEntryIndex: the entry index to start keeping from\n * - turnStartIndex: if cutting mid-turn, the user message that started that turn\n * - isSplitTurn: whether we're cutting in the middle of a turn\n *\n * Only considers entries between `startIndex` and `endIndex` (exclusive).\n */\nexport function findCutPoint(\n\tentries: SessionEntry[],\n\tstartIndex: number,\n\tendIndex: number,\n\tkeepRecentTokens: number,\n): CutPointResult {\n\tconst cutPoints = findValidCutPoints(entries, startIndex, endIndex);\n\n\tif (cutPoints.length === 0) {\n\t\treturn { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };\n\t}\n\n\t// Walk backwards from newest, accumulating estimated message sizes\n\tlet accumulatedTokens = 0;\n\tlet cutIndex = cutPoints[0]; // Default: keep from first message (not header)\n\n\tfor (let i = endIndex - 1; i >= startIndex; i--) {\n\t\tconst entry = entries[i];\n\t\tconst messageTokens = sessionEntryToContextMessages(entry).reduce(\n\t\t\t(sum, message) => sum + estimateTokens(message),\n\t\t\t0,\n\t\t);\n\t\tif (messageTokens === 0) continue;\n\t\taccumulatedTokens += messageTokens;\n\n\t\t// Check if we've exceeded the budget\n\t\tif (accumulatedTokens >= keepRecentTokens) {\n\t\t\t// Prefer the closest valid cut point at or after this entry. If trailing\n\t\t\t// tool results exceed the budget by themselves, keep their preceding\n\t\t\t// assistant tool call instead of falling back to the first message.\n\t\t\tcutIndex = cutPoints.find((candidate) => candidate >= i) ?? cutPoints[cutPoints.length - 1];\n\t\t\tbreak;\n\t\t}\n\t}\n\n\t// Scan backwards from cutIndex to include adjacent metadata entries that do not affect context.\n\twhile (cutIndex > startIndex) {\n\t\tconst prevEntry = entries[cutIndex - 1];\n\t\t// Stop at compaction boundaries or context-visible entries.\n\t\tif (prevEntry.type === \"compaction\" || sessionEntryToContextMessages(prevEntry).length > 0) {\n\t\t\tbreak;\n\t\t}\n\t\tcutIndex--;\n\t}\n\n\t// Determine if this is a split turn\n\tconst cutEntry = entries[cutIndex];\n\tconst startsTurn = isTurnStartEntry(cutEntry);\n\tconst turnStartIndex = startsTurn ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);\n\n\treturn {\n\t\tfirstKeptEntryIndex: cutIndex,\n\t\tturnStartIndex,\n\t\tisSplitTurn: !startsTurn && turnStartIndex !== -1,\n\t};\n}\n\n// ============================================================================\n// Summarization\n// ============================================================================\n\nconst SUMMARIZATION_PROMPT = `The messages above are a conversation to summarize. Create a structured context checkpoint summary that another LLM will use to continue the work.\n\nUse this EXACT format:\n\n## Goal\n[What is the user trying to accomplish? Can be multiple items if the session covers different tasks.]\n\n## Constraints & Preferences\n- [Any constraints, preferences, or requirements mentioned by user]\n- [Or \"(none)\" if none were mentioned]\n\n## Progress\n### Done\n- [x] [Completed tasks/changes]\n\n### In Progress\n- [ ] [Current work]\n\n### Blocked\n- [Issues preventing progress, if any]\n\n## Key Decisions\n- **[Decision]**: [Brief rationale]\n\n## Next Steps\n1. [Ordered list of what should happen next]\n\n## Critical Context\n- [Any data, examples, or references needed to continue]\n- [Or \"(none)\" if not applicable]\n\nKeep each section concise. Preserve exact file paths, function names, and error messages.`;\n\nconst UPDATE_SUMMARIZATION_INSTRUCTIONS = `Update the existing structured summary with new information. RULES:\n- PRESERVE all existing information from the previous summary\n- ADD new progress, decisions, and context from the new messages\n- UPDATE the Progress section: move items from \"In Progress\" to \"Done\" when completed\n- UPDATE \"Next Steps\" based on what was accomplished\n- PRESERVE exact file paths, function names, and error messages\n- If something is no longer relevant, you may remove it\n\nUse this EXACT format:\n\n## Goal\n[Preserve existing goals, add new ones if the task expanded]\n\n## Constraints & Preferences\n- [Preserve existing, add new ones discovered]\n\n## Progress\n### Done\n- [x] [Include previously done items AND newly completed items]\n\n### In Progress\n- [ ] [Current work - update based on progress]\n\n### Blocked\n- [Current blockers - remove if resolved]\n\n## Key Decisions\n- **[Decision]**: [Brief rationale] (preserve all previous, add new)\n\n## Next Steps\n1. [Update based on current state]\n\n## Critical Context\n- [Preserve important context, add new if needed]\n\nKeep each section concise. Preserve exact file paths, function names, and error messages.`;\n\nconst UPDATE_SUMMARIZATION_PROMPT = `The messages above are NEW conversation messages to incorporate into the existing summary provided in <previous-summary> tags.\n\n${UPDATE_SUMMARIZATION_INSTRUCTIONS}`;\n\n/**\n * Returns an error message when a summarization response cannot safely be persisted.\n * A length stop contains partial text and must not become a session checkpoint.\n */\nexport function getSummarizationFailure(response: AssistantMessage, label: string): string | undefined {\n\tif (response.stopReason === \"error\") {\n\t\treturn `${label} failed: ${response.errorMessage || \"Unknown error\"}`;\n\t}\n\tif (response.stopReason === \"length\") {\n\t\treturn `${label} failed: generation hit the token cap and the summary is incomplete`;\n\t}\n\treturn undefined;\n}\n\nfunction createSummarizationOptions(\n\tmodel: Model<any>,\n\tmaxTokens: number,\n\tapiKey: string | undefined,\n\theaders: Record<string, string> | undefined,\n\tenv: Record<string, string> | undefined,\n\tsignal: AbortSignal | undefined,\n\tthinkingLevel: ThinkingLevel | undefined,\n\tsessionId: string | undefined,\n): SimpleStreamOptions {\n\tconst options: SimpleStreamOptions = { maxTokens, signal, apiKey, headers, env, sessionId };\n\tif (model.reasoning && thinkingLevel && thinkingLevel !== \"off\") {\n\t\toptions.reasoning = thinkingLevel;\n\t}\n\treturn options;\n}\n\n/**\n * Shared choke point for every compaction/branch-summary summarization call. Wraps the\n * single LLM call in {@link retryAssistantCall} so transient stream drops (e.g.\n * `terminated`, socket close) honor the configured retry policy instead of failing\n * the whole compaction on the first attempt. Deterministic errors and aborts return\n * immediately (see {@link retryAssistantCall}).\n */\nexport async function completeSummarization(\n\tmodel: Model<any>,\n\tcontext: TranscriptContext,\n\toptions: SimpleStreamOptions,\n\tstreamFn?: StreamFn,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n): Promise<AssistantMessage> {\n\t// Avoid cache writes for one-off summaries. Reuse caller-supplied routing when available;\n\t// callers without a session ID, including branch summaries, receive a fresh routing ID.\n\tconst requestOptions: SimpleStreamOptions = {\n\t\t...options,\n\t\tcacheRetention: \"none\",\n\t\tsessionId: options.sessionId ?? uuidv7(),\n\t};\n\tconst produce = async (): Promise<AssistantMessage> =>\n\t\tstreamFn\n\t\t\t? (await streamFn(model, context, requestOptions)).result()\n\t\t\t: completeSimple(model, context, requestOptions);\n\treturn retryAssistantCall(produce, retry, requestOptions.signal, callbacks);\n}\n\n/**\n * Generate a summary of the conversation using the LLM.\n * If previousSummary is provided, uses the update prompt to merge.\n */\nexport async function generateSummary(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<string> {\n\treturn (\n\t\tawait generateSummaryWithUsage(\n\t\t\tcurrentMessages,\n\t\t\tmodel,\n\t\t\treserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tcustomInstructions,\n\t\t\tpreviousSummary,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tenv,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t)\n\t).text;\n}\n\n/** Build the provider context for a standalone summary request. */\nfunction buildSummarizationContext(promptText: string): TranscriptContext {\n\treturn normalizeContext({\n\t\tsystemPrompt: SUMMARIZATION_SYSTEM_PROMPT,\n\t\tmessages: [\n\t\t\t{\n\t\t\t\trole: \"user\",\n\t\t\t\tcontent: [{ type: \"text\", text: promptText }],\n\t\t\t\ttimestamp: Date.now(),\n\t\t\t},\n\t\t],\n\t});\n}\n\n/** Generate or update a conversation summary and return its provider usage. */\nexport async function generateSummaryWithUsage(\n\tcurrentMessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tcustomInstructions?: string,\n\tpreviousSummary?: string,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<{ text: string; usage: Usage }> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.8 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t);\n\n\t// Use update prompt if we have a previous summary, otherwise initial prompt\n\tlet basePrompt = previousSummary ? UPDATE_SUMMARIZATION_PROMPT : SUMMARIZATION_PROMPT;\n\tif (customInstructions) {\n\t\tbasePrompt = `${basePrompt}\\n\\nAdditional focus: ${customInstructions}`;\n\t}\n\n\t// Serialize conversation to text so model doesn't try to continue it\n\t// Convert to LLM messages first (handles custom types like bashExecution, custom, etc.)\n\tconst llmMessages = convertToLlm(currentMessages);\n\tconst conversationText = serializeConversation(llmMessages);\n\n\t// Build the prompt with conversation wrapped in tags\n\tlet promptText = `<conversation>\\n${conversationText}\\n</conversation>\\n\\n`;\n\tif (previousSummary) {\n\t\tpromptText += `<previous-summary>\\n${previousSummary}\\n</previous-summary>\\n\\n`;\n\t}\n\tpromptText += basePrompt;\n\n\tconst completionOptions = createSummarizationOptions(\n\t\tmodel,\n\t\tmaxTokens,\n\t\tapiKey,\n\t\theaders,\n\t\tenv,\n\t\tsignal,\n\t\tthinkingLevel,\n\t\tsessionId,\n\t);\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\tbuildSummarizationContext(promptText),\n\t\tcompletionOptions,\n\t\tstreamFn,\n\t\tretry,\n\t\tcallbacks,\n\t);\n\n\tconst failure = getSummarizationFailure(response, \"Summarization\");\n\tif (failure) {\n\t\tthrow new Error(failure);\n\t}\n\tif (response.content.some((block) => block.type === \"toolCall\")) {\n\t\tthrow new Error(\"Summarization attempted to call a tool\");\n\t}\n\n\tconst textContent = contentText(response.content);\n\n\treturn { text: textContent, usage: response.usage };\n}\n\n// ============================================================================\n// Compaction Preparation (for extensions)\n// ============================================================================\n\nexport interface CompactionPreparation {\n\t/** UUID of first entry to keep */\n\tfirstKeptEntryId: string;\n\t/** Messages that will be summarized and discarded */\n\tmessagesToSummarize: AgentMessage[];\n\t/** Messages that will be turned into turn prefix summary (if splitting) */\n\tturnPrefixMessages: AgentMessage[];\n\t/** Whether this is a split turn (cut point in middle of turn) */\n\tisSplitTurn: boolean;\n\ttokensBefore: number;\n\t/** Summary from previous compaction, for iterative update */\n\tpreviousSummary?: string;\n\t/** File operations extracted from messagesToSummarize */\n\tfileOps: FileOperations;\n\t/** Compaction settions from settings.jsonl\t*/\n\tsettings: CompactionSettings;\n}\n\nfunction isProjectedTurnStart(entry: ProjectedSessionEntry): boolean {\n\tif (entry.sourceEntry.type === \"compaction\") return false;\n\treturn entry.messages.some(isTurnStartMessage);\n}\n\nfunction findProjectedTurnStartIndex(entries: ProjectedSessionEntry[], entryIndex: number, startIndex: number): number {\n\tfor (let i = entryIndex; i >= startIndex; i--) {\n\t\tif (isProjectedTurnStart(entries[i])) return i;\n\t}\n\treturn -1;\n}\n\nfunction findProjectedCutPoint(\n\tentries: ProjectedSessionEntry[],\n\tstartIndex: number,\n\tendIndex: number,\n\tkeepRecentTokens: number,\n): CutPointResult {\n\tconst cutPoints: number[] = [];\n\tfor (let i = startIndex; i < endIndex; i++) {\n\t\tconst entry = entries[i];\n\t\tif (entry.sourceEntry.type !== \"compaction\" && entry.messages.some(isCutPointMessage)) cutPoints.push(i);\n\t}\n\tif (cutPoints.length === 0) {\n\t\treturn { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };\n\t}\n\n\tlet accumulatedTokens = 0;\n\tlet exceededBudget = false;\n\tlet cutIndex = cutPoints[0];\n\tfor (let i = endIndex - 1; i >= startIndex; i--) {\n\t\tconst messageTokens = entries[i].messages.reduce((sum, message) => sum + estimateTokens(message), 0);\n\t\tif (messageTokens === 0) continue;\n\t\taccumulatedTokens += messageTokens;\n\t\tif (accumulatedTokens >= keepRecentTokens) {\n\t\t\texceededBudget = true;\n\t\t\tcutIndex = cutPoints.find((candidate) => candidate >= i) ?? cutPoints[cutPoints.length - 1];\n\t\t\tbreak;\n\t\t}\n\t}\n\n\t// A recovery attempt and its omission edits are context-invisible after the last\n\t// visible input. Advance only for a closed suffix containing an omitted assistant\n\t// attempt; arbitrary metadata must not move the cut past unsent input.\n\tconst suffix = entries.slice(cutIndex + 1, endIndex);\n\tconst isIntrinsicallyVisible = (entry: ProjectedSessionEntry): boolean =>\n\t\tentry.sourceEntry.type !== \"context_edit\" && sessionEntryToContextMessages(entry.sourceEntry).length > 0;\n\tconst isOmitted = (entry: ProjectedSessionEntry): boolean =>\n\t\tisIntrinsicallyVisible(entry) && entry.messages.length === 0;\n\tconst omittedSuffixIds = new Set(suffix.filter(isOmitted).map((entry) => entry.sourceEntry.id));\n\tconst hasExternalReplacement = suffix.some(\n\t\t(entry) =>\n\t\t\tentry.sourceEntry.type === \"context_edit\" &&\n\t\t\tentry.sourceEntry.replacement !== null &&\n\t\t\t!omittedSuffixIds.has(entry.sourceEntry.targetId),\n\t);\n\tconst isRecoveryOmissionSuffix =\n\t\texceededBudget &&\n\t\t!hasExternalReplacement &&\n\t\tsuffix.some(\n\t\t\t(entry) =>\n\t\t\t\tentry.sourceEntry.type === \"message\" && entry.sourceEntry.message.role === \"assistant\" && isOmitted(entry),\n\t\t) &&\n\t\tsuffix.every(\n\t\t\t(entry) => entry.sourceEntry.type !== \"compaction\" && (!isIntrinsicallyVisible(entry) || isOmitted(entry)),\n\t\t);\n\tif (isRecoveryOmissionSuffix) cutIndex++;\n\n\twhile (cutIndex > startIndex) {\n\t\tconst previous = entries[cutIndex - 1];\n\t\tif (previous.sourceEntry.type === \"compaction\" || previous.messages.length > 0) break;\n\t\tcutIndex--;\n\t}\n\tconst startsTurn = isProjectedTurnStart(entries[cutIndex]);\n\tconst turnStartIndex = startsTurn ? -1 : findProjectedTurnStartIndex(entries, cutIndex, startIndex);\n\treturn {\n\t\tfirstKeptEntryIndex: cutIndex,\n\t\tturnStartIndex,\n\t\tisSplitTurn: !startsTurn && turnStartIndex !== -1,\n\t};\n}\n\nexport function prepareCompaction(\n\tpathEntries: SessionEntry[],\n\tsettings: CompactionSettings,\n): CompactionPreparation | undefined {\n\tif (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === \"compaction\") {\n\t\treturn undefined;\n\t}\n\n\tconst projection = buildSessionProjection(pathEntries);\n\tconst projectedEntries = projection.entries;\n\tconst sourceEntries = projectedEntries.map((entry) => entry.sourceEntry);\n\t// The newest compaction is projected first. Older compaction entries can still\n\t// occur in its retained raw range, but their projected contribution is empty.\n\tconst prevCompactionIndex = projectedEntries.findIndex(\n\t\t(entry) => entry.sourceEntry.type === \"compaction\" && entry.messages.length > 0,\n\t);\n\n\tlet previousSummary: string | undefined;\n\tlet boundaryStart = 0;\n\tif (prevCompactionIndex >= 0) {\n\t\tpreviousSummary = (projectedEntries[prevCompactionIndex].sourceEntry as CompactionEntry).summary;\n\t\t// The canonical projection has already selected the previous compaction's retained tail.\n\t\tboundaryStart = prevCompactionIndex + 1;\n\t}\n\tconst boundaryEnd = projectedEntries.length;\n\tconst tokensBefore = estimateProjectedContextTokens(projection, pathEntries).tokens;\n\tconst cutPoint = findProjectedCutPoint(projectedEntries, boundaryStart, boundaryEnd, settings.keepRecentTokens);\n\n\tconst firstKeptEntry = projectedEntries[cutPoint.firstKeptEntryIndex]?.sourceEntry;\n\tif (!firstKeptEntry?.id) return undefined;\n\tconst firstKeptEntryId = firstKeptEntry.id;\n\tconst historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;\n\n\tconst messagesToSummarize = projectedEntries\n\t\t.slice(boundaryStart, historyEnd)\n\t\t.flatMap(getMessagesFromProjectedEntryForCompaction);\n\tconst turnPrefixMessages = cutPoint.isSplitTurn\n\t\t? projectedEntries\n\t\t\t\t.slice(cutPoint.turnStartIndex, cutPoint.firstKeptEntryIndex)\n\t\t\t\t.flatMap(getMessagesFromProjectedEntryForCompaction)\n\t\t: [];\n\n\tif (messagesToSummarize.length === 0 && turnPrefixMessages.length === 0) return undefined;\n\n\t// Extract file operations from edited model-visible messages and the previous compaction.\n\tconst fileOps = extractFileOperations(messagesToSummarize, sourceEntries, prevCompactionIndex);\n\n\t// Also extract file ops from turn prefix if splitting\n\tif (cutPoint.isSplitTurn) {\n\t\tfor (const msg of turnPrefixMessages) {\n\t\t\textractFileOpsFromMessage(msg, fileOps);\n\t\t}\n\t}\n\n\treturn {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn: cutPoint.isSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t};\n}\n\n// ============================================================================\n// Main compaction function\n// ============================================================================\n\nconst TURN_PREFIX_SUMMARIZATION_PROMPT = `The messages above are earlier context from an ongoing conversation. Later messages are stored separately and do not need to be reconstructed.\n\nCreate a concise checkpoint of the user's request and the progress shown above. This checkpoint will be placed before the later messages so the conversation can continue with the necessary context.\n\n## Original Request\n[What did the user ask for?]\n\n## Progress So Far\n- [Key decisions and work completed in these messages]\n\n## Context Needed to Continue\n- [Information from these messages needed to understand the later work]\n\nOnly summarize information explicitly present above. Do not infer or recreate later messages.`;\n\n/**\n * Generate summaries for compaction using prepared data.\n * Returns CompactionResult - SessionManager adds uuid/parentUuid when saving.\n *\n * @param preparation - Pre-calculated preparation from prepareCompaction()\n * @param customInstructions - Optional custom focus for the summary\n * @param sessionId - Optional routing session ID forwarded without enabling prompt caching\n */\nexport async function compact(\n\tpreparation: CompactionPreparation,\n\tmodel: Model<any>,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tcustomInstructions?: string,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tenv?: Record<string, string>,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<CompactionResult> {\n\tconst {\n\t\tfirstKeptEntryId,\n\t\tmessagesToSummarize,\n\t\tturnPrefixMessages,\n\t\tisSplitTurn,\n\t\ttokensBefore,\n\t\tpreviousSummary,\n\t\tfileOps,\n\t\tsettings,\n\t} = preparation;\n\n\t// Generate summaries and merge into one\n\tlet summary: string;\n\tlet summaryUsage: Usage;\n\n\tif (isSplitTurn && turnPrefixMessages.length > 0) {\n\t\tlet historyText = previousSummary ?? \"No prior history.\";\n\t\tlet historyUsage: Usage | undefined;\n\t\tif (messagesToSummarize.length > 0) {\n\t\t\tconst historyResult = await generateSummaryWithUsage(\n\t\t\t\tmessagesToSummarize,\n\t\t\t\tmodel,\n\t\t\t\tsettings.reserveTokens,\n\t\t\t\tapiKey,\n\t\t\t\theaders,\n\t\t\t\tsignal,\n\t\t\t\tcustomInstructions,\n\t\t\t\tpreviousSummary,\n\t\t\t\tthinkingLevel,\n\t\t\t\tstreamFn,\n\t\t\t\tenv,\n\t\t\t\tretry,\n\t\t\t\tcallbacks,\n\t\t\t\tsessionId,\n\t\t\t);\n\t\t\thistoryText = historyResult.text;\n\t\t\thistoryUsage = historyResult.usage;\n\t\t}\n\t\tconst turnPrefixResult = await generateTurnPrefixSummary(\n\t\t\tturnPrefixMessages,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tenv,\n\t\t\tsignal,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t);\n\t\t// Merge into single summary\n\t\tsummary = `${historyText}\\n\\n---\\n\\n**Turn Context (split turn):**\\n\\n${turnPrefixResult.text}`;\n\t\tsummaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage;\n\t} else {\n\t\t// Just generate history summary\n\t\tconst result = await generateSummaryWithUsage(\n\t\t\tmessagesToSummarize,\n\t\t\tmodel,\n\t\t\tsettings.reserveTokens,\n\t\t\tapiKey,\n\t\t\theaders,\n\t\t\tsignal,\n\t\t\tcustomInstructions,\n\t\t\tpreviousSummary,\n\t\t\tthinkingLevel,\n\t\t\tstreamFn,\n\t\t\tenv,\n\t\t\tretry,\n\t\t\tcallbacks,\n\t\t\tsessionId,\n\t\t);\n\t\tsummary = result.text;\n\t\tsummaryUsage = result.usage;\n\t}\n\n\t// Compute file lists and append to summary\n\tconst { readFiles, modifiedFiles } = computeFileLists(fileOps);\n\tsummary += formatFileOperations(readFiles, modifiedFiles);\n\n\tif (!firstKeptEntryId) {\n\t\tthrow new Error(\"First kept entry has no UUID - session may need migration\");\n\t}\n\n\treturn {\n\t\tsummary,\n\t\tfirstKeptEntryId,\n\t\ttokensBefore,\n\t\tusage: summaryUsage,\n\t\tdetails: { readFiles, modifiedFiles } as CompactionDetails,\n\t};\n}\n\n/**\n * Generate a summary for a turn prefix (when splitting a turn).\n */\nasync function generateTurnPrefixSummary(\n\tmessages: AgentMessage[],\n\tmodel: Model<any>,\n\treserveTokens: number,\n\tapiKey: string | undefined,\n\theaders?: Record<string, string>,\n\tenv?: Record<string, string>,\n\tsignal?: AbortSignal,\n\tthinkingLevel?: ThinkingLevel,\n\tstreamFn?: StreamFn,\n\tretry?: RetryPolicy,\n\tcallbacks?: RetryCallbacks,\n\tsessionId?: string,\n): Promise<{ text: string; usage: Usage }> {\n\tconst maxTokens = Math.min(\n\t\tMath.floor(0.5 * reserveTokens),\n\t\tmodel.maxTokens > 0 ? model.maxTokens : Number.POSITIVE_INFINITY,\n\t); // Smaller budget for turn prefix\n\tconst llmMessages = convertToLlm(messages);\n\tconst conversationText = serializeConversation(llmMessages);\n\tconst promptText = `# Conversation\\n${conversationText}\\n\\n# Instructions\\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;\n\n\tconst response = await completeSummarization(\n\t\tmodel,\n\t\tbuildSummarizationContext(promptText),\n\t\tcreateSummarizationOptions(model, maxTokens, apiKey, headers, env, signal, thinkingLevel, sessionId),\n\t\tstreamFn,\n\t\tretry,\n\t\tcallbacks,\n\t);\n\n\tconst failure = getSummarizationFailure(response, \"Turn prefix summarization\");\n\tif (failure) {\n\t\tthrow new Error(failure);\n\t}\n\tif (response.content.some((block) => block.type === \"toolCall\")) {\n\t\tthrow new Error(\"Turn prefix summarization attempted to call a tool\");\n\t}\n\n\treturn {\n\t\ttext: contentText(response.content),\n\t\tusage: response.usage,\n\t};\n}\n"]}