pi-smart-compact 9.7.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (211) hide show
  1. package/ARCHITECTURE.md +979 -372
  2. package/CHANGELOG.md +720 -0
  3. package/LICENSE +8 -0
  4. package/README.md +130 -640
  5. package/SECURITY.md +34 -12
  6. package/SUPPORT.md +26 -9
  7. package/assets/DejaVu-LICENSE.txt +187 -0
  8. package/assets/DejaVuSansMono.ttf +0 -0
  9. package/assets/README.md +26 -0
  10. package/assets/skills/context-management/SKILL.md +34 -0
  11. package/dist/app/anchor-cache.d.ts +36 -0
  12. package/dist/app/anchor-cache.d.ts.map +1 -0
  13. package/dist/app/artifact-storage.d.ts +47 -0
  14. package/dist/app/artifact-storage.d.ts.map +1 -0
  15. package/dist/app/background-preparation.d.ts +39 -0
  16. package/dist/app/background-preparation.d.ts.map +1 -0
  17. package/dist/app/compaction-commit-store.d.ts +5 -1
  18. package/dist/app/compaction-commit-store.d.ts.map +1 -1
  19. package/dist/app/context-evidence.d.ts +57 -0
  20. package/dist/app/context-evidence.d.ts.map +1 -0
  21. package/dist/app/context-guide.d.ts +3 -0
  22. package/dist/app/context-guide.d.ts.map +1 -0
  23. package/dist/app/context-operations.d.ts +106 -0
  24. package/dist/app/context-operations.d.ts.map +1 -0
  25. package/dist/app/effective-state.d.ts +23 -0
  26. package/dist/app/effective-state.d.ts.map +1 -0
  27. package/dist/app/extension-conflicts.d.ts +15 -0
  28. package/dist/app/extension-conflicts.d.ts.map +1 -0
  29. package/dist/app/global-settings-runtime.d.ts +3 -3
  30. package/dist/app/global-settings-runtime.d.ts.map +1 -1
  31. package/dist/app/hindsight-memory.d.ts +100 -0
  32. package/dist/app/hindsight-memory.d.ts.map +1 -0
  33. package/dist/app/host-cache-ledger.d.ts +68 -0
  34. package/dist/app/host-cache-ledger.d.ts.map +1 -0
  35. package/dist/app/lazy-tools.d.ts +36 -0
  36. package/dist/app/lazy-tools.d.ts.map +1 -0
  37. package/dist/app/memory-backend.d.ts +58 -0
  38. package/dist/app/memory-backend.d.ts.map +1 -0
  39. package/dist/app/mnemopi-memory.d.ts +13 -0
  40. package/dist/app/mnemopi-memory.d.ts.map +1 -0
  41. package/dist/app/mnemopi-protocol.d.ts +78 -0
  42. package/dist/app/mnemopi-protocol.d.ts.map +1 -0
  43. package/dist/app/mnemopi-worker.d.ts +2 -0
  44. package/dist/app/mnemopi-worker.d.ts.map +1 -0
  45. package/dist/app/model-feasibility.d.ts +20 -0
  46. package/dist/app/model-feasibility.d.ts.map +1 -0
  47. package/dist/app/native-compaction.d.ts +88 -0
  48. package/dist/app/native-compaction.d.ts.map +1 -0
  49. package/dist/app/native-continuity-bridge.d.ts.map +1 -1
  50. package/dist/app/navigation-data.d.ts +28 -0
  51. package/dist/app/navigation-data.d.ts.map +1 -0
  52. package/dist/app/navigation-types.d.ts +60 -0
  53. package/dist/app/navigation-types.d.ts.map +1 -0
  54. package/dist/app/pending-slot.d.ts +11 -1
  55. package/dist/app/pending-slot.d.ts.map +1 -1
  56. package/dist/app/preflight.d.ts.map +1 -1
  57. package/dist/app/register-context-tools.d.ts +16 -3
  58. package/dist/app/register-context-tools.d.ts.map +1 -1
  59. package/dist/app/register-navigation.d.ts +20 -0
  60. package/dist/app/register-navigation.d.ts.map +1 -0
  61. package/dist/app/register-smart-compact-command.d.ts +17 -2
  62. package/dist/app/register-smart-compact-command.d.ts.map +1 -1
  63. package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
  64. package/dist/app/register-smart-context-tool.d.ts +55 -0
  65. package/dist/app/register-smart-context-tool.d.ts.map +1 -0
  66. package/dist/app/run-context.d.ts +4 -1
  67. package/dist/app/run-context.d.ts.map +1 -1
  68. package/dist/app/run-smart-compact.d.ts +3 -3
  69. package/dist/app/run-smart-compact.d.ts.map +1 -1
  70. package/dist/app/session-handoff.d.ts +64 -0
  71. package/dist/app/session-handoff.d.ts.map +1 -0
  72. package/dist/app/session-lineage.d.ts +17 -0
  73. package/dist/app/session-lineage.d.ts.map +1 -0
  74. package/dist/app/session-run-lock.d.ts +0 -2
  75. package/dist/app/session-run-lock.d.ts.map +1 -1
  76. package/dist/app/settled-auto-trigger.d.ts +2 -0
  77. package/dist/app/settled-auto-trigger.d.ts.map +1 -1
  78. package/dist/app/smart-compact-input.d.ts +1 -1
  79. package/dist/app/smart-compact-input.d.ts.map +1 -1
  80. package/dist/app/smart-compact-policy.d.ts +1 -1
  81. package/dist/app/smart-compact-policy.d.ts.map +1 -1
  82. package/dist/app/steps/extract.d.ts +45 -1
  83. package/dist/app/steps/extract.d.ts.map +1 -1
  84. package/dist/app/steps/metrics.d.ts +1 -0
  85. package/dist/app/steps/metrics.d.ts.map +1 -1
  86. package/dist/app/steps/persist.d.ts.map +1 -1
  87. package/dist/app/steps/prepare.d.ts.map +1 -1
  88. package/dist/app/steps/recover.d.ts +9 -0
  89. package/dist/app/steps/recover.d.ts.map +1 -1
  90. package/dist/app/steps/state.d.ts.map +1 -1
  91. package/dist/app/steps/synthesize.d.ts.map +1 -1
  92. package/dist/app/steps/tier.d.ts.map +1 -1
  93. package/dist/app/steps/verify.d.ts.map +1 -1
  94. package/dist/app/steps/visual.d.ts +4 -0
  95. package/dist/app/steps/visual.d.ts.map +1 -0
  96. package/dist/app/steps/window.d.ts.map +1 -1
  97. package/dist/app/tool-artifacts.d.ts +27 -0
  98. package/dist/app/tool-artifacts.d.ts.map +1 -0
  99. package/dist/app/visual-archive.d.ts +29 -0
  100. package/dist/app/visual-archive.d.ts.map +1 -0
  101. package/dist/constants.d.ts +96 -1
  102. package/dist/constants.d.ts.map +1 -1
  103. package/dist/domain/compaction-usage.d.ts +16 -0
  104. package/dist/domain/compaction-usage.d.ts.map +1 -0
  105. package/dist/domain/model-capacity.d.ts +12 -0
  106. package/dist/domain/model-capacity.d.ts.map +1 -0
  107. package/dist/domain/provider-evaluation.d.ts +7 -0
  108. package/dist/domain/provider-evaluation.d.ts.map +1 -1
  109. package/dist/domain/telemetry.d.ts +43 -2
  110. package/dist/domain/telemetry.d.ts.map +1 -1
  111. package/dist/domain/tool-semantics.d.ts +23 -0
  112. package/dist/domain/tool-semantics.d.ts.map +1 -1
  113. package/dist/index.d.ts.map +1 -1
  114. package/dist/index.js +17298 -8331
  115. package/dist/infra/ai-messages.d.ts +1 -1
  116. package/dist/infra/ai-messages.d.ts.map +1 -1
  117. package/dist/infra/context-graph.d.ts +38 -7
  118. package/dist/infra/context-graph.d.ts.map +1 -1
  119. package/dist/infra/fs.d.ts.map +1 -1
  120. package/dist/infra/hindsight-client.d.ts +73 -0
  121. package/dist/infra/hindsight-client.d.ts.map +1 -0
  122. package/dist/infra/hindsight-receipts.d.ts +68 -0
  123. package/dist/infra/hindsight-receipts.d.ts.map +1 -0
  124. package/dist/infra/llm-client.d.ts +26 -23
  125. package/dist/infra/llm-client.d.ts.map +1 -1
  126. package/dist/infra/memory-ref.d.ts +27 -0
  127. package/dist/infra/memory-ref.d.ts.map +1 -0
  128. package/dist/infra/native-protocol.d.ts +54 -0
  129. package/dist/infra/native-protocol.d.ts.map +1 -0
  130. package/dist/infra/optional-components.d.ts +15 -0
  131. package/dist/infra/optional-components.d.ts.map +1 -0
  132. package/dist/infra/paths.d.ts +2 -0
  133. package/dist/infra/paths.d.ts.map +1 -1
  134. package/dist/infra/services.d.ts +15 -5
  135. package/dist/infra/services.d.ts.map +1 -1
  136. package/dist/infra/visual-renderer.d.ts +16 -0
  137. package/dist/infra/visual-renderer.d.ts.map +1 -0
  138. package/dist/mnemopi-worker.js +213 -0
  139. package/dist/phases/explore.d.ts +12 -9
  140. package/dist/phases/explore.d.ts.map +1 -1
  141. package/dist/phases/synthesize.d.ts +18 -3
  142. package/dist/phases/synthesize.d.ts.map +1 -1
  143. package/dist/phases/verify.d.ts +20 -2
  144. package/dist/phases/verify.d.ts.map +1 -1
  145. package/dist/rtk.d.ts +7 -0
  146. package/dist/rtk.d.ts.map +1 -0
  147. package/dist/rtk.js +767 -0
  148. package/dist/types.d.ts +128 -4
  149. package/dist/types.d.ts.map +1 -1
  150. package/dist/ui/dashboard-format.d.ts +2 -1
  151. package/dist/ui/dashboard-format.d.ts.map +1 -1
  152. package/dist/ui/dashboard-insights.d.ts +9 -1
  153. package/dist/ui/dashboard-insights.d.ts.map +1 -1
  154. package/dist/ui/error-format.d.ts +7 -2
  155. package/dist/ui/error-format.d.ts.map +1 -1
  156. package/dist/ui/handoff-overlay.d.ts +26 -0
  157. package/dist/ui/handoff-overlay.d.ts.map +1 -0
  158. package/dist/ui/home-overlay.d.ts +54 -0
  159. package/dist/ui/home-overlay.d.ts.map +1 -0
  160. package/dist/ui/metrics-dashboard-overlay.d.ts.map +1 -1
  161. package/dist/ui/metrics-report.d.ts.map +1 -1
  162. package/dist/ui/navigation-overlay.d.ts +92 -0
  163. package/dist/ui/navigation-overlay.d.ts.map +1 -0
  164. package/dist/ui/overlays.d.ts +12 -2
  165. package/dist/ui/overlays.d.ts.map +1 -1
  166. package/dist/ui/profiles.d.ts +51 -0
  167. package/dist/ui/profiles.d.ts.map +1 -0
  168. package/dist/ui/settings-complex.d.ts +49 -3
  169. package/dist/ui/settings-complex.d.ts.map +1 -1
  170. package/dist/ui/settings-list.d.ts +28 -0
  171. package/dist/ui/settings-list.d.ts.map +1 -0
  172. package/dist/ui/settings-overlay.d.ts +13 -6
  173. package/dist/ui/settings-overlay.d.ts.map +1 -1
  174. package/dist/ui/storage-report.d.ts +4 -0
  175. package/dist/ui/storage-report.d.ts.map +1 -0
  176. package/dist/utils/backups.d.ts.map +1 -1
  177. package/dist/utils/cache.d.ts +6 -2
  178. package/dist/utils/cache.d.ts.map +1 -1
  179. package/dist/utils/config.d.ts +12 -0
  180. package/dist/utils/config.d.ts.map +1 -1
  181. package/dist/utils/extraction.d.ts +7 -0
  182. package/dist/utils/extraction.d.ts.map +1 -1
  183. package/dist/utils/helpers.d.ts.map +1 -1
  184. package/dist/utils/id-fingerprint.d.ts +3 -1
  185. package/dist/utils/id-fingerprint.d.ts.map +1 -1
  186. package/dist/utils/issues.d.ts +61 -0
  187. package/dist/utils/issues.d.ts.map +1 -0
  188. package/dist/utils/pruning.d.ts.map +1 -1
  189. package/dist/utils/session-log.d.ts +0 -2
  190. package/dist/utils/session-log.d.ts.map +1 -1
  191. package/dist/utils/state.d.ts +21 -2
  192. package/dist/utils/state.d.ts.map +1 -1
  193. package/dist/utils/tokens.d.ts +10 -2
  194. package/dist/utils/tokens.d.ts.map +1 -1
  195. package/docs/MIGRATING_TO_V8.md +7 -1
  196. package/docs/README.md +69 -0
  197. package/docs/RELEASE.md +174 -56
  198. package/docs/assets/banner.png +0 -0
  199. package/docs/assets/banner.svg +1158 -70
  200. package/docs/assets/pi-smart-compact.png +0 -0
  201. package/docs/assets/pi-smart-compact.svg +24 -0
  202. package/docs/configuration.md +637 -0
  203. package/docs/evaluation.md +409 -0
  204. package/docs/guide.md +879 -0
  205. package/docs/hindsight-memory.md +314 -0
  206. package/docs/identity.md +124 -0
  207. package/package.json +44 -11
  208. package/dist/provider-eval.js +0 -2107
  209. package/dist/provider-scenario-eval.js +0 -2884
  210. package/dist/telemetry-report.js +0 -1958
  211. package/docs/provider-evaluation-2026-08-06.md +0 -63
package/ARCHITECTURE.md CHANGED
@@ -1,54 +1,430 @@
1
1
  # Architecture
2
2
 
3
- System-level design of `pi-smart-compact`. This is the maintainer-facing
4
- companion to the user-facing [`README.md`](./README.md).
3
+ Maintainer-facing system design for **Pi Continuity**, published as the npm
4
+ package `pi-smart-compact`. The product name is documentation branding only:
5
+ the package name, `/smart-compact` command, `smart_*` tools, `smartCompact`
6
+ configuration key, runtime and UI names, stored paths and ref prefixes are
7
+ unchanged.
8
+
9
+ Usage lives in the [user guide](./docs/guide.md), every setting in
10
+ [configuration](./docs/configuration.md), and evidence rules in
11
+ [evaluation](./docs/evaluation.md). This page explains how the parts fit and
12
+ which invariants they must keep. Contributor workflow and the repository map
13
+ are in [`CONTRIBUTING.md`](https://github.com/alpertarhan/pi-smart-compact/blob/main/CONTRIBUTING.md).
14
+
15
+ **Contents:** [product layers](#product-layers) ·
16
+ [integration surfaces](#integration-surfaces) ·
17
+ [1. context hygiene](#1-context-hygiene) ·
18
+ [2. recoverable continuity](#2-recoverable-continuity) ·
19
+ [3. verified compaction](#3-verified-compaction) ·
20
+ [4. cross-session memory](#4-optional-cross-session-memory) ·
21
+ [state and persistence](#state-caching-and-persistence) ·
22
+ [concurrency and safety](#concurrency-and-safety-model) ·
23
+ [provider awareness](#provider-awareness) ·
24
+ [layer responsibilities](#layer-responsibilities) ·
25
+ [host dependency boundary](#host-dependency-boundary) ·
26
+ [extending the system](#extending-the-system)
27
+
28
+ ## Product layers
29
+
30
+ > **Job:** keep the agent's working set useful and quiet, and keep the session
31
+ > continuous across research, cleanup, compaction, reload and, optionally,
32
+ > later sessions. Compaction and memory are mechanisms, not the product
33
+ > boundary.
34
+
35
+ Prefer cheaper, recoverable operations before lossy compaction when they fit
36
+ the task. This is a design preference, not an enforced execution chain: the
37
+ features can be selected independently, and memory does not require compaction.
5
38
 
6
- > **Job:** not to produce a generic recap, but to preserve the agent's working
7
- > state so the next turn can continue with minimal loss.
39
+ ```mermaid
40
+ flowchart LR
41
+ H[1. Context hygiene] --> R[2. Recoverable continuity]
42
+ R --> C[3. Verified compaction]
43
+ C -. optional .-> M[4. Cross-session memory]
44
+ ```
8
45
 
9
- ## Design ideas
46
+ | Layer | Question it answers | Main modules | Core invariant |
47
+ | --- | --- | --- | --- |
48
+ | [1. Context hygiene](#1-context-hygiene) | How do we keep noise out of the working set? | `rtk.ts`, `app/tool-artifacts.ts`, `app/context-operations.ts` | Preserve protected content and retain retrieval paths for eligible offloaded output |
49
+ | [2. Recoverable continuity](#2-recoverable-continuity) | How does removed or old evidence stay reachable on the same branch? | `app/register-smart-context-tool.ts`, `app/context-evidence.ts`, `app/artifact-storage.ts` | Access follows active-branch provenance; the host session log stays authoritative |
50
+ | [3. Verified compaction](#3-verified-compaction) | How do we replace history when pressure demands it? | `app/run-smart-compact.ts`, `app/steps/*`, `phases/*` | Facts first, synthesis second, verification before apply; rejected custom summaries are not staged or applied |
51
+ | [4. Cross-session memory](#4-optional-cross-session-memory) | What should a later session know? | `app/memory-backend.ts`, `infra/context-graph.ts`, Hindsight and Mnemopi modules | One selected backend; explicit fact saves require confirmation, while enabled local indexing follows host-confirmed compaction |
10
52
 
11
- The design combines three ideas:
53
+ Quality means retained constraints, trustworthy failure evidence and low
54
+ retrieval churn, not merely fewer tokens. There are no recurring
55
+ model-visible status prompts, no automatic error deletion, and no destructive
56
+ file rollback. The [2026-09-24 context hygiene report](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/context-hygiene-2026-09-24.md)
57
+ (repository only, historical) records the original experiments and acceptance
58
+ criteria.
12
59
 
13
- - **Agentic compaction** — let the system inspect the session, not just summarize it.
14
- - **Kamradt-style chunking** — segment large conversations into coherent units before synthesis.
15
- - **EESV** — **Extract → Explore → Synthesize → Verify**: facts first, synthesis second, verification last.
60
+ ## Design ideas
61
+
62
+ - **Agentic compaction.** The system may inspect the session through bounded
63
+ tools instead of summarizing a flat transcript.
64
+ - **Coverage across the whole conversation.** A design intuition borrowed from
65
+ Greg Kamradt's public work on semantic chunking and long-context retrieval:
66
+ important facts can sit anywhere in a long history, so an early constraint
67
+ or a mid-session decision deserves the same chance to survive as the latest
68
+ error. In this codebase that intuition shows up as deterministic extraction
69
+ over the entire compacted prefix, topic-aware segmentation before
70
+ hierarchical synthesis, and verification against the extracted facts
71
+ regardless of where they occurred. It is not a formal sampling algorithm or
72
+ benchmark result, and it is not a claim about how Pi's own compactor
73
+ selects content.
74
+ - **EESV:** Extract, Explore, Synthesize, Verify. Facts first, synthesis
75
+ second, verification last.
76
+
77
+ Verification measures coverage of deterministic facts, structure and grounded
78
+ claims. It is a strong regression guard, not proof of semantic truth, and a
79
+ compacted summary is not a lossless copy of the history it replaces.
16
80
 
17
81
  ## Integration surfaces
18
82
 
19
- Registered in [`src/index.ts`](./src/index.ts). See the README for usage; this
20
- section is about lifecycle.
83
+ Registered in [`src/index.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/index.ts).
21
84
 
22
85
  | Surface | Lifecycle |
23
86
  | --- | --- |
24
- | `/smart-compact` | Manual command. Explainable target-first preflight or direct args. Bypasses the adaptive pressure gate, not yield/verification gates. |
25
- | `session_before_compact` | Auto hook. Returns/stages a pending summary or runs under pressure; durable commit waits for matching `session_compact`. |
26
- | `session_compact_failed` | Pi 0.85 failure hook. Clears extension-owned staged state and records one error/cancellation outcome; older hosts simply never emit it. |
27
- | `agent_settled` | Opt-in pressure monitor. Requests `ctx.compact()` only; it never runs EESV or consumes/stages pending state. |
87
+ | `/smart-compact` | Manual command. A bare TUI invocation opens the keyboard Home (`ui/home-overlay.ts`); `Compact now` opens the target-first preflight. Direct arguments bypass Home. `trim` and `storage` expose the shared cleanup controller and the read-only storage inventory; `context` opens the session-navigation panel (`ui/navigation-overlay.ts`). Bypasses the adaptive pressure gate, not yield or verification gates. |
88
+ | `session_before_compact` | Auto hook. Returns or stages a pending summary, or runs under pressure; the durable commit waits for the matching `session_compact`. |
89
+ | `session_compact_failed` | Clears extension-owned staged state and records one error or cancellation outcome. |
90
+ | `turn_end` | Commits queued local context edits first; background mode tries bounded trimming before non-blocking speculative preparation. |
91
+ | `agent_settled` | Checks background preparation and requests native `ctx.compact()` at the idle pressure gate; does not commit itself. |
92
+ | `session_shutdown` | Cancels speculative work and awaits late preparation plus discard-metric writes; never applies an unconfirmed candidate. |
93
+ | `context` | Optionally rehydrates validated bitmap evidence beside the matching text summary; does not mutate session history. |
94
+ | `tool_result` | Opt-in early spill of large safe read-only text: verify persistence, then replace content with a preview and reference; errors leave the original result intact. |
95
+ | `before_provider_request` | Replays provider-native compaction state on a matching route; no I/O until the session has native state. |
96
+ | `before_provider_request` (anchor cache) | `app/anchor-cache.ts`: on Anthropic Messages routes, moves one `cache_control` marker to the newest anchor on the branch so the prefix before it stays cacheable while later turns change. |
97
+ | `session_before_tree` / `session_tree` | Own pivots supply the branch summary (carryover) for the exact prepared target; a foreign navigation cancels a queued pivot. Boundaries re-decide lazy tool exposure and refresh the anchor footer. |
98
+ | `smart_tools` tool | Agent-callable loader (`app/lazy-tools.ts`): `load`/`unload` one tool group (`navigation`, `history`, `memory`, `compaction`), `status`, or `guide` (reads `assets/skills/context-management/SKILL.md` on demand; never injected). |
99
+ | `smart_navigation` tool | `view`/`recall` anchors, `anchor` this point, or `pivot` to an anchor with required carryover; a pivot terminates the turn and is applied by the host only after the batch settles and revalidation passes. |
28
100
  | `smart_compact` tool | Agent-callable. Prepares a pending summary; never compacts mid-turn. |
29
- | `smart_recall` tool | Searches only the current project's bounded context graph; same session/branch ranks first. |
30
- | `smart_save_memory` tool | Persists one explicit user-confirmed project fact after secret/PII scrubbing. |
31
-
32
- A short-lived pending compaction is staged in the [`PendingSlot`](#pending-compaction-slot)
33
- and handed to Pi when compaction is applied.
101
+ | `smart_context` tool | Session-local checkpoint/rewind, safe trimming, and bounded original-output retrieval via native boundary drafts. |
102
+ | `smart_recall` tool | Searches the selected memory backend only. `scope: "session"` is local-graph-only; remote backends skip it and read nothing else. |
103
+ | `smart_save_memory` tool | Saves a host-confirmed scrubbed fact through the selected backend, or resolves the exact store named by a target-bound ref. |
104
+
105
+ The table lists the owning surfaces. Other host events support them:
106
+ session switch/fork/tree and `model_select` cancel speculative
107
+ preparation and queued edits; `session_start` also names known conflicting
108
+ compaction or context-editing extensions (`app/extension-conflicts.ts`,
109
+ name-based evidence from the live command/tool registry, one notice);
110
+ `before_agent_start` injects the one-shot native continuity bridge;
111
+ `message_end` feeds damage monitoring and the host prompt-cache ledger;
112
+ `context_with_system` and `cache_warming_decision` serve held automatic trims
113
+ (see [recoverable trimming](#recoverable-trimming)); the RTK companion uses
114
+ `tool_call`.
115
+
116
+ Tool exposure is owned by `app/lazy-tools.ts`. With `toolLoading: "lazy"`
117
+ (default) only the `smart_tools` loader is active; a group becomes active when
118
+ the agent loads it, and loaded groups are forgotten at `session_start`,
119
+ `session_tree` and `session_compact`. `eager` activates every permitted group;
120
+ `off` removes all owned tools while the human UI keeps working. Group
121
+ permissions (`agentToolAccess`, memory backend, `contextNavigationEnabled`)
122
+ apply in every mode. The exposure removes only its own tools, restores only
123
+ what it removed itself, and treats a `/tools` change by the user as final:
124
+ hidden tools are never re-shown by a loader request. Artifact offload runs
125
+ only while `smart_context` is reachable (active or loadable), so an archived
126
+ output can always be read back by the agent that lost it.
34
127
 
35
- With `autoTriggerStrategy: "settled"`, the idle hook applies finite
36
- token/percentage, queue, in-flight, and per-session cooldown guards, then asks
37
- Pi for a normal host compaction. Pi re-enters `session_before_compact`, which
38
- reuses an already tool-staged summary or runs EESV exactly once under the
39
- host's signal and timeout. The matching `session_compact` event remains the
40
- only durable commit authority; `session_compact_failed` discards the staged
41
- candidate without committing it. This keeps proactive triggering out of the
42
- pending/commit state machine and preserves branch-provenance checks.
43
-
44
- ### Host dependency boundary
45
-
46
- Pi core modules and `typebox` are wildcard peers supplied by the running host;
47
- they are neither bundled nor duplicated as versioned development dependencies.
48
- The lockfile pins a reproducible local baseline, and `bun run compat:pi [version]`
49
- validates another Pi release in an isolated temporary workspace.
50
-
51
- ## Pipeline at a glance
128
+ `app/smart-compact-policy.ts` keeps agent-tool visibility separate from
129
+ automatic compaction. The tool remains registered for immediate re-enable, but
130
+ Pi's active-tool set controls whether its schema and guidance reach the agent.
131
+ Agent access is tri-state: `inherit` leaves host `/tools` and allowlists in
132
+ control, while explicit `enabled`/`disabled` mutate only `smart_compact` and
133
+ then report the effective host state. Policy snapshots are custom branch
134
+ entries restored on `session_start` and `session_tree`; the manual command is
135
+ never gated.
136
+
137
+ ### Home, presets and readiness
138
+
139
+ `ui/home-overlay.ts` shows five rows: **Compact now**, **Clean up tool
140
+ output**, **Settings**, **History & recovery**, and **Status & help**. Context
141
+ and the effective automatic/agent policy stay visible above the list.
142
+ Compact-picker cancellation returns to Home; no menu-open path changes
143
+ settings.
144
+
145
+ `ui/profiles.ts` derives presets from exact persisted flags. Behavior presets
146
+ are **Manual only**, **Manual + agent**, **Cleanup only** and **Fully
147
+ automatic**; the built-in defaults, which match none of them, are labeled
148
+ **With Pi (default)**. Summary formats are **Verified text**, **Text +
149
+ images** and **Provider (experimental)**; provider output needs a second
150
+ confirming Enter, and capacity-ineligible models cannot be selected. Fully
151
+ automatic selects the `settled` strategy. Presets atomically patch existing
152
+ keys; existing settings are never migrated, and branch overrides stay separate.
153
+
154
+ `app/effective-state.ts` gives Home (`Status & help → Readiness & details`),
155
+ preflight, metrics and dashboard one local-evidence view. It resolves routes,
156
+ reports credential presence without calling `getApiKey`, checks local backend
157
+ prerequisites and reads pure runtime state. It never refreshes OAuth, probes a
158
+ provider or server, creates a store or acquires a lease. Pi's effective
159
+ auto-compaction setting is not available through the public extension API, so
160
+ for `native-hook` that prerequisite is reported as unknown. Memory readiness is
161
+ informational: an unready optional backend warns but never blocks compaction.
162
+ `app/model-feasibility.ts` may disable model rows from a local estimate of the
163
+ planned stage requests; it never refreshes credentials.
164
+
165
+ Preflight and result overlays size their own viewports from terminal rows,
166
+ because Pi 0.87.1 renders overlays with `render(width)` and no height. Actions
167
+ stay outside scrolling content. Result approval is explicit (`A` only);
168
+ technical details are opt-in, while verification and fallback warnings stay
169
+ visible.
170
+
171
+ ## 1. Context hygiene
172
+
173
+ Hygiene keeps the working set small before summarization is needed. It never
174
+ calls a model.
175
+
176
+ ### Command hygiene (optional RTK companion)
177
+
178
+ `src/rtk.ts` is a separate entry point, absent from `pi.extensions`. It
179
+ delegates rewrite rules to the external RTK binary and never executes the
180
+ original command itself. Eligibility is only bare `git status`, `cargo test`
181
+ and `bun test`, tested against RTK 0.50: `bun test` joined after a paired
182
+ native/filtered runner check showed exit-code and failure/load-error parity
183
+ plus full recall of the filtered output, with one execution; `git diff`, `tsc`
184
+ and vitest 5 measured lossy or growing, and `npm test`/`node --test` have no
185
+ rule. Arguments, unknown syntax/flags and compound commands pass through.
186
+ Missing binary, unknown version, cancellation, session invalidation and CLI
187
+ failure all retain the input. The host bash tool keeps execution and result
188
+ ownership. No retries or output rewrite hooks are added. Permission hooks must
189
+ run after this input mutation. RTK's recall store and retention are not Smart
190
+ Compact artifacts and are not covered by its scrubbing.
191
+
192
+ ### Early tool-output artifacts
193
+
194
+ `artifactOffloadEnabled` defaults to false. `app/tool-artifacts.ts` handles
195
+ `tool_result` before the next model request, using the shared conservative
196
+ read-only allowlist. Errors, images, commands, mutations, unknown tools,
197
+ instruction/skill reads, file/symbol read deliveries (`read`, `read_symbol`,
198
+ `read_enclosing`) and the extension's own recovery tool stay inline. File reads
199
+ are excluded until read guards can honor actual delivered coverage rather than
200
+ the original call's implied full range. Earlier hooks' content is the capture
201
+ boundary; no arbitrary full-output path is read. Native details and status are
202
+ preserved.
203
+
204
+ After scrubbing, text of 16k+ characters (maximum 2 MiB) is content-addressed
205
+ in a session-origin directory under `smart-compact-artifacts/`, separate from
206
+ disposable caches. Async atomic-write and cross-process lock helpers protect
207
+ writes and quota checks. The final file is read back with bounded I/O, size and
208
+ SHA-256 verification before a preview is returned. Directory symlinks and file
209
+ symlinks/hardlinks are rejected. Storage, cancellation and quota failures are
210
+ fail-open for the original result and are never advertised as recoverable
211
+ output.
212
+
213
+ A small `details.smartCompactArtifact` record carries scope/content/preview
214
+ hashes, size, tool and source description. Only active-branch provenance
215
+ authorizes lookup; a content/preview mismatch or unowned context edit revokes
216
+ recovery. Forks may retain their parent reference; they do not duplicate
217
+ files. Files are not expired or evicted while refs may exist: 256 files/32 MiB
218
+ per origin is a stop-spilling quota, not LRU. Global usage across sessions is
219
+ not capped. Explicit cleanup must account for dependent forks; missing evidence
220
+ is an error, not a live-file reread. Interrupted writes can leave unreferenced
221
+ files charged against that origin's cap. Branch authorization is indexed by
222
+ entry ID, not content hash, so byte deduplication cannot overwrite source
223
+ provenance. The catalog collapses only repeated (tool, source label, payload
224
+ hash) rows; distinct labels share the hash read alias and keep direct entry-ID
225
+ retrieval. Foreign edits revoke the affected occurrence, not independent
226
+ references. No new on-disk format or migration is involved.
227
+
228
+ ### Recoverable trimming
229
+
230
+ `app/context-operations.ts` plans edits against native
231
+ `buildSessionProjection()`. `app/register-smart-context-tool.ts` queues one
232
+ mutation and returns drafts from `turn_end`, only after a successful
233
+ originating tool result, on the same branch, with no competing drafts or
234
+ pending user input. The native host owns persistence; there are no mid-tool
235
+ session mutations, tree-navigation hacks, or summarizer calls.
236
+
237
+ `requestManualTrim` on that controller backs both `/smart-compact trim` and the
238
+ Home **Clean up tool output** row. Queueing performs no model call and forces
239
+ no turn, so the next provider request is still sent untrimmed and the queued
240
+ edit applies at the next natural completed-turn boundary. A pending return to
241
+ an anchor or newer boundary change cancels it with a visible message.
242
+
243
+ Trimming protects four recent assistant turns and the active checkpoint prefix,
244
+ replacing up to 32 old read-only results of at least 4096 characters with a
245
+ bounded reference marker. Already-edited entries are not rewritten. Automatic
246
+ trimming runs with `contextHygieneEnabled` independently of compaction, or with
247
+ the effective `background` strategy, and requires `smart_context` reachable
248
+ by the model (`canAutoTrim` checks it like artifact offload does).
249
+ `plan` gives an on-demand non-mutating preview. Automatic batches require at
250
+ least 16,384 net saved characters and eight assistant turns since the last
251
+ owned trim/rewind/compaction; branch history supplies that cooldown across
252
+ reloads and forks. At a completed, uncontested turn boundary a ready batch
253
+ commits with cause `pressure` (early pressure gate reached) or `break-even`
254
+ (catalog prices say it pays back within `AUTO_TRIM_BREAK_EVEN_REQUESTS` = 24
255
+ requests). Otherwise it is held (`deferredTrim` in `smart_context` status):
256
+ once the prompt cache has expired, `context_with_system` sends the trimmed
257
+ results request-locally, byte-identical to the future `context_edit`, and the
258
+ edits commit with cause `cold` at the next completed turn. While a batch is
259
+ held, `cache_warming_decision` may stop Pi's cache warming when a refresh no
260
+ longer pays. Unknown prices allow only `pressure` and `cold`. The formula and
261
+ drop conditions are in [configuration](./docs/configuration.md);
262
+ `app/host-cache-ledger.ts` attributes observed prefix rebuilds. These are
263
+ catalog-price estimates, not measured cache billing. Manual and agent trims
264
+ commit at the next boundary with cause `manual` or `agent`; explicit trim
265
+ bypasses batching, not safety.
266
+
267
+ Instruction and skill sources stay inline through trim, rewind, artifact,
268
+ bitmap and pre-compaction pruning; recovery tool output is not recursively
269
+ re-archived. Pre-compaction dedup requires identical text content plus
270
+ arguments and no intervening changed observation or unsafe operation.
271
+ Unproven status-looking user text is never a deletion signal. Hygiene yields to
272
+ existing EESV work and staged candidates. Its handler precedes background
273
+ observation, and projected edits in the accumulated drafts prevent speculative
274
+ work from capturing a stale pre-edit branch. Explicit edits cancel background
275
+ work and invalidate the shared pending slot. Context fingerprints and
276
+ compaction guards prevent removed evidence from being resurrected.
277
+
278
+ ## 2. Recoverable continuity
279
+
280
+ Continuity means the session can find what it needs again on the same branch,
281
+ across reloads and forks, without copying the transcript to a second store.
282
+
283
+ ### Checkpoint, rewind and archived output
284
+
285
+ Small `custom` records (`smart-compact-context`, version 1) hold one active
286
+ checkpoint and lists of archived output IDs. Status is rebuilt from active
287
+ ancestry on demand, including after reload; these records do not enter model
288
+ context. A checkpoint captures session/origin IDs and a projected-prefix
289
+ fingerprint. New user messages, an intervening compaction or branch summary, or
290
+ a changed prefix invalidate it. The agent-written handoff is a bounded
291
+ `custom_message`, not an authoritative user instruction or an EESV
292
+ verification result.
293
+
294
+ A conservative read-only tool allowlist plus shared nested-call normalization
295
+ protects side effects. Rewind removes only whole successful, complete,
296
+ text-only read exchanges and successful assistant prose after the checkpoint.
297
+ Mixed batches, errors, commands, writes, unknown tools and image results stay
298
+ raw. Context edits omit messages in the projection, not in session history. The
299
+ mutation cap is 512; no partial rewind is applied when it is exceeded. Files
300
+ and processes are untouched.
301
+
302
+ Reference retrieval is restricted to this extension's branch-local archive
303
+ records and original textual tool results; foreign context edits revoke access
304
+ until an owned archive record explicitly re-authorizes it. Assistant reasoning
305
+ and arbitrary entries are never exposed. Full text is scrubbed before a bounded
306
+ page is sliced, avoiding cross-page credential reconstruction.
307
+
308
+ ### Evidence search
309
+
310
+ `app/context-evidence.ts` unifies native-history refs, bounded visual excerpts
311
+ and artifact files behind `smart_context`. Source descriptions and bounded
312
+ literal search make old evidence discoverable without knowing its ID.
313
+ Native-history labels use the shared `extractToolPath` alias rules. Search
314
+ checks at most 32 sources/4 Mi characters and returns at most ten
315
+ first-per-source matches plus a continuation cursor; no source bodies are
316
+ injected just to list them. Read supports character or line selection and a
317
+ 4096-character response cap. All representations are re-scrubbed in full
318
+ before searching or paging. Context, recovery and compaction see only the
319
+ stored preview; full files are retrieved only on explicit agent calls.
320
+
321
+ ### Storage inventory
322
+
323
+ `app/artifact-storage.ts` backs `/smart-compact storage` with a strictly
324
+ read-only inventory of the spill store. It streams every `*.jsonl` under the
325
+ native sessions root with bounded memory and classifies each origin directory
326
+ against two lineage anchors that only Pi maintains: session headers
327
+ (`sha256(id)` = owner scope) and `details.smartCompactArtifact.owner`
328
+ references that forks copy. Outcomes are `live`, `unreferenced-in-scan`, or
329
+ `unknown` when the scan is incomplete. `unreferenced-in-scan` is deliberately
330
+ not safe-to-delete: sessions outside the scanned root are undiscoverable, and a
331
+ running session can add references after the scan. That is why no `--clean` or
332
+ artifact GC exists; `ui/storage-report.ts` renders totals, status, bytes and
333
+ retention without deletion verbs. Durability tests age real `SessionManager`
334
+ artifacts past 20 days by timestamp (deterministic, not a wall-clock soak),
335
+ reload and fork through the public consumer, fail closed on missing or tampered
336
+ bytes, and hold the per-origin caps without losing earlier evidence.
337
+
338
+ ### Session navigation and apply-time validation
339
+
340
+ `app/register-navigation.ts` owns anchors, cross-session recall, pivots, the
341
+ anchor footer and the Anthropic anchor cache marker (`app/anchor-cache.ts`;
342
+ semantics adapted from pi-toolkit's auto-context under its MIT license).
343
+ `app/navigation-data.ts` reads both owned anchors (`custom_message` entries of
344
+ type `smart-context-anchor`, plus `smart_navigation` tool results) and legacy
345
+ `context` tool anchors recorded by pi-toolkit; recall scans other sessions'
346
+ JSONL read-only. pi-toolkit's auto-context extension is not a supported
347
+ co-resident: two anchor and pruning owners are unsafe in any load order, and
348
+ `piToolkit.context.thinningEnabled` is no longer read.
349
+
350
+ A human anchor is a `sendMessage` custom message followed by a native label on
351
+ that entry; a human pivot navigates to the label so Pi's `custom_message`
352
+ handling cannot drop the anchor text. An agent pivot is queued by the tool,
353
+ revalidated at `turn_end` (uncontested successful batch, same session, same
354
+ leaf, permission unchanged), dispatched at `agent_settled` through a nonce-bound
355
+ `/smart-compact` apply command, and supplied to `session_before_tree` as the
356
+ branch summary. New input, a session switch, a foreign tree navigation or a
357
+ permission change cancels it with a visible notice. While a pivot is queued or
358
+ applying, `session_before_compact` returns `{ cancel: true }`, automatic
359
+ preparation stops, and both new and queued `smart_context` mutations are
360
+ refused; evidence reads keep working. Files, processes and Git state are never
361
+ rolled back.
362
+
363
+ Reader identity and limits are captured before preparation starts.
364
+ `revalidatePending` checks that snapshot, the active projected prefix, current
365
+ usage/growth, native reserve, response headroom, target and yield for
366
+ foreground and background candidates. New compact instructions invalidate old
367
+ staged work. Navigation and model events clear speculative, pending and
368
+ staged-commit candidates; a generation guard rejects late native-hook work if
369
+ invalidation happened during its provider call. Pi remains the apply owner.
370
+
371
+ ### Continuity state across compactions
372
+
373
+ After a confirmed compaction, `utils/state.ts` carries structured state
374
+ forward: open loops, a continuity ledger in which prior facts persist until
375
+ positive resolution evidence or an explicit override, non-destructive goal
376
+ breadcrumbs, and a "Changes Since Last Compaction" delta. Snapshots are scoped
377
+ to project, session and branch head; see
378
+ [state, caching and persistence](#state-caching-and-persistence).
379
+
380
+ ## 3. Verified compaction
381
+
382
+ ### Automatic strategies
383
+
384
+ `autoTrigger` gates every automatic path; with it off, neither `settled` nor
385
+ `background` runs and the native hook does not replace Pi's summaries. Pi's own
386
+ compactor is not affected. Hygiene can still run.
387
+
388
+ - **`native-hook`** (default) is passive. It participates only when Pi starts
389
+ a compaction; its percentage setting is a minimum replacement gate, not a
390
+ scheduler, and Pi auto-compaction must itself be enabled for host-driven
391
+ runs.
392
+ - **`settled`** applies finite token/percentage, queue, in-flight and
393
+ per-session cooldown guards at idle `agent_settled`, then asks Pi for a
394
+ normal host compaction. Pi re-enters `session_before_compact`, which reuses an
395
+ already tool-staged summary or runs EESV exactly once under the host's signal
396
+ and timeout. `session_compact` stays the only durable commit authority;
397
+ `session_compact_failed` discards the candidate. This keeps proactive
398
+ triggering out of the pending/commit state machine and preserves
399
+ branch-provenance checks.
400
+ - **`background`** (`app/background-preparation.ts`) snapshots branch, model,
401
+ session and usage before asynchronous work. `prepareContextPercent` sets an
402
+ independent early gate with the existing minimum token floor; null preserves
403
+ `applyTokens - clamp(floor(applyTokens * 0.125), 8192, 32000)`. An explicit
404
+ prepare percentage must be below the effective `minContextPercent` apply gate.
405
+ Only pipeline admission is lowered; targets, yield and apply-time safety are
406
+ unchanged. A private pending slot isolates cancellation from foreground
407
+ staging. One task, a ten-minute attempt cooldown and a five-minute ready TTL
408
+ bound speculation. Model/config changes, branch navigation, new compaction or
409
+ context edits, switch/shutdown and native compaction invalidate unfinished
410
+ work, and late completion cannot publish after cancellation. Before reuse, a
411
+ content fingerprint proves the captured prefix is unchanged; appended tail
412
+ growth must still satisfy the verified target, minimum yield and response
413
+ reserve.
414
+
415
+ Preparation never waits inside `turn_end`. Proactive application waits for
416
+ idle `agent_settled`; an ongoing tool loop relies on Pi's native maintenance
417
+ boundary. No boundary draft bypasses the `session_compact` commit protocol. The
418
+ native hook falls back normally if speculative work is unavailable. Completed
419
+ discarded preparation emits one cost-only record with reason and ready/wait
420
+ timing, never an applied outcome. Shutdown cancels and drains tracked work and
421
+ writes so late completion cannot lose its accounting on graceful exit.
422
+
423
+ A short-lived pending compaction is staged in the
424
+ [`PendingSlot`](#pending-compaction-slot) and handed to Pi when compaction is
425
+ applied.
426
+
427
+ ### Pipeline at a glance
52
428
 
53
429
  ```mermaid
54
430
  flowchart LR
@@ -66,7 +442,7 @@ flowchart LR
66
442
  Y -- Yes --> J[Pending compaction returned to Pi]
67
443
  ```
68
444
 
69
- The orchestrator ([`src/app/run-smart-compact.ts`](./src/app/run-smart-compact.ts))
445
+ The orchestrator ([`src/app/run-smart-compact.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-smart-compact.ts))
70
446
  threads a typed context through ten stages:
71
447
 
72
448
  | # | Stage | Module | Transition |
@@ -82,12 +458,15 @@ threads a typed context through ten stages:
82
458
  | 9 | persist | `app/steps/persist.ts` | stage pending, apply compaction |
83
459
  | 10 | metrics | `app/steps/metrics.ts` | success / failure record |
84
460
 
85
- ## The typed stage machine
461
+ `app/steps/visual.ts` may run after verification when the experimental
462
+ [visual evidence](#experimental-visual-evidence) path is enabled.
463
+
464
+ ### The typed stage machine
86
465
 
87
- [`src/app/run-context.ts`](./src/app/run-context.ts) models the pipeline context
88
- as a **state machine of branded intersection types**. Each step accepts the
466
+ [`src/app/run-context.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-context.ts) models the pipeline
467
+ context as a state machine of branded intersection types. Each step accepts the
89
468
  previous stage type and returns the next, so reordering or skipping a step is a
90
- **compile-time error**, not a runtime crash:
469
+ compile-time error:
91
470
 
92
471
  ```text
93
472
  RcBase
@@ -101,246 +480,427 @@ RcBase
101
480
  → StatedRc (after state)
102
481
  ```
103
482
 
104
- Each stage adds a `_prepared` / `_windowed` / … discriminator field that
105
- carries the type-level proof and is checked by `advance()` at runtime.
106
- Mutation is preserved: a step mutates its input object and casts it to the
107
- next stage (no per-step copy of ~30 fields). The final alias
108
- `RunContext = StatedRc` keeps post-`buildState` consumers readable.
109
-
110
- This is what lets `applyCompaction` read `rc.details` with zero `!` non-null
111
- assertions: the type system proves `buildState` has run.
112
-
113
- ## Core execution model
114
-
115
- ### Entry and context gate
116
-
117
- `src/index.ts` owns host lifecycle wiring. Cohesive adapters in
118
- `src/app/register-smart-compact-command.ts`, `register-smart-compact-tool.ts`,
119
- and `model-routing.ts` validate input, resolve models, and route work into
120
- `runSmartCompact()`. Before any expensive work, the system checks context size
121
- against the threshold in `src/constants.ts`. Auto / tool runs are skipped while
122
- context is small; manual `/smart-compact` uses an absolute adaptive safety tail
123
- rather than a percentage of large model windows. Its decision-card preflight
124
- is built from the same config snapshot, calibrated estimator, adaptive profile,
125
- active branch, and pure window planner as execution. It compares exactly Fast,
126
- Balanced, and Thorough; `M` changes the summary route and replans all three,
127
- while `D` reveals technical estimator/boundary details. A plan must meet the
128
- tail target and at least 10% projected net savings before any model call. A
129
- pending summary for the same session is reused instead of invoking the pipeline
130
- again. `auto` is not a fourth policy: it selects one of the three from context
131
- pressure and deterministic extraction risk.
132
-
133
- `app/smart-compact-policy.ts` keeps agent-tool visibility separate from
134
- automatic compaction. The tool remains registered for immediate re-enable, but
135
- Pi's active-tool set controls whether its schema and prompt guidance reach the
136
- agent. Agent access is tri-state: `inherit` leaves host `/tools` and allowlists
137
- in control, while explicit `enabled`/`disabled` choices mutate only
138
- `smart_compact` and then report the effective host state. Full policy snapshots
139
- are custom branch entries restored on `session_start` and `session_tree`; the
140
- manual command is never gated.
141
-
142
- Model routes are stage-specific but never inferred from mode. With no explicit
143
- configuration, Explore, Synthesize, and Verify all use the selected Pi model.
144
- `segmentationModel`, `summaryModel`, and `verificationModel` can independently
145
- override those routes. Credentials resolve lazily immediately before each
146
- stage's first network call and are reused for equivalent routes; call metrics
147
- preserve the actual route.
483
+ Each stage adds a `_prepared` / `_windowed` / … discriminator that carries the
484
+ type-level proof and is checked by `advance()` at runtime. A step mutates its
485
+ input and casts it to the next stage (no per-step copy of ~30 fields). The
486
+ final alias `RunContext = StatedRc` lets `applyCompaction` read `rc.details`
487
+ with no non-null assertions: the type system proves `buildState` has run.
488
+
489
+ ### Entry, modes and routing
490
+
491
+ `src/index.ts` owns host lifecycle wiring. `app/register-smart-compact-command.ts`,
492
+ `register-smart-compact-tool.ts`, `smart-compact-input.ts` and `model-routing.ts`
493
+ validate input, resolve models and route work into `runSmartCompact()`. Before
494
+ expensive work, context size is checked against the thresholds in
495
+ `src/constants.ts`. Auto and tool runs are skipped while context is small;
496
+ manual `/smart-compact` uses an absolute adaptive safety tail rather than a
497
+ percentage of large model windows.
498
+
499
+ `app/preflight.ts` builds the decision card from the same config snapshot,
500
+ calibrated estimator, adaptive profile, active branch and pure window planner
501
+ as execution. It compares exactly Fast, Balanced and Thorough; `M` changes the
502
+ summary route and replans all three, and `D` reveals estimator/boundary
503
+ details. A plan must meet the tail target and at least 10% projected net
504
+ savings before any model call. A pending summary for the same session is reused
505
+ instead of running the pipeline again. `auto` is a selector, not a fourth
506
+ policy (`app/mode-policy.ts`): it chooses one of the three from context pressure
507
+ and deterministic extraction risk. Legacy `aggressive` maps to Fast.
508
+
509
+ Model routes are stage-specific and never inferred from mode. With no explicit
510
+ configuration, Explore, Synthesize and Verify use the selected Pi model;
511
+ `segmentationModel`, `summaryModel` and `verificationModel` override them
512
+ independently. `app/stage-auth.ts` checks credential availability just before
513
+ each stage's first network call and reuses the answer for equivalent routes;
514
+ the session runtime resolves the actual auth per request, and call metrics
515
+ keep the actual route. How routing evidence is gathered is described in
516
+ [evaluation](./docs/evaluation.md#provider-routing-evidence).
148
517
 
149
518
  ### Keep window and preprocessing
150
519
 
151
- `app/steps/window.ts` starts from Pi's compaction-aware
152
- `buildContextEntries()` view, never the append-only session history, and builds
153
- a content-free `CompactionWindowPlan` from the selected mode budget. The shared
154
- `contextMessageEntries()` adapter uses Pi's `sessionEntryToContextMessages()` and
155
- `convertToLlm()`, retaining host-visible custom/branch/compaction summaries and
156
- original entry IDs while excluding private or context-disabled entries:
157
-
158
- - **hard `toolCall` / `toolResult` guard** — never orphan a result from its call
159
- - **soft recent-user/checkpoint/topical preferences** — retain raw only when the resulting suffix still fits the planned budget
160
- - **yield contract** — projected replacement must meet its target and save at least 10% after reserving the summary budget
161
-
162
- The same pure planner powers manual preflight and execution. A soft boundary is
163
- recorded as relaxed rather than silently overriding the target. Long turns may
164
- be summarized through their older prefix; if the nominal cut lands inside a
165
- tool exchange, the planner either retains the complete pair within budget or
166
- advances past it so the complete exchange is summarized. It also advances past
167
- complete historical exchanges whose tool names violate the portable provider
168
- contract; this keeps model switches from exposing an unsendable raw tail. If no
169
- provider-safe hard boundary can meet the target, automatic/tool runs normally
170
- return control to Pi's native compactor before any LLM call. An
171
- already-overflowed context is
172
- the safety exception: measured usage is mapped across active messages and EESV
173
- keeps chunked recovery instead of sending an oversized one-shot prompt to
174
- native summarization. Manual runs use the profile's absolute adaptive tail, so
175
- model-window size cannot turn an explicit command into a full-context no-op.
520
+ `app/steps/window.ts` reads the selected ancestry via `getBranch()` and passes
521
+ it to native `buildSessionProjection()`, never converting raw entries
522
+ individually. `contextMessageEntries()` converts that projection with
523
+ `convertToLlm()`, keeping source IDs and host-visible custom, branch and
524
+ compaction summaries while honoring `context_edit` replacements and omissions.
525
+ Intentional edits are marked against raw-log recovery. The active view drives a
526
+ content-free `CompactionWindowPlan` from the selected mode budget:
527
+
528
+ - **hard `toolCall` / `toolResult` guard**: never orphan a result from its call;
529
+ - **soft recent-user/checkpoint/topical preferences**: keep raw only when the
530
+ suffix still fits the planned budget;
531
+ - **yield contract**: the projected replacement must meet its target and save
532
+ at least 10% after reserving the summary budget.
533
+
534
+ A relaxed soft boundary is recorded rather than silently overriding the
535
+ target. Long turns may be summarized through their older prefix; a cut inside a
536
+ tool exchange either keeps the complete pair within budget or advances past it.
537
+ The planner also advances past complete historical exchanges whose tool names
538
+ violate the portable provider contract, so model switches cannot expose an
539
+ unsendable raw tail. If no provider-safe hard boundary meets the target,
540
+ automatic and tool runs normally return control to Pi's native compactor before
541
+ any LLM call. An already-overflowed context is the exception: measured usage is
542
+ mapped across active messages and EESV keeps chunked recovery instead of
543
+ sending an oversized one-shot prompt. Manual runs use the profile's absolute
544
+ adaptive tail, so model-window size cannot turn an explicit command into a
545
+ full-context no-op.
176
546
 
177
547
  Before summarization the pipeline keeps a deferred reference to the recovered
178
- pre-prune messages, prunes redundant messages, loads prior continuity, checks the
179
- extraction cache, and loads the project fingerprint. Both synthesis and backup
548
+ pre-prune messages, prunes redundant messages, loads prior continuity, checks
549
+ the extraction cache and loads the project fingerprint. Synthesis and backup
180
550
  text use `serializeConversationText()` without the host summarizer's implicit
181
- 2,000-character tool-result cap. Structural and text redaction remain enforced;
182
- binary attachment archival is outside this text format. The backup materializes
183
- and writes atomically only after the matching native compaction is confirmed.
551
+ 2,000-character tool-result cap. Structural and text redaction stay enforced;
552
+ binary attachment archival is outside this format. The backup is materialized
553
+ and written atomically only after the matching native compaction is confirmed.
184
554
 
185
555
  ### Extract
186
556
 
187
- Primary: [`src/utils/extraction.ts`](./src/utils/extraction.ts). **Zero LLM calls.**
188
-
189
- Deterministically pulls: modified / read / deleted files, tool and bash-like
190
- errors, retry / resolution signals, explicit & implicit decisions, constraints
191
- and preferences, heuristic topic segments, timeline events, the main goal, and
192
- open loops. **This is the ground truth** that synthesis and verification trust.
557
+ [`src/utils/extraction.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/utils/extraction.ts). Zero LLM calls.
558
+ Deterministically pulls modified/read/deleted files, tool and bash-like errors,
559
+ retry/resolution signals, explicit and implicit decisions, constraints and
560
+ preferences, heuristic topic segments, timeline events, the main goal and open
561
+ loops across the whole compacted prefix. This is the ground truth that
562
+ synthesis and verification trust.
193
563
 
194
564
  ### Explore
195
565
 
196
- Primary: [`src/phases/explore.ts`](./src/phases/explore.ts). Optional — runs only
197
- in `thorough` mode or when `auto` selects that policy from deterministic risk. The model inspects
198
- the conversation through a small toolset: message ranges, conversation search,
199
- recent user messages, local context around an index, file-change lookups, and
200
- error chains. Tool support is runtime-probed once and cached per run; if a
201
- provider has no function calling, the system falls back to a direct structured
202
- analysis path. The growing tool conversation is capped at three rounds, each
203
- response is capped where the provider supports output limits, and the shared
204
- prefix uses short-lived prompt caching.
566
+ [`src/phases/explore.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/explore.ts). Runs only in `thorough`
567
+ mode or when `auto` selects it from deterministic risk. The model inspects the
568
+ conversation through a small toolset: message ranges, conversation search,
569
+ recent user messages, local context around an index, file-change lookups and
570
+ error chains. Tool support is probed once and cached per run; without function
571
+ calling the system falls back to a direct structured analysis. The tool
572
+ conversation is capped at three rounds, each response is capped where the
573
+ provider supports output limits, and the shared prefix uses short-lived prompt
574
+ caching.
205
575
 
206
576
  ### Synthesize
207
577
 
208
- Primary: [`src/phases/synthesize.ts`](./src/phases/synthesize.ts). Three paths:
578
+ [`src/phases/synthesize.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/synthesize.ts). Three paths:
209
579
 
210
- - **Deterministic zero-call** — high-confidence Fast extractions.
211
- - **Single-pass** — when the compacted conversation fits under the configured threshold.
212
- - **Hierarchical** — for larger sessions: merge available boundaries → split oversized semantic chunks → batch by token budget → summarize batches → assemble.
580
+ - **Deterministic zero-call** for high-confidence Fast extractions.
581
+ - **Single-pass** when the compacted conversation fits under the configured
582
+ threshold.
583
+ - **Hierarchical** for larger sessions: merge available boundaries, split
584
+ oversized semantic chunks, batch by token budget, summarize batches,
585
+ assemble. Every batch is summarized, so coverage does not depend on a
586
+ fragment's position in the history.
213
587
 
214
- Behaviors: session-aware prompting, decision propagation across later batches,
215
- mode-specific single-pass thresholds and output limits, provider-aware
216
- concurrency (wave scheduling), aggregate prompt-token reservation, and a
217
- deterministic fallback assembly when any budget or LLM call fails.
588
+ Session-aware prompting, decision propagation across later batches,
589
+ mode-specific thresholds and output limits, provider-aware wave concurrency,
590
+ aggregate prompt-token reservation and deterministic fallback assembly when any
591
+ budget or LLM call fails.
218
592
 
219
593
  ### Verify
220
594
 
221
- Primary: [`src/phases/verify.ts`](./src/phases/verify.ts). Scores the summary
222
- against deterministic extraction, continuity, explicit focus/note steering,
223
- and source messages. It checks missing modified/read/deleted files, unresolved
224
- errors, high-confidence constraints, weak goal coverage, missing structure,
225
- suspicious fabricated paths, done/unresolved inconsistencies, explicit
226
- decisions, open loops, and unsupported high-risk outcome claims. Claims such as
227
- “tests passed” require matching source prose or a successful related tool
228
- result.
229
-
230
- **Repair order is intentional:** (1) deterministic patch first (free,
231
- idempotent) → (2) one LLM patch only in `thorough` mode if still insufficient
232
- → (3) replace lower-scoring output with a deterministic quality floor built
233
- only from extraction, continuity, and steering → (4) reject unless final
234
- verification has no gaps and meets the verified threshold. Untrusted chunk
235
- prose is never an input to the quality floor.
236
- Final verification runs again after continuity injection. The final scalar is
237
- reported as repaired **verification coverage**, alongside the pre-repair source
238
- score and fallback provenance; it is not labeled as raw synthesis quality.
239
- Verification failures retain only exhaustive content-free gap kinds and the
240
- rejecting gate (`post-synthesis` or `post-state`) in local telemetry. Summary
241
- evidence is never copied into failure metrics. Both gates remain mandatory:
242
- summary-derived continuity fields cannot become evidence for their own initial
243
- verification.
595
+ [`src/phases/verify.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/verify.ts) scores the summary against
596
+ deterministic extraction, continuity, explicit focus/note steering and source
597
+ messages. It checks missing modified/read/deleted files, unresolved errors,
598
+ high-confidence constraints, weak goal coverage, missing structure, suspicious
599
+ fabricated paths, done/unresolved inconsistencies, explicit decisions, open
600
+ loops and unsupported high-risk outcome claims. Claims such as "tests passed"
601
+ need matching source prose or a successful related tool result.
602
+
603
+ Repair order is intentional: (1) deterministic patch first (free, idempotent);
604
+ (2) one LLM patch only in `thorough` mode if still insufficient; (3) replace
605
+ lower-scoring output with a deterministic quality floor built only from
606
+ extraction, continuity and steering; (4) reject unless final verification has
607
+ no gaps and meets the verified threshold. Untrusted chunk prose never feeds the
608
+ quality floor. Final verification runs again after continuity injection. The
609
+ final scalar is reported as repaired **verification coverage**, alongside the
610
+ pre-repair score and fallback provenance, never as raw synthesis quality.
611
+ Failures keep only exhaustive content-free gap kinds and the rejecting gate
612
+ (`post-synthesis` or `post-state`) in local telemetry. Summary-derived
613
+ continuity fields cannot become evidence for their own initial verification.
614
+
244
615
  Polarity checks are symmetric: adding negation to a positive fact is rejected
245
616
  just as removing negation from a prohibition is. Short negation tokens such as
246
617
  `no` survive token filtering. Verbatim source clauses are not compared against
247
- the whole instruction's polarity, while additional contradictory clauses remain
618
+ the whole instruction's polarity, while additional contradictory clauses stay
248
619
  checked. Exact grounded path representations are not outcome claims; prose in
249
620
  file sections is still verified. Synthesis and post-state verification use the
250
- same summary budget for path encoding. Unresolved-error source
251
- snippets and fallback-rendered evidence share `summaryEvidenceLine()`, so
252
- Markdown prefixes and multiline wrapping cannot create false missing-error gaps.
253
-
254
- ## EESV hardening and control surfaces
255
-
256
- - **Canonical summary IR** accepts recognized H1/H2/H3 headings outside fenced code, preserves Progress subsections, and merges duplicate canonical kinds before state mutation.
257
- - **Typed verification gaps** drive mandatory deterministic repair; collision-aware path needles prevent basename cross-satisfaction. Tool/file provenance and normalized semantic evidence are indexed once per verification pass, and truncated or delimiter-incomplete LLM patches are rejected. Provenance is persisted and shown before optional approval.
258
- - **Fine tool semantics** separate read/search/list/mutate/delete/execute operations. Pruning deduplicates only identical idempotent access signatures.
259
- - **Unified token planning** uses a run-bound estimator with bounded process-shared provider/model calibration, counts structured tool-call arguments, preserves an adaptive recent tail, targets mode-specific post-compaction headroom, reserves bounded deterministic post-summary state sections, clamps every provider request to the model's advertised output limit, and reserves/reconciles every request against aggregate prompt/output-token caps. Missing provider usage is estimated conservatively. Tool exchanges remain atomic; oversized result bodies are head/tail bounded only for synthesis after full deterministic extraction.
260
- - **Security boundaries** recursively scrub structured messages before host serialization or provider calls, redact secret-bearing primitive values, and scrub plus hard-cap exploration tool feedback. PII scrubbing is opt-in. Backups remain unmaterialized until confirmed apply.
261
- - **Policy controls** include focus weighting, exact call/latency budgets, default fail-closed manual approval, online damage monitoring, and persisted open-loop overrides. Interactive review time is outside the pipeline deadline.
262
- - **Release gates** (`bun run gate`, `bun run bench`) cover adversarial parser, verification, tool, cache, budget, scrub and damage fixtures plus bounded p95 regressions for extraction, pruning, chunking, summary parsing, and path matching.
263
-
264
- ## State, caching & persistence
265
-
266
- Post-verification, `app/steps/state.ts` + `src/utils/state.ts` enrich the
621
+ same summary budget for path encoding. Unresolved-error snippets and
622
+ fallback-rendered evidence share `summaryEvidenceLine()`, so Markdown prefixes
623
+ and wrapping cannot create false missing-error gaps. Fallback constraints keep
624
+ the full extraction bound (`TRUNC.CONSTRAINT_TEXT`, 300 characters) without a
625
+ second preview cut or category label, because decorating or shortening a
626
+ faithful compound instruction can defeat exact-source matching.
627
+ `domain/keywords.ts` supplies the shared salient-keyword check used by
628
+ verification and damage detection.
629
+
630
+ ### EESV hardening and control surfaces
631
+
632
+ - **Canonical summary IR** accepts recognized H1/H2/H3 headings outside fenced
633
+ code, preserves Progress subsections, and merges duplicate canonical kinds
634
+ before state mutation.
635
+ - **Typed verification gaps** drive mandatory deterministic repair;
636
+ collision-aware path needles prevent basename cross-satisfaction.
637
+ Provenance and normalized semantic evidence are indexed once per pass, and
638
+ truncated or delimiter-incomplete LLM patches are rejected. Provenance is
639
+ persisted and shown before optional approval.
640
+ - **Fine tool semantics** separate read/search/list/mutate/delete/execute.
641
+ Pruning deduplicates only identical idempotent access signatures.
642
+ - **Unified token planning** uses a run-bound estimator with bounded
643
+ process-shared provider/model calibration, counts structured tool-call
644
+ arguments, keeps an adaptive recent tail, targets mode-specific
645
+ post-compaction headroom, reserves bounded post-summary state sections,
646
+ clamps every request to the model's advertised output limit, and reconciles
647
+ every request against aggregate prompt/output caps. Missing provider usage is
648
+ estimated conservatively. Tool exchanges stay atomic; oversized result bodies
649
+ are head/tail bounded only for synthesis, after full deterministic
650
+ extraction.
651
+ - **Per-dispatch capacity revalidation** (`domain/model-capacity.ts`) never
652
+ compares the whole conversation to a stage model's window: `trackedComplete`
653
+ estimates each actual serialized request, with output clamped to the model's
654
+ limit and Pi-AI's 4,096-token safety margin, and throws an actionable
655
+ `ModelCapacityError` before the provider is contacted. UI feasibility rows are
656
+ advisory snapshots; sizes that only exist after generation rely on this
657
+ runtime guard.
658
+ - **Security boundaries** recursively scrub structured messages before host
659
+ serialization or provider calls, redact secret-bearing primitive values, and
660
+ scrub plus hard-cap exploration tool feedback. PII scrubbing is opt-in.
661
+ Backups stay unmaterialized until confirmed apply.
662
+ - **Policy controls** include focus weighting, exact call/latency budgets,
663
+ default fail-closed manual approval, online damage monitoring and persisted
664
+ open-loop overrides. Interactive review time is outside the pipeline
665
+ deadline.
666
+ - **Release gates** (`bun run gate`, `bun run bench`) cover adversarial parser,
667
+ verification, tool, cache, budget, scrub and damage fixtures plus bounded p95
668
+ regressions for extraction, pruning, chunking, summary parsing and path
669
+ matching.
670
+
671
+ ### Provider-native compaction engine
672
+
673
+ Provider-native compaction is an optional engine on stock Pi, using public
674
+ extension APIs only. Pi 0.87.1's adapters do not parse or replay signed
675
+ Anthropic blocks or opaque OpenAI items, so this extension does both:
676
+
677
+ - `run-smart-compact.ts` runs the `compactionEngines` list after the window
678
+ step. Each engine is `applied`, `skipped` or `failed` (`EngineAttempt`); all
679
+ share one provider-call budget; if none applies, `EngineChainError` lists
680
+ every outcome and nothing is staged. The default list is `["eesv"]`.
681
+ - `app/native-compaction.ts` gates on `isNativeApi(ctx.model.api)` and uses only
682
+ the current session model. It moves the cut back to a clean turn boundary
683
+ (kept tail starts at a user message, prefix ends with a completed assistant
684
+ reply, no open tool call).
685
+ - Request: one nested `ctx.modelRegistry.streamSimple(model, { systemPrompt,
686
+ messages, tools }, { fetch, transport: "sse", maxRetries: 0, signal,
687
+ sessionId })`. Active tools are included because Anthropic rejects tool_use
688
+ history without them. Pi's adapter builds the ordinary request; the `fetch`
689
+ from `infra/native-protocol.ts` (`createCompactionFetch`) rewrites it into the
690
+ provider's compaction request, sends it once and answers Pi with a
691
+ non-retryable 400. Without a result no request was sent and Pi's error is
692
+ reported literally. A prefix that starts with an earlier native compaction of
693
+ the same route replays that state (`prior`); failure to replay aborts before
694
+ the provider call.
695
+ - Codex `response.incomplete` is a failure even if an item and usage arrived.
696
+ Results and persisted state require a signed Anthropic compaction block or
697
+ OpenAI compaction items with encrypted content; the opaque bytes are
698
+ preserved, not cryptographically verified locally. Results are also rejected
699
+ for another route or when not smaller than the prefix estimate. Accepted
700
+ state is staged in the ordinary pending slot and stored only in
701
+ `details.native` (`NativeState`, validated with `isNativeState` on every
702
+ read).
703
+ - Replay: `before_provider_request` takes the latest compaction entry on the
704
+ branch; if its `details.native` matches the current api/provider/model,
705
+ `replayNativeState` returns a payload copy replacing only Pi's exact wrapped
706
+ summary. A bare substring match never authorizes replay. When replay is
707
+ impossible on a matching route, OpenAI routes get a one-time warning per entry
708
+ and Anthropic is recorded only. The hook does no I/O and does not walk the
709
+ branch until the session has native state (a flag recomputed from in-memory
710
+ entries at `session_start`, `session_tree` and `session_compact`).
711
+ - Details record `method: "native"`, `nativeApi` and the engine attempts;
712
+ notices never claim EESV verification for native state.
713
+ - Requests made by the extension itself (EESV stages and the native
714
+ compaction body) go through the requesting session's public model runtime
715
+ (`ctx.modelRegistry.stream`/`streamSimple`, `infra/llm-client.ts`), never
716
+ pi-ai's standalone completers. The runtime applies request-time auth and any
717
+ provider registered by another extension with `pi.registerProvider`, such as
718
+ the separate `pi-claude-oauth-adapter`; stock Pi still skips
719
+ `before_provider_request` for these requests, so an adapter must normalize
720
+ the final payload inside its own provider (the published `0.2.2` does not;
721
+ [upstream PR #10](https://github.com/minzique/pi-claude-oauth-adapter/pull/10)).
722
+ Caller `apiKey`/`headers` are stripped so an explicit key never bypasses
723
+ stored OAuth; `app/stage-auth.ts` is an availability preflight only. The
724
+ Anthropic prior-state replay runs in the caller's `onPayload` before any
725
+ adapter normalization, and the final on-demand request drops
726
+ `context_management` because Anthropic prohibits combining it with
727
+ on-demand compaction.
728
+
729
+ `test/native-compaction-compat.test.ts` pins what stock adapters drop by
730
+ themselves. The design research and measured runs are in the
731
+ [2026-09-24 research report](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/hindsight-native-compaction-research-2026-09-24.md)
732
+ (repository only, historical).
733
+
734
+ ### Experimental visual evidence
735
+
736
+ `visualArchiveEnabled` defaults to false. After `buildState` has verified text
737
+ and post-compaction yield, `steps/visual.ts` may add a bounded supplementary
738
+ archive. It reuses the conservative read-only tool-batch selector, skips
739
+ intentional context edits, and scrubs text before clipping and rendering. A
740
+ Latin/Turkish glyph scope, at most eight 3k-character excerpts, two 1280-wide
741
+ pages (74 rows each) and 1 MB total PNG bytes bound local work. Oldest whole
742
+ excerpts are dropped until the image allowance plus reading guide fits both the
743
+ original target and response reserve. Representation comparisons use a
744
+ reader-bound estimator from the shared calibration store; summarizer
745
+ accounting stays separate. The yield gate runs again, and any failure keeps the
746
+ unchanged text. This does not avoid the EESV call or promise cheaper
747
+ compaction.
748
+
749
+ The optional resvg renderer is imported only on demand, uses one shipped
750
+ licensed font with system-font discovery disabled, and receives only
751
+ XML-escaped text in a fixed generated SVG, with no user-controlled SVG,
752
+ resource paths or URLs. It is external to both bundles; Node loads the default
753
+ extension without it. Rendering uses the run's abort signal plus a five-second
754
+ cancellation limit, with a fresh composed signal per page because resvg's
755
+ native abort binding cannot be reused.
756
+
757
+ `SmartCompactDetails.visualArchive` stores versioned bounded source excerpts
758
+ and PNGs in the native compaction entry; no file cache or second transcript
759
+ store is created. Later hybrid compaction re-renders source text, never OCR.
760
+ `context` validates frame bounds/signatures, source ancestry and revocation,
761
+ reader route, privacy, summary identity and request headroom, then inserts a
762
+ request-local custom image message; it never replaces the verified text
763
+ summary. Changing model/provider/API or using a text-only model withholds
764
+ images; stricter scrubbing also withholds old pixels that cannot be
765
+ retroactively redacted. Only the latest compaction's archive is eligible, and
766
+ native fallback can discard it. Metrics record only visual token estimates and
767
+ frame counts. The [2026-09-24 visual pilot](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/visual-pilot-2026-09-24.md)
768
+ (repository only) is a dated single-model synthetic sample, not production or
769
+ cross-model accuracy.
770
+
771
+ ## 4. Optional cross-session memory
772
+
773
+ Memory is exactly one selected backend at a time (`app/memory-backend.ts`):
774
+ `local`, `hindsight` or `mnemopi`. The selected backend is the only store read
775
+ or written. With Hindsight or Mnemopi selected, the local context graph is
776
+ neither indexed by compaction nor consulted, no local copy exists, and inactive
777
+ stores are preserved untouched. Hindsight means the user's existing configured
778
+ server; nothing installs or starts one. Continuity state, backups and artifact
779
+ spill are session mechanisms outside this choice. Manual saves require host
780
+ confirmation of the complete scrubbed content. Only the local backend also
781
+ indexes derived facts, and only from apply-confirmed compactions; remote
782
+ backends receive nothing automatically.
783
+
784
+ - **Local** (`infra/context-graph.ts`): project-partitioned SQLite FTS5 facts
785
+ and file edges, indexed from apply-confirmed compactions plus confirmed
786
+ manual memories. Details are in
787
+ [state, caching and persistence](#state-caching-and-persistence).
788
+ - **Hindsight** (`app/hindsight-memory.ts`, `infra/hindsight-client.ts`,
789
+ `infra/hindsight-receipts.ts`): four fixed routes (retain, status, recall,
790
+ delete one document), no generic request, origin/bank/project-scoped receipts
791
+ that never evict unconfirmed operations. Data flow, consent and receipt states
792
+ are in [Hindsight memory backend](./docs/hindsight-memory.md).
793
+ - **Mnemopi** is an optional Bun-only dependency, never imported by the Node
794
+ host. `app/mnemopi-memory.ts` starts a bounded worker and waits for a
795
+ readiness line after its imports before sending any content over stdin.
796
+ `resolveBunExecutable()` resolves the worker runtime read-only: the
797
+ optional `bun` component installed beside the extension first (manifest bin
798
+ plus a real-file check that rejects the postinstall placeholder), then the
799
+ platform `@oven/*` package, then a supported PATH Bun (>=1.3.14). No shell,
800
+ download or self-install is involved, so missing components fail before a memory
801
+ request; interrupted submitted writes remain uncertain. TypeBox validates both
802
+ IPC directions and persisted engine metadata. Project-isolated files,
803
+ author/kind filters and checked provenance prevent cross-project recall. The
804
+ cross-process lock covers identity lookup and mutation through child exit;
805
+ Mnemopi manages its own SQLite transactions. Stable per-fact engine sessions
806
+ make duplicate saves and metadata-id resolution work without a sidecar index.
807
+ No shared default bank, embeddings, LLM extraction, consolidation, runtime
808
+ model download or silent backend fallback is enabled.
809
+
810
+ `infra/memory-ref.ts` supplies opaque backend/id refs with mandatory 96-bit
811
+ target digests. Local and Mnemopi refs bind their store path; Hindsight also
812
+ binds server, bank, project and document. Resolution compares current target
813
+ configuration rather than reading a destination from untrusted input. These are
814
+ routing checks, not authorization tokens or remote-content attestations; host
815
+ confirmation stays mandatory. Local and Mnemopi resolve inspects the stored
816
+ fact; Hindsight does not claim its confirmation is a document read. Unknown
817
+ retain outcomes, including lost operation status, block remote deletion until
818
+ terminal evidence; bounded recall refresh never evicts uncertainty.
819
+
820
+ ## State, caching and persistence
821
+
822
+ After verification, `app/steps/state.ts` and `utils/state.ts` enrich the
267
823
  summary, then `domain/yield-gate.ts` measures the final replacement. Planning
268
- has already reserved the bounded post-synthesis enrichment band by reducing the
269
- retained tail; missing the original target or 10% net-saving floor still throws
270
- before a `StatedRc` can reach staging/apply. `session_before_compact` only stages
271
- a passing candidate; after the host emits the matching `session_compact`,
272
- `app/steps/persist.ts` commits reusable state, the prepared conversation backup,
273
- and success telemetry. Aborted/unconfirmed candidates write none of them. The
274
- UI reports `Applied` only after that correlated commit and emits a separate
275
- warning if any durable side effect was partial.
276
- `ui/error-format.ts` converts verification/yield failures to one bounded,
277
- content-free diagnostic and next action, including unknown/provider errors.
278
- Per-call categories survive in aggregated route metrics even when fallback
279
- succeeds; raw errors never enter that telemetry. Full stacks are suppressed by
280
- default and require restarting Pi with explicit `DEBUG=smart-compact`.
281
- Manual execution uses a two-line widget: a colored EESV phase chain plus a
282
- phase-specific action brief. Before Apply it explicitly says the conversation
283
- is unchanged. Routine info toasts are hidden unless `verbose`; handled provider,
284
- watchdog, Explore, batch, and assembly failures switch to deterministic fallback
285
- without printing raw messages. Auto-trigger rejection logs are also debug-only,
286
- leaving one content-free safe-fallback notice in the UI.
824
+ has already reserved the bounded enrichment band by reducing the retained tail;
825
+ missing the original target or the 10% net-saving floor still throws before a
826
+ `StatedRc` can reach staging or apply. `session_before_compact` only stages a
827
+ passing candidate (`app/compaction-commit-store.ts` holds it between the two
828
+ events). After the host emits the matching `session_compact`,
829
+ `app/steps/persist.ts` commits reusable state, the prepared conversation backup
830
+ and success telemetry. Aborted or unconfirmed candidates write none of them.
831
+ The UI reports `Applied` only after that correlated commit and warns separately
832
+ if any durable side effect was partial. Cost-only discarded-preparation records
833
+ cannot commit reusable state, backups or applied-canary evidence.
834
+
835
+ `ui/error-format.ts` turns verification/yield failures into one bounded,
836
+ content-free diagnostic and next action. Per-call categories survive in route
837
+ metrics even when fallback succeeds; raw errors never enter telemetry. Full
838
+ stacks require restarting Pi with `DEBUG=smart-compact`. Manual execution shows
839
+ a two-line widget: a colored EESV phase chain plus a phase-specific brief that
840
+ says the conversation is unchanged until Apply. Routine info toasts are hidden
841
+ unless `verbose`; handled provider, watchdog, Explore, batch and assembly
842
+ failures switch to deterministic fallback without printing raw messages.
843
+ Auto-trigger rejection logs are debug-only, leaving one content-free notice.
844
+ `utils/issues.ts` deduplicates user-facing problems once per session.
287
845
 
288
846
  | Concern | Where | Notes |
289
847
  | --- | --- | --- |
290
848
  | Open-loop injection | `utils/state.ts` | inserted before Next Steps via the canonical parser |
291
849
  | `CompactionState` | `utils/state.ts` | immutable project/session/branch-head snapshots; descendants resolve the newest matching ancestor and siblings never overwrite each other |
292
- | Continuity Ledger | `utils/state.ts` | prior facts carry forward until positive resolution evidence or an explicit override; goal shifts become non-destructive breadcrumbs |
850
+ | Continuity ledger | `utils/state.ts` | prior facts carry forward until positive resolution evidence or an explicit override; goal shifts become non-destructive breadcrumbs |
293
851
  | Cross-compaction delta | `utils/state.ts` | "Changes Since Last Compaction" section |
294
- | Incremental extraction cache | `utils/cache.ts` + `utils/id-fingerprint.ts` | bounded SHA-256 prefix fingerprint + tail; an exact pruned-payload match is reused directly, while incremental reuse is allowed only when the pruned prefix still matches |
295
- | Synthesis cache | `infra/synthesis-cache.ts` | behavior key includes normalized focus, route, mode, profile limits, run-level call/input/latency limits, and reasoning |
296
- | Session-log recovery | `utils/session-log.ts` | async bounded-memory JSONL scan with event-loop yields and a bounded path cache; bypasses pi-toolkit truncation by entry-id mapping without dropping late active-branch IDs at a fixed byte cap |
297
- | Project fingerprint | `utils/fingerprint.ts` | locked read/merge/write; language/framework/key dirs stay bounded and `sessionCount` tracks distinct hashed session identities |
852
+ | Native continuity handoff | `app/native-continuity-bridge.ts` | one-shot, bounded, keyed by project + session + branch head |
853
+ | Incremental extraction cache | `utils/cache.ts` + `utils/id-fingerprint.ts` | bounded entry-ID fingerprint plus projected/recovered content hash; reuse requires both prefix proofs |
854
+ | Synthesis cache | `infra/synthesis-cache.ts` | key includes normalized focus, route, mode, profile limits, run-level call/input/latency limits and reasoning |
855
+ | Session-log recovery | `utils/session-log.ts` | async bounded-memory JSONL scan; recovers only truncated, unedited messages by entry ID without resurrecting intentional replacements or omissions |
856
+ | Project fingerprint | `utils/fingerprint.ts` | locked read/merge/write; bounded language/framework/key dirs; `sessionCount` tracks distinct hashed sessions |
298
857
  | Damage detection | `utils/damage.ts` | best-effort post-compaction regression signals |
299
- | Context graph | `infra/context-graph.ts` | SQLite FTS5 facts + file edges; 2,000 non-structural nodes per project |
300
-
301
- Apply-confirmed verified state is queued and duplicate updates coalesce only
302
- for the exact project/session/branch head. Replacing an existing key refreshes
303
- that pending value even at capacity; a divergent 65th key is rejected rather
304
- than evicting accepted work. A microtask drains the accepted batch through one
305
- reused SQLite connection. Every caller awaits the transaction result, so
306
- persistence telemetry is complete only after indexing succeeds; permanent
307
- open/write failures settle once as `context graph` failures and are never
308
- zero-delay retried. All graph surfaces
309
- (recall, manual memory, stats, indexing) share one process-wide, path-keyed
310
- cached connection, so connection setup, schema checks, and migration probes
311
- are paid once per process instead of per call; the cache is keyed by database
312
- path so environments that relocate the cache directory (tests swapping HOME)
313
- reopen cleanly.
314
- `infra/context-graph.ts` adapts the same fail-closed transaction contract to
315
- `bun:sqlite` in Bun tests and `node:sqlite` `DatabaseSync` in Pi's Node runtime;
316
- the packed release audit exercises both. Graph data is derived and a later
317
- cumulative state safely supersedes a failed update. Fact
318
- occurrences are branch-head scoped; state, recall, and resolution use the
319
- complete host-visible branch ancestry before equivalent facts are deduplicated.
320
- Schema v1 preserves user-confirmed manual memory but resets older derived
321
- compaction nodes once so sibling branches cannot inherit a last-writer identity.
322
- Recall starts from FTS5 lexical matches whose rowids are the owning
323
- `context_nodes` rowids, expands one hop through file-reference edges, then
324
- applies session, branch, fact-kind, confidence, recency, and explicit-memory
325
- weights. Exact equivalent facts are deduplicated before bounded output.
326
- Resolved/superseded state is removed from the active FTS index; another
327
- project's rows are never eligible.
328
-
329
- **Important retention limits:** pending in-memory compaction 5 min · exploration
858
+ | Context graph | `infra/context-graph.ts` | SQLite FTS5 facts + file edges; 2,000 active derived nodes and 2,000 resolved/superseded tombstones per project |
859
+
860
+ Apply-confirmed state is queued, and duplicate updates coalesce only for the
861
+ exact project/session/branch head. Replacing an existing key refreshes that
862
+ pending value even at capacity; a divergent 65th key is rejected rather than
863
+ evicting accepted work. A microtask drains the batch through one reused SQLite
864
+ connection. Every caller awaits the transaction result, so persistence
865
+ telemetry completes only after indexing succeeds; permanent open/write failures
866
+ settle once as `context graph` failures and are never zero-delay retried. All
867
+ graph surfaces share one process-wide connection cached by database path, so
868
+ environments that relocate the cache directory reopen cleanly. The same
869
+ fail-closed transaction contract runs on `bun:sqlite` in Bun tests and
870
+ `node:sqlite` `DatabaseSync` in Pi's Node runtime; the packed release audit
871
+ exercises both.
872
+
873
+ Later cumulative state can supersede a failed derived update; user-confirmed
874
+ memory is not derived. Fact occurrences are branch-head scoped; state, recall
875
+ and resolution use the complete host-visible branch ancestry before equivalent
876
+ facts are deduplicated. Schema v1 preserves user-confirmed manual memory but
877
+ resets older derived compaction nodes once so sibling branches cannot inherit a
878
+ last-writer identity. Recall starts from FTS5 lexical matches whose rowids are
879
+ the owning `context_nodes` rowids, expands one hop through file-reference
880
+ edges, then weights session, branch, fact kind, confidence, recency and
881
+ explicit memory. Resolved or superseded state leaves the active FTS index;
882
+ another project's rows are never eligible. The forget command distinguishes a
883
+ derived-only reset from confirmed all-project graph deletion; neither changes
884
+ Mnemopi/Hindsight, compaction state or backups. Closing a local ref only marks
885
+ its node resolved.
886
+
887
+ **Retention limits:** pending in-memory compaction 5 min · exploration
330
888
  tool-support cache 1 h / 128 routes · token calibration 128 routes · extraction
331
889
  cache 1 h · compaction state 7 d / 64 snapshots · context graph 2,000
332
- non-structural fact nodes, 64 pending branch-head updates, and 500 active manual
890
+ active derived fact nodes, 2,000 tombstones (resolved facts and closed manual
891
+ memories), 64 pending branch-head updates and 500 active manual
333
892
  memories per project · remediation hints 7 d · metrics and damage JSONL logs
334
- 5 MiB each · one exploration tool result 12,000 characters.
893
+ 5 MiB each · one exploration tool result 12,000 characters. File locations are
894
+ listed in the guide's [storage and privacy](./docs/guide.md#storage-and-privacy)
895
+ section.
335
896
 
336
- ## Concurrency & safety model
897
+ ## Concurrency and safety model
337
898
 
338
- The extension is built to run safely alongside other Pi sessions and other
339
- extensions.
899
+ The extension runs alongside other Pi sessions and other extensions.
340
900
 
341
901
  ### Pending-compaction slot
342
902
 
343
- [`src/app/pending-slot.ts`](./src/app/pending-slot.ts) is an encapsulated,
903
+ [`src/app/pending-slot.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/pending-slot.ts) is an encapsulated,
344
904
  host-agnostic state cell (one producer, one consumer, single-threaded event
345
905
  loop). `consume()` returns a discriminated result:
346
906
 
@@ -351,49 +911,49 @@ loop). `consume()` returns a discriminated result:
351
911
  | `expired` | older than the 5-minute TTL |
352
912
  | `mismatch` | staged by a different session, project, or non-ancestor branch head |
353
913
 
354
- Session identity comes from [`infra/session-identity.ts`](./src/infra/session-identity.ts):
355
- a real id when the host exposes one, otherwise a per-call unforgeable
356
- `unresolved:<uuid>` — two unresolved sessions can never collide. Apply also
357
- requires the staged branch head to be the current head or one of its visible
358
- ancestors, so navigation to a sibling branch cannot consume stale payload.
914
+ Session identity comes from
915
+ [`infra/session-identity.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/session-identity.ts): a real ID when
916
+ the host exposes one, otherwise a per-call unforgeable `unresolved:<uuid>`, so
917
+ two unresolved sessions never collide. Apply also requires the staged branch
918
+ head to be the current head or one of its visible ancestors, so navigation to a
919
+ sibling branch cannot consume a stale payload. Payloads fingerprint projected
920
+ message IDs and content: append-only growth is allowed, same-ID context edits
921
+ invalidate.
359
922
 
360
923
  ### Cancellation deadlines
361
924
 
362
925
  Automatic compaction combines the host event's `AbortSignal` with its own
363
- deadline through a shared [`ExternalCancellation`](./src/app/run-smart-compact.ts)
364
- handle. Either source calls `abort()`, and every side-effect gate in the
365
- orchestrator checks the shared state before writing or applying compaction.
366
- The caller waits for safe pipeline unwind; no unsafe `Promise.race` hard return
367
- can leave work running past the hook lifecycle.
368
-
369
- ### Filesystem & concurrency
370
-
371
- JSON/text cache writes use [`src/infra/fs.ts`](./src/infra/fs.ts): private
372
- artifact directories are 0700 and files are 0600; atomic temp-file + rename
373
- prevents half-truncated readers but intentionally does not claim fsync/power-loss
374
- durability. Append/trim operations run asynchronously, yield before synchronous
375
- filesystem work, and hold a `mkdir`-based cross-process lock for the complete
376
- transaction. Lock ownership is reclaimed by atomic rename, never by deleting a
377
- possibly renewed lease in place. Sessions therefore cannot interleave bytes or
378
- steal a live successor's lock. SQLite supplies its own WAL durability. Native
379
- continuity handoffs are one-shot, bounded, and keyed by project + session +
380
- branch head.
381
-
382
- The session run lock uses file leases reclaimed when the owning PID dies.
383
- Reclaim has a deliberate, documented TOCTOU window: two processes reclaiming
384
- the same stale lease within milliseconds can, in one interleaving, unlink the
385
- other's freshly created lease. The double re-read (token + inode metadata)
386
- narrows but cannot atomically close this without an O_EXCL rename protocol;
387
- the lock is best-effort serialization of a normally single-writer flow, not a
388
- mutual-exclusion guarantee, and the surrounding pipeline remains fail-closed
389
- when the lock cannot be acquired.
926
+ deadline through a shared [`ExternalCancellation`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-smart-compact.ts)
927
+ handle. Either source calls `abort()`, and every side-effect gate checks the
928
+ shared state before writing or applying. The caller waits for a safe pipeline
929
+ unwind; no `Promise.race` hard return can leave work running past the hook
930
+ lifecycle.
931
+
932
+ ### Filesystem and locks
933
+
934
+ JSON/text cache writes use [`src/infra/fs.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/fs.ts): private
935
+ artifact directories are 0700 and files 0600; atomic temp-file + rename
936
+ prevents half-truncated readers but does not claim fsync/power-loss
937
+ durability. Append/trim operations run asynchronously, yield before
938
+ synchronous filesystem work, and hold a `mkdir`-based cross-process lock for
939
+ the whole transaction. Lock ownership is reclaimed by atomic rename, never by
940
+ deleting a possibly renewed lease in place. SQLite supplies its own WAL
941
+ durability.
942
+
943
+ The session run lock (`app/session-run-lock.ts`) uses file leases reclaimed
944
+ when the owning PID dies. Reclaim has a deliberate, documented TOCTOU window:
945
+ two processes reclaiming the same stale lease within milliseconds can, in one
946
+ interleaving, unlink the other's fresh lease. A double re-read (token + inode
947
+ metadata) narrows but cannot atomically close this without an O_EXCL rename
948
+ protocol. The lock is best-effort serialization of a normally single-writer
949
+ flow, not a mutual-exclusion guarantee, and the pipeline stays fail-closed when
950
+ the lock cannot be acquired.
390
951
 
391
952
  ## Provider awareness
392
953
 
393
- [`src/utils/tokens.ts`](./src/utils/tokens.ts) keeps a per-provider capability
954
+ [`src/utils/tokens.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/utils/tokens.ts) keeps a per-provider capability
394
955
  table (Anthropic, OpenAI, Google, DeepSeek, MiniMax, Xiaomi, Mistral, xAI, …)
395
- with unknowns falling back to a safe default + fuzzy alias matching. Each entry
396
- drives pipeline behavior:
956
+ with a safe default and fuzzy alias matching for unknowns:
397
957
 
398
958
  | Capability | Drives |
399
959
  | --- | --- |
@@ -403,61 +963,44 @@ drives pipeline behavior:
403
963
  | `cacheStrategy` | prompt-cache retention per call |
404
964
  | `timeoutMultiplier` | auto-trigger hard-timeout headroom |
405
965
  | `singlePassTokenMultiplier` | single-pass vs chunked threshold |
406
- | `tokenRatioEstimate` | token estimation; refined by per-(provider,model) **EMA calibration** |
407
-
408
- Every provider call is raced against one aborting hard deadline so a transport
409
- that ignores cancellation cannot keep the run lock indefinitely. Custom Codex
410
- endpoints also receive `max_output_tokens` through Pi AI's payload hook. The
411
- ChatGPT subscription endpoint rejects every wire output-cap field, so its
412
- deadline is derived from the requested output allowance (15–90s) and paired
413
- with a visible-output ceiling. Deadline failures route to the phase's
414
- deterministic fallback.
415
-
416
- ### Provider evaluation and routing evidence
417
-
418
- [`src/domain/provider-evaluation.ts`](./src/domain/provider-evaluation.ts)
419
- collapses call telemetry into Explore/Synthesize/Verify routes and compares
420
- provider/models across a deterministic context-pressure × tool-density matrix.
421
- Only an explicitly attributed pre-repair synthesis score contributes route
422
- quality; a run's final verifier score is never copied into Explore/Verify.
423
- Legacy or operational-only routes still contribute latency and reliability.
424
- Recommendations require minimum samples, ≥80% call reliability, and ≥50%
425
- stage-local quality coverage, shrink toward neutral under low confidence, and
426
- are advisory only.
427
- They never mutate config or replace the selected model. The opt-in live harness
428
- runs three identical bounded continuity scenarios across explicitly named
429
- models.
430
-
431
- ### Privacy-safe telemetry and canary decisions
432
-
433
- [`src/domain/telemetry.ts`](./src/domain/telemetry.ts) maps raw exceptions to a
434
- content-free failure taxonomy, aggregates schema-v2 run quality without IDs or
435
- conversation data, and compares an explicitly tagged `canary` cohort against
436
- `stable` history. Reports expose total/applied counts; only non-dry,
437
- host-confirmed applied outcomes count toward promotion. A deterministic green
438
- release check is not promotion evidence. The gate returns Hold, Rollback, or
439
- Promote from applied sample/quality coverage plus failure, verifier quality, p95
440
- latency, token, heuristic-fallback, and post-compaction-damage thresholds.
441
- Damage observations join their originating compaction by local run id, dedupe
442
- per run, and require ≥70% stable/canary coverage before promotion. It is
443
- advisory: rollout selection, configuration changes, and rollback remain external.
444
-
445
- Dashboard trust calculations live in
446
- [`src/ui/dashboard-insights.ts`](./src/ui/dashboard-insights.ts). Data
447
- Confidence is an auditable 100-point score over sample size, schema-v2
448
- coverage, verifier-quality coverage, required-field completeness, and
449
- freshness; ≥85 is the high-confidence target. TUI and HTML surfaces share the
450
- same quality-repair, stage/provider/model, failure-taxonomy, and
451
- stable-vs-canary aggregates. Missing/legacy evidence lowers the score and
452
- produces remediation guidance instead of being imputed.
966
+ | `tokenRatioEstimate` | token estimation; refined by per-(provider, model) EMA calibration |
967
+
968
+ Every provider call is raced against one aborting hard deadline, so a
969
+ transport that ignores cancellation cannot hold the run lock indefinitely.
970
+ Custom Codex endpoints receive `max_output_tokens` through Pi AI's payload
971
+ hook. The ChatGPT subscription endpoint rejects every wire output-cap field, so
972
+ its deadline is derived from the requested output allowance (15–90 s) and
973
+ paired with a visible-output ceiling. A per-call deadline may use the phase's
974
+ deterministic fallback while the run stays active. Run-wide timeout or host
975
+ cancellation propagates instead: it cannot schedule more synthesis or repair,
976
+ return a successful dry run, or publish a pending summary. The single run
977
+ outcome is `timeout` for the deadline or neutral `cancelled` for a host abort.
978
+ The run lock and pending slot are released. Native recovery is the requesting
979
+ host's decision, never an implicit fallback promised to manual callers.
980
+
981
+ ### Evaluation and telemetry
982
+
983
+ [`src/domain/provider-evaluation.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/domain/provider-evaluation.ts)
984
+ aggregates call telemetry into an advisory stage × context-pressure ×
985
+ tool-density matrix; it never mutates configuration.
986
+ [`src/domain/telemetry.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/domain/telemetry.ts) maps exceptions to a
987
+ content-free failure taxonomy, aggregates schema-v2 quality without IDs or
988
+ conversation data, and compares an explicit `canary` cohort with `stable`
989
+ history. `src/ui/dashboard-insights.ts` computes the dashboard's Data
990
+ Confidence heuristic. `scripts/task-eval.ts` and `task-eval-case.ts` pair the
991
+ same task across no-compaction, hygiene, EESV and hybrid stock-Pi sessions;
992
+ `scripts/replay-eval.ts` replays recorded sessions under alternative trim
993
+ policies and reports estimates only.
994
+ Commands, exact decision thresholds and evidence limits are documented once, in
995
+ [evaluation](./docs/evaluation.md).
453
996
 
454
997
  ## Dependency injection
455
998
 
456
- [`src/infra/services.ts`](./src/infra/services.ts) is a per-`runSmartCompact`
457
- service bag. Metrics, budgets, scrubbers, and prompt namespaces are isolated per
999
+ [`src/infra/services.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/services.ts) is a per-`runSmartCompact`
1000
+ service bag. Metrics, budgets, scrubbers and prompt namespaces are isolated per
458
1001
  run. Production shares only bounded provider/model capability and calibration
459
- knowledge, which contains no conversation/session data; tests use isolated
460
- stores by default:
1002
+ knowledge, which contains no conversation or session data; tests use isolated
1003
+ stores by default.
461
1004
 
462
1005
  | Service | Role |
463
1006
  | --- | --- |
@@ -466,58 +1009,91 @@ stores by default:
466
1009
  | `toolSupport` | process-shared in production; explicit unsupported capability, 1 h TTL / 128 routes |
467
1010
  | `metrics` | bounded metrics sink |
468
1011
  | `extractionCacheStats` | hit / miss counters |
469
- | `tokenCalibration` | process-shared bounded per-(provider,model) EMA factors |
1012
+ | `tokenCalibration` | process-shared bounded per-(provider, model) EMA factors |
470
1013
  | `compactSessionId` | per-run prompt-cache namespace |
471
1014
 
472
1015
  ## Layer responsibilities
473
1016
 
474
- The code is organized into six layers, each with a single responsibility.
475
-
476
1017
  ### Entry layer
477
1018
 
478
1019
  | File | Responsibility |
479
1020
  | --- | --- |
480
1021
  | `src/index.ts` | extension composition root and host lifecycle hooks |
1022
+ | `src/rtk.ts` | optional RTK companion entry point (not in `pi.extensions`) |
481
1023
  | `src/constants.ts` | version, thresholds, prompts, config keys |
482
1024
  | `src/types.ts` | shared types and discriminated unions |
483
- | `domain/provider-evaluation.ts` | advisory provider scenario matrix and route telemetry aggregation |
484
- | `domain/telemetry.ts` | privacy-safe aggregates, failure taxonomy, and canary rollback gates |
485
1025
 
486
1026
  ### Orchestration layer (`src/app/`)
487
1027
 
488
1028
  | File | Responsibility |
489
1029
  | --- | --- |
490
- | `app/run-smart-compact.ts` | top-level pipeline orchestrator |
491
- | `app/register-smart-compact-command.ts` | manual command adapter and restore/loop actions |
1030
+ | `app/run-smart-compact.ts` | top-level pipeline orchestrator and engine chain |
1031
+ | `app/register-smart-compact-command.ts` | manual command adapter: Home, preflight args, trim/storage/forget/restore/loops actions |
492
1032
  | `app/register-smart-compact-tool.ts` | bounded agent-tool adapter |
1033
+ | `app/smart-compact-input.ts` | command and tool argument parsing |
493
1034
  | `app/register-context-tools.ts` | project-scoped recall/save-memory adapters |
1035
+ | `app/register-smart-context-tool.ts` | session-control tool and native turn-boundary lifecycle |
1036
+ | `app/context-operations.ts` | pure checkpoint validation, pair-safe edit planning, branch-scoped archived-output access |
1037
+ | `app/tool-artifacts.ts` | safe early tool-output spill, private storage quotas/integrity, branch-owned references |
1038
+ | `app/context-evidence.ts` | common bounded listing/search/read for session output, visual excerpts and artifacts, active branch first then loaded lineage |
1039
+ | `app/session-lineage.ts` | read-only in-memory load of `parentSession` ancestors (depth, size and cycle bounds) |
1040
+ | `app/session-handoff.ts` | handoff seed from recorded state only; preview and `ctx.newSession` seeding |
1041
+ | `app/host-cache-ledger.ts` | session-local ledger of Pi's own requests: rebuild detection, cause attribution, cache lifetime |
1042
+ | `app/extension-conflicts.ts` | startup detection of known conflicting compaction/context-editing extensions; name-based, one notice |
1043
+ | `app/artifact-storage.ts` | read-only storage inventory and lineage classification |
1044
+ | `app/visual-archive.ts` | bounded evidence selection, persisted archive validation, request-local image rehydration |
1045
+ | `app/native-compaction.ts` | native engine: nested-request compaction, clean-turn cut, route/size checks, replay |
1046
+ | `app/native-continuity-bridge.ts` | one-shot continuity handoff keyed by project, session and branch head |
1047
+ | `app/memory-backend.ts` | exclusive memory-backend policy, read-only readiness, Mnemopi runtime evidence and Bun resolution |
1048
+ | `app/hindsight-memory.ts` | confirmed Hindsight save/resolve/recall flow and honest outcome reporting |
1049
+ | `app/mnemopi-memory.ts` / `mnemopi-worker.ts` / `mnemopi-protocol.ts` | Node-safe optional Bun engine, project-isolated confirmed memory and validated bounded IPC |
1050
+ | `app/effective-state.ts` | shared local-only readiness, effective policy and runtime-state view |
1051
+ | `app/model-feasibility.ts` | lazy local estimate of planned stage requests for model rows |
1052
+ | `app/lazy-tools.ts` | tool exposure: on-demand groups, eager and off modes, user `/tools` precedence, reachability for offload |
1053
+ | `app/register-navigation.ts` | anchors, recall, queued/revalidated pivots, footer status and the `smart_navigation` tool |
1054
+ | `app/navigation-data.ts` / `navigation-types.ts` | anchor and recall data over owned and legacy `context` anchors; read-only session scans |
1055
+ | `app/anchor-cache.ts` | Anthropic prompt-cache marker on the newest anchor |
1056
+ | `app/context-guide.ts` | on-demand read of the context-management guide |
494
1057
  | `app/model-routing.ts` | stage model resolution and explicit-model precedence |
1058
+ | `app/stage-auth.ts` | per-stage credential availability preflight; the session runtime resolves auth per request |
495
1059
  | `app/smart-compact-policy.ts` | branch-scoped agent visibility and auto-trigger policy; owns active-tool updates |
1060
+ | `app/global-settings-runtime.ts` | refresh each runtime owner once after an atomic global-settings patch |
1061
+ | `app/preflight.ts` | shared deterministic preparation for preview and real run |
496
1062
  | `app/run-context.ts` | typed stage chain (`RcBase → … → StatedRc`) |
497
1063
  | `app/mode-policy.ts` | Auto selector and finite Fast/Balanced/Thorough policies; legacy Aggressive maps to Fast |
498
1064
  | `app/pending-slot.ts` | encapsulated pending-compaction state cell |
1065
+ | `app/compaction-commit-store.ts` | holds summaries between `session_before_compact` and `session_compact` |
1066
+ | `app/session-run-lock.ts` | same-session serialization plus process-global file lease |
499
1067
  | `app/settled-auto-trigger.ts` | guarded proactive host compact requests; no EESV or pending-state ownership |
500
- | `app/steps/prepare.ts` | resolve config, provider caps, budgets, and cancellation; stage auth resolves lazily |
501
- | `app/steps/window.ts` | pick the prefix using provider-calibrated synthesis and deterministic post-processing bounds |
1068
+ | `app/background-preparation.ts` | early snapshots, invalidation, validated handoff, discard accounting and shutdown drain |
1069
+ | `app/steps/prepare.ts` | resolve config, provider caps, budgets and cancellation |
1070
+ | `app/steps/window.ts` | pick the prefix using calibrated synthesis and post-processing bounds |
502
1071
  | `app/steps/recover.ts` | recover full content for log-truncated messages |
503
- | `app/steps/tier.ts` | admission gate + context-pressure label (none / light / full); modes own execution depth |
1072
+ | `app/steps/tier.ts` | admission gate + context-pressure label; modes own execution depth |
504
1073
  | `app/steps/extract.ts` | pruning + deterministic extraction with incremental cache |
505
1074
  | `app/steps/synthesize.ts` | single-pass / EESV synthesis |
506
1075
  | `app/steps/verify.ts` | structural verification + repair with tool-result trust boundaries |
507
- | `app/steps/state.ts` | enrich summary with state, open loops, and recent resolved-error history |
1076
+ | `app/steps/state.ts` | enrich summary with state, open loops and resolved-error history |
1077
+ | `app/steps/visual.ts` | optional post-verification bitmap evidence inside the same yield target |
508
1078
  | `app/steps/persist.ts` | apply compaction, save fingerprint, persist state |
509
1079
  | `app/steps/metrics.ts` | record success / failure metrics |
510
1080
 
511
1081
  ### Domain layer (`src/domain/`)
512
1082
 
513
- Pure semantics — no I/O, no async, no globals.
1083
+ Pure semantics: no I/O, no async, no globals.
514
1084
 
515
1085
  | File | Responsibility |
516
1086
  | --- | --- |
517
1087
  | `domain/summary-schema.ts` | canonical section kinds + heading classification |
518
- | `domain/summary-parse.ts` | parse/render canonical H1/H2/H3 sections; merge duplicates; placement (`before`/`after`) |
519
- | `domain/tool-semantics.ts` | fine tool operation taxonomy with broad compatibility wrapper |
1088
+ | `domain/summary-parse.ts` | parse/render canonical H1/H2/H3 sections; merge duplicates; placement |
1089
+ | `domain/tool-semantics.ts` | fine tool operation taxonomy with broad compatibility wrapper; file-operation paths for superseded ordering |
1090
+ | `domain/compaction-usage.ts` | applied run's provider usage in Pi's `Usage` shape, priced per route |
520
1091
  | `domain/scrub.ts` | pure secret/PII redaction primitives and run-scoped scrubber |
1092
+ | `domain/keywords.ts` | salient-keyword extraction shared by verify and damage detection |
1093
+ | `domain/model-capacity.ts` | per-request output clamping and capacity reasons |
1094
+ | `domain/yield-gate.ts` | final yield proof and content-free `YieldGateError` |
1095
+ | `domain/provider-evaluation.ts` | advisory provider scenario matrix and route telemetry aggregation |
1096
+ | `domain/telemetry.ts` | privacy-safe aggregates, failure taxonomy and canary decision rules |
521
1097
 
522
1098
  ### Algorithm layer (`src/phases/`)
523
1099
 
@@ -533,14 +1109,22 @@ All external-world interaction.
533
1109
 
534
1110
  | File | Responsibility |
535
1111
  | --- | --- |
536
- | `infra/fs.ts` | atomic writes, advisory locks, and yielding async append/trim |
1112
+ | `infra/fs.ts` | atomic writes, advisory locks, yielding async append/trim |
537
1113
  | `infra/paths.ts` | canonical cache/session/backup paths |
538
1114
  | `infra/git.ts` | cached git-root discovery |
539
1115
  | `infra/clock.ts` | injectable wall clock |
540
- | `infra/llm-client.ts` | LLM seam, custom-Codex wire cap, and ChatGPT Codex stream watchdog |
1116
+ | `infra/llm-client.ts` | LLM seam over the session's public model runtime, custom-Codex wire cap, ChatGPT Codex stream watchdog |
541
1117
  | `infra/services.ts` | per-run services container |
542
- | `infra/session-identity.ts` | robust session-id resolution with opaque `unresolved:` fallback |
1118
+ | `infra/session-identity.ts` | session-ID resolution with opaque `unresolved:` fallback |
543
1119
  | `infra/ai-messages.ts` | validated message upcasts and recursive pre-serialization redaction |
1120
+ | `infra/context-graph.ts` | local SQLite FTS5 context graph |
1121
+ | `infra/synthesis-cache.ts` | behavior-keyed synthesis cache |
1122
+ | `infra/native-protocol.ts` | provider wire formats for native compaction and replay; no Pi imports |
1123
+ | `infra/hindsight-client.ts` | four fixed Hindsight routes; no generic request |
1124
+ | `infra/hindsight-receipts.ts` | origin/bank/project-scoped submission receipts; never evicts unconfirmed ones |
1125
+ | `infra/optional-components.ts` | read-only presence checks and exact install commands for optional peers (Mnemopi, Bun, resvg); no shell or network |
1126
+ | `infra/memory-ref.ts` | opaque backend/id refs and target-binding checks |
1127
+ | `infra/visual-renderer.ts` | lazy optional resvg renderer using `assets/DejaVuSansMono.ttf` |
544
1128
 
545
1129
  ### Utility layer (`src/utils/`)
546
1130
 
@@ -550,53 +1134,76 @@ All external-world interaction.
550
1134
  | `utils/pruning.ts` | redundancy removal on the message list |
551
1135
  | `utils/state.ts` | structured state, open loops, delta, pinned-path preservation |
552
1136
  | `utils/config.ts` | validated config loading and mtime-keyed cache |
553
- | `utils/helpers.ts` | batching, compaction boundaries, and extraction rendering helpers |
554
- | `utils/backups.ts` | backup persistence, listing, and restore message construction |
1137
+ | `utils/helpers.ts` | batching, compaction boundaries, extraction rendering helpers |
1138
+ | `utils/backups.ts` | backup persistence, listing and restore message construction |
555
1139
  | `utils/cache.ts` | metrics log + extraction prefix cache |
556
1140
  | `utils/fingerprint.ts` | project fingerprinting (language, framework, deps) |
557
1141
  | `utils/damage.ts` | post-compaction regression signals + remediation hints |
558
- | `utils/id-fingerprint.ts` | compact SHA-256 fingerprint of entry-id arrays |
1142
+ | `utils/id-fingerprint.ts` | compact SHA-256 fingerprint of entry-ID arrays |
559
1143
  | `utils/file-needles.ts` | path-suffix needles for error→file attribution |
560
1144
  | `utils/file-ref-detect.ts` | fabricated file-reference detection (SemVer-rejecting) |
561
1145
  | `utils/session-log.ts` | streaming JSONL parser for the Pi session log |
562
- | `utils/tokens.ts` | per-(provider,model) token estimation with EMA calibration |
1146
+ | `utils/tokens.ts` | per-(provider, model) token estimation with EMA calibration |
563
1147
  | `utils/type-guards.ts` | runtime validators for cross-version compatibility |
564
- | `utils/logger.ts` | stderr-prefixed log shim |
565
- | `utils/lru.ts` | small bounded LRU cache primitive |
1148
+ | `utils/logger.ts` | debug-only trace shim |
1149
+ | `utils/issues.ts` | user-facing problem reporting: once-per-session dedupe, scrubbed one-line messages, recent-issue history |
1150
+ | `utils/lru.ts` | small bounded LRU primitive |
566
1151
 
567
1152
  ### UI layer (`src/ui/`)
568
1153
 
569
1154
  | File | Responsibility |
570
1155
  | --- | --- |
571
- | `ui/overlays.ts` | progressive preflight, semantic phase progress, and approval review |
572
- | `ui/metrics-dashboard-overlay.ts` | interactive metrics dashboard screen |
573
- | `ui/backup-overlays.ts` | backup picker, viewer, and restore action screen |
1156
+ | `ui/home-overlay.ts` | keyboard Home: five task rows and readiness panel |
1157
+ | `ui/profiles.ts` | presets derived from exact persisted flags; model feasibility snapshot type |
1158
+ | `ui/overlays.ts` | progressive preflight, phase progress and approval review |
1159
+ | `ui/storage-report.ts` | read-only storage inventory rendering; no deletion verbs |
1160
+ | `ui/navigation-overlay.ts` | human session navigation: browse anchors, mark a point, search earlier sessions, confirmed return |
1161
+ | `ui/metrics-dashboard-overlay.ts` | interactive metrics dashboard |
1162
+ | `ui/backup-overlays.ts` | backup picker, viewer and restore action |
574
1163
  | `ui/open-loops-overlay.ts` | persisted open-loop manager |
575
- | `ui/settings-overlay.ts` | session/branch policy editor for agent access and automatic compaction |
1164
+ | `ui/handoff-overlay.ts` | Home handoff panel: note, seed preview, open |
1165
+ | `ui/settings-overlay.ts` | settings TUI: task-grouped categories, named values, dependency rules, branch overrides |
1166
+ | `ui/settings-complex.ts` | input, model and profile-budget rows with inline validation |
1167
+ | `ui/settings-list.ts` | settings list with per-row `r` reset and dimmed inactive rows |
1168
+ | `ui/error-format.ts` | one-line failure text with the scrubbed first provider error line |
576
1169
  | `ui/dashboard-format.ts` | shared pure formatters for metrics surfaces |
577
- | `ui/dashboard-insights.ts` | Data Confidence, quality/provider drilldowns, and canary trust views |
1170
+ | `ui/dashboard-insights.ts` | Data Confidence, quality/provider drilldowns, canary trust views |
578
1171
  | `ui/metrics-report.ts` | text report + local HTML metrics dashboard |
579
1172
 
580
- ## Design principles
1173
+ ## Host dependency boundary
1174
+
1175
+ Pi core modules are host-supplied peers requiring 0.87.1+; `typebox` is a
1176
+ wildcard peer. Neither is bundled. Development dependencies pin Pi 0.87.1 so
1177
+ native context projection is checked against the minimum supported API.
1178
+ `bun run compat:pi [version]` validates another release in an isolated
1179
+ workspace. The optional resvg renderer and the Mnemopi engine stay external to
1180
+ the bundles.
581
1181
 
582
- The architecture intentionally biases toward safety:
1182
+ ## Design principles
583
1183
 
584
- - deterministic extraction before any synthesis
585
- - adaptive exploration instead of always-on tool use
586
- - verified file lists and error context
587
- - deterministic repair before additional LLM calls
588
- - hallucinated file-reference detection
589
- - stateful tracking of open loops and cross-compaction deltas
590
- - tool-driven compaction never compacts mid-turn
591
- - summaries preserve exact file paths and identifiers where possible; saturated file lists use budgeted path tails plus collision-checked digests while scoped state retains full paths
592
- - the recent tail stays live outside the compacted region
1184
+ - Prefer hygiene and recoverable references over summarization.
1185
+ - Deterministic extraction before any synthesis; deterministic repair before
1186
+ additional LLM calls.
1187
+ - Adaptive exploration instead of always-on tool use.
1188
+ - Verified file lists and error context; hallucinated file-reference
1189
+ detection.
1190
+ - Stateful open loops and cross-compaction deltas.
1191
+ - Tool-driven compaction never compacts mid-turn; the host owns apply.
1192
+ - Summaries keep exact paths and identifiers where possible; saturated file
1193
+ lists use budgeted path tails plus collision-checked digests while scoped
1194
+ state keeps full paths.
1195
+ - The recent tail stays live outside the compacted region.
1196
+ - Memory is confined to one selected backend; explicit saves need per-fact
1197
+ host confirmation, and only the local backend indexes derived facts, from
1198
+ apply-confirmed compactions.
593
1199
 
594
1200
  ## Extending the system
595
1201
 
596
- When adding features, prefer this order:
1202
+ Prefer this order:
597
1203
 
598
- 1. extract more deterministic signal if possible
599
- 2. enrich exploration only when needed
600
- 3. keep synthesis prompts structured and bounded
601
- 4. strengthen verification before increasing model dependence
602
- 5. update tests and docs in the same change
1204
+ 1. avoid the noise or keep it recoverable before summarizing it;
1205
+ 2. extract more deterministic signal;
1206
+ 3. enrich exploration only when needed;
1207
+ 4. keep synthesis prompts structured and bounded;
1208
+ 5. strengthen verification before increasing model dependence;
1209
+ 6. update tests and docs in the same change.