pi-smart-compact 9.7.1 → 10.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/ARCHITECTURE.md +973 -372
  2. package/CHANGELOG.md +721 -0
  3. package/LICENSE +8 -0
  4. package/README.md +128 -640
  5. package/SECURITY.md +34 -12
  6. package/SUPPORT.md +26 -9
  7. package/assets/DejaVu-LICENSE.txt +187 -0
  8. package/assets/DejaVuSansMono.ttf +0 -0
  9. package/assets/README.md +26 -0
  10. package/assets/skills/context-management/SKILL.md +34 -0
  11. package/dist/app/anchor-cache.d.ts +36 -0
  12. package/dist/app/anchor-cache.d.ts.map +1 -0
  13. package/dist/app/artifact-storage.d.ts +47 -0
  14. package/dist/app/artifact-storage.d.ts.map +1 -0
  15. package/dist/app/background-preparation.d.ts +39 -0
  16. package/dist/app/background-preparation.d.ts.map +1 -0
  17. package/dist/app/compaction-commit-store.d.ts +5 -1
  18. package/dist/app/compaction-commit-store.d.ts.map +1 -1
  19. package/dist/app/context-evidence.d.ts +57 -0
  20. package/dist/app/context-evidence.d.ts.map +1 -0
  21. package/dist/app/context-guide.d.ts +3 -0
  22. package/dist/app/context-guide.d.ts.map +1 -0
  23. package/dist/app/context-operations.d.ts +106 -0
  24. package/dist/app/context-operations.d.ts.map +1 -0
  25. package/dist/app/effective-state.d.ts +23 -0
  26. package/dist/app/effective-state.d.ts.map +1 -0
  27. package/dist/app/global-settings-runtime.d.ts +3 -3
  28. package/dist/app/global-settings-runtime.d.ts.map +1 -1
  29. package/dist/app/hindsight-memory.d.ts +100 -0
  30. package/dist/app/hindsight-memory.d.ts.map +1 -0
  31. package/dist/app/host-cache-ledger.d.ts +68 -0
  32. package/dist/app/host-cache-ledger.d.ts.map +1 -0
  33. package/dist/app/lazy-tools.d.ts +36 -0
  34. package/dist/app/lazy-tools.d.ts.map +1 -0
  35. package/dist/app/memory-backend.d.ts +58 -0
  36. package/dist/app/memory-backend.d.ts.map +1 -0
  37. package/dist/app/mnemopi-memory.d.ts +13 -0
  38. package/dist/app/mnemopi-memory.d.ts.map +1 -0
  39. package/dist/app/mnemopi-protocol.d.ts +78 -0
  40. package/dist/app/mnemopi-protocol.d.ts.map +1 -0
  41. package/dist/app/mnemopi-worker.d.ts +2 -0
  42. package/dist/app/mnemopi-worker.d.ts.map +1 -0
  43. package/dist/app/model-feasibility.d.ts +20 -0
  44. package/dist/app/model-feasibility.d.ts.map +1 -0
  45. package/dist/app/native-compaction.d.ts +88 -0
  46. package/dist/app/native-compaction.d.ts.map +1 -0
  47. package/dist/app/native-continuity-bridge.d.ts.map +1 -1
  48. package/dist/app/navigation-data.d.ts +28 -0
  49. package/dist/app/navigation-data.d.ts.map +1 -0
  50. package/dist/app/navigation-types.d.ts +60 -0
  51. package/dist/app/navigation-types.d.ts.map +1 -0
  52. package/dist/app/pending-slot.d.ts +11 -1
  53. package/dist/app/pending-slot.d.ts.map +1 -1
  54. package/dist/app/preflight.d.ts.map +1 -1
  55. package/dist/app/register-context-tools.d.ts +16 -3
  56. package/dist/app/register-context-tools.d.ts.map +1 -1
  57. package/dist/app/register-navigation.d.ts +20 -0
  58. package/dist/app/register-navigation.d.ts.map +1 -0
  59. package/dist/app/register-smart-compact-command.d.ts +17 -2
  60. package/dist/app/register-smart-compact-command.d.ts.map +1 -1
  61. package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
  62. package/dist/app/register-smart-context-tool.d.ts +55 -0
  63. package/dist/app/register-smart-context-tool.d.ts.map +1 -0
  64. package/dist/app/run-context.d.ts +1 -0
  65. package/dist/app/run-context.d.ts.map +1 -1
  66. package/dist/app/run-smart-compact.d.ts +3 -3
  67. package/dist/app/run-smart-compact.d.ts.map +1 -1
  68. package/dist/app/session-handoff.d.ts +64 -0
  69. package/dist/app/session-handoff.d.ts.map +1 -0
  70. package/dist/app/session-lineage.d.ts +17 -0
  71. package/dist/app/session-lineage.d.ts.map +1 -0
  72. package/dist/app/session-run-lock.d.ts +0 -2
  73. package/dist/app/session-run-lock.d.ts.map +1 -1
  74. package/dist/app/settled-auto-trigger.d.ts +2 -0
  75. package/dist/app/settled-auto-trigger.d.ts.map +1 -1
  76. package/dist/app/smart-compact-input.d.ts +1 -1
  77. package/dist/app/smart-compact-input.d.ts.map +1 -1
  78. package/dist/app/smart-compact-policy.d.ts +1 -1
  79. package/dist/app/smart-compact-policy.d.ts.map +1 -1
  80. package/dist/app/steps/extract.d.ts +45 -1
  81. package/dist/app/steps/extract.d.ts.map +1 -1
  82. package/dist/app/steps/metrics.d.ts +1 -0
  83. package/dist/app/steps/metrics.d.ts.map +1 -1
  84. package/dist/app/steps/persist.d.ts.map +1 -1
  85. package/dist/app/steps/prepare.d.ts.map +1 -1
  86. package/dist/app/steps/recover.d.ts +9 -0
  87. package/dist/app/steps/recover.d.ts.map +1 -1
  88. package/dist/app/steps/synthesize.d.ts.map +1 -1
  89. package/dist/app/steps/tier.d.ts.map +1 -1
  90. package/dist/app/steps/verify.d.ts.map +1 -1
  91. package/dist/app/steps/visual.d.ts +4 -0
  92. package/dist/app/steps/visual.d.ts.map +1 -0
  93. package/dist/app/steps/window.d.ts.map +1 -1
  94. package/dist/app/tool-artifacts.d.ts +27 -0
  95. package/dist/app/tool-artifacts.d.ts.map +1 -0
  96. package/dist/app/visual-archive.d.ts +29 -0
  97. package/dist/app/visual-archive.d.ts.map +1 -0
  98. package/dist/constants.d.ts +96 -1
  99. package/dist/constants.d.ts.map +1 -1
  100. package/dist/domain/compaction-usage.d.ts +16 -0
  101. package/dist/domain/compaction-usage.d.ts.map +1 -0
  102. package/dist/domain/model-capacity.d.ts +12 -0
  103. package/dist/domain/model-capacity.d.ts.map +1 -0
  104. package/dist/domain/provider-evaluation.d.ts +7 -0
  105. package/dist/domain/provider-evaluation.d.ts.map +1 -1
  106. package/dist/domain/telemetry.d.ts +43 -2
  107. package/dist/domain/telemetry.d.ts.map +1 -1
  108. package/dist/domain/tool-semantics.d.ts +23 -0
  109. package/dist/domain/tool-semantics.d.ts.map +1 -1
  110. package/dist/index.d.ts.map +1 -1
  111. package/dist/index.js +15757 -6942
  112. package/dist/infra/ai-messages.d.ts +1 -1
  113. package/dist/infra/ai-messages.d.ts.map +1 -1
  114. package/dist/infra/context-graph.d.ts +38 -7
  115. package/dist/infra/context-graph.d.ts.map +1 -1
  116. package/dist/infra/fs.d.ts.map +1 -1
  117. package/dist/infra/hindsight-client.d.ts +73 -0
  118. package/dist/infra/hindsight-client.d.ts.map +1 -0
  119. package/dist/infra/hindsight-receipts.d.ts +68 -0
  120. package/dist/infra/hindsight-receipts.d.ts.map +1 -0
  121. package/dist/infra/llm-client.d.ts +26 -23
  122. package/dist/infra/llm-client.d.ts.map +1 -1
  123. package/dist/infra/memory-ref.d.ts +27 -0
  124. package/dist/infra/memory-ref.d.ts.map +1 -0
  125. package/dist/infra/native-protocol.d.ts +54 -0
  126. package/dist/infra/native-protocol.d.ts.map +1 -0
  127. package/dist/infra/optional-components.d.ts +15 -0
  128. package/dist/infra/optional-components.d.ts.map +1 -0
  129. package/dist/infra/paths.d.ts +2 -0
  130. package/dist/infra/paths.d.ts.map +1 -1
  131. package/dist/infra/services.d.ts +15 -5
  132. package/dist/infra/services.d.ts.map +1 -1
  133. package/dist/infra/visual-renderer.d.ts +16 -0
  134. package/dist/infra/visual-renderer.d.ts.map +1 -0
  135. package/dist/mnemopi-worker.js +213 -0
  136. package/dist/phases/explore.d.ts +12 -9
  137. package/dist/phases/explore.d.ts.map +1 -1
  138. package/dist/phases/synthesize.d.ts +18 -3
  139. package/dist/phases/synthesize.d.ts.map +1 -1
  140. package/dist/phases/verify.d.ts +5 -1
  141. package/dist/phases/verify.d.ts.map +1 -1
  142. package/dist/rtk.d.ts +7 -0
  143. package/dist/rtk.d.ts.map +1 -0
  144. package/dist/rtk.js +767 -0
  145. package/dist/types.d.ts +128 -4
  146. package/dist/types.d.ts.map +1 -1
  147. package/dist/ui/dashboard-format.d.ts +2 -1
  148. package/dist/ui/dashboard-format.d.ts.map +1 -1
  149. package/dist/ui/dashboard-insights.d.ts +9 -1
  150. package/dist/ui/dashboard-insights.d.ts.map +1 -1
  151. package/dist/ui/error-format.d.ts +7 -2
  152. package/dist/ui/error-format.d.ts.map +1 -1
  153. package/dist/ui/handoff-overlay.d.ts +26 -0
  154. package/dist/ui/handoff-overlay.d.ts.map +1 -0
  155. package/dist/ui/home-overlay.d.ts +54 -0
  156. package/dist/ui/home-overlay.d.ts.map +1 -0
  157. package/dist/ui/metrics-dashboard-overlay.d.ts.map +1 -1
  158. package/dist/ui/metrics-report.d.ts.map +1 -1
  159. package/dist/ui/navigation-overlay.d.ts +92 -0
  160. package/dist/ui/navigation-overlay.d.ts.map +1 -0
  161. package/dist/ui/overlays.d.ts +12 -2
  162. package/dist/ui/overlays.d.ts.map +1 -1
  163. package/dist/ui/profiles.d.ts +51 -0
  164. package/dist/ui/profiles.d.ts.map +1 -0
  165. package/dist/ui/settings-complex.d.ts +49 -3
  166. package/dist/ui/settings-complex.d.ts.map +1 -1
  167. package/dist/ui/settings-list.d.ts +28 -0
  168. package/dist/ui/settings-list.d.ts.map +1 -0
  169. package/dist/ui/settings-overlay.d.ts +13 -6
  170. package/dist/ui/settings-overlay.d.ts.map +1 -1
  171. package/dist/ui/storage-report.d.ts +4 -0
  172. package/dist/ui/storage-report.d.ts.map +1 -0
  173. package/dist/utils/backups.d.ts.map +1 -1
  174. package/dist/utils/cache.d.ts +6 -2
  175. package/dist/utils/cache.d.ts.map +1 -1
  176. package/dist/utils/config.d.ts +12 -0
  177. package/dist/utils/config.d.ts.map +1 -1
  178. package/dist/utils/helpers.d.ts.map +1 -1
  179. package/dist/utils/id-fingerprint.d.ts +3 -1
  180. package/dist/utils/id-fingerprint.d.ts.map +1 -1
  181. package/dist/utils/issues.d.ts +61 -0
  182. package/dist/utils/issues.d.ts.map +1 -0
  183. package/dist/utils/pruning.d.ts.map +1 -1
  184. package/dist/utils/session-log.d.ts +0 -2
  185. package/dist/utils/session-log.d.ts.map +1 -1
  186. package/dist/utils/state.d.ts +3 -1
  187. package/dist/utils/state.d.ts.map +1 -1
  188. package/dist/utils/tokens.d.ts +10 -2
  189. package/dist/utils/tokens.d.ts.map +1 -1
  190. package/docs/MIGRATING_TO_V8.md +7 -1
  191. package/docs/README.md +69 -0
  192. package/docs/RELEASE.md +173 -56
  193. package/docs/assets/banner.png +0 -0
  194. package/docs/assets/banner.svg +1158 -70
  195. package/docs/assets/pi-smart-compact.png +0 -0
  196. package/docs/assets/pi-smart-compact.svg +24 -0
  197. package/docs/configuration.md +637 -0
  198. package/docs/evaluation.md +408 -0
  199. package/docs/guide.md +860 -0
  200. package/docs/hindsight-memory.md +314 -0
  201. package/docs/identity.md +124 -0
  202. package/package.json +44 -11
  203. package/dist/provider-eval.js +0 -2122
  204. package/dist/provider-scenario-eval.js +0 -2900
  205. package/dist/telemetry-report.js +0 -1973
  206. package/docs/provider-evaluation-2026-08-06.md +0 -63
package/ARCHITECTURE.md CHANGED
@@ -1,54 +1,425 @@
1
1
  # Architecture
2
2
 
3
- System-level design of `pi-smart-compact`. This is the maintainer-facing
4
- companion to the user-facing [`README.md`](./README.md).
3
+ Maintainer-facing system design for **Pi Continuity**, published as the npm
4
+ package `pi-smart-compact`. The product name is documentation branding only:
5
+ the package name, `/smart-compact` command, `smart_*` tools, `smartCompact`
6
+ configuration key, runtime and UI names, stored paths and ref prefixes are
7
+ unchanged.
8
+
9
+ Usage lives in the [user guide](./docs/guide.md), every setting in
10
+ [configuration](./docs/configuration.md), and evidence rules in
11
+ [evaluation](./docs/evaluation.md). This page explains how the parts fit and
12
+ which invariants they must keep. Contributor workflow and the repository map
13
+ are in [`CONTRIBUTING.md`](https://github.com/alpertarhan/pi-smart-compact/blob/main/CONTRIBUTING.md).
14
+
15
+ **Contents:** [product layers](#product-layers) ·
16
+ [integration surfaces](#integration-surfaces) ·
17
+ [1. context hygiene](#1-context-hygiene) ·
18
+ [2. recoverable continuity](#2-recoverable-continuity) ·
19
+ [3. verified compaction](#3-verified-compaction) ·
20
+ [4. cross-session memory](#4-optional-cross-session-memory) ·
21
+ [state and persistence](#state-caching-and-persistence) ·
22
+ [concurrency and safety](#concurrency-and-safety-model) ·
23
+ [provider awareness](#provider-awareness) ·
24
+ [layer responsibilities](#layer-responsibilities) ·
25
+ [host dependency boundary](#host-dependency-boundary) ·
26
+ [extending the system](#extending-the-system)
27
+
28
+ ## Product layers
29
+
30
+ > **Job:** keep the agent's working set useful and quiet, and keep the session
31
+ > continuous across research, cleanup, compaction, reload and, optionally,
32
+ > later sessions. Compaction and memory are mechanisms, not the product
33
+ > boundary.
34
+
35
+ Prefer cheaper, recoverable operations before lossy compaction when they fit
36
+ the task. This is a design preference, not an enforced execution chain: the
37
+ features can be selected independently, and memory does not require compaction.
5
38
 
6
- > **Job:** not to produce a generic recap, but to preserve the agent's working
7
- > state so the next turn can continue with minimal loss.
39
+ ```mermaid
40
+ flowchart LR
41
+ H[1. Context hygiene] --> R[2. Recoverable continuity]
42
+ R --> C[3. Verified compaction]
43
+ C -. optional .-> M[4. Cross-session memory]
44
+ ```
8
45
 
9
- ## Design ideas
46
+ | Layer | Question it answers | Main modules | Core invariant |
47
+ | --- | --- | --- | --- |
48
+ | [1. Context hygiene](#1-context-hygiene) | How do we keep noise out of the working set? | `rtk.ts`, `app/tool-artifacts.ts`, `app/context-operations.ts` | Preserve protected content and retain retrieval paths for eligible offloaded output |
49
+ | [2. Recoverable continuity](#2-recoverable-continuity) | How does removed or old evidence stay reachable on the same branch? | `app/register-smart-context-tool.ts`, `app/context-evidence.ts`, `app/artifact-storage.ts` | Access follows active-branch provenance; the host session log stays authoritative |
50
+ | [3. Verified compaction](#3-verified-compaction) | How do we replace history when pressure demands it? | `app/run-smart-compact.ts`, `app/steps/*`, `phases/*` | Facts first, synthesis second, verification before apply; rejected custom summaries are not staged or applied |
51
+ | [4. Cross-session memory](#4-optional-cross-session-memory) | What should a later session know? | `app/memory-backend.ts`, `infra/context-graph.ts`, Hindsight and Mnemopi modules | One selected backend; explicit fact saves require confirmation, while enabled local indexing follows host-confirmed compaction |
10
52
 
11
- The design combines three ideas:
53
+ Quality means retained constraints, trustworthy failure evidence and low
54
+ retrieval churn, not merely fewer tokens. There are no recurring
55
+ model-visible status prompts, no automatic error deletion, and no destructive
56
+ file rollback. The [2026-09-24 context hygiene report](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/context-hygiene-2026-09-24.md)
57
+ (repository only, historical) records the original experiments and acceptance
58
+ criteria.
12
59
 
13
- - **Agentic compaction** — let the system inspect the session, not just summarize it.
14
- - **Kamradt-style chunking** — segment large conversations into coherent units before synthesis.
15
- - **EESV** — **Extract → Explore → Synthesize → Verify**: facts first, synthesis second, verification last.
60
+ ## Design ideas
61
+
62
+ - **Agentic compaction.** The system may inspect the session through bounded
63
+ tools instead of summarizing a flat transcript.
64
+ - **Coverage across the whole conversation.** A design intuition borrowed from
65
+ Greg Kamradt's public work on semantic chunking and long-context retrieval:
66
+ important facts can sit anywhere in a long history, so an early constraint
67
+ or a mid-session decision deserves the same chance to survive as the latest
68
+ error. In this codebase that intuition shows up as deterministic extraction
69
+ over the entire compacted prefix, topic-aware segmentation before
70
+ hierarchical synthesis, and verification against the extracted facts
71
+ regardless of where they occurred. It is not a formal sampling algorithm or
72
+ benchmark result, and it is not a claim about how Pi's own compactor
73
+ selects content.
74
+ - **EESV:** Extract, Explore, Synthesize, Verify. Facts first, synthesis
75
+ second, verification last.
76
+
77
+ Verification measures coverage of deterministic facts, structure and grounded
78
+ claims. It is a strong regression guard, not proof of semantic truth, and a
79
+ compacted summary is not a lossless copy of the history it replaces.
16
80
 
17
81
  ## Integration surfaces
18
82
 
19
- Registered in [`src/index.ts`](./src/index.ts). See the README for usage; this
20
- section is about lifecycle.
83
+ Registered in [`src/index.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/index.ts).
21
84
 
22
85
  | Surface | Lifecycle |
23
86
  | --- | --- |
24
- | `/smart-compact` | Manual command. Explainable target-first preflight or direct args. Bypasses the adaptive pressure gate, not yield/verification gates. |
25
- | `session_before_compact` | Auto hook. Returns/stages a pending summary or runs under pressure; durable commit waits for matching `session_compact`. |
26
- | `session_compact_failed` | Pi 0.85 failure hook. Clears extension-owned staged state and records one error/cancellation outcome; older hosts simply never emit it. |
27
- | `agent_settled` | Opt-in pressure monitor. Requests `ctx.compact()` only; it never runs EESV or consumes/stages pending state. |
87
+ | `/smart-compact` | Manual command. A bare TUI invocation opens the keyboard Home (`ui/home-overlay.ts`); `Compact now` opens the target-first preflight. Direct arguments bypass Home. `trim` and `storage` expose the shared cleanup controller and the read-only storage inventory; `context` opens the session-navigation panel (`ui/navigation-overlay.ts`). Bypasses the adaptive pressure gate, not yield or verification gates. |
88
+ | `session_before_compact` | Auto hook. Returns or stages a pending summary, or runs under pressure; the durable commit waits for the matching `session_compact`. |
89
+ | `session_compact_failed` | Clears extension-owned staged state and records one error or cancellation outcome. |
90
+ | `turn_end` | Commits queued local context edits first; background mode tries bounded trimming before non-blocking speculative preparation. |
91
+ | `agent_settled` | Checks background preparation and requests native `ctx.compact()` at the idle pressure gate; does not commit itself. |
92
+ | `session_shutdown` | Cancels speculative work and awaits late preparation plus discard-metric writes; never applies an unconfirmed candidate. |
93
+ | `context` | Optionally rehydrates validated bitmap evidence beside the matching text summary; does not mutate session history. |
94
+ | `tool_result` | Opt-in early spill of large safe read-only text: verify persistence, then replace content with a preview and reference; errors leave the original result intact. |
95
+ | `before_provider_request` | Replays provider-native compaction state on a matching route; no I/O until the session has native state. |
96
+ | `before_provider_request` (anchor cache) | `app/anchor-cache.ts`: on Anthropic Messages routes, moves one `cache_control` marker to the newest anchor on the branch so the prefix before it stays cacheable while later turns change. |
97
+ | `session_before_tree` / `session_tree` | Own pivots supply the branch summary (carryover) for the exact prepared target; a foreign navigation cancels a queued pivot. Boundaries re-decide lazy tool exposure and refresh the anchor footer. |
98
+ | `smart_tools` tool | Agent-callable loader (`app/lazy-tools.ts`): `load`/`unload` one tool group (`navigation`, `history`, `memory`, `compaction`), `status`, or `guide` (reads `assets/skills/context-management/SKILL.md` on demand; never injected). |
99
+ | `smart_navigation` tool | `view`/`recall` anchors, `anchor` this point, or `pivot` to an anchor with required carryover; a pivot terminates the turn and is applied by the host only after the batch settles and revalidation passes. |
28
100
  | `smart_compact` tool | Agent-callable. Prepares a pending summary; never compacts mid-turn. |
29
- | `smart_recall` tool | Searches only the current project's bounded context graph; same session/branch ranks first. |
30
- | `smart_save_memory` tool | Persists one explicit user-confirmed project fact after secret/PII scrubbing. |
31
-
32
- A short-lived pending compaction is staged in the [`PendingSlot`](#pending-compaction-slot)
33
- and handed to Pi when compaction is applied.
101
+ | `smart_context` tool | Session-local checkpoint/rewind, safe trimming, and bounded original-output retrieval via native boundary drafts. |
102
+ | `smart_recall` tool | Searches the selected memory backend only. `scope: "session"` is local-graph-only; remote backends skip it and read nothing else. |
103
+ | `smart_save_memory` tool | Saves a host-confirmed scrubbed fact through the selected backend, or resolves the exact store named by a target-bound ref. |
104
+
105
+ The table lists the owning surfaces. Other host events support them:
106
+ session switch/fork/tree and `model_select` cancel speculative
107
+ preparation and queued edits; `session_start` initializes replay, cache,
108
+ tool exposure, policy and navigation state;
109
+ `before_agent_start` injects the one-shot native continuity bridge;
110
+ `message_end` feeds damage monitoring and the host prompt-cache ledger;
111
+ `context_with_system` and `cache_warming_decision` serve held automatic trims
112
+ (see [recoverable trimming](#recoverable-trimming)); the RTK companion uses
113
+ `tool_call`.
114
+
115
+ Tool exposure is owned by `app/lazy-tools.ts`. With `toolLoading: "lazy"`
116
+ (default) only the `smart_tools` loader is active; a group becomes active when
117
+ the agent loads it, and loaded groups are forgotten at `session_start`,
118
+ `session_tree` and `session_compact`. `eager` activates every permitted group;
119
+ `off` removes all owned tools while the human UI keeps working. Group
120
+ permissions (`agentToolAccess`, memory backend, `contextNavigationEnabled`)
121
+ apply in every mode. The exposure removes only its own tools, restores only
122
+ what it removed itself, and treats a `/tools` change by the user as final:
123
+ hidden tools are never re-shown by a loader request. Artifact offload runs
124
+ only while `smart_context` is reachable (active or loadable), so an archived
125
+ output can always be read back by the agent that lost it.
34
126
 
35
- With `autoTriggerStrategy: "settled"`, the idle hook applies finite
36
- token/percentage, queue, in-flight, and per-session cooldown guards, then asks
37
- Pi for a normal host compaction. Pi re-enters `session_before_compact`, which
38
- reuses an already tool-staged summary or runs EESV exactly once under the
39
- host's signal and timeout. The matching `session_compact` event remains the
40
- only durable commit authority; `session_compact_failed` discards the staged
41
- candidate without committing it. This keeps proactive triggering out of the
42
- pending/commit state machine and preserves branch-provenance checks.
43
-
44
- ### Host dependency boundary
45
-
46
- Pi core modules and `typebox` are wildcard peers supplied by the running host;
47
- they are neither bundled nor duplicated as versioned development dependencies.
48
- The lockfile pins a reproducible local baseline, and `bun run compat:pi [version]`
49
- validates another Pi release in an isolated temporary workspace.
50
-
51
- ## Pipeline at a glance
127
+ `app/smart-compact-policy.ts` keeps agent-tool visibility separate from
128
+ automatic compaction. The tool remains registered for immediate re-enable, but
129
+ Pi's active-tool set controls whether its schema and guidance reach the agent.
130
+ Agent access is tri-state: `inherit` leaves host `/tools` and allowlists in
131
+ control, while explicit `enabled`/`disabled` mutate only `smart_compact` and
132
+ then report the effective host state. Policy snapshots are custom branch
133
+ entries restored on `session_start` and `session_tree`; the manual command is
134
+ never gated.
135
+
136
+ ### Home, presets and readiness
137
+
138
+ `ui/home-overlay.ts` shows five rows: **Compact now**, **Clean up tool
139
+ output**, **Settings**, **History & recovery**, and **Status & help**. Context
140
+ and the effective automatic/agent policy stay visible above the list.
141
+ Compact-picker cancellation returns to Home; no menu-open path changes
142
+ settings.
143
+
144
+ `ui/profiles.ts` derives presets from exact persisted flags. Behavior presets
145
+ are **Manual only**, **Manual + agent**, **Cleanup only** and **Fully
146
+ automatic**; the built-in defaults, which match none of them, are labeled
147
+ **With Pi (default)**. Summary formats are **Verified text**, **Text +
148
+ images** and **Provider (experimental)**; provider output needs a second
149
+ confirming Enter, and capacity-ineligible models cannot be selected. Fully
150
+ automatic selects the `settled` strategy. Presets atomically patch existing
151
+ keys; existing settings are never migrated, and branch overrides stay separate.
152
+
153
+ `app/effective-state.ts` gives Home (`Status & help → Readiness & details`),
154
+ preflight, metrics and dashboard one local-evidence view. It resolves routes,
155
+ reports credential presence without calling `getApiKey`, checks local backend
156
+ prerequisites and reads pure runtime state. It never refreshes OAuth, probes a
157
+ provider or server, creates a store or acquires a lease. Pi's effective
158
+ auto-compaction setting is not available through the public extension API, so
159
+ for `native-hook` that prerequisite is reported as unknown. Memory readiness is
160
+ informational: an unready optional backend warns but never blocks compaction.
161
+ `app/model-feasibility.ts` may disable model rows from a local estimate of the
162
+ planned stage requests; it never refreshes credentials.
163
+
164
+ Preflight and result overlays size their own viewports from terminal rows,
165
+ because Pi 0.87.1 renders overlays with `render(width)` and no height. Actions
166
+ stay outside scrolling content. Result approval is explicit (`A` only);
167
+ technical details are opt-in, while verification and fallback warnings stay
168
+ visible.
169
+
170
+ ## 1. Context hygiene
171
+
172
+ Hygiene keeps the working set small before summarization is needed. It never
173
+ calls a model.
174
+
175
+ ### Command hygiene (optional RTK companion)
176
+
177
+ `src/rtk.ts` is a separate entry point, absent from `pi.extensions`. It
178
+ delegates rewrite rules to the external RTK binary and never executes the
179
+ original command itself. Eligibility is only bare `git status`, `cargo test`
180
+ and `bun test`, tested against RTK 0.50: `bun test` joined after a paired
181
+ native/filtered runner check showed exit-code and failure/load-error parity
182
+ plus full recall of the filtered output, with one execution; `git diff`, `tsc`
183
+ and vitest 5 measured lossy or growing, and `npm test`/`node --test` have no
184
+ rule. Arguments, unknown syntax/flags and compound commands pass through.
185
+ Missing binary, unknown version, cancellation, session invalidation and CLI
186
+ failure all retain the input. The host bash tool keeps execution and result
187
+ ownership. No retries or output rewrite hooks are added. Permission hooks must
188
+ run after this input mutation. RTK's recall store and retention are not Smart
189
+ Compact artifacts and are not covered by its scrubbing.
190
+
191
+ ### Early tool-output artifacts
192
+
193
+ `artifactOffloadEnabled` defaults to false. `app/tool-artifacts.ts` handles
194
+ `tool_result` before the next model request, using the shared conservative
195
+ read-only allowlist. Errors, images, commands, mutations, unknown tools,
196
+ instruction/skill reads, file/symbol read deliveries (`read`, `read_symbol`,
197
+ `read_enclosing`) and the extension's own recovery tool stay inline. File reads
198
+ are excluded until read guards can honor actual delivered coverage rather than
199
+ the original call's implied full range. Earlier hooks' content is the capture
200
+ boundary; no arbitrary full-output path is read. Native details and status are
201
+ preserved.
202
+
203
+ After scrubbing, text of 16k+ characters (maximum 2 MiB) is content-addressed
204
+ in a session-origin directory under `smart-compact-artifacts/`, separate from
205
+ disposable caches. Async atomic-write and cross-process lock helpers protect
206
+ writes and quota checks. The final file is read back with bounded I/O, size and
207
+ SHA-256 verification before a preview is returned. Directory symlinks and file
208
+ symlinks/hardlinks are rejected. Storage, cancellation and quota failures are
209
+ fail-open for the original result and are never advertised as recoverable
210
+ output.
211
+
212
+ A small `details.smartCompactArtifact` record carries scope/content/preview
213
+ hashes, size, tool and source description. Only active-branch provenance
214
+ authorizes lookup; a content/preview mismatch or unowned context edit revokes
215
+ recovery. Forks may retain their parent reference; they do not duplicate
216
+ files. Files are not expired or evicted while refs may exist: 256 files/32 MiB
217
+ per origin is a stop-spilling quota, not LRU. Global usage across sessions is
218
+ not capped. Explicit cleanup must account for dependent forks; missing evidence
219
+ is an error, not a live-file reread. Interrupted writes can leave unreferenced
220
+ files charged against that origin's cap. Branch authorization is indexed by
221
+ entry ID, not content hash, so byte deduplication cannot overwrite source
222
+ provenance. The catalog collapses only repeated (tool, source label, payload
223
+ hash) rows; distinct labels share the hash read alias and keep direct entry-ID
224
+ retrieval. Foreign edits revoke the affected occurrence, not independent
225
+ references. No new on-disk format or migration is involved.
226
+
227
+ ### Recoverable trimming
228
+
229
+ `app/context-operations.ts` plans edits against native
230
+ `buildSessionProjection()`. `app/register-smart-context-tool.ts` queues one
231
+ mutation and returns drafts from `turn_end`, only after a successful
232
+ originating tool result, on the same branch, with no competing drafts or
233
+ pending user input. The native host owns persistence; there are no mid-tool
234
+ session mutations, tree-navigation hacks, or summarizer calls.
235
+
236
+ `requestManualTrim` on that controller backs both `/smart-compact trim` and the
237
+ Home **Clean up tool output** row. Queueing performs no model call and forces
238
+ no turn, so the next provider request is still sent untrimmed and the queued
239
+ edit applies at the next natural completed-turn boundary. A pending return to
240
+ an anchor or newer boundary change cancels it with a visible message.
241
+
242
+ Trimming protects four recent assistant turns and the active checkpoint prefix,
243
+ replacing up to 32 old read-only results of at least 4096 characters with a
244
+ bounded reference marker. Already-edited entries are not rewritten. Automatic
245
+ trimming runs with `contextHygieneEnabled` independently of compaction, or with
246
+ the effective `background` strategy, and requires `smart_context` reachable
247
+ by the model (`canAutoTrim` checks it like artifact offload does).
248
+ `plan` gives an on-demand non-mutating preview. Automatic batches require at
249
+ least 16,384 net saved characters and eight assistant turns since the last
250
+ owned trim/rewind/compaction; branch history supplies that cooldown across
251
+ reloads and forks. At a completed, uncontested turn boundary a ready batch
252
+ commits with cause `pressure` (early pressure gate reached) or `break-even`
253
+ (catalog prices say it pays back within `AUTO_TRIM_BREAK_EVEN_REQUESTS` = 24
254
+ requests). Otherwise it is held (`deferredTrim` in `smart_context` status):
255
+ once the prompt cache has expired, `context_with_system` sends the trimmed
256
+ results request-locally, byte-identical to the future `context_edit`, and the
257
+ edits commit with cause `cold` at the next completed turn. While a batch is
258
+ held, `cache_warming_decision` may stop Pi's cache warming when a refresh no
259
+ longer pays. Unknown prices allow only `pressure` and `cold`. The formula and
260
+ drop conditions are in [configuration](./docs/configuration.md);
261
+ `app/host-cache-ledger.ts` attributes observed prefix rebuilds. These are
262
+ catalog-price estimates, not measured cache billing. Manual and agent trims
263
+ commit at the next boundary with cause `manual` or `agent`; explicit trim
264
+ bypasses batching, not safety.
265
+
266
+ Instruction and skill sources stay inline through trim, rewind, artifact,
267
+ bitmap and pre-compaction pruning; recovery tool output is not recursively
268
+ re-archived. Pre-compaction dedup requires identical text content plus
269
+ arguments and no intervening changed observation or unsafe operation.
270
+ Unproven status-looking user text is never a deletion signal. Hygiene yields to
271
+ existing EESV work and staged candidates. Its handler precedes background
272
+ observation, and projected edits in the accumulated drafts prevent speculative
273
+ work from capturing a stale pre-edit branch. Explicit edits cancel background
274
+ work and invalidate the shared pending slot. Context fingerprints and
275
+ compaction guards prevent removed evidence from being resurrected.
276
+
277
+ ## 2. Recoverable continuity
278
+
279
+ Continuity means the session can find what it needs again on the same branch,
280
+ across reloads and forks, without copying the transcript to a second store.
281
+
282
+ ### Checkpoint, rewind and archived output
283
+
284
+ Small `custom` records (`smart-compact-context`, version 1) hold one active
285
+ checkpoint and lists of archived output IDs. Status is rebuilt from active
286
+ ancestry on demand, including after reload; these records do not enter model
287
+ context. A checkpoint captures session/origin IDs and a projected-prefix
288
+ fingerprint. New user messages, an intervening compaction or branch summary, or
289
+ a changed prefix invalidate it. The agent-written handoff is a bounded
290
+ `custom_message`, not an authoritative user instruction or an EESV
291
+ verification result.
292
+
293
+ A conservative read-only tool allowlist plus shared nested-call normalization
294
+ protects side effects. Rewind removes only whole successful, complete,
295
+ text-only read exchanges and successful assistant prose after the checkpoint.
296
+ Mixed batches, errors, commands, writes, unknown tools and image results stay
297
+ raw. Context edits omit messages in the projection, not in session history. The
298
+ mutation cap is 512; no partial rewind is applied when it is exceeded. Files
299
+ and processes are untouched.
300
+
301
+ Reference retrieval is restricted to this extension's branch-local archive
302
+ records and original textual tool results; foreign context edits revoke access
303
+ until an owned archive record explicitly re-authorizes it. Assistant reasoning
304
+ and arbitrary entries are never exposed. Full text is scrubbed before a bounded
305
+ page is sliced, avoiding cross-page credential reconstruction.
306
+
307
+ ### Evidence search
308
+
309
+ `app/context-evidence.ts` unifies native-history refs, bounded visual excerpts
310
+ and artifact files behind `smart_context`. Source descriptions and bounded
311
+ literal search make old evidence discoverable without knowing its ID.
312
+ Native-history labels use the shared `extractToolPath` alias rules. Search
313
+ checks at most 32 sources/4 Mi characters and returns at most ten
314
+ first-per-source matches plus a continuation cursor; no source bodies are
315
+ injected just to list them. Read supports character or line selection and a
316
+ 4096-character response cap. All representations are re-scrubbed in full
317
+ before searching or paging. Context, recovery and compaction see only the
318
+ stored preview; full files are retrieved only on explicit agent calls.
319
+
320
+ ### Storage inventory
321
+
322
+ `app/artifact-storage.ts` backs `/smart-compact storage` with a strictly
323
+ read-only inventory of the spill store. It streams every `*.jsonl` under the
324
+ native sessions root with bounded memory and classifies each origin directory
325
+ against two lineage anchors that only Pi maintains: session headers
326
+ (`sha256(id)` = owner scope) and `details.smartCompactArtifact.owner`
327
+ references that forks copy. Outcomes are `live`, `unreferenced-in-scan`, or
328
+ `unknown` when the scan is incomplete. `unreferenced-in-scan` is deliberately
329
+ not safe-to-delete: sessions outside the scanned root are undiscoverable, and a
330
+ running session can add references after the scan. That is why no `--clean` or
331
+ artifact GC exists; `ui/storage-report.ts` renders totals, status, bytes and
332
+ retention without deletion verbs. Durability tests age real `SessionManager`
333
+ artifacts past 20 days by timestamp (deterministic, not a wall-clock soak),
334
+ reload and fork through the public consumer, fail closed on missing or tampered
335
+ bytes, and hold the per-origin caps without losing earlier evidence.
336
+
337
+ ### Session navigation and apply-time validation
338
+
339
+ `app/register-navigation.ts` owns anchors, cross-session recall, pivots, the
340
+ anchor footer and the Anthropic anchor cache marker (`app/anchor-cache.ts`).
341
+ `app/navigation-data.ts` reads both owned anchors (`custom_message` entries of
342
+ type `smart-context-anchor`, plus `smart_navigation` tool results) and legacy
343
+ `context` tool anchors; recall scans other sessions' JSONL read-only.
344
+
345
+ A human anchor is a `sendMessage` custom message followed by a native label on
346
+ that entry; a human pivot navigates to the label so Pi's `custom_message`
347
+ handling cannot drop the anchor text. An agent pivot is queued by the tool,
348
+ revalidated at `turn_end` (uncontested successful batch, same session, same
349
+ leaf, permission unchanged), dispatched at `agent_settled` through a nonce-bound
350
+ `/smart-compact` apply command, and supplied to `session_before_tree` as the
351
+ branch summary. New input, a session switch, a foreign tree navigation or a
352
+ permission change cancels it with a visible notice. While a pivot is queued or
353
+ applying, `session_before_compact` returns `{ cancel: true }`, automatic
354
+ preparation stops, and both new and queued `smart_context` mutations are
355
+ refused; evidence reads keep working. Files, processes and Git state are never
356
+ rolled back.
357
+
358
+ Reader identity and limits are captured before preparation starts.
359
+ `revalidatePending` checks that snapshot, the active projected prefix, current
360
+ usage/growth, native reserve, response headroom, target and yield for
361
+ foreground and background candidates. New compact instructions invalidate old
362
+ staged work. Navigation and model events clear speculative, pending and
363
+ staged-commit candidates; a generation guard rejects late native-hook work if
364
+ invalidation happened during its provider call. Pi remains the apply owner.
365
+
366
+ ### Continuity state across compactions
367
+
368
+ After a confirmed compaction, `utils/state.ts` carries structured state
369
+ forward: open loops, a continuity ledger in which prior facts persist until
370
+ positive resolution evidence or an explicit override, non-destructive goal
371
+ breadcrumbs, and a "Changes Since Last Compaction" delta. Snapshots are scoped
372
+ to project, session and branch head; see
373
+ [state, caching and persistence](#state-caching-and-persistence).
374
+
375
+ ## 3. Verified compaction
376
+
377
+ ### Automatic strategies
378
+
379
+ `autoTrigger` gates every automatic path; with it off, neither `settled` nor
380
+ `background` runs and the native hook does not replace Pi's summaries. Pi's own
381
+ compactor is not affected. Hygiene can still run.
382
+
383
+ - **`native-hook`** (default) is passive. It participates only when Pi starts
384
+ a compaction; its percentage setting is a minimum replacement gate, not a
385
+ scheduler, and Pi auto-compaction must itself be enabled for host-driven
386
+ runs.
387
+ - **`settled`** applies finite token/percentage, queue, in-flight and
388
+ per-session cooldown guards at idle `agent_settled`, then asks Pi for a
389
+ normal host compaction. Pi re-enters `session_before_compact`, which reuses an
390
+ already tool-staged summary or runs EESV exactly once under the host's signal
391
+ and timeout. `session_compact` stays the only durable commit authority;
392
+ `session_compact_failed` discards the candidate. This keeps proactive
393
+ triggering out of the pending/commit state machine and preserves
394
+ branch-provenance checks.
395
+ - **`background`** (`app/background-preparation.ts`) snapshots branch, model,
396
+ session and usage before asynchronous work. `prepareContextPercent` sets an
397
+ independent early gate with the existing minimum token floor; null preserves
398
+ `applyTokens - clamp(floor(applyTokens * 0.125), 8192, 32000)`. An explicit
399
+ prepare percentage must be below the effective `minContextPercent` apply gate.
400
+ Only pipeline admission is lowered; targets, yield and apply-time safety are
401
+ unchanged. A private pending slot isolates cancellation from foreground
402
+ staging. One task, a ten-minute attempt cooldown and a five-minute ready TTL
403
+ bound speculation. Model/config changes, branch navigation, new compaction or
404
+ context edits, switch/shutdown and native compaction invalidate unfinished
405
+ work, and late completion cannot publish after cancellation. Before reuse, a
406
+ content fingerprint proves the captured prefix is unchanged; appended tail
407
+ growth must still satisfy the verified target, minimum yield and response
408
+ reserve.
409
+
410
+ Preparation never waits inside `turn_end`. Proactive application waits for
411
+ idle `agent_settled`; an ongoing tool loop relies on Pi's native maintenance
412
+ boundary. No boundary draft bypasses the `session_compact` commit protocol. The
413
+ native hook falls back normally if speculative work is unavailable. Completed
414
+ discarded preparation emits one cost-only record with reason and ready/wait
415
+ timing, never an applied outcome. Shutdown cancels and drains tracked work and
416
+ writes so late completion cannot lose its accounting on graceful exit.
417
+
418
+ A short-lived pending compaction is staged in the
419
+ [`PendingSlot`](#pending-compaction-slot) and handed to Pi when compaction is
420
+ applied.
421
+
422
+ ### Pipeline at a glance
52
423
 
53
424
  ```mermaid
54
425
  flowchart LR
@@ -66,7 +437,7 @@ flowchart LR
66
437
  Y -- Yes --> J[Pending compaction returned to Pi]
67
438
  ```
68
439
 
69
- The orchestrator ([`src/app/run-smart-compact.ts`](./src/app/run-smart-compact.ts))
440
+ The orchestrator ([`src/app/run-smart-compact.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-smart-compact.ts))
70
441
  threads a typed context through ten stages:
71
442
 
72
443
  | # | Stage | Module | Transition |
@@ -82,12 +453,15 @@ threads a typed context through ten stages:
82
453
  | 9 | persist | `app/steps/persist.ts` | stage pending, apply compaction |
83
454
  | 10 | metrics | `app/steps/metrics.ts` | success / failure record |
84
455
 
85
- ## The typed stage machine
456
+ `app/steps/visual.ts` may run after verification when the experimental
457
+ [visual evidence](#experimental-visual-evidence) path is enabled.
458
+
459
+ ### The typed stage machine
86
460
 
87
- [`src/app/run-context.ts`](./src/app/run-context.ts) models the pipeline context
88
- as a **state machine of branded intersection types**. Each step accepts the
461
+ [`src/app/run-context.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-context.ts) models the pipeline
462
+ context as a state machine of branded intersection types. Each step accepts the
89
463
  previous stage type and returns the next, so reordering or skipping a step is a
90
- **compile-time error**, not a runtime crash:
464
+ compile-time error:
91
465
 
92
466
  ```text
93
467
  RcBase
@@ -101,246 +475,427 @@ RcBase
101
475
  → StatedRc (after state)
102
476
  ```
103
477
 
104
- Each stage adds a `_prepared` / `_windowed` / … discriminator field that
105
- carries the type-level proof and is checked by `advance()` at runtime.
106
- Mutation is preserved: a step mutates its input object and casts it to the
107
- next stage (no per-step copy of ~30 fields). The final alias
108
- `RunContext = StatedRc` keeps post-`buildState` consumers readable.
109
-
110
- This is what lets `applyCompaction` read `rc.details` with zero `!` non-null
111
- assertions: the type system proves `buildState` has run.
112
-
113
- ## Core execution model
114
-
115
- ### Entry and context gate
116
-
117
- `src/index.ts` owns host lifecycle wiring. Cohesive adapters in
118
- `src/app/register-smart-compact-command.ts`, `register-smart-compact-tool.ts`,
119
- and `model-routing.ts` validate input, resolve models, and route work into
120
- `runSmartCompact()`. Before any expensive work, the system checks context size
121
- against the threshold in `src/constants.ts`. Auto / tool runs are skipped while
122
- context is small; manual `/smart-compact` uses an absolute adaptive safety tail
123
- rather than a percentage of large model windows. Its decision-card preflight
124
- is built from the same config snapshot, calibrated estimator, adaptive profile,
125
- active branch, and pure window planner as execution. It compares exactly Fast,
126
- Balanced, and Thorough; `M` changes the summary route and replans all three,
127
- while `D` reveals technical estimator/boundary details. A plan must meet the
128
- tail target and at least 10% projected net savings before any model call. A
129
- pending summary for the same session is reused instead of invoking the pipeline
130
- again. `auto` is not a fourth policy: it selects one of the three from context
131
- pressure and deterministic extraction risk.
132
-
133
- `app/smart-compact-policy.ts` keeps agent-tool visibility separate from
134
- automatic compaction. The tool remains registered for immediate re-enable, but
135
- Pi's active-tool set controls whether its schema and prompt guidance reach the
136
- agent. Agent access is tri-state: `inherit` leaves host `/tools` and allowlists
137
- in control, while explicit `enabled`/`disabled` choices mutate only
138
- `smart_compact` and then report the effective host state. Full policy snapshots
139
- are custom branch entries restored on `session_start` and `session_tree`; the
140
- manual command is never gated.
141
-
142
- Model routes are stage-specific but never inferred from mode. With no explicit
143
- configuration, Explore, Synthesize, and Verify all use the selected Pi model.
144
- `segmentationModel`, `summaryModel`, and `verificationModel` can independently
145
- override those routes. Credentials resolve lazily immediately before each
146
- stage's first network call and are reused for equivalent routes; call metrics
147
- preserve the actual route.
478
+ Each stage adds a `_prepared` / `_windowed` / … discriminator that carries the
479
+ type-level proof and is checked by `advance()` at runtime. A step mutates its
480
+ input and casts it to the next stage (no per-step copy of ~30 fields). The
481
+ final alias `RunContext = StatedRc` lets `applyCompaction` read `rc.details`
482
+ with no non-null assertions: the type system proves `buildState` has run.
483
+
484
+ ### Entry, modes and routing
485
+
486
+ `src/index.ts` owns host lifecycle wiring. `app/register-smart-compact-command.ts`,
487
+ `register-smart-compact-tool.ts`, `smart-compact-input.ts` and `model-routing.ts`
488
+ validate input, resolve models and route work into `runSmartCompact()`. Before
489
+ expensive work, context size is checked against the thresholds in
490
+ `src/constants.ts`. Auto and tool runs are skipped while context is small;
491
+ manual `/smart-compact` uses an absolute adaptive safety tail rather than a
492
+ percentage of large model windows.
493
+
494
+ `app/preflight.ts` builds the decision card from the same config snapshot,
495
+ calibrated estimator, adaptive profile, active branch and pure window planner
496
+ as execution. It compares exactly Fast, Balanced and Thorough; `M` changes the
497
+ summary route and replans all three, and `D` reveals estimator/boundary
498
+ details. A plan must meet the tail target and at least 10% projected net
499
+ savings before any model call. A pending summary for the same session is reused
500
+ instead of running the pipeline again. `auto` is a selector, not a fourth
501
+ policy (`app/mode-policy.ts`): it chooses one of the three from context pressure
502
+ and deterministic extraction risk. Legacy `aggressive` maps to Fast.
503
+
504
+ Model routes are stage-specific and never inferred from mode. With no explicit
505
+ configuration, Explore, Synthesize and Verify use the selected Pi model;
506
+ `segmentationModel`, `summaryModel` and `verificationModel` override them
507
+ independently. `app/stage-auth.ts` checks credential availability just before
508
+ each stage's first network call and reuses the answer for equivalent routes;
509
+ the session runtime resolves the actual auth per request, and call metrics
510
+ keep the actual route. How routing evidence is gathered is described in
511
+ [evaluation](./docs/evaluation.md#provider-routing-evidence).
148
512
 
149
513
  ### Keep window and preprocessing
150
514
 
151
- `app/steps/window.ts` starts from Pi's compaction-aware
152
- `buildContextEntries()` view, never the append-only session history, and builds
153
- a content-free `CompactionWindowPlan` from the selected mode budget. The shared
154
- `contextMessageEntries()` adapter uses Pi's `sessionEntryToContextMessages()` and
155
- `convertToLlm()`, retaining host-visible custom/branch/compaction summaries and
156
- original entry IDs while excluding private or context-disabled entries:
157
-
158
- - **hard `toolCall` / `toolResult` guard** — never orphan a result from its call
159
- - **soft recent-user/checkpoint/topical preferences** — retain raw only when the resulting suffix still fits the planned budget
160
- - **yield contract** — projected replacement must meet its target and save at least 10% after reserving the summary budget
161
-
162
- The same pure planner powers manual preflight and execution. A soft boundary is
163
- recorded as relaxed rather than silently overriding the target. Long turns may
164
- be summarized through their older prefix; if the nominal cut lands inside a
165
- tool exchange, the planner either retains the complete pair within budget or
166
- advances past it so the complete exchange is summarized. It also advances past
167
- complete historical exchanges whose tool names violate the portable provider
168
- contract; this keeps model switches from exposing an unsendable raw tail. If no
169
- provider-safe hard boundary can meet the target, automatic/tool runs normally
170
- return control to Pi's native compactor before any LLM call. An
171
- already-overflowed context is
172
- the safety exception: measured usage is mapped across active messages and EESV
173
- keeps chunked recovery instead of sending an oversized one-shot prompt to
174
- native summarization. Manual runs use the profile's absolute adaptive tail, so
175
- model-window size cannot turn an explicit command into a full-context no-op.
515
+ `app/steps/window.ts` reads the selected ancestry via `getBranch()` and passes
516
+ it to native `buildSessionProjection()`, never converting raw entries
517
+ individually. `contextMessageEntries()` converts that projection with
518
+ `convertToLlm()`, keeping source IDs and host-visible custom, branch and
519
+ compaction summaries while honoring `context_edit` replacements and omissions.
520
+ Intentional edits are marked against raw-log recovery. The active view drives a
521
+ content-free `CompactionWindowPlan` from the selected mode budget:
522
+
523
+ - **hard `toolCall` / `toolResult` guard**: never orphan a result from its call;
524
+ - **soft recent-user/checkpoint/topical preferences**: keep raw only when the
525
+ suffix still fits the planned budget;
526
+ - **yield contract**: the projected replacement must meet its target and save
527
+ at least 10% after reserving the summary budget.
528
+
529
+ A relaxed soft boundary is recorded rather than silently overriding the
530
+ target. Long turns may be summarized through their older prefix; a cut inside a
531
+ tool exchange either keeps the complete pair within budget or advances past it.
532
+ The planner also advances past complete historical exchanges whose tool names
533
+ violate the portable provider contract, so model switches cannot expose an
534
+ unsendable raw tail. If no provider-safe hard boundary meets the target,
535
+ automatic and tool runs normally return control to Pi's native compactor before
536
+ any LLM call. An already-overflowed context is the exception: measured usage is
537
+ mapped across active messages and EESV keeps chunked recovery instead of
538
+ sending an oversized one-shot prompt. Manual runs use the profile's absolute
539
+ adaptive tail, so model-window size cannot turn an explicit command into a
540
+ full-context no-op.
176
541
 
177
542
  Before summarization the pipeline keeps a deferred reference to the recovered
178
- pre-prune messages, prunes redundant messages, loads prior continuity, checks the
179
- extraction cache, and loads the project fingerprint. Both synthesis and backup
543
+ pre-prune messages, prunes redundant messages, loads prior continuity, checks
544
+ the extraction cache and loads the project fingerprint. Synthesis and backup
180
545
  text use `serializeConversationText()` without the host summarizer's implicit
181
- 2,000-character tool-result cap. Structural and text redaction remain enforced;
182
- binary attachment archival is outside this text format. The backup materializes
183
- and writes atomically only after the matching native compaction is confirmed.
546
+ 2,000-character tool-result cap. Structural and text redaction stay enforced;
547
+ binary attachment archival is outside this format. The backup is materialized
548
+ and written atomically only after the matching native compaction is confirmed.
184
549
 
185
550
  ### Extract
186
551
 
187
- Primary: [`src/utils/extraction.ts`](./src/utils/extraction.ts). **Zero LLM calls.**
188
-
189
- Deterministically pulls: modified / read / deleted files, tool and bash-like
190
- errors, retry / resolution signals, explicit & implicit decisions, constraints
191
- and preferences, heuristic topic segments, timeline events, the main goal, and
192
- open loops. **This is the ground truth** that synthesis and verification trust.
552
+ [`src/utils/extraction.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/utils/extraction.ts). Zero LLM calls.
553
+ Deterministically pulls modified/read/deleted files, tool and bash-like errors,
554
+ retry/resolution signals, explicit and implicit decisions, constraints and
555
+ preferences, heuristic topic segments, timeline events, the main goal and open
556
+ loops across the whole compacted prefix. This is the ground truth that
557
+ synthesis and verification trust.
193
558
 
194
559
  ### Explore
195
560
 
196
- Primary: [`src/phases/explore.ts`](./src/phases/explore.ts). Optional — runs only
197
- in `thorough` mode or when `auto` selects that policy from deterministic risk. The model inspects
198
- the conversation through a small toolset: message ranges, conversation search,
199
- recent user messages, local context around an index, file-change lookups, and
200
- error chains. Tool support is runtime-probed once and cached per run; if a
201
- provider has no function calling, the system falls back to a direct structured
202
- analysis path. The growing tool conversation is capped at three rounds, each
203
- response is capped where the provider supports output limits, and the shared
204
- prefix uses short-lived prompt caching.
561
+ [`src/phases/explore.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/explore.ts). Runs only in `thorough`
562
+ mode or when `auto` selects it from deterministic risk. The model inspects the
563
+ conversation through a small toolset: message ranges, conversation search,
564
+ recent user messages, local context around an index, file-change lookups and
565
+ error chains. Tool support is probed once and cached per run; without function
566
+ calling the system falls back to a direct structured analysis. The tool
567
+ conversation is capped at three rounds, each response is capped where the
568
+ provider supports output limits, and the shared prefix uses short-lived prompt
569
+ caching.
205
570
 
206
571
  ### Synthesize
207
572
 
208
- Primary: [`src/phases/synthesize.ts`](./src/phases/synthesize.ts). Three paths:
573
+ [`src/phases/synthesize.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/synthesize.ts). Three paths:
209
574
 
210
- - **Deterministic zero-call** — high-confidence Fast extractions.
211
- - **Single-pass** — when the compacted conversation fits under the configured threshold.
212
- - **Hierarchical** — for larger sessions: merge available boundaries → split oversized semantic chunks → batch by token budget → summarize batches → assemble.
575
+ - **Deterministic zero-call** for high-confidence Fast extractions.
576
+ - **Single-pass** when the compacted conversation fits under the configured
577
+ threshold.
578
+ - **Hierarchical** for larger sessions: merge available boundaries, split
579
+ oversized semantic chunks, batch by token budget, summarize batches,
580
+ assemble. Every batch is summarized, so coverage does not depend on a
581
+ fragment's position in the history.
213
582
 
214
- Behaviors: session-aware prompting, decision propagation across later batches,
215
- mode-specific single-pass thresholds and output limits, provider-aware
216
- concurrency (wave scheduling), aggregate prompt-token reservation, and a
217
- deterministic fallback assembly when any budget or LLM call fails.
583
+ Session-aware prompting, decision propagation across later batches,
584
+ mode-specific thresholds and output limits, provider-aware wave concurrency,
585
+ aggregate prompt-token reservation and deterministic fallback assembly when any
586
+ budget or LLM call fails.
218
587
 
219
588
  ### Verify
220
589
 
221
- Primary: [`src/phases/verify.ts`](./src/phases/verify.ts). Scores the summary
222
- against deterministic extraction, continuity, explicit focus/note steering,
223
- and source messages. It checks missing modified/read/deleted files, unresolved
224
- errors, high-confidence constraints, weak goal coverage, missing structure,
225
- suspicious fabricated paths, done/unresolved inconsistencies, explicit
226
- decisions, open loops, and unsupported high-risk outcome claims. Claims such as
227
- “tests passed” require matching source prose or a successful related tool
228
- result.
229
-
230
- **Repair order is intentional:** (1) deterministic patch first (free,
231
- idempotent) → (2) one LLM patch only in `thorough` mode if still insufficient
232
- → (3) replace lower-scoring output with a deterministic quality floor built
233
- only from extraction, continuity, and steering → (4) reject unless final
234
- verification has no gaps and meets the verified threshold. Untrusted chunk
235
- prose is never an input to the quality floor.
236
- Final verification runs again after continuity injection. The final scalar is
237
- reported as repaired **verification coverage**, alongside the pre-repair source
238
- score and fallback provenance; it is not labeled as raw synthesis quality.
239
- Verification failures retain only exhaustive content-free gap kinds and the
240
- rejecting gate (`post-synthesis` or `post-state`) in local telemetry. Summary
241
- evidence is never copied into failure metrics. Both gates remain mandatory:
242
- summary-derived continuity fields cannot become evidence for their own initial
243
- verification.
590
+ [`src/phases/verify.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/phases/verify.ts) scores the summary against
591
+ deterministic extraction, continuity, explicit focus/note steering and source
592
+ messages. It checks missing modified/read/deleted files, unresolved errors,
593
+ high-confidence constraints, weak goal coverage, missing structure, suspicious
594
+ fabricated paths, done/unresolved inconsistencies, explicit decisions, open
595
+ loops and unsupported high-risk outcome claims. Claims such as "tests passed"
596
+ need matching source prose or a successful related tool result.
597
+
598
+ Repair order is intentional: (1) deterministic patch first (free, idempotent);
599
+ (2) one LLM patch only in `thorough` mode if still insufficient; (3) replace
600
+ lower-scoring output with a deterministic quality floor built only from
601
+ extraction, continuity and steering; (4) reject unless final verification has
602
+ no gaps and meets the verified threshold. Untrusted chunk prose never feeds the
603
+ quality floor. Final verification runs again after continuity injection. The
604
+ final scalar is reported as repaired **verification coverage**, alongside the
605
+ pre-repair score and fallback provenance, never as raw synthesis quality.
606
+ Failures keep only exhaustive content-free gap kinds and the rejecting gate
607
+ (`post-synthesis` or `post-state`) in local telemetry. Summary-derived
608
+ continuity fields cannot become evidence for their own initial verification.
609
+
244
610
  Polarity checks are symmetric: adding negation to a positive fact is rejected
245
611
  just as removing negation from a prohibition is. Short negation tokens such as
246
612
  `no` survive token filtering. Verbatim source clauses are not compared against
247
- the whole instruction's polarity, while additional contradictory clauses remain
613
+ the whole instruction's polarity, while additional contradictory clauses stay
248
614
  checked. Exact grounded path representations are not outcome claims; prose in
249
615
  file sections is still verified. Synthesis and post-state verification use the
250
- same summary budget for path encoding. Unresolved-error source
251
- snippets and fallback-rendered evidence share `summaryEvidenceLine()`, so
252
- Markdown prefixes and multiline wrapping cannot create false missing-error gaps.
253
-
254
- ## EESV hardening and control surfaces
255
-
256
- - **Canonical summary IR** accepts recognized H1/H2/H3 headings outside fenced code, preserves Progress subsections, and merges duplicate canonical kinds before state mutation.
257
- - **Typed verification gaps** drive mandatory deterministic repair; collision-aware path needles prevent basename cross-satisfaction. Tool/file provenance and normalized semantic evidence are indexed once per verification pass, and truncated or delimiter-incomplete LLM patches are rejected. Provenance is persisted and shown before optional approval.
258
- - **Fine tool semantics** separate read/search/list/mutate/delete/execute operations. Pruning deduplicates only identical idempotent access signatures.
259
- - **Unified token planning** uses a run-bound estimator with bounded process-shared provider/model calibration, counts structured tool-call arguments, preserves an adaptive recent tail, targets mode-specific post-compaction headroom, reserves bounded deterministic post-summary state sections, clamps every provider request to the model's advertised output limit, and reserves/reconciles every request against aggregate prompt/output-token caps. Missing provider usage is estimated conservatively. Tool exchanges remain atomic; oversized result bodies are head/tail bounded only for synthesis after full deterministic extraction.
260
- - **Security boundaries** recursively scrub structured messages before host serialization or provider calls, redact secret-bearing primitive values, and scrub plus hard-cap exploration tool feedback. PII scrubbing is opt-in. Backups remain unmaterialized until confirmed apply.
261
- - **Policy controls** include focus weighting, exact call/latency budgets, default fail-closed manual approval, online damage monitoring, and persisted open-loop overrides. Interactive review time is outside the pipeline deadline.
262
- - **Release gates** (`bun run gate`, `bun run bench`) cover adversarial parser, verification, tool, cache, budget, scrub and damage fixtures plus bounded p95 regressions for extraction, pruning, chunking, summary parsing, and path matching.
263
-
264
- ## State, caching & persistence
265
-
266
- Post-verification, `app/steps/state.ts` + `src/utils/state.ts` enrich the
616
+ same summary budget for path encoding. Unresolved-error snippets and
617
+ fallback-rendered evidence share `summaryEvidenceLine()`, so Markdown prefixes
618
+ and wrapping cannot create false missing-error gaps. Fallback constraints keep
619
+ the full extraction bound (`TRUNC.CONSTRAINT_TEXT`, 300 characters) without a
620
+ second preview cut or category label, because decorating or shortening a
621
+ faithful compound instruction can defeat exact-source matching.
622
+ `domain/keywords.ts` supplies the shared salient-keyword check used by
623
+ verification and damage detection.
624
+
625
+ ### EESV hardening and control surfaces
626
+
627
+ - **Canonical summary IR** accepts recognized H1/H2/H3 headings outside fenced
628
+ code, preserves Progress subsections, and merges duplicate canonical kinds
629
+ before state mutation.
630
+ - **Typed verification gaps** drive mandatory deterministic repair;
631
+ collision-aware path needles prevent basename cross-satisfaction.
632
+ Provenance and normalized semantic evidence are indexed once per pass, and
633
+ truncated or delimiter-incomplete LLM patches are rejected. Provenance is
634
+ persisted and shown before optional approval.
635
+ - **Fine tool semantics** separate read/search/list/mutate/delete/execute.
636
+ Pruning deduplicates only identical idempotent access signatures.
637
+ - **Unified token planning** uses a run-bound estimator with bounded
638
+ process-shared provider/model calibration, counts structured tool-call
639
+ arguments, keeps an adaptive recent tail, targets mode-specific
640
+ post-compaction headroom, reserves bounded post-summary state sections,
641
+ clamps every request to the model's advertised output limit, and reconciles
642
+ every request against aggregate prompt/output caps. Missing provider usage is
643
+ estimated conservatively. Tool exchanges stay atomic; oversized result bodies
644
+ are head/tail bounded only for synthesis, after full deterministic
645
+ extraction.
646
+ - **Per-dispatch capacity revalidation** (`domain/model-capacity.ts`) never
647
+ compares the whole conversation to a stage model's window: `trackedComplete`
648
+ estimates each actual serialized request, with output clamped to the model's
649
+ limit and Pi-AI's 4,096-token safety margin, and throws an actionable
650
+ `ModelCapacityError` before the provider is contacted. UI feasibility rows are
651
+ advisory snapshots; sizes that only exist after generation rely on this
652
+ runtime guard.
653
+ - **Security boundaries** recursively scrub structured messages before host
654
+ serialization or provider calls, redact secret-bearing primitive values, and
655
+ scrub plus hard-cap exploration tool feedback. PII scrubbing is opt-in.
656
+ Backups stay unmaterialized until confirmed apply.
657
+ - **Policy controls** include focus weighting, exact call/latency budgets,
658
+ default fail-closed manual approval, online damage monitoring and persisted
659
+ open-loop overrides. Interactive review time is outside the pipeline
660
+ deadline.
661
+ - **Release gates** (`bun run gate`, `bun run bench`) cover adversarial parser,
662
+ verification, tool, cache, budget, scrub and damage fixtures plus bounded p95
663
+ regressions for extraction, pruning, chunking, summary parsing and path
664
+ matching.
665
+
666
+ ### Provider-native compaction engine
667
+
668
+ Provider-native compaction is an optional engine on stock Pi, using public
669
+ extension APIs only. Pi 0.87.1's adapters do not parse or replay signed
670
+ Anthropic blocks or opaque OpenAI items, so this extension does both:
671
+
672
+ - `run-smart-compact.ts` runs the `compactionEngines` list after the window
673
+ step. Each engine is `applied`, `skipped` or `failed` (`EngineAttempt`); all
674
+ share one provider-call budget; if none applies, `EngineChainError` lists
675
+ every outcome and nothing is staged. The default list is `["eesv"]`.
676
+ - `app/native-compaction.ts` gates on `isNativeApi(ctx.model.api)` and uses only
677
+ the current session model. It moves the cut back to a clean turn boundary
678
+ (kept tail starts at a user message, prefix ends with a completed assistant
679
+ reply, no open tool call).
680
+ - Request: one nested `ctx.modelRegistry.streamSimple(model, { systemPrompt,
681
+ messages, tools }, { fetch, transport: "sse", maxRetries: 0, signal,
682
+ sessionId })`. Active tools are included because Anthropic rejects tool_use
683
+ history without them. Pi's adapter builds the ordinary request; the `fetch`
684
+ from `infra/native-protocol.ts` (`createCompactionFetch`) rewrites it into the
685
+ provider's compaction request, sends it once and answers Pi with a
686
+ non-retryable 400. Without a result no request was sent and Pi's error is
687
+ reported literally. A prefix that starts with an earlier native compaction of
688
+ the same route replays that state (`prior`); failure to replay aborts before
689
+ the provider call.
690
+ - Codex `response.incomplete` is a failure even if an item and usage arrived.
691
+ Results and persisted state require a signed Anthropic compaction block or
692
+ OpenAI compaction items with encrypted content; the opaque bytes are
693
+ preserved, not cryptographically verified locally. Results are also rejected
694
+ for another route or when not smaller than the prefix estimate. Accepted
695
+ state is staged in the ordinary pending slot and stored only in
696
+ `details.native` (`NativeState`, validated with `isNativeState` on every
697
+ read).
698
+ - Replay: `before_provider_request` takes the latest compaction entry on the
699
+ branch; if its `details.native` matches the current api/provider/model,
700
+ `replayNativeState` returns a payload copy replacing only Pi's exact wrapped
701
+ summary. A bare substring match never authorizes replay. When replay is
702
+ impossible on a matching route, OpenAI routes get a one-time warning per entry
703
+ and Anthropic is recorded only. The hook does no I/O and does not walk the
704
+ branch until the session has native state (a flag recomputed from in-memory
705
+ entries at `session_start`, `session_tree` and `session_compact`).
706
+ - Details record `method: "native"`, `nativeApi` and the engine attempts;
707
+ notices never claim EESV verification for native state.
708
+ - Requests made by the extension itself (EESV stages and the native
709
+ compaction body) go through the requesting session's public model runtime
710
+ (`ctx.modelRegistry.stream`/`streamSimple`, `infra/llm-client.ts`), never
711
+ pi-ai's standalone completers. The runtime applies request-time auth and any
712
+ provider registered by another extension with `pi.registerProvider`, such as
713
+ the separate `pi-claude-oauth-adapter`; stock Pi still skips
714
+ `before_provider_request` for these requests, so an adapter must normalize
715
+ the final payload inside its own provider (the published `0.2.2` does not;
716
+ [upstream PR #10](https://github.com/minzique/pi-claude-oauth-adapter/pull/10)).
717
+ Caller `apiKey`/`headers` are stripped so an explicit key never bypasses
718
+ stored OAuth; `app/stage-auth.ts` is an availability preflight only. The
719
+ Anthropic prior-state replay runs in the caller's `onPayload` before any
720
+ adapter normalization, and the final on-demand request drops
721
+ `context_management` because Anthropic prohibits combining it with
722
+ on-demand compaction.
723
+
724
+ `test/native-compaction-compat.test.ts` pins what stock adapters drop by
725
+ themselves. The design research and measured runs are in the
726
+ [2026-09-24 research report](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/hindsight-native-compaction-research-2026-09-24.md)
727
+ (repository only, historical).
728
+
729
+ ### Experimental visual evidence
730
+
731
+ `visualArchiveEnabled` defaults to false. After `buildState` has verified text
732
+ and post-compaction yield, `steps/visual.ts` may add a bounded supplementary
733
+ archive. It reuses the conservative read-only tool-batch selector, skips
734
+ intentional context edits, and scrubs text before clipping and rendering. A
735
+ Latin/Turkish glyph scope, at most eight 3k-character excerpts, two 1280-wide
736
+ pages (74 rows each) and 1 MB total PNG bytes bound local work. Oldest whole
737
+ excerpts are dropped until the image allowance plus reading guide fits both the
738
+ original target and response reserve. Representation comparisons use a
739
+ reader-bound estimator from the shared calibration store; summarizer
740
+ accounting stays separate. The yield gate runs again, and any failure keeps the
741
+ unchanged text. This does not avoid the EESV call or promise cheaper
742
+ compaction.
743
+
744
+ The optional resvg renderer is imported only on demand, uses one shipped
745
+ licensed font with system-font discovery disabled, and receives only
746
+ XML-escaped text in a fixed generated SVG, with no user-controlled SVG,
747
+ resource paths or URLs. It is external to both bundles; Node loads the default
748
+ extension without it. Rendering uses the run's abort signal plus a five-second
749
+ cancellation limit, with a fresh composed signal per page because resvg's
750
+ native abort binding cannot be reused.
751
+
752
+ `SmartCompactDetails.visualArchive` stores versioned bounded source excerpts
753
+ and PNGs in the native compaction entry; no file cache or second transcript
754
+ store is created. Later hybrid compaction re-renders source text, never OCR.
755
+ `context` validates frame bounds/signatures, source ancestry and revocation,
756
+ reader route, privacy, summary identity and request headroom, then inserts a
757
+ request-local custom image message; it never replaces the verified text
758
+ summary. Changing model/provider/API or using a text-only model withholds
759
+ images; stricter scrubbing also withholds old pixels that cannot be
760
+ retroactively redacted. Only the latest compaction's archive is eligible, and
761
+ native fallback can discard it. Metrics record only visual token estimates and
762
+ frame counts. The [2026-09-24 visual pilot](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/visual-pilot-2026-09-24.md)
763
+ (repository only) is a dated single-model synthetic sample, not production or
764
+ cross-model accuracy.
765
+
766
+ ## 4. Optional cross-session memory
767
+
768
+ Memory is exactly one selected backend at a time (`app/memory-backend.ts`):
769
+ `local`, `hindsight` or `mnemopi`. The selected backend is the only store read
770
+ or written. With Hindsight or Mnemopi selected, the local context graph is
771
+ neither indexed by compaction nor consulted, no local copy exists, and inactive
772
+ stores are preserved untouched. Hindsight means the user's existing configured
773
+ server; nothing installs or starts one. Continuity state, backups and artifact
774
+ spill are session mechanisms outside this choice. Manual saves require host
775
+ confirmation of the complete scrubbed content. Only the local backend also
776
+ indexes derived facts, and only from apply-confirmed compactions; remote
777
+ backends receive nothing automatically.
778
+
779
+ - **Local** (`infra/context-graph.ts`): project-partitioned SQLite FTS5 facts
780
+ and file edges, indexed from apply-confirmed compactions plus confirmed
781
+ manual memories. Details are in
782
+ [state, caching and persistence](#state-caching-and-persistence).
783
+ - **Hindsight** (`app/hindsight-memory.ts`, `infra/hindsight-client.ts`,
784
+ `infra/hindsight-receipts.ts`): four fixed routes (retain, status, recall,
785
+ delete one document), no generic request, origin/bank/project-scoped receipts
786
+ that never evict unconfirmed operations. Data flow, consent and receipt states
787
+ are in [Hindsight memory backend](./docs/hindsight-memory.md).
788
+ - **Mnemopi** is an optional Bun-only dependency, never imported by the Node
789
+ host. `app/mnemopi-memory.ts` starts a bounded worker and waits for a
790
+ readiness line after its imports before sending any content over stdin.
791
+ `resolveBunExecutable()` resolves the worker runtime read-only: the
792
+ optional `bun` component installed beside the extension first (manifest bin
793
+ plus a real-file check that rejects the postinstall placeholder), then the
794
+ platform `@oven/*` package, then a supported PATH Bun (>=1.3.14). No shell,
795
+ download or self-install is involved, so missing components fail before a memory
796
+ request; interrupted submitted writes remain uncertain. TypeBox validates both
797
+ IPC directions and persisted engine metadata. Project-isolated files,
798
+ author/kind filters and checked provenance prevent cross-project recall. The
799
+ cross-process lock covers identity lookup and mutation through child exit;
800
+ Mnemopi manages its own SQLite transactions. Stable per-fact engine sessions
801
+ make duplicate saves and metadata-id resolution work without a sidecar index.
802
+ No shared default bank, embeddings, LLM extraction, consolidation, runtime
803
+ model download or silent backend fallback is enabled.
804
+
805
+ `infra/memory-ref.ts` supplies opaque backend/id refs with mandatory 96-bit
806
+ target digests. Local and Mnemopi refs bind their store path; Hindsight also
807
+ binds server, bank, project and document. Resolution compares current target
808
+ configuration rather than reading a destination from untrusted input. These are
809
+ routing checks, not authorization tokens or remote-content attestations; host
810
+ confirmation stays mandatory. Local and Mnemopi resolve inspects the stored
811
+ fact; Hindsight does not claim its confirmation is a document read. Unknown
812
+ retain outcomes, including lost operation status, block remote deletion until
813
+ terminal evidence; bounded recall refresh never evicts uncertainty.
814
+
815
+ ## State, caching and persistence
816
+
817
+ After verification, `app/steps/state.ts` and `utils/state.ts` enrich the
267
818
  summary, then `domain/yield-gate.ts` measures the final replacement. Planning
268
- has already reserved the bounded post-synthesis enrichment band by reducing the
269
- retained tail; missing the original target or 10% net-saving floor still throws
270
- before a `StatedRc` can reach staging/apply. `session_before_compact` only stages
271
- a passing candidate; after the host emits the matching `session_compact`,
272
- `app/steps/persist.ts` commits reusable state, the prepared conversation backup,
273
- and success telemetry. Aborted/unconfirmed candidates write none of them. The
274
- UI reports `Applied` only after that correlated commit and emits a separate
275
- warning if any durable side effect was partial.
276
- `ui/error-format.ts` converts verification/yield failures to one bounded,
277
- content-free diagnostic and next action, including unknown/provider errors.
278
- Per-call categories survive in aggregated route metrics even when fallback
279
- succeeds; raw errors never enter that telemetry. Full stacks are suppressed by
280
- default and require restarting Pi with explicit `DEBUG=smart-compact`.
281
- Manual execution uses a two-line widget: a colored EESV phase chain plus a
282
- phase-specific action brief. Before Apply it explicitly says the conversation
283
- is unchanged. Routine info toasts are hidden unless `verbose`; handled provider,
284
- watchdog, Explore, batch, and assembly failures switch to deterministic fallback
285
- without printing raw messages. Auto-trigger rejection logs are also debug-only,
286
- leaving one content-free safe-fallback notice in the UI.
819
+ has already reserved the bounded enrichment band by reducing the retained tail;
820
+ missing the original target or the 10% net-saving floor still throws before a
821
+ `StatedRc` can reach staging or apply. `session_before_compact` only stages a
822
+ passing candidate (`app/compaction-commit-store.ts` holds it between the two
823
+ events). After the host emits the matching `session_compact`,
824
+ `app/steps/persist.ts` commits reusable state, the prepared conversation backup
825
+ and success telemetry. Aborted or unconfirmed candidates write none of them.
826
+ The UI reports `Applied` only after that correlated commit and warns separately
827
+ if any durable side effect was partial. Cost-only discarded-preparation records
828
+ cannot commit reusable state, backups or applied-canary evidence.
829
+
830
+ `ui/error-format.ts` turns verification/yield failures into one bounded,
831
+ content-free diagnostic and next action. Per-call categories survive in route
832
+ metrics even when fallback succeeds; raw errors never enter telemetry. Full
833
+ stacks require restarting Pi with `DEBUG=smart-compact`. Manual execution shows
834
+ a two-line widget: a colored EESV phase chain plus a phase-specific brief that
835
+ says the conversation is unchanged until Apply. Routine info toasts are hidden
836
+ unless `verbose`; handled provider, watchdog, Explore, batch and assembly
837
+ failures switch to deterministic fallback without printing raw messages.
838
+ Auto-trigger rejection logs are debug-only, leaving one content-free notice.
839
+ `utils/issues.ts` deduplicates user-facing problems once per session.
287
840
 
288
841
  | Concern | Where | Notes |
289
842
  | --- | --- | --- |
290
843
  | Open-loop injection | `utils/state.ts` | inserted before Next Steps via the canonical parser |
291
844
  | `CompactionState` | `utils/state.ts` | immutable project/session/branch-head snapshots; descendants resolve the newest matching ancestor and siblings never overwrite each other |
292
- | Continuity Ledger | `utils/state.ts` | prior facts carry forward until positive resolution evidence or an explicit override; goal shifts become non-destructive breadcrumbs |
845
+ | Continuity ledger | `utils/state.ts` | prior facts carry forward until positive resolution evidence or an explicit override; goal shifts become non-destructive breadcrumbs |
293
846
  | Cross-compaction delta | `utils/state.ts` | "Changes Since Last Compaction" section |
294
- | Incremental extraction cache | `utils/cache.ts` + `utils/id-fingerprint.ts` | bounded SHA-256 prefix fingerprint + tail; an exact pruned-payload match is reused directly, while incremental reuse is allowed only when the pruned prefix still matches |
295
- | Synthesis cache | `infra/synthesis-cache.ts` | behavior key includes normalized focus, route, mode, profile limits, run-level call/input/latency limits, and reasoning |
296
- | Session-log recovery | `utils/session-log.ts` | async bounded-memory JSONL scan with event-loop yields and a bounded path cache; bypasses pi-toolkit truncation by entry-id mapping without dropping late active-branch IDs at a fixed byte cap |
297
- | Project fingerprint | `utils/fingerprint.ts` | locked read/merge/write; language/framework/key dirs stay bounded and `sessionCount` tracks distinct hashed session identities |
847
+ | Native continuity handoff | `app/native-continuity-bridge.ts` | one-shot, bounded, keyed by project + session + branch head |
848
+ | Incremental extraction cache | `utils/cache.ts` + `utils/id-fingerprint.ts` | bounded entry-ID fingerprint plus projected/recovered content hash; reuse requires both prefix proofs |
849
+ | Synthesis cache | `infra/synthesis-cache.ts` | key includes normalized focus, route, mode, profile limits, run-level call/input/latency limits and reasoning |
850
+ | Session-log recovery | `utils/session-log.ts` | async bounded-memory JSONL scan; recovers only truncated, unedited messages by entry ID without resurrecting intentional replacements or omissions |
851
+ | Project fingerprint | `utils/fingerprint.ts` | locked read/merge/write; bounded language/framework/key dirs; `sessionCount` tracks distinct hashed sessions |
298
852
  | Damage detection | `utils/damage.ts` | best-effort post-compaction regression signals |
299
- | Context graph | `infra/context-graph.ts` | SQLite FTS5 facts + file edges; 2,000 non-structural nodes per project |
300
-
301
- Apply-confirmed verified state is queued and duplicate updates coalesce only
302
- for the exact project/session/branch head. Replacing an existing key refreshes
303
- that pending value even at capacity; a divergent 65th key is rejected rather
304
- than evicting accepted work. A microtask drains the accepted batch through one
305
- reused SQLite connection. Every caller awaits the transaction result, so
306
- persistence telemetry is complete only after indexing succeeds; permanent
307
- open/write failures settle once as `context graph` failures and are never
308
- zero-delay retried. All graph surfaces
309
- (recall, manual memory, stats, indexing) share one process-wide, path-keyed
310
- cached connection, so connection setup, schema checks, and migration probes
311
- are paid once per process instead of per call; the cache is keyed by database
312
- path so environments that relocate the cache directory (tests swapping HOME)
313
- reopen cleanly.
314
- `infra/context-graph.ts` adapts the same fail-closed transaction contract to
315
- `bun:sqlite` in Bun tests and `node:sqlite` `DatabaseSync` in Pi's Node runtime;
316
- the packed release audit exercises both. Graph data is derived and a later
317
- cumulative state safely supersedes a failed update. Fact
318
- occurrences are branch-head scoped; state, recall, and resolution use the
319
- complete host-visible branch ancestry before equivalent facts are deduplicated.
320
- Schema v1 preserves user-confirmed manual memory but resets older derived
321
- compaction nodes once so sibling branches cannot inherit a last-writer identity.
322
- Recall starts from FTS5 lexical matches whose rowids are the owning
323
- `context_nodes` rowids, expands one hop through file-reference edges, then
324
- applies session, branch, fact-kind, confidence, recency, and explicit-memory
325
- weights. Exact equivalent facts are deduplicated before bounded output.
326
- Resolved/superseded state is removed from the active FTS index; another
327
- project's rows are never eligible.
328
-
329
- **Important retention limits:** pending in-memory compaction 5 min · exploration
853
+ | Context graph | `infra/context-graph.ts` | SQLite FTS5 facts + file edges; 2,000 active derived nodes and 2,000 resolved/superseded tombstones per project |
854
+
855
+ Apply-confirmed state is queued, and duplicate updates coalesce only for the
856
+ exact project/session/branch head. Replacing an existing key refreshes that
857
+ pending value even at capacity; a divergent 65th key is rejected rather than
858
+ evicting accepted work. A microtask drains the batch through one reused SQLite
859
+ connection. Every caller awaits the transaction result, so persistence
860
+ telemetry completes only after indexing succeeds; permanent open/write failures
861
+ settle once as `context graph` failures and are never zero-delay retried. All
862
+ graph surfaces share one process-wide connection cached by database path, so
863
+ environments that relocate the cache directory reopen cleanly. The same
864
+ fail-closed transaction contract runs on `bun:sqlite` in Bun tests and
865
+ `node:sqlite` `DatabaseSync` in Pi's Node runtime; the packed release audit
866
+ exercises both.
867
+
868
+ Later cumulative state can supersede a failed derived update; user-confirmed
869
+ memory is not derived. Fact occurrences are branch-head scoped; state, recall
870
+ and resolution use the complete host-visible branch ancestry before equivalent
871
+ facts are deduplicated. Schema v1 preserves user-confirmed manual memory but
872
+ resets older derived compaction nodes once so sibling branches cannot inherit a
873
+ last-writer identity. Recall starts from FTS5 lexical matches whose rowids are
874
+ the owning `context_nodes` rowids, expands one hop through file-reference
875
+ edges, then weights session, branch, fact kind, confidence, recency and
876
+ explicit memory. Resolved or superseded state leaves the active FTS index;
877
+ another project's rows are never eligible. The forget command distinguishes a
878
+ derived-only reset from confirmed all-project graph deletion; neither changes
879
+ Mnemopi/Hindsight, compaction state or backups. Closing a local ref only marks
880
+ its node resolved.
881
+
882
+ **Retention limits:** pending in-memory compaction 5 min · exploration
330
883
  tool-support cache 1 h / 128 routes · token calibration 128 routes · extraction
331
884
  cache 1 h · compaction state 7 d / 64 snapshots · context graph 2,000
332
- non-structural fact nodes, 64 pending branch-head updates, and 500 active manual
885
+ active derived fact nodes, 2,000 tombstones (resolved facts and closed manual
886
+ memories), 64 pending branch-head updates and 500 active manual
333
887
  memories per project · remediation hints 7 d · metrics and damage JSONL logs
334
- 5 MiB each · one exploration tool result 12,000 characters.
888
+ 5 MiB each · one exploration tool result 12,000 characters. File locations are
889
+ listed in the guide's [storage and privacy](./docs/guide.md#storage-and-privacy)
890
+ section.
335
891
 
336
- ## Concurrency & safety model
892
+ ## Concurrency and safety model
337
893
 
338
- The extension is built to run safely alongside other Pi sessions and other
339
- extensions.
894
+ The extension runs alongside other Pi sessions and other extensions.
340
895
 
341
896
  ### Pending-compaction slot
342
897
 
343
- [`src/app/pending-slot.ts`](./src/app/pending-slot.ts) is an encapsulated,
898
+ [`src/app/pending-slot.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/pending-slot.ts) is an encapsulated,
344
899
  host-agnostic state cell (one producer, one consumer, single-threaded event
345
900
  loop). `consume()` returns a discriminated result:
346
901
 
@@ -351,49 +906,49 @@ loop). `consume()` returns a discriminated result:
351
906
  | `expired` | older than the 5-minute TTL |
352
907
  | `mismatch` | staged by a different session, project, or non-ancestor branch head |
353
908
 
354
- Session identity comes from [`infra/session-identity.ts`](./src/infra/session-identity.ts):
355
- a real id when the host exposes one, otherwise a per-call unforgeable
356
- `unresolved:<uuid>` — two unresolved sessions can never collide. Apply also
357
- requires the staged branch head to be the current head or one of its visible
358
- ancestors, so navigation to a sibling branch cannot consume stale payload.
909
+ Session identity comes from
910
+ [`infra/session-identity.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/session-identity.ts): a real ID when
911
+ the host exposes one, otherwise a per-call unforgeable `unresolved:<uuid>`, so
912
+ two unresolved sessions never collide. Apply also requires the staged branch
913
+ head to be the current head or one of its visible ancestors, so navigation to a
914
+ sibling branch cannot consume a stale payload. Payloads fingerprint projected
915
+ message IDs and content: append-only growth is allowed, same-ID context edits
916
+ invalidate.
359
917
 
360
918
  ### Cancellation deadlines
361
919
 
362
920
  Automatic compaction combines the host event's `AbortSignal` with its own
363
- deadline through a shared [`ExternalCancellation`](./src/app/run-smart-compact.ts)
364
- handle. Either source calls `abort()`, and every side-effect gate in the
365
- orchestrator checks the shared state before writing or applying compaction.
366
- The caller waits for safe pipeline unwind; no unsafe `Promise.race` hard return
367
- can leave work running past the hook lifecycle.
368
-
369
- ### Filesystem & concurrency
370
-
371
- JSON/text cache writes use [`src/infra/fs.ts`](./src/infra/fs.ts): private
372
- artifact directories are 0700 and files are 0600; atomic temp-file + rename
373
- prevents half-truncated readers but intentionally does not claim fsync/power-loss
374
- durability. Append/trim operations run asynchronously, yield before synchronous
375
- filesystem work, and hold a `mkdir`-based cross-process lock for the complete
376
- transaction. Lock ownership is reclaimed by atomic rename, never by deleting a
377
- possibly renewed lease in place. Sessions therefore cannot interleave bytes or
378
- steal a live successor's lock. SQLite supplies its own WAL durability. Native
379
- continuity handoffs are one-shot, bounded, and keyed by project + session +
380
- branch head.
381
-
382
- The session run lock uses file leases reclaimed when the owning PID dies.
383
- Reclaim has a deliberate, documented TOCTOU window: two processes reclaiming
384
- the same stale lease within milliseconds can, in one interleaving, unlink the
385
- other's freshly created lease. The double re-read (token + inode metadata)
386
- narrows but cannot atomically close this without an O_EXCL rename protocol;
387
- the lock is best-effort serialization of a normally single-writer flow, not a
388
- mutual-exclusion guarantee, and the surrounding pipeline remains fail-closed
389
- when the lock cannot be acquired.
921
+ deadline through a shared [`ExternalCancellation`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/app/run-smart-compact.ts)
922
+ handle. Either source calls `abort()`, and every side-effect gate checks the
923
+ shared state before writing or applying. The caller waits for a safe pipeline
924
+ unwind; no `Promise.race` hard return can leave work running past the hook
925
+ lifecycle.
926
+
927
+ ### Filesystem and locks
928
+
929
+ JSON/text cache writes use [`src/infra/fs.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/fs.ts): private
930
+ artifact directories are 0700 and files 0600; atomic temp-file + rename
931
+ prevents half-truncated readers but does not claim fsync/power-loss
932
+ durability. Append/trim operations run asynchronously, yield before
933
+ synchronous filesystem work, and hold a `mkdir`-based cross-process lock for
934
+ the whole transaction. Lock ownership is reclaimed by atomic rename, never by
935
+ deleting a possibly renewed lease in place. SQLite supplies its own WAL
936
+ durability.
937
+
938
+ The session run lock (`app/session-run-lock.ts`) uses file leases reclaimed
939
+ when the owning PID dies. Reclaim has a deliberate, documented TOCTOU window:
940
+ two processes reclaiming the same stale lease within milliseconds can, in one
941
+ interleaving, unlink the other's fresh lease. A double re-read (token + inode
942
+ metadata) narrows but cannot atomically close this without an O_EXCL rename
943
+ protocol. The lock is best-effort serialization of a normally single-writer
944
+ flow, not a mutual-exclusion guarantee, and the pipeline stays fail-closed when
945
+ the lock cannot be acquired.
390
946
 
391
947
  ## Provider awareness
392
948
 
393
- [`src/utils/tokens.ts`](./src/utils/tokens.ts) keeps a per-provider capability
949
+ [`src/utils/tokens.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/utils/tokens.ts) keeps a per-provider capability
394
950
  table (Anthropic, OpenAI, Google, DeepSeek, MiniMax, Xiaomi, Mistral, xAI, …)
395
- with unknowns falling back to a safe default + fuzzy alias matching. Each entry
396
- drives pipeline behavior:
951
+ with a safe default and fuzzy alias matching for unknowns:
397
952
 
398
953
  | Capability | Drives |
399
954
  | --- | --- |
@@ -403,61 +958,44 @@ drives pipeline behavior:
403
958
  | `cacheStrategy` | prompt-cache retention per call |
404
959
  | `timeoutMultiplier` | auto-trigger hard-timeout headroom |
405
960
  | `singlePassTokenMultiplier` | single-pass vs chunked threshold |
406
- | `tokenRatioEstimate` | token estimation; refined by per-(provider,model) **EMA calibration** |
407
-
408
- Every provider call is raced against one aborting hard deadline so a transport
409
- that ignores cancellation cannot keep the run lock indefinitely. Custom Codex
410
- endpoints also receive `max_output_tokens` through Pi AI's payload hook. The
411
- ChatGPT subscription endpoint rejects every wire output-cap field, so its
412
- deadline is derived from the requested output allowance (15–90s) and paired
413
- with a visible-output ceiling. Deadline failures route to the phase's
414
- deterministic fallback.
415
-
416
- ### Provider evaluation and routing evidence
417
-
418
- [`src/domain/provider-evaluation.ts`](./src/domain/provider-evaluation.ts)
419
- collapses call telemetry into Explore/Synthesize/Verify routes and compares
420
- provider/models across a deterministic context-pressure × tool-density matrix.
421
- Only an explicitly attributed pre-repair synthesis score contributes route
422
- quality; a run's final verifier score is never copied into Explore/Verify.
423
- Legacy or operational-only routes still contribute latency and reliability.
424
- Recommendations require minimum samples, ≥80% call reliability, and ≥50%
425
- stage-local quality coverage, shrink toward neutral under low confidence, and
426
- are advisory only.
427
- They never mutate config or replace the selected model. The opt-in live harness
428
- runs three identical bounded continuity scenarios across explicitly named
429
- models.
430
-
431
- ### Privacy-safe telemetry and canary decisions
432
-
433
- [`src/domain/telemetry.ts`](./src/domain/telemetry.ts) maps raw exceptions to a
434
- content-free failure taxonomy, aggregates schema-v2 run quality without IDs or
435
- conversation data, and compares an explicitly tagged `canary` cohort against
436
- `stable` history. Reports expose total/applied counts; only non-dry,
437
- host-confirmed applied outcomes count toward promotion. A deterministic green
438
- release check is not promotion evidence. The gate returns Hold, Rollback, or
439
- Promote from applied sample/quality coverage plus failure, verifier quality, p95
440
- latency, token, heuristic-fallback, and post-compaction-damage thresholds.
441
- Damage observations join their originating compaction by local run id, dedupe
442
- per run, and require ≥70% stable/canary coverage before promotion. It is
443
- advisory: rollout selection, configuration changes, and rollback remain external.
444
-
445
- Dashboard trust calculations live in
446
- [`src/ui/dashboard-insights.ts`](./src/ui/dashboard-insights.ts). Data
447
- Confidence is an auditable 100-point score over sample size, schema-v2
448
- coverage, verifier-quality coverage, required-field completeness, and
449
- freshness; ≥85 is the high-confidence target. TUI and HTML surfaces share the
450
- same quality-repair, stage/provider/model, failure-taxonomy, and
451
- stable-vs-canary aggregates. Missing/legacy evidence lowers the score and
452
- produces remediation guidance instead of being imputed.
961
+ | `tokenRatioEstimate` | token estimation; refined by per-(provider, model) EMA calibration |
962
+
963
+ Every provider call is raced against one aborting hard deadline, so a
964
+ transport that ignores cancellation cannot hold the run lock indefinitely.
965
+ Custom Codex endpoints receive `max_output_tokens` through Pi AI's payload
966
+ hook. The ChatGPT subscription endpoint rejects every wire output-cap field, so
967
+ its deadline is derived from the requested output allowance (15–90 s) and
968
+ paired with a visible-output ceiling. A per-call deadline may use the phase's
969
+ deterministic fallback while the run stays active. Run-wide timeout or host
970
+ cancellation propagates instead: it cannot schedule more synthesis or repair,
971
+ return a successful dry run, or publish a pending summary. The single run
972
+ outcome is `timeout` for the deadline or neutral `cancelled` for a host abort.
973
+ The run lock and pending slot are released. Native recovery is the requesting
974
+ host's decision, never an implicit fallback promised to manual callers.
975
+
976
+ ### Evaluation and telemetry
977
+
978
+ [`src/domain/provider-evaluation.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/domain/provider-evaluation.ts)
979
+ aggregates call telemetry into an advisory stage × context-pressure ×
980
+ tool-density matrix; it never mutates configuration.
981
+ [`src/domain/telemetry.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/domain/telemetry.ts) maps exceptions to a
982
+ content-free failure taxonomy, aggregates schema-v2 quality without IDs or
983
+ conversation data, and compares an explicit `canary` cohort with `stable`
984
+ history. `src/ui/dashboard-insights.ts` computes the dashboard's Data
985
+ Confidence heuristic. `scripts/task-eval.ts` and `task-eval-case.ts` pair the
986
+ same task across no-compaction, hygiene, EESV and hybrid stock-Pi sessions;
987
+ `scripts/replay-eval.ts` replays recorded sessions under alternative trim
988
+ policies and reports estimates only.
989
+ Commands, exact decision thresholds and evidence limits are documented once, in
990
+ [evaluation](./docs/evaluation.md).
453
991
 
454
992
  ## Dependency injection
455
993
 
456
- [`src/infra/services.ts`](./src/infra/services.ts) is a per-`runSmartCompact`
457
- service bag. Metrics, budgets, scrubbers, and prompt namespaces are isolated per
994
+ [`src/infra/services.ts`](https://github.com/alpertarhan/pi-smart-compact/blob/main/src/infra/services.ts) is a per-`runSmartCompact`
995
+ service bag. Metrics, budgets, scrubbers and prompt namespaces are isolated per
458
996
  run. Production shares only bounded provider/model capability and calibration
459
- knowledge, which contains no conversation/session data; tests use isolated
460
- stores by default:
997
+ knowledge, which contains no conversation or session data; tests use isolated
998
+ stores by default.
461
999
 
462
1000
  | Service | Role |
463
1001
  | --- | --- |
@@ -466,58 +1004,90 @@ stores by default:
466
1004
  | `toolSupport` | process-shared in production; explicit unsupported capability, 1 h TTL / 128 routes |
467
1005
  | `metrics` | bounded metrics sink |
468
1006
  | `extractionCacheStats` | hit / miss counters |
469
- | `tokenCalibration` | process-shared bounded per-(provider,model) EMA factors |
1007
+ | `tokenCalibration` | process-shared bounded per-(provider, model) EMA factors |
470
1008
  | `compactSessionId` | per-run prompt-cache namespace |
471
1009
 
472
1010
  ## Layer responsibilities
473
1011
 
474
- The code is organized into six layers, each with a single responsibility.
475
-
476
1012
  ### Entry layer
477
1013
 
478
1014
  | File | Responsibility |
479
1015
  | --- | --- |
480
1016
  | `src/index.ts` | extension composition root and host lifecycle hooks |
1017
+ | `src/rtk.ts` | optional RTK companion entry point (not in `pi.extensions`) |
481
1018
  | `src/constants.ts` | version, thresholds, prompts, config keys |
482
1019
  | `src/types.ts` | shared types and discriminated unions |
483
- | `domain/provider-evaluation.ts` | advisory provider scenario matrix and route telemetry aggregation |
484
- | `domain/telemetry.ts` | privacy-safe aggregates, failure taxonomy, and canary rollback gates |
485
1020
 
486
1021
  ### Orchestration layer (`src/app/`)
487
1022
 
488
1023
  | File | Responsibility |
489
1024
  | --- | --- |
490
- | `app/run-smart-compact.ts` | top-level pipeline orchestrator |
491
- | `app/register-smart-compact-command.ts` | manual command adapter and restore/loop actions |
1025
+ | `app/run-smart-compact.ts` | top-level pipeline orchestrator and engine chain |
1026
+ | `app/register-smart-compact-command.ts` | manual command adapter: Home, preflight args, trim/storage/forget/restore/loops actions |
492
1027
  | `app/register-smart-compact-tool.ts` | bounded agent-tool adapter |
1028
+ | `app/smart-compact-input.ts` | command and tool argument parsing |
493
1029
  | `app/register-context-tools.ts` | project-scoped recall/save-memory adapters |
1030
+ | `app/register-smart-context-tool.ts` | session-control tool and native turn-boundary lifecycle |
1031
+ | `app/context-operations.ts` | pure checkpoint validation, pair-safe edit planning, branch-scoped archived-output access |
1032
+ | `app/tool-artifacts.ts` | safe early tool-output spill, private storage quotas/integrity, branch-owned references |
1033
+ | `app/context-evidence.ts` | common bounded listing/search/read for session output, visual excerpts and artifacts, active branch first then loaded lineage |
1034
+ | `app/session-lineage.ts` | read-only in-memory load of `parentSession` ancestors (depth, size and cycle bounds) |
1035
+ | `app/session-handoff.ts` | handoff seed from recorded state only; preview and `ctx.newSession` seeding |
1036
+ | `app/host-cache-ledger.ts` | session-local ledger of Pi's own requests: rebuild detection, cause attribution, cache lifetime |
1037
+ | `app/artifact-storage.ts` | read-only storage inventory and lineage classification |
1038
+ | `app/visual-archive.ts` | bounded evidence selection, persisted archive validation, request-local image rehydration |
1039
+ | `app/native-compaction.ts` | native engine: nested-request compaction, clean-turn cut, route/size checks, replay |
1040
+ | `app/native-continuity-bridge.ts` | one-shot continuity handoff keyed by project, session and branch head |
1041
+ | `app/memory-backend.ts` | exclusive memory-backend policy, read-only readiness, Mnemopi runtime evidence and Bun resolution |
1042
+ | `app/hindsight-memory.ts` | confirmed Hindsight save/resolve/recall flow and honest outcome reporting |
1043
+ | `app/mnemopi-memory.ts` / `mnemopi-worker.ts` / `mnemopi-protocol.ts` | Node-safe optional Bun engine, project-isolated confirmed memory and validated bounded IPC |
1044
+ | `app/effective-state.ts` | shared local-only readiness, effective policy and runtime-state view |
1045
+ | `app/model-feasibility.ts` | lazy local estimate of planned stage requests for model rows |
1046
+ | `app/lazy-tools.ts` | tool exposure: on-demand groups, eager and off modes, user `/tools` precedence, reachability for offload |
1047
+ | `app/register-navigation.ts` | anchors, recall, queued/revalidated pivots, footer status and the `smart_navigation` tool |
1048
+ | `app/navigation-data.ts` / `navigation-types.ts` | anchor and recall data over owned and legacy `context` anchors; read-only session scans |
1049
+ | `app/anchor-cache.ts` | Anthropic prompt-cache marker on the newest anchor |
1050
+ | `app/context-guide.ts` | on-demand read of the context-management guide |
494
1051
  | `app/model-routing.ts` | stage model resolution and explicit-model precedence |
1052
+ | `app/stage-auth.ts` | per-stage credential availability preflight; the session runtime resolves auth per request |
495
1053
  | `app/smart-compact-policy.ts` | branch-scoped agent visibility and auto-trigger policy; owns active-tool updates |
1054
+ | `app/global-settings-runtime.ts` | refresh each runtime owner once after an atomic global-settings patch |
1055
+ | `app/preflight.ts` | shared deterministic preparation for preview and real run |
496
1056
  | `app/run-context.ts` | typed stage chain (`RcBase → … → StatedRc`) |
497
1057
  | `app/mode-policy.ts` | Auto selector and finite Fast/Balanced/Thorough policies; legacy Aggressive maps to Fast |
498
1058
  | `app/pending-slot.ts` | encapsulated pending-compaction state cell |
1059
+ | `app/compaction-commit-store.ts` | holds summaries between `session_before_compact` and `session_compact` |
1060
+ | `app/session-run-lock.ts` | same-session serialization plus process-global file lease |
499
1061
  | `app/settled-auto-trigger.ts` | guarded proactive host compact requests; no EESV or pending-state ownership |
500
- | `app/steps/prepare.ts` | resolve config, provider caps, budgets, and cancellation; stage auth resolves lazily |
501
- | `app/steps/window.ts` | pick the prefix using provider-calibrated synthesis and deterministic post-processing bounds |
1062
+ | `app/background-preparation.ts` | early snapshots, invalidation, validated handoff, discard accounting and shutdown drain |
1063
+ | `app/steps/prepare.ts` | resolve config, provider caps, budgets and cancellation |
1064
+ | `app/steps/window.ts` | pick the prefix using calibrated synthesis and post-processing bounds |
502
1065
  | `app/steps/recover.ts` | recover full content for log-truncated messages |
503
- | `app/steps/tier.ts` | admission gate + context-pressure label (none / light / full); modes own execution depth |
1066
+ | `app/steps/tier.ts` | admission gate + context-pressure label; modes own execution depth |
504
1067
  | `app/steps/extract.ts` | pruning + deterministic extraction with incremental cache |
505
1068
  | `app/steps/synthesize.ts` | single-pass / EESV synthesis |
506
1069
  | `app/steps/verify.ts` | structural verification + repair with tool-result trust boundaries |
507
- | `app/steps/state.ts` | enrich summary with state, open loops, and recent resolved-error history |
1070
+ | `app/steps/state.ts` | enrich summary with state, open loops and resolved-error history |
1071
+ | `app/steps/visual.ts` | optional post-verification bitmap evidence inside the same yield target |
508
1072
  | `app/steps/persist.ts` | apply compaction, save fingerprint, persist state |
509
1073
  | `app/steps/metrics.ts` | record success / failure metrics |
510
1074
 
511
1075
  ### Domain layer (`src/domain/`)
512
1076
 
513
- Pure semantics — no I/O, no async, no globals.
1077
+ Pure semantics: no I/O, no async, no globals.
514
1078
 
515
1079
  | File | Responsibility |
516
1080
  | --- | --- |
517
1081
  | `domain/summary-schema.ts` | canonical section kinds + heading classification |
518
- | `domain/summary-parse.ts` | parse/render canonical H1/H2/H3 sections; merge duplicates; placement (`before`/`after`) |
519
- | `domain/tool-semantics.ts` | fine tool operation taxonomy with broad compatibility wrapper |
1082
+ | `domain/summary-parse.ts` | parse/render canonical H1/H2/H3 sections; merge duplicates; placement |
1083
+ | `domain/tool-semantics.ts` | fine tool operation taxonomy with broad compatibility wrapper; file-operation paths for superseded ordering |
1084
+ | `domain/compaction-usage.ts` | applied run's provider usage in Pi's `Usage` shape, priced per route |
520
1085
  | `domain/scrub.ts` | pure secret/PII redaction primitives and run-scoped scrubber |
1086
+ | `domain/keywords.ts` | salient-keyword extraction shared by verify and damage detection |
1087
+ | `domain/model-capacity.ts` | per-request output clamping and capacity reasons |
1088
+ | `domain/yield-gate.ts` | final yield proof and content-free `YieldGateError` |
1089
+ | `domain/provider-evaluation.ts` | advisory provider scenario matrix and route telemetry aggregation |
1090
+ | `domain/telemetry.ts` | privacy-safe aggregates, failure taxonomy and canary decision rules |
521
1091
 
522
1092
  ### Algorithm layer (`src/phases/`)
523
1093
 
@@ -533,14 +1103,22 @@ All external-world interaction.
533
1103
 
534
1104
  | File | Responsibility |
535
1105
  | --- | --- |
536
- | `infra/fs.ts` | atomic writes, advisory locks, and yielding async append/trim |
1106
+ | `infra/fs.ts` | atomic writes, advisory locks, yielding async append/trim |
537
1107
  | `infra/paths.ts` | canonical cache/session/backup paths |
538
1108
  | `infra/git.ts` | cached git-root discovery |
539
1109
  | `infra/clock.ts` | injectable wall clock |
540
- | `infra/llm-client.ts` | LLM seam, custom-Codex wire cap, and ChatGPT Codex stream watchdog |
1110
+ | `infra/llm-client.ts` | LLM seam over the session's public model runtime, custom-Codex wire cap, ChatGPT Codex stream watchdog |
541
1111
  | `infra/services.ts` | per-run services container |
542
- | `infra/session-identity.ts` | robust session-id resolution with opaque `unresolved:` fallback |
1112
+ | `infra/session-identity.ts` | session-ID resolution with opaque `unresolved:` fallback |
543
1113
  | `infra/ai-messages.ts` | validated message upcasts and recursive pre-serialization redaction |
1114
+ | `infra/context-graph.ts` | local SQLite FTS5 context graph |
1115
+ | `infra/synthesis-cache.ts` | behavior-keyed synthesis cache |
1116
+ | `infra/native-protocol.ts` | provider wire formats for native compaction and replay; no Pi imports |
1117
+ | `infra/hindsight-client.ts` | four fixed Hindsight routes; no generic request |
1118
+ | `infra/hindsight-receipts.ts` | origin/bank/project-scoped submission receipts; never evicts unconfirmed ones |
1119
+ | `infra/optional-components.ts` | read-only presence checks and exact install commands for optional peers (Mnemopi, Bun, resvg); no shell or network |
1120
+ | `infra/memory-ref.ts` | opaque backend/id refs and target-binding checks |
1121
+ | `infra/visual-renderer.ts` | lazy optional resvg renderer using `assets/DejaVuSansMono.ttf` |
544
1122
 
545
1123
  ### Utility layer (`src/utils/`)
546
1124
 
@@ -550,53 +1128,76 @@ All external-world interaction.
550
1128
  | `utils/pruning.ts` | redundancy removal on the message list |
551
1129
  | `utils/state.ts` | structured state, open loops, delta, pinned-path preservation |
552
1130
  | `utils/config.ts` | validated config loading and mtime-keyed cache |
553
- | `utils/helpers.ts` | batching, compaction boundaries, and extraction rendering helpers |
554
- | `utils/backups.ts` | backup persistence, listing, and restore message construction |
1131
+ | `utils/helpers.ts` | batching, compaction boundaries, extraction rendering helpers |
1132
+ | `utils/backups.ts` | backup persistence, listing and restore message construction |
555
1133
  | `utils/cache.ts` | metrics log + extraction prefix cache |
556
1134
  | `utils/fingerprint.ts` | project fingerprinting (language, framework, deps) |
557
1135
  | `utils/damage.ts` | post-compaction regression signals + remediation hints |
558
- | `utils/id-fingerprint.ts` | compact SHA-256 fingerprint of entry-id arrays |
1136
+ | `utils/id-fingerprint.ts` | compact SHA-256 fingerprint of entry-ID arrays |
559
1137
  | `utils/file-needles.ts` | path-suffix needles for error→file attribution |
560
1138
  | `utils/file-ref-detect.ts` | fabricated file-reference detection (SemVer-rejecting) |
561
1139
  | `utils/session-log.ts` | streaming JSONL parser for the Pi session log |
562
- | `utils/tokens.ts` | per-(provider,model) token estimation with EMA calibration |
1140
+ | `utils/tokens.ts` | per-(provider, model) token estimation with EMA calibration |
563
1141
  | `utils/type-guards.ts` | runtime validators for cross-version compatibility |
564
- | `utils/logger.ts` | stderr-prefixed log shim |
565
- | `utils/lru.ts` | small bounded LRU cache primitive |
1142
+ | `utils/logger.ts` | debug-only trace shim |
1143
+ | `utils/issues.ts` | user-facing problem reporting: once-per-session dedupe, scrubbed one-line messages, recent-issue history |
1144
+ | `utils/lru.ts` | small bounded LRU primitive |
566
1145
 
567
1146
  ### UI layer (`src/ui/`)
568
1147
 
569
1148
  | File | Responsibility |
570
1149
  | --- | --- |
571
- | `ui/overlays.ts` | progressive preflight, semantic phase progress, and approval review |
572
- | `ui/metrics-dashboard-overlay.ts` | interactive metrics dashboard screen |
573
- | `ui/backup-overlays.ts` | backup picker, viewer, and restore action screen |
1150
+ | `ui/home-overlay.ts` | keyboard Home: five task rows and readiness panel |
1151
+ | `ui/profiles.ts` | presets derived from exact persisted flags; model feasibility snapshot type |
1152
+ | `ui/overlays.ts` | progressive preflight, phase progress and approval review |
1153
+ | `ui/storage-report.ts` | read-only storage inventory rendering; no deletion verbs |
1154
+ | `ui/navigation-overlay.ts` | human session navigation: browse anchors, mark a point, search earlier sessions, confirmed return |
1155
+ | `ui/metrics-dashboard-overlay.ts` | interactive metrics dashboard |
1156
+ | `ui/backup-overlays.ts` | backup picker, viewer and restore action |
574
1157
  | `ui/open-loops-overlay.ts` | persisted open-loop manager |
575
- | `ui/settings-overlay.ts` | session/branch policy editor for agent access and automatic compaction |
1158
+ | `ui/handoff-overlay.ts` | Home handoff panel: note, seed preview, open |
1159
+ | `ui/settings-overlay.ts` | settings TUI: task-grouped categories, named values, dependency rules, branch overrides |
1160
+ | `ui/settings-complex.ts` | input, model and profile-budget rows with inline validation |
1161
+ | `ui/settings-list.ts` | settings list with per-row `r` reset and dimmed inactive rows |
1162
+ | `ui/error-format.ts` | one-line failure text with the scrubbed first provider error line |
576
1163
  | `ui/dashboard-format.ts` | shared pure formatters for metrics surfaces |
577
- | `ui/dashboard-insights.ts` | Data Confidence, quality/provider drilldowns, and canary trust views |
1164
+ | `ui/dashboard-insights.ts` | Data Confidence, quality/provider drilldowns, canary trust views |
578
1165
  | `ui/metrics-report.ts` | text report + local HTML metrics dashboard |
579
1166
 
580
- ## Design principles
1167
+ ## Host dependency boundary
1168
+
1169
+ Pi core modules are host-supplied peers requiring 0.87.1+; `typebox` is a
1170
+ wildcard peer. Neither is bundled. Development dependencies pin Pi 0.87.1 so
1171
+ native context projection is checked against the minimum supported API.
1172
+ `bun run compat:pi [version]` validates another release in an isolated
1173
+ workspace. The optional resvg renderer and the Mnemopi engine stay external to
1174
+ the bundles.
581
1175
 
582
- The architecture intentionally biases toward safety:
1176
+ ## Design principles
583
1177
 
584
- - deterministic extraction before any synthesis
585
- - adaptive exploration instead of always-on tool use
586
- - verified file lists and error context
587
- - deterministic repair before additional LLM calls
588
- - hallucinated file-reference detection
589
- - stateful tracking of open loops and cross-compaction deltas
590
- - tool-driven compaction never compacts mid-turn
591
- - summaries preserve exact file paths and identifiers where possible; saturated file lists use budgeted path tails plus collision-checked digests while scoped state retains full paths
592
- - the recent tail stays live outside the compacted region
1178
+ - Prefer hygiene and recoverable references over summarization.
1179
+ - Deterministic extraction before any synthesis; deterministic repair before
1180
+ additional LLM calls.
1181
+ - Adaptive exploration instead of always-on tool use.
1182
+ - Verified file lists and error context; hallucinated file-reference
1183
+ detection.
1184
+ - Stateful open loops and cross-compaction deltas.
1185
+ - Tool-driven compaction never compacts mid-turn; the host owns apply.
1186
+ - Summaries keep exact paths and identifiers where possible; saturated file
1187
+ lists use budgeted path tails plus collision-checked digests while scoped
1188
+ state keeps full paths.
1189
+ - The recent tail stays live outside the compacted region.
1190
+ - Memory is confined to one selected backend; explicit saves need per-fact
1191
+ host confirmation, and only the local backend indexes derived facts, from
1192
+ apply-confirmed compactions.
593
1193
 
594
1194
  ## Extending the system
595
1195
 
596
- When adding features, prefer this order:
1196
+ Prefer this order:
597
1197
 
598
- 1. extract more deterministic signal if possible
599
- 2. enrich exploration only when needed
600
- 3. keep synthesis prompts structured and bounded
601
- 4. strengthen verification before increasing model dependence
602
- 5. update tests and docs in the same change
1198
+ 1. avoid the noise or keep it recoverable before summarizing it;
1199
+ 2. extract more deterministic signal;
1200
+ 3. enrich exploration only when needed;
1201
+ 4. keep synthesis prompts structured and bounded;
1202
+ 5. strengthen verification before increasing model dependence;
1203
+ 6. update tests and docs in the same change.