pi-smart-compact 9.7.1 → 10.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/ARCHITECTURE.md +973 -372
  2. package/CHANGELOG.md +721 -0
  3. package/LICENSE +8 -0
  4. package/README.md +128 -640
  5. package/SECURITY.md +34 -12
  6. package/SUPPORT.md +26 -9
  7. package/assets/DejaVu-LICENSE.txt +187 -0
  8. package/assets/DejaVuSansMono.ttf +0 -0
  9. package/assets/README.md +26 -0
  10. package/assets/skills/context-management/SKILL.md +34 -0
  11. package/dist/app/anchor-cache.d.ts +36 -0
  12. package/dist/app/anchor-cache.d.ts.map +1 -0
  13. package/dist/app/artifact-storage.d.ts +47 -0
  14. package/dist/app/artifact-storage.d.ts.map +1 -0
  15. package/dist/app/background-preparation.d.ts +39 -0
  16. package/dist/app/background-preparation.d.ts.map +1 -0
  17. package/dist/app/compaction-commit-store.d.ts +5 -1
  18. package/dist/app/compaction-commit-store.d.ts.map +1 -1
  19. package/dist/app/context-evidence.d.ts +57 -0
  20. package/dist/app/context-evidence.d.ts.map +1 -0
  21. package/dist/app/context-guide.d.ts +3 -0
  22. package/dist/app/context-guide.d.ts.map +1 -0
  23. package/dist/app/context-operations.d.ts +106 -0
  24. package/dist/app/context-operations.d.ts.map +1 -0
  25. package/dist/app/effective-state.d.ts +23 -0
  26. package/dist/app/effective-state.d.ts.map +1 -0
  27. package/dist/app/global-settings-runtime.d.ts +3 -3
  28. package/dist/app/global-settings-runtime.d.ts.map +1 -1
  29. package/dist/app/hindsight-memory.d.ts +100 -0
  30. package/dist/app/hindsight-memory.d.ts.map +1 -0
  31. package/dist/app/host-cache-ledger.d.ts +68 -0
  32. package/dist/app/host-cache-ledger.d.ts.map +1 -0
  33. package/dist/app/lazy-tools.d.ts +36 -0
  34. package/dist/app/lazy-tools.d.ts.map +1 -0
  35. package/dist/app/memory-backend.d.ts +58 -0
  36. package/dist/app/memory-backend.d.ts.map +1 -0
  37. package/dist/app/mnemopi-memory.d.ts +13 -0
  38. package/dist/app/mnemopi-memory.d.ts.map +1 -0
  39. package/dist/app/mnemopi-protocol.d.ts +78 -0
  40. package/dist/app/mnemopi-protocol.d.ts.map +1 -0
  41. package/dist/app/mnemopi-worker.d.ts +2 -0
  42. package/dist/app/mnemopi-worker.d.ts.map +1 -0
  43. package/dist/app/model-feasibility.d.ts +20 -0
  44. package/dist/app/model-feasibility.d.ts.map +1 -0
  45. package/dist/app/native-compaction.d.ts +88 -0
  46. package/dist/app/native-compaction.d.ts.map +1 -0
  47. package/dist/app/native-continuity-bridge.d.ts.map +1 -1
  48. package/dist/app/navigation-data.d.ts +28 -0
  49. package/dist/app/navigation-data.d.ts.map +1 -0
  50. package/dist/app/navigation-types.d.ts +60 -0
  51. package/dist/app/navigation-types.d.ts.map +1 -0
  52. package/dist/app/pending-slot.d.ts +11 -1
  53. package/dist/app/pending-slot.d.ts.map +1 -1
  54. package/dist/app/preflight.d.ts.map +1 -1
  55. package/dist/app/register-context-tools.d.ts +16 -3
  56. package/dist/app/register-context-tools.d.ts.map +1 -1
  57. package/dist/app/register-navigation.d.ts +20 -0
  58. package/dist/app/register-navigation.d.ts.map +1 -0
  59. package/dist/app/register-smart-compact-command.d.ts +17 -2
  60. package/dist/app/register-smart-compact-command.d.ts.map +1 -1
  61. package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
  62. package/dist/app/register-smart-context-tool.d.ts +55 -0
  63. package/dist/app/register-smart-context-tool.d.ts.map +1 -0
  64. package/dist/app/run-context.d.ts +1 -0
  65. package/dist/app/run-context.d.ts.map +1 -1
  66. package/dist/app/run-smart-compact.d.ts +3 -3
  67. package/dist/app/run-smart-compact.d.ts.map +1 -1
  68. package/dist/app/session-handoff.d.ts +64 -0
  69. package/dist/app/session-handoff.d.ts.map +1 -0
  70. package/dist/app/session-lineage.d.ts +17 -0
  71. package/dist/app/session-lineage.d.ts.map +1 -0
  72. package/dist/app/session-run-lock.d.ts +0 -2
  73. package/dist/app/session-run-lock.d.ts.map +1 -1
  74. package/dist/app/settled-auto-trigger.d.ts +2 -0
  75. package/dist/app/settled-auto-trigger.d.ts.map +1 -1
  76. package/dist/app/smart-compact-input.d.ts +1 -1
  77. package/dist/app/smart-compact-input.d.ts.map +1 -1
  78. package/dist/app/smart-compact-policy.d.ts +1 -1
  79. package/dist/app/smart-compact-policy.d.ts.map +1 -1
  80. package/dist/app/steps/extract.d.ts +45 -1
  81. package/dist/app/steps/extract.d.ts.map +1 -1
  82. package/dist/app/steps/metrics.d.ts +1 -0
  83. package/dist/app/steps/metrics.d.ts.map +1 -1
  84. package/dist/app/steps/persist.d.ts.map +1 -1
  85. package/dist/app/steps/prepare.d.ts.map +1 -1
  86. package/dist/app/steps/recover.d.ts +9 -0
  87. package/dist/app/steps/recover.d.ts.map +1 -1
  88. package/dist/app/steps/synthesize.d.ts.map +1 -1
  89. package/dist/app/steps/tier.d.ts.map +1 -1
  90. package/dist/app/steps/verify.d.ts.map +1 -1
  91. package/dist/app/steps/visual.d.ts +4 -0
  92. package/dist/app/steps/visual.d.ts.map +1 -0
  93. package/dist/app/steps/window.d.ts.map +1 -1
  94. package/dist/app/tool-artifacts.d.ts +27 -0
  95. package/dist/app/tool-artifacts.d.ts.map +1 -0
  96. package/dist/app/visual-archive.d.ts +29 -0
  97. package/dist/app/visual-archive.d.ts.map +1 -0
  98. package/dist/constants.d.ts +96 -1
  99. package/dist/constants.d.ts.map +1 -1
  100. package/dist/domain/compaction-usage.d.ts +16 -0
  101. package/dist/domain/compaction-usage.d.ts.map +1 -0
  102. package/dist/domain/model-capacity.d.ts +12 -0
  103. package/dist/domain/model-capacity.d.ts.map +1 -0
  104. package/dist/domain/provider-evaluation.d.ts +7 -0
  105. package/dist/domain/provider-evaluation.d.ts.map +1 -1
  106. package/dist/domain/telemetry.d.ts +43 -2
  107. package/dist/domain/telemetry.d.ts.map +1 -1
  108. package/dist/domain/tool-semantics.d.ts +23 -0
  109. package/dist/domain/tool-semantics.d.ts.map +1 -1
  110. package/dist/index.d.ts.map +1 -1
  111. package/dist/index.js +15757 -6942
  112. package/dist/infra/ai-messages.d.ts +1 -1
  113. package/dist/infra/ai-messages.d.ts.map +1 -1
  114. package/dist/infra/context-graph.d.ts +38 -7
  115. package/dist/infra/context-graph.d.ts.map +1 -1
  116. package/dist/infra/fs.d.ts.map +1 -1
  117. package/dist/infra/hindsight-client.d.ts +73 -0
  118. package/dist/infra/hindsight-client.d.ts.map +1 -0
  119. package/dist/infra/hindsight-receipts.d.ts +68 -0
  120. package/dist/infra/hindsight-receipts.d.ts.map +1 -0
  121. package/dist/infra/llm-client.d.ts +26 -23
  122. package/dist/infra/llm-client.d.ts.map +1 -1
  123. package/dist/infra/memory-ref.d.ts +27 -0
  124. package/dist/infra/memory-ref.d.ts.map +1 -0
  125. package/dist/infra/native-protocol.d.ts +54 -0
  126. package/dist/infra/native-protocol.d.ts.map +1 -0
  127. package/dist/infra/optional-components.d.ts +15 -0
  128. package/dist/infra/optional-components.d.ts.map +1 -0
  129. package/dist/infra/paths.d.ts +2 -0
  130. package/dist/infra/paths.d.ts.map +1 -1
  131. package/dist/infra/services.d.ts +15 -5
  132. package/dist/infra/services.d.ts.map +1 -1
  133. package/dist/infra/visual-renderer.d.ts +16 -0
  134. package/dist/infra/visual-renderer.d.ts.map +1 -0
  135. package/dist/mnemopi-worker.js +213 -0
  136. package/dist/phases/explore.d.ts +12 -9
  137. package/dist/phases/explore.d.ts.map +1 -1
  138. package/dist/phases/synthesize.d.ts +18 -3
  139. package/dist/phases/synthesize.d.ts.map +1 -1
  140. package/dist/phases/verify.d.ts +5 -1
  141. package/dist/phases/verify.d.ts.map +1 -1
  142. package/dist/rtk.d.ts +7 -0
  143. package/dist/rtk.d.ts.map +1 -0
  144. package/dist/rtk.js +767 -0
  145. package/dist/types.d.ts +128 -4
  146. package/dist/types.d.ts.map +1 -1
  147. package/dist/ui/dashboard-format.d.ts +2 -1
  148. package/dist/ui/dashboard-format.d.ts.map +1 -1
  149. package/dist/ui/dashboard-insights.d.ts +9 -1
  150. package/dist/ui/dashboard-insights.d.ts.map +1 -1
  151. package/dist/ui/error-format.d.ts +7 -2
  152. package/dist/ui/error-format.d.ts.map +1 -1
  153. package/dist/ui/handoff-overlay.d.ts +26 -0
  154. package/dist/ui/handoff-overlay.d.ts.map +1 -0
  155. package/dist/ui/home-overlay.d.ts +54 -0
  156. package/dist/ui/home-overlay.d.ts.map +1 -0
  157. package/dist/ui/metrics-dashboard-overlay.d.ts.map +1 -1
  158. package/dist/ui/metrics-report.d.ts.map +1 -1
  159. package/dist/ui/navigation-overlay.d.ts +92 -0
  160. package/dist/ui/navigation-overlay.d.ts.map +1 -0
  161. package/dist/ui/overlays.d.ts +12 -2
  162. package/dist/ui/overlays.d.ts.map +1 -1
  163. package/dist/ui/profiles.d.ts +51 -0
  164. package/dist/ui/profiles.d.ts.map +1 -0
  165. package/dist/ui/settings-complex.d.ts +49 -3
  166. package/dist/ui/settings-complex.d.ts.map +1 -1
  167. package/dist/ui/settings-list.d.ts +28 -0
  168. package/dist/ui/settings-list.d.ts.map +1 -0
  169. package/dist/ui/settings-overlay.d.ts +13 -6
  170. package/dist/ui/settings-overlay.d.ts.map +1 -1
  171. package/dist/ui/storage-report.d.ts +4 -0
  172. package/dist/ui/storage-report.d.ts.map +1 -0
  173. package/dist/utils/backups.d.ts.map +1 -1
  174. package/dist/utils/cache.d.ts +6 -2
  175. package/dist/utils/cache.d.ts.map +1 -1
  176. package/dist/utils/config.d.ts +12 -0
  177. package/dist/utils/config.d.ts.map +1 -1
  178. package/dist/utils/helpers.d.ts.map +1 -1
  179. package/dist/utils/id-fingerprint.d.ts +3 -1
  180. package/dist/utils/id-fingerprint.d.ts.map +1 -1
  181. package/dist/utils/issues.d.ts +61 -0
  182. package/dist/utils/issues.d.ts.map +1 -0
  183. package/dist/utils/pruning.d.ts.map +1 -1
  184. package/dist/utils/session-log.d.ts +0 -2
  185. package/dist/utils/session-log.d.ts.map +1 -1
  186. package/dist/utils/state.d.ts +3 -1
  187. package/dist/utils/state.d.ts.map +1 -1
  188. package/dist/utils/tokens.d.ts +10 -2
  189. package/dist/utils/tokens.d.ts.map +1 -1
  190. package/docs/MIGRATING_TO_V8.md +7 -1
  191. package/docs/README.md +69 -0
  192. package/docs/RELEASE.md +173 -56
  193. package/docs/assets/banner.png +0 -0
  194. package/docs/assets/banner.svg +1158 -70
  195. package/docs/assets/pi-smart-compact.png +0 -0
  196. package/docs/assets/pi-smart-compact.svg +24 -0
  197. package/docs/configuration.md +637 -0
  198. package/docs/evaluation.md +408 -0
  199. package/docs/guide.md +860 -0
  200. package/docs/hindsight-memory.md +314 -0
  201. package/docs/identity.md +124 -0
  202. package/package.json +44 -11
  203. package/dist/provider-eval.js +0 -2122
  204. package/dist/provider-scenario-eval.js +0 -2900
  205. package/dist/telemetry-report.js +0 -1973
  206. package/docs/provider-evaluation-2026-08-06.md +0 -63
@@ -0,0 +1,408 @@
1
+ # Evaluation and evidence
2
+
3
+ How Pi Continuity (published as the `pi-smart-compact` package) is evaluated,
4
+ which commands exist, and what each kind of evidence can and cannot support.
5
+
6
+ This page is for maintainers and evaluators working from a **development
7
+ checkout**. The evaluation and report CLIs run directly from source with Bun;
8
+ no build is needed. The installed npm package contains only the extension, the
9
+ optional RTK entry, the Mnemopi worker and declarations in `dist/`, not these
10
+ tools.
11
+
12
+ Related pages: [user guide](./guide.md) · [configuration](./configuration.md) ·
13
+ [architecture](../ARCHITECTURE.md) · [release checklist](./RELEASE.md) ·
14
+ [Hindsight memory backend](./hindsight-memory.md).
15
+
16
+ **Contents:** [evidence classes](#offline-and-live-evidence) ·
17
+ [release gates](#deterministic-release-gates) ·
18
+ [provider routing](#provider-routing-evidence) ·
19
+ [paired task evaluation](#paired-continuation-and-memory-evaluation) ·
20
+ [telemetry and canary gates](#telemetry-and-canary-gates) ·
21
+ [replay estimates](#replay-estimates) ·
22
+ [pilots and dated reports](#pilots-and-dated-reports)
23
+
24
+ Prerequisites for a source checkout: Bun 1.4.2 (the `packageManager` pin),
25
+ Node >=22.19 with npm on `PATH` for the Node-host smokes in `release:audit`,
26
+ and `rg` (ripgrep) on `PATH` for offline `task-eval` arms, which set
27
+ `PI_OFFLINE=1` and refuse to let Pi download tools.
28
+
29
+ ## Offline and live evidence
30
+
31
+ Every result belongs to exactly one evidence class. Do not promote a claim from
32
+ one class to another.
33
+
34
+ | Class | Examples | Can show | Cannot show |
35
+ | --- | --- | --- | --- |
36
+ | Deterministic checks | `release:check`, `gate`, `bench`, unit tests | Contracts, invariants, bounded hot paths, packed-install behavior | Model quality, real savings, provider behavior |
37
+ | Offline lifecycle runs | `task-eval` (default), `session-pilot`, `context-compat-pilot`, `native-host-pilot` (default) | Real Pi `AgentSession`/tool/storage lifecycle with scripted model transport; oracle plumbing | Autonomous model decisions, live token cost or billing |
38
+ | Opt-in live probes | `provider-eval:live`, `task-eval --live`, `visual-pilot --live`, `PSC_NATIVE_LIVE=1`, Hindsight live canary | One bounded sample on one date, model and account | Production quality, other models, invoice-level cost |
39
+ | Local telemetry | `provider-eval`, `telemetry-report`, dashboards | Aggregates over runs recorded on this machine | Anything about runs not recorded, or statistical confidence |
40
+ | Replay estimates | `replay-eval` | Estimated prompt tokens and catalog-priced deltas for recorded sessions under alternative trim policies | Real savings, provider cache behavior, billing |
41
+
42
+ Rules that apply everywhere:
43
+
44
+ - A green deterministic or offline result never implies live quality, real
45
+ token savings, or a `PROMOTE` decision.
46
+ - Live runs are never part of `release:check`. Each live run needs a fresh,
47
+ explicit approval of the model, request count and token/cost exposure before
48
+ it starts. No approval carries over from an earlier run or document.
49
+ - Input guard counts are local estimates. Output reservations and wire caps
50
+ are distinct from provider-reported usage; missing usage stays unknown.
51
+ - Subscription (OAuth) usage is quota, not pay-as-you-go spend, and is never
52
+ priced at API rates.
53
+ - The verifier checks coverage of deterministic facts, structure and grounded
54
+ claims. It is a regression signal, not proof of semantic truth or of
55
+ lossless preservation.
56
+ - Dated reports are historical snapshots. Their measurements are not re-run
57
+ when the code changes; see [dated reports](#pilots-and-dated-reports).
58
+
59
+ ## Deterministic release gates
60
+
61
+ ```bash
62
+ bun install --frozen-lockfile
63
+ bun run release:check # typecheck + tests + gate + bench + build + release:audit + compat:pi latest
64
+ bun run gate # adversarial parser/verify/tool/cache/budget/scrub/damage fixtures
65
+ bun run bench # standalone hot-path p95 regression gate
66
+ bun run compat:pi 0.87.1 # locked minimum host, isolated workspace
67
+ bun run compat:pi latest # latest Pi host, isolated workspace
68
+ ```
69
+
70
+ `release:audit` packs the tarball, installs it in an isolated frozen
71
+ workspace, checks the manifest, peers and packed file list, registers the
72
+ extension and its tools under stock Node, exercises Node SQLite and the
73
+ optional Mnemopi worker on the user-installed `bun` component with no Bun on
74
+ `PATH`, and
75
+ runs the offline source CLIs (`provider-eval`, `telemetry-report`, and all four
76
+ offline `task-eval` arms) under a temporary `HOME`. Apart from package
77
+ installation and a local loopback tripwire, it makes no network or model
78
+ request. The exact release procedure is in
79
+ [the release checklist](./RELEASE.md).
80
+
81
+ A previous green run does not carry over: rerun the full `release:check` after
82
+ any change to the candidate.
83
+
84
+ Pull-request CI runs frozen install, `typecheck`, `bun test`, `gate`, `bench`,
85
+ `build` and `release:audit`. The latest-Pi compatibility job runs only on a
86
+ schedule or manual dispatch. A green CI badge therefore does not replace a
87
+ full local `release:check` on the exact candidate.
88
+
89
+ ## Provider routing evidence
90
+
91
+ All stages use the selected Pi model by default. Routing is explicit,
92
+ independent of modes, and never inferred or changed automatically:
93
+
94
+ | Stage | Config key | Fallback when unset |
95
+ | --- | --- | --- |
96
+ | Explore / segmentation | `segmentationModel` | resolved summary model |
97
+ | Synthesis / assembly | `summaryModel` | explicitly selected model, otherwise the chat model |
98
+ | Verification repair | `verificationModel` | resolved summary model |
99
+
100
+ Every run records per-stage provider, model, reliability, latency and token
101
+ telemetry, with schema-versioned verifier quality. Failed dispatched calls keep
102
+ content-free categories (authentication, rate limit, timeout, and so on), even
103
+ when a deterministic fallback completed the run. Older records without
104
+ categories stay unclassified.
105
+
106
+ ### Advisory matrix from local telemetry
107
+
108
+ ```bash
109
+ bun run provider-eval # text report, --min-samples defaults to 5
110
+ bun run provider-eval --min-samples=10 --json
111
+ ```
112
+
113
+ Reads the local metrics log and groups routes by stage, context pressure and
114
+ tool density. Only an explicitly attributed pre-repair synthesis score counts
115
+ as route quality; a run's final verifier score is never copied to Explore or
116
+ Verify. Legacy rows contribute latency and reliability, not quality. A cell is
117
+ eligible for a recommendation only with at least `--min-samples` runs, at least
118
+ 80% call reliability, at least 50% stage-local quality coverage and an average
119
+ attributed quality of at least 85; scores shrink toward neutral under low
120
+ confidence. The report is advisory only: it never edits configuration or
121
+ selects a model.
122
+
123
+ ### Opt-in live scenario probe
124
+
125
+ ```bash
126
+ # Paid API or subscription quota. Run only with explicit approval.
127
+ bun run provider-eval:live --live \
128
+ --models=provider/model-a,provider/model-b
129
+ ```
130
+
131
+ Refuses to run without both `--live` and `--models` (1 to 8 models). Each model
132
+ receives the same three bounded coding-continuity scenarios (`implementation`,
133
+ `debugging`, `continuity`) sequentially, with a 1,500-token output cap and a
134
+ 60-second timeout per call, scored by the deterministic verifier. It reports
135
+ score, latency and reported usage. Apply a route manually, and only after
136
+ representative evidence; one probe is not that evidence. The dated
137
+ [2026-08-06 baseline](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/provider-evaluation-2026-08-06.md)
138
+ (repository only) is an example of this output, not a current ranking.
139
+
140
+ ## Paired continuation and memory evaluation
141
+
142
+ `task-eval` runs the same synthetic coding task through four real stock-Pi
143
+ `AgentSession` arms:
144
+
145
+ | Arm | Meaning |
146
+ | --- | --- |
147
+ | `no-compaction` | Baseline; context only grows |
148
+ | `recoverable-hygiene` | Recoverable trimming and retrieval, no summary |
149
+ | `eesv` | Real verified compactions |
150
+ | `hybrid` | Hygiene plus verified compactions |
151
+
152
+ ```bash
153
+ bun run task-eval --help
154
+ bun run task-eval --out=/tmp/psc-task-eval-new # offline, all arms
155
+ bun run task-eval --arms=eesv,hybrid --repeats=3 --out=/tmp/psc-task-eval-eesv
156
+ bun run task-eval --repeats=1 --json --out=/tmp/psc-task-eval-smoke
157
+ ```
158
+
159
+ Options (from `--help`): `--arms`, `--repeats=1..8` (default five rounds, probes
160
+ after rounds two and five), `--out` (absolute directory; defaults to
161
+ `./task-eval-reports/<timestamp>`), `--json`, and two offline-only
162
+ cost-accounting fixtures, `--cache-warming=off|streaming|idle` and
163
+ `--background-prep`. Every arm runs with `toolLoading: "eager"`, so the arms
164
+ differ only in hygiene/offload knobs, not in on-demand tool discovery. With
165
+ `--cache-warming=idle`, arms that have automatic cleanup on may commit a
166
+ break-even trim at an idle boundary; the rebuilt context ends Pi's warming for
167
+ that entry, so zero warm replays there is expected, while the `no-compaction`
168
+ baseline keeps its refreshes.
169
+
170
+ The default transport is scripted and offline: the arm sets `PI_OFFLINE=1` so
171
+ Pi never downloads tools, and it fails before the first round when `rg`
172
+ (ripgrep, used by Pi's grep tool) is not on PATH. Independent oracles execute the
173
+ changed store/server and test process, check preserved constraints, errors and
174
+ side effects, require the newest decision, test unknown and false-premise
175
+ answers, and verify archive retrieval plus actual saved-memory recall and use.
176
+ Every arm runs the same explicitly approved synthetic memory task in temporary
177
+ stores. Reports compare requests, reported usage and cache classes, tool
178
+ interactions, compactions, retrieval, latency, hygiene and preparation
179
+ measurements. A green offline report proves lifecycle and oracle behavior, not
180
+ autonomous model quality, real token savings, or a production promotion.
181
+
182
+ ### Live mode
183
+
184
+ ```bash
185
+ # Prepare once, before approving any provider spend. The empty context sends
186
+ # no checkout, credentials or host files to the builder. Pull/apt needs network.
187
+ EMPTY_CONTEXT=$(mktemp -d)
188
+ docker build --file scripts/task-eval.Dockerfile \
189
+ --tag pi-smart-compact-task-eval:runtime-1 "$EMPTY_CONTEXT"
190
+ rmdir "$EMPTY_CONTEXT"
191
+
192
+ # Only with fresh explicit approval. PRIVATE_HOME contains only the approved
193
+ # frozen credential/selected model; PRIVATE_TMPDIR is caller-owned and private.
194
+ # Freeze the local endpoint before replacing HOME; do not copy Docker config.
195
+ DOCKER_ENDPOINT=$(docker context inspect --format '{{.Endpoints.docker.Host}}')
196
+ env -i HOME="$PRIVATE_HOME" TMPDIR="$PRIVATE_TMPDIR" PATH="$PATH" LANG=en_US.UTF-8 \
197
+ DOCKER_HOST="$DOCKER_ENDPOINT" bun run task-eval --live --models=provider/model \
198
+ --main-max-tokens=4096 --context-window=200000 \
199
+ --budget-requests=N --budget-input-tokens=N --budget-output-tokens=N \
200
+ --out=/absolute/private/report-directory
201
+ ```
202
+
203
+ Live mode requires the selected model and three positive budgets.
204
+ `--summary-model` may select another model on the same approved provider.
205
+ `--main-max-tokens` defaults to 4096. The context window is native unless an
206
+ explicit `--context-window` fixture is supplied; a fixture cannot exceed the
207
+ model catalog. Record both the fixture and native window when comparing arms.
208
+ Offline warming/preparation fixtures are refused in live mode.
209
+
210
+ Only the selected provider and selected custom model definitions enter the
211
+ runtime. API keys may come from that provider's stored credential or a literal
212
+ `models.json` key; commands and environment templates must already be resolved.
213
+ The credential is held by a read-only in-memory store. OAuth refresh material
214
+ is removed, mutation is refused, and expiring access tokens fail closed. Never
215
+ copy an entire daily auth/models file or source an ambient `.env` for a pilot.
216
+
217
+ Live tools require a local Unix-socket Docker daemon running Linux containers
218
+ and the prebuilt image above. On macOS, use a VM-backed daemon. The image is
219
+ resolved to its immutable ID before execution; startup probes must pass before
220
+ a provider call. No native fallback exists: raw `KERN_PROCARGS2` defeated the
221
+ macOS sandbox used by the rejected canary.5 candidate, even with a restricted
222
+ sysctl allowlist.
223
+
224
+ Model bash/read/write/grep, fixture snapshots and oracle processes run in fresh
225
+ containers. Only the synthetic project and owned HOME are bound; the runtime
226
+ image is read-only. No host credentials, Docker socket or ambient environment
227
+ are mounted/passed. Network and PID namespaces isolate host/sibling processes.
228
+ Capabilities are dropped, privilege elevation is disabled, and each container
229
+ is bounded to 64 PIDs, 512 MiB and one CPU. Calls are serialized. Tools execute
230
+ on Linux even when the evaluator is on macOS; record the image ID and policy
231
+ hash with the evidence. Whole-arm latency includes container startup/teardown.
232
+ These are evaluation prerequisites, not extension runtime dependencies.
233
+
234
+ Each container is created before it is started, avoiding a cancellation race
235
+ that could start a late container. Normal exit, timeout, abort and disposal
236
+ remove its whole process tree, including detached children. File reads have a
237
+ 30-second command deadline and an 8 MiB output ceiling; local Docker control
238
+ operations have a separate 15-second deadline. Normal completion/failure
239
+ removes owned arm and child scratch directories. The caller must remove its
240
+ frozen-credential HOME/TMPDIR on signals; SIGKILL or a failed daemon can leave
241
+ owned resources requiring manual cleanup. Never delete unrelated containers.
242
+
243
+ The SDK fetch guard allows only the selected origin, refuses redirects, and
244
+ reserves a request's full output cap before dispatch. If that cap does not fit,
245
+ it refuses rather than shortening the response. Input is a character-derived
246
+ estimate, not a billed-token hard limit. Reports snapshot each arm after all
247
+ usage accounting settles, including failed dispatched calls, main/summary
248
+ classes, HTTP status, provider input/output/cache fields and per-request
249
+ elapsed time. Anthropic input excludes its separate cache fields; missing
250
+ fields remain null. Whole-arm time also includes tools and local oracles.
251
+ `--json` writes the same report to stdout and `task-eval-report.json`.
252
+
253
+ ChatGPT/Codex is refused by default: `--accept-codex-soft-cap` explicitly
254
+ selects unbounded output and can never satisfy a hard output-token budget.
255
+ Neither an offline loopback smoke nor one live paired sample satisfies the
256
+ production promotion gates below.
257
+
258
+ ## Telemetry and canary gates
259
+
260
+ Raw local JSONL stays available to the interactive dashboard. The aggregate
261
+ report contains no session or project IDs, prompts, summaries, paths or error
262
+ text:
263
+
264
+ ```bash
265
+ bun run telemetry-report # --min-canary-runs defaults to 20 (minimum 5)
266
+ bun run telemetry-report --min-canary-runs=20 --json
267
+ ```
268
+
269
+ Failures use a stable content-free taxonomy (cancelled, timeout, rate limit,
270
+ authentication, budget, output limit, provider, persistence, validation,
271
+ verification, yield, internal). Verification and yield failures keep only
272
+ content-free diagnostics.
273
+
274
+ ### Cohorts
275
+
276
+ Set `telemetryChannel: "canary"` only on an externally selected canary
277
+ installation; the default is `stable`. Canary evidence is limited to schema-v2
278
+ runs of the version under evaluation. Entries without an explicit release
279
+ channel are reported as unattributed and excluded from both cohorts, never
280
+ treated as stable. Reports separate total, attempted and host-confirmed applied
281
+ runs. Dry runs, staged or discarded preparations and voluntary cancellations are
282
+ not applied evidence; cancellations are neutral, while real timeouts and
283
+ provider failures count.
284
+
285
+ ### Decision rules
286
+
287
+ The report returns `ROLLBACK`, `HOLD` or `PROMOTE` (implemented in
288
+ `src/domain/telemetry.ts`). It never deploys, rolls back or edits
289
+ configuration; promotion authority stays with the release owner.
290
+
291
+ Rollback is evaluated once the canary has at least three attempted runs. Any
292
+ trigger returns `ROLLBACK`:
293
+
294
+ | Metric | Trigger |
295
+ | --- | --- |
296
+ | Failure rate | Canary above 5% absolute, or at least 5pp above stable |
297
+ | Verifier quality | Canary average below 85, or at least 5 points below stable |
298
+ | p95 duration | At least +50% versus stable (stable p95 at least 1 s) |
299
+ | Average tokens | At least +50% versus stable (stable average at least 1,000) |
300
+ | Fallback rate | At least +10pp versus stable |
301
+ | Damage rate | At least +10pp versus stable |
302
+
303
+ Without a trigger, `PROMOTE` requires all of the following; otherwise the
304
+ result is `HOLD` with the first missing reason:
305
+
306
+ - at least `--min-canary-runs` (default 20) host-applied canary runs;
307
+ - a stable baseline of at least `max(20, --min-canary-runs)` applied runs;
308
+ - at least 70% verifier-quality coverage in both cohorts;
309
+ - at least 70% run-correlated damage-observation coverage in both cohorts;
310
+ - canary average verifier quality of at least 85 and success of at least 95%;
311
+ - canary data confidence of at least 85.
312
+
313
+ Damage observations join their originating compaction by run ID and are
314
+ deduplicated per run. Missing observations are missing evidence, never clean
315
+ runs.
316
+
317
+ ### Two confidence scores
318
+
319
+ Both are completeness heuristics, not statistical confidence:
320
+
321
+ | Score | Where | Components |
322
+ | --- | --- | --- |
323
+ | Canary data confidence | `telemetry-report` | canary sample 25, stable sample 15, canary quality coverage 20, canary damage coverage 20, stable damage coverage 20 |
324
+ | Dashboard Data Confidence | `/smart-compact dashboard` | recent sample 25, schema-v2 share 25, quality coverage 20, field completeness 20, freshness 10 (last complete run within 7 days) |
325
+
326
+ Legacy or incompatible evidence stays missing and lowers both. Dashboards also
327
+ show initial score, patch, LLM-repair and deterministic-floor provenance instead
328
+ of hiding repair behind the final score.
329
+
330
+ ### Preparation and cost measurements
331
+
332
+ Completed background work that is discarded is recorded once with its reason
333
+ and cost, never as applied evidence; graceful shutdown waits for that record.
334
+ Reports show used and discarded preparation, time to ready, wait to use or
335
+ discard, reuse rate and discarded spend. Stage routes keep input, cache-read,
336
+ cache-write and output classes and mark estimated usage. These are measurements
337
+ only: savings floors, cooldowns, pressure gates and TTLs are unchanged by them.
338
+
339
+ ### Replay estimates
340
+
341
+ `bun run replay-eval --sessions=<dir|file[,file…]> [--out=/abs/dir] [--json]
342
+ [--break-even=8,16,24,48] [--rebuild-min=16384] [--limit=N] [--since=DAYS]
343
+ [--progress]` replays recorded session files in memory (never written; a file
344
+ that changes while it is read, such as a live session, is skipped and counted;
345
+ `--since` keeps files modified in the last `DAYS` days; `--progress` prints one
346
+ line per file to stderr) and judges the automatic-trim timing constants
347
+ `AUTO_TRIM_BREAK_EVEN_REQUESTS` and `REBUILD_MIN_TOKENS`. Recorded automatic trims are removed first so every
348
+ policy starts from the same history. Policies:
349
+
350
+ - `none`: no automatic trim.
351
+ - `pressure`: the old rule; a ready batch commits at a turn boundary only when
352
+ the estimated prompt reaches 0.8 × the catalog context window. Live, the gate
353
+ is the configured start percentage of `min(window, maxContextTokens)` against
354
+ Pi's reported usage, which includes the system prompt and tool definitions, so
355
+ pressure fires later in replay than live.
356
+ - `timed-<N>`: the current rule with `N` in place of the break-even limit;
357
+ pressure commits, `N* ≤ N` commits (`break-even`), otherwise the batch is held
358
+ and applied at the first request after the previous request's cache lifetime
359
+ (`cold`). Planning, cooldown and protected prefixes use the extension's own
360
+ `planContextTrim`/`trimEntries`/`trimTokens`.
361
+
362
+ Cost model per request: the projected context is estimated per message; the
363
+ cached prefix is the longest run of identical projected messages shared with
364
+ the previous request (0 after the cache lifetime or a model switch);
365
+ `uncached = prompt − cached`; a rebuild is `uncached ≥ max(--rebuild-min,
366
+ 0.5 × prompt)`; price = `cacheRead × cached + (cacheWrite, else input) ×
367
+ uncached` at catalog rates. System prompt, tool definitions and output are
368
+ identical across policies and excluded. The recorded baseline (usage and
369
+ `usage.cost.total`) is the only measured figure; subscription requests report
370
+ tokens only. `--json` writes `<out>/replay-eval.json` with session ids and
371
+ numbers, no message text or paths. Absolute estimates are not calibrated to
372
+ recorded usage (a first run over three Codex sessions estimated about 4× the
373
+ recorded `input + cacheRead`); compare policies by their Δ, never by the
374
+ absolute column.
375
+
376
+ ## Pilots and dated reports
377
+
378
+ Pilot scripts are development tools with narrow purposes. Offline defaults make
379
+ no provider request.
380
+
381
+ | Command | Default | Purpose |
382
+ | --- | --- | --- |
383
+ | `bun scripts/session-pilot.ts` | Offline | Real `AgentSession` with scripted model transport, Pi Continuity only: tools, anchor and pivot with carryover, trimming/retrieval, rewind, compaction and reopen |
384
+ | `PSC_CLAUDE_OAUTH_EXTENSION=<pi-claude-oauth-adapter>/extensions/index.ts bun scripts/native-host-pilot.ts` | Offline fake provider | Provider-native compaction on stock Pi; the Anthropic OAuth route needs the standalone adapter (patched final-payload build for billing on nested requests). `PSC_NATIVE_ROUTES=codex-oauth,openai-api-key` runs without it and `PSC_PILOT_SHORT=1` shortens each route. `PSC_NATIVE_LIVE=1` sends real, ledger-capped requests and needs explicit approval |
385
+ | `bun run scripts/rtk-pilot.ts /absolute/path/to/rtk` | Local only | Synthetic RTK rewrite contract; characters, not provider tokens |
386
+ | `bun run scripts/visual-pilot.ts --model=provider/id` | Offline planning | Bitmap versus text evidence; `--live` authorizes at most 9 sequential requests and needs explicit approval |
387
+ | `PSC_HINDSIGHT_LIVE=1 … bun run test/hindsight-live.canary.ts` | Not run by `bun test` | Live Hindsight contract with a hard call budget; see [Hindsight memory](./hindsight-memory.md#tests) |
388
+
389
+ Dated reports record what was measured on their date, with the code and host
390
+ versions stated inside. They are kept for provenance and are not updated
391
+ retroactively; later findings are added as dated addenda. Current behavior is
392
+ described in the [guide](./guide.md), [configuration](./configuration.md) and
393
+ [architecture](../ARCHITECTURE.md). Reports live in the repository only; the
394
+ npm package does not include `docs/reports/` or `docs/findings/`. Pilot
395
+ reports have a machine-readable `.json` companion beside them.
396
+
397
+ | Report | Scope |
398
+ | --- | --- |
399
+ | [Provider evaluation baseline, 2026-08-06](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/provider-evaluation-2026-08-06.md) | Live three-scenario probe across five models; advisory; 2026-09-25 addendum on route-report token semantics |
400
+ | [Context hygiene and continuity, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/context-hygiene-2026-09-24.md) | Hygiene design and offline experiments |
401
+ | [Hindsight and provider-native compaction research, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/hindsight-native-compaction-research-2026-09-24.md) | Pre-implementation research plus later measured results |
402
+ | [Full AgentSession offline pilot, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/session-pilot-2026-09-24.md) | Scripted-transport lifecycle pilot; 2026-09-25 candidate and `9.8.0-canary.1` follow-ups |
403
+ | [Visual evidence pilot, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/visual-pilot-2026-09-24.md) | Live synthetic bitmap-versus-text reading pilot on one model |
404
+
405
+ External review findings, one folder per reviewer, are indexed in
406
+ [`docs/findings/`](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/findings/README.md).
407
+ They are advisory analyses of a specific revision range, not release gates or
408
+ evidence of current behavior.