pi-ui-extend 1.0.39 → 1.0.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/README.md +1 -1
  2. package/dist/app/commands/command-registry.js +2 -2
  3. package/dist/app/commands/command-session-actions.d.ts +0 -1
  4. package/dist/app/commands/command-session-actions.js +22 -13
  5. package/dist/app/icons.d.ts +14 -0
  6. package/dist/app/icons.js +33 -0
  7. package/dist/app/rendering/conversation-tool-renderer.js +2 -2
  8. package/dist/app/rendering/dcp-stats.d.ts +6 -1
  9. package/dist/app/rendering/dcp-stats.js +214 -46
  10. package/dist/app/rendering/editor-panels.js +8 -5
  11. package/dist/app/session/lazy-session-manager.js +12 -1
  12. package/dist/app/session/tabs-controller.d.ts +2 -5
  13. package/dist/app/session/tabs-controller.js +12 -21
  14. package/dist/app/subagents/subagents-model.d.ts +14 -1
  15. package/dist/app/subagents/subagents-model.js +34 -15
  16. package/dist/app/types.d.ts +2 -0
  17. package/dist/bundled-extensions/session-title/config.js +1 -1
  18. package/dist/markdown-format.js +27 -9
  19. package/dist/schemas/pi-tools-suite-schema.d.ts +29 -16
  20. package/dist/schemas/pi-tools-suite-schema.js +46 -31
  21. package/external/pi-tools-suite/README.md +392 -52
  22. package/external/pi-tools-suite/docs/browser-qa-subagent.md +31 -21
  23. package/external/pi-tools-suite/docs/context-gateway-p00-adr.md +216 -0
  24. package/external/pi-tools-suite/docs/context-gateway-p01n-gate-review.md +122 -0
  25. package/external/pi-tools-suite/docs/context-gateway-p01n-measurement.md +133 -0
  26. package/external/pi-tools-suite/docs/context-gateway-p01r-ra-evidence.md +111 -0
  27. package/external/pi-tools-suite/docs/context-gateway-p01r-rb-evidence.md +100 -0
  28. package/external/pi-tools-suite/docs/context-gateway-p01r-rc-evidence.md +69 -0
  29. package/external/pi-tools-suite/docs/context-gateway-p01r-rd-evidence.md +100 -0
  30. package/external/pi-tools-suite/docs/context-gateway-p01r-re-evidence.md +74 -0
  31. package/external/pi-tools-suite/docs/context-gateway-p01r-rf-evidence.md +153 -0
  32. package/external/pi-tools-suite/docs/context-gateway-p01r-rg-evidence.md +235 -0
  33. package/external/pi-tools-suite/docs/evals.md +684 -0
  34. package/external/pi-tools-suite/docs/subagent-model-pools.md +109 -0
  35. package/external/pi-tools-suite/package.json +10 -3
  36. package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/scripts/browser-qa-runner.mjs +82 -1
  37. package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/SKILL.md → agents/browser-qa.md} +261 -12
  38. package/external/pi-tools-suite/src/async-subagents/agents/implement.md +20 -0
  39. package/external/pi-tools-suite/src/async-subagents/agents/oracle.md +16 -0
  40. package/external/pi-tools-suite/src/async-subagents/agents/research.md +18 -0
  41. package/external/pi-tools-suite/src/async-subagents/agents/verify.md +18 -0
  42. package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +27 -243
  43. package/external/pi-tools-suite/src/async-subagents/commands.ts +6 -2
  44. package/external/pi-tools-suite/src/async-subagents/core/agent-catalog.ts +41 -0
  45. package/external/pi-tools-suite/src/async-subagents/core/agent-strategy.ts +13 -93
  46. package/external/pi-tools-suite/src/async-subagents/core/agents-dir.ts +494 -0
  47. package/external/pi-tools-suite/src/async-subagents/core/browser-qa.ts +9 -0
  48. package/external/pi-tools-suite/src/async-subagents/core/config.ts +200 -143
  49. package/external/pi-tools-suite/src/async-subagents/core/model-fallback.ts +1 -1
  50. package/external/pi-tools-suite/src/async-subagents/core/model-selection.ts +54 -0
  51. package/external/pi-tools-suite/src/async-subagents/core/prompt.ts +7 -6
  52. package/external/pi-tools-suite/src/async-subagents/core/routing.ts +52 -45
  53. package/external/pi-tools-suite/src/async-subagents/core/spawn.ts +12 -4
  54. package/external/pi-tools-suite/src/async-subagents/index.ts +11 -1
  55. package/external/pi-tools-suite/src/async-subagents/lib.ts +6 -2
  56. package/external/pi-tools-suite/src/async-subagents/tools/spawn.ts +46 -18
  57. package/external/pi-tools-suite/src/async-subagents/tools/subagents.ts +3 -2
  58. package/external/pi-tools-suite/src/async-subagents/types.ts +2 -0
  59. package/external/pi-tools-suite/src/coding-discipline/index.ts +41 -142
  60. package/external/pi-tools-suite/src/config.ts +1 -22
  61. package/external/pi-tools-suite/src/context-gateway/accounting.ts +151 -0
  62. package/external/pi-tools-suite/src/context-gateway/config.ts +111 -0
  63. package/external/pi-tools-suite/src/context-gateway/index.ts +160 -0
  64. package/external/pi-tools-suite/src/context-gateway/metadata-normalization.ts +88 -0
  65. package/external/pi-tools-suite/src/context-gateway/storeless-capabilities.ts +89 -0
  66. package/external/pi-tools-suite/src/context-gateway/telemetry.ts +429 -0
  67. package/external/pi-tools-suite/src/context-gateway/test-output-parser.ts +326 -0
  68. package/external/pi-tools-suite/src/context-gateway/types.ts +152 -0
  69. package/external/pi-tools-suite/src/dcp/auto-compress-budget.ts +106 -0
  70. package/external/pi-tools-suite/src/dcp/auto-compress.ts +810 -106
  71. package/external/pi-tools-suite/src/dcp/commands.ts +64 -139
  72. package/external/pi-tools-suite/src/dcp/compress-tool.ts +369 -35
  73. package/external/pi-tools-suite/src/dcp/compression-blocks.ts +510 -64
  74. package/external/pi-tools-suite/src/dcp/compression-preview.ts +113 -0
  75. package/external/pi-tools-suite/src/dcp/compression-progress.ts +70 -0
  76. package/external/pi-tools-suite/src/dcp/config.ts +36 -61
  77. package/external/pi-tools-suite/src/dcp/conversation-index.ts +421 -0
  78. package/external/pi-tools-suite/src/dcp/debug-log.ts +7 -5
  79. package/external/pi-tools-suite/src/dcp/index.ts +617 -203
  80. package/external/pi-tools-suite/src/dcp/journal.ts +566 -0
  81. package/external/pi-tools-suite/src/dcp/progress-controller.ts +244 -0
  82. package/external/pi-tools-suite/src/dcp/prompts.ts +10 -7
  83. package/external/pi-tools-suite/src/dcp/provider-tool-results.ts +189 -0
  84. package/external/pi-tools-suite/src/dcp/pruner-candidates.ts +298 -78
  85. package/external/pi-tools-suite/src/dcp/pruner-compression-blocks.ts +173 -281
  86. package/external/pi-tools-suite/src/dcp/pruner-emergency.ts +2 -4
  87. package/external/pi-tools-suite/src/dcp/pruner-message-ids.ts +17 -5
  88. package/external/pi-tools-suite/src/dcp/pruner-metadata.ts +11 -1
  89. package/external/pi-tools-suite/src/dcp/pruner-nudge.ts +30 -82
  90. package/external/pi-tools-suite/src/dcp/pruner-tools.ts +22 -133
  91. package/external/pi-tools-suite/src/dcp/pruner.ts +18 -33
  92. package/external/pi-tools-suite/src/dcp/recovery.ts +129 -0
  93. package/external/pi-tools-suite/src/dcp/shadow-plan.ts +127 -0
  94. package/external/pi-tools-suite/src/dcp/state-transaction.ts +102 -0
  95. package/external/pi-tools-suite/src/dcp/state.ts +158 -580
  96. package/external/pi-tools-suite/src/dcp/ui.ts +1 -0
  97. package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +55 -220
  98. package/external/pi-tools-suite/src/index.ts +9 -0
  99. package/external/pi-tools-suite/src/model-tools/index.ts +76 -42
  100. package/external/pi-tools-suite/src/repo-discovery/index.ts +84 -18
  101. package/external/pi-tools-suite/src/repo-discovery/native-compact.ts +458 -0
  102. package/external/pi-tools-suite/src/session-recovery/index.ts +189 -43
  103. package/external/pi-tools-suite/src/tool-descriptions.ts +43 -38
  104. package/external/pi-tools-suite/src/truncation-metadata-normalizer/index.ts +17 -0
  105. package/package.json +6 -6
  106. package/schemas/pi-tools-suite.json +159 -78
  107. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/references/auth-scaffold-spec.md +0 -78
  108. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/references/qa-design.md +0 -223
  109. package/external/pi-tools-suite/src/dcp/state-persistence.ts +0 -195
  110. /package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/references → agents/browser-qa/examples}/qa-auth.example.jsonc +0 -0
  111. /package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/references → agents/browser-qa/examples}/qa-flow.example.jsonc +0 -0
  112. /package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/vendor/fflate.LICENSE +0 -0
  113. /package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/vendor/fflate.mjs +0 -0
@@ -0,0 +1,216 @@
1
+ # Context Gateway P00 ADR: tool-result stage ordering
2
+
3
+ <!-- markdownlint-disable MD013 -->
4
+
5
+ > Status: accepted implementation decision for future Context Gateway work.
6
+ > This ADR does not enable Gateway, change user configuration, or certify rollout.
7
+ > Evidence baseline: repository `daa1b06`, installed Pi SDK `0.85.1`, 7 September 2026.
8
+
9
+ ## Context
10
+
11
+ The installed extension runner chains `tool_result` handlers in registration
12
+ order. The suite currently has four independent `tool_result` registrations:
13
+ LSP, comment-checker, DCP, and the opt-in credential firewall. Their module load
14
+ order is LSP → comment-checker → DCP → credential firewall.
15
+
16
+ That order is unsuitable for Gateway enforce mode:
17
+
18
+ - mutation diagnostics must be present before Gateway renders a bounded result;
19
+ - session-hygiene redaction, when enabled, must happen before durable capture;
20
+ - Gateway must shape before DCP records the delivered result;
21
+ - changing the global `MODULES` order would also change unrelated lifecycle and
22
+ provider hooks, including the intentionally late provider firewall.
23
+
24
+ P00 tests also prove that throwing from a result handler is fail-open, and that
25
+ an in-flight extension tool can cross an extension reload: the old wrapper then
26
+ fails on its stale runner while the new runner handles the resulting error.
27
+ Therefore current active-runner state is not an origin binding.
28
+
29
+ ## Decision
30
+
31
+ For the current **P01-R storeless scope**, keep the existing independent
32
+ `tool_result` chain. A suite-local coordinator is **not** introduced merely to
33
+ make the architecture look uniform. It becomes conditional future work only if
34
+ a store-backed/enforce stage is authorised and a concrete conflict between
35
+ enabled suite-owned result handlers cannot be expressed safely through the
36
+ verified event-specific order.
37
+
38
+ The verified storeless order is:
39
+
40
+ 1. LSP and comment-checker result enrichers.
41
+ 2. Context Gateway `observe` (passive; no result patch).
42
+ 3. Optional `truncation-metadata-normalizer` (details-only duplicate cleanup).
43
+ 4. Other downstream result observers/modifiers; the current legacy DCP happens
44
+ to be in this position but P01-R does not depend on DCP state or persistence.
45
+ 5. Opt-in credential-firewall session-hygiene redaction.
46
+
47
+ For provider hooks, credential-firewall redaction runs before
48
+ `codex-reasoning-fix`, and `codex-reasoning-fix` remains the final
49
+ `before_provider_request` sanitizer.
50
+
51
+ This order is covered both by the `MODULES` contract and by an executable
52
+ `ExtensionRunner` chain with a DCP-free fake downstream observer. The optional
53
+ normalizer removes only redundant metadata; the later firewall still owns secret
54
+ redaction. With firewall session hygiene disabled, the normalizer does not redact
55
+ or otherwise alter visible content.
56
+
57
+ If a future enforce/store stage is authorised, the previously proposed
58
+ coordinator remains the candidate design: modules participating in it must not
59
+ also register an independent `tool_result` handler, while their non-result hooks
60
+ remain registered normally. That future decision must be revalidated against the
61
+ then-current DCP/session-recovery implementation rather than copied mechanically
62
+ from the legacy chain.
63
+
64
+ The **future enforce coordinator candidate** order is:
65
+
66
+ 1. **Enrich** — LSP mutation diagnostics, then comment-checker output.
67
+ 2. **Session-hygiene sanitize** — the credential-firewall tool-result transform
68
+ when that existing opt-in policy is enabled.
69
+ 3. **Gateway sanitize/capture/render** — apply Gateway's archive-specific
70
+ permitted-snapshot sanitizer, consume any trusted pre-truncation capture
71
+ handle, publish if policy requires it, and return passthrough/exact/compact/
72
+ degraded delivery.
73
+ 4. **DCP observe** — record only the result that will be delivered after the
74
+ preceding stages.
75
+
76
+ The credential firewall's `before_provider_request` hook stays in its existing
77
+ late provider position. `codex-reasoning-fix` remains the last provider-payload
78
+ sanitizer. The coordinator is event-specific; it is not a replacement for
79
+ global module ordering.
80
+
81
+ ### P01-R storeless ordering before any coordinator exists
82
+
83
+ The accepted coordinator above is conditional future **store-backed enforce**
84
+ work. P01-R does not add it merely to clean redundant SDK metadata. The checked
85
+ storeless path keeps independent handlers and the current suite registration
86
+ order:
87
+
88
+ 1. suite-owned result enrichers such as LSP/comment-checker;
89
+ 2. Context Gateway `observe`, which records the original result boundary and
90
+ does not transform it;
91
+ 3. optional `truncation-metadata-normalizer`, which may remove only a proven
92
+ duplicate `details.truncation.content` while preserving the delivered body,
93
+ outcome and every other details field;
94
+ 4. downstream result observers. The R-B contract uses a fake observer with no
95
+ DCP imports so this boundary is not coupled to the current or future DCP
96
+ implementation;
97
+ 5. optional credential-firewall `tool_result` hygiene, which remains the
98
+ security transform for delivered result content/details when that module is
99
+ enabled;
100
+ 6. provider hooks later run in their normal order: credential firewall before
101
+ the final `codex-reasoning-fix` payload sanitizer.
102
+
103
+ This ordering intentionally lets `observe` measure the pre-cleanup metadata
104
+ boundary. The normalizer is not a security boundary and does not restore or
105
+ create source bytes. If firewall hygiene is enabled after it, secrets still
106
+ present in delivered content/remaining details are redacted before JSONL/next
107
+ provider context. When hygiene is disabled, the normalizer must not silently
108
+ pretend those bytes were redacted.
109
+
110
+ The normalizer's tool-name/SDK-shape allowlist is a compatibility guard, not
111
+ trusted provenance. `tool_result` carries no authenticated extension owner, so
112
+ a replacement tool can reuse `Read`/shell/`ast_grep` names. Exact duplicate
113
+ cleanup can remain semantics-preserving while all stronger capability/lifetime
114
+ claims for such a replacement stay **limited**. A future capture/store path
115
+ cannot use this name/shape test as permission to open a path or publish a
116
+ snapshot.
117
+
118
+ R-B does **not** introduce a coordinator because no conflict requiring one is
119
+ present in the storeless path. Existing tests prove independent handler
120
+ composition, exception fail-open behavior and the fact that a later extension
121
+ can reinsert raw content. Therefore hard enforce remains unsupported; if a
122
+ future store-backed stage needs different security ordering, it must implement
123
+ the event-specific coordinator without leaving duplicate independent result
124
+ registrations active.
125
+
126
+ ## P01-R same-name replacement limitation
127
+
128
+ The metadata normalizer can conservatively reject unknown tool names and
129
+ malformed/non-matching truncation metadata, but the generic `tool_result` event
130
+ does not carry a cryptographic or host-owned identity proving which tool
131
+ definition produced a result. A separately loaded replacement that deliberately
132
+ uses a measured name such as `read` plus the exact SDK truncation shape is
133
+ therefore **not independently provenance-certified** by name+shape alone. P01-R
134
+ support claims are limited to the verified suite/SDK tool combinations. Strict
135
+ enforce must not generalise this heuristic to arbitrary replacements without a
136
+ host-owned tool-definition identity seam.
137
+
138
+ ## Failure contract
139
+
140
+ Gateway stage failure must not rely on `throw`. A Gateway failure returns an
141
+ explicit bounded degraded result that preserves the real execution outcome and
142
+ states that archival/retrieval is unavailable. A successful mutation must never
143
+ be reported as "not executed" merely because publication failed.
144
+
145
+ Enrichment or optional session-hygiene failures keep their existing semantics
146
+ until their coordinator adapters are implemented and tested. DCP observation
147
+ must never be allowed to restore a raw source after Gateway shaping.
148
+
149
+ ## Origin binding and reload
150
+
151
+ `toolCallId` is necessary but not sufficient. Gateway execution identity must
152
+ also bind the originating session/workspace and attempt/runtime epoch before an
153
+ await boundary. The active tab or current extension runner after execution is
154
+ not authoritative.
155
+
156
+ The installed SDK currently makes extension reload during an in-flight custom
157
+ tool a limited path: `wrapRegisteredTool()` touches the old runner after the
158
+ tool returns and can turn the original completion into a stale-runner error.
159
+ Gateway strict-enforce support for such cross-reload executions is therefore
160
+ **not claimed**. Wrapper-level capture may preserve permitted bytes as an
161
+ orphaned source, but it must not fabricate a successful delivered result.
162
+
163
+ At the app layer, tab ownership already uses runtime/session/generation guards.
164
+ Gateway bindings should use equivalent host-owned identity rather than the
165
+ currently active tab.
166
+
167
+ ## Capture implications
168
+
169
+ - Built-in `read`: generic `tool_result` is after truncation and there is no
170
+ full-output handle. Treat as limited unless a supported execution wrapper is
171
+ introduced; exact native paging remains the primary path.
172
+ - Built-in `bash`: generic result is bounded, but a trusted `fullOutputPath`
173
+ exists on successful truncated output. Timeout/abort throw paths preserve a
174
+ text prefix/status but lose structured details.
175
+ - `repo_*`: add any future capture seam inside the suite wrapper before
176
+ `truncateOutput`; the generic result hook cannot recover omitted bytes.
177
+ - `ast_grep`: the suite wrapper can capture before truncation and already
178
+ exposes a full-output temp path when truncated.
179
+ - Unknown/custom tools remain limited unless their concrete execution path is
180
+ separately proven.
181
+
182
+ ## Unsupported integrations in the current evidence
183
+
184
+ The current Pix ACP `session/new` implementation consumes `cwd` and `_meta` and
185
+ does not plumb the protocol `mcpServers` field into a local MCP execution path.
186
+ MCP result capture is therefore unsupported, not implicitly covered by the
187
+ generic hook.
188
+
189
+ Browser QA runs in a child Pi process launched with `--no-extensions` plus a
190
+ restricted extension set. The parent Gateway cannot observe Playwright DOM,
191
+ network, or console bytes as parent `tool_result` events. Future integration may
192
+ reuse completed subagent artifacts; it is not a direct browser adapter.
193
+
194
+ ## Storage/platform boundary
195
+
196
+ P00 chooses no durable publication primitive. Current tests run on macOS and do
197
+ not certify directory rename/fsync/no-clobber or cross-process quota behavior on
198
+ Linux or Windows. Until P02 fault/platform tests exist, Gateway must not describe
199
+ its storage as crash-durable across the supported platform matrix.
200
+
201
+ ## Third-party extensions
202
+
203
+ The suite coordinator controls only suite-owned stages. A separately loaded
204
+ third-party extension may register a later `tool_result` modifier and reinsert
205
+ large/raw data. Strict enforce is unsupported for an unverified extension
206
+ combination until a final session/provider-boundary test proves that the raw
207
+ sentinel cannot reappear.
208
+
209
+ ## Evidence
210
+
211
+ - `test/context-gateway/sdk-pipeline.test.ts`
212
+ - `test/context-gateway/capture-contracts.test.ts`
213
+ - `test/context-gateway/lifecycle-contracts.test.ts`
214
+ - `test/context-gateway/provider-serialization.test.ts`
215
+ - `test/context-gateway/p00-capabilities.md`
216
+ - root `tests/tabs-controller.test.ts`, late origin-tab result contract
@@ -0,0 +1,122 @@
1
+ # Context Gateway P01-N gate review
2
+
3
+ <!-- markdownlint-disable MD013 -->
4
+
5
+ > Decision date: 7 September 2026.
6
+ > Baseline: repository `daa1b06`, Pi SDK `0.85.1`.
7
+ > Decision: **Do not start P02 immutable-store implementation now. The independent P01-R storeless track in plan 31 was subsequently completed without unlocking P02.**
8
+
9
+ ## Scope of this decision
10
+
11
+ This is a gate decision, not a claim that a Gateway store will never be useful. The question is narrower: does the current evidence justify paying the P02 storage/security/quota complexity now?
12
+
13
+ It does not. Native Compact already removes a large fraction of deterministic repo-result delivery, while the remaining large result classes have not yet been shown to require durable artifact semantics rather than existing native paging/temp-output mechanisms. The authorized live paired gate also does not show a consistent total-cost win large enough to justify adding a store.
14
+
15
+ ## Current Native Compact evidence
16
+
17
+ The existing paired synthetic corpus in `test/context-gateway/benchmark.test.ts` reports:
18
+
19
+ | Metric | Baseline / Prompt runtime | Native Compact | Delta |
20
+ | --- | ---: | ---: | ---: |
21
+ | Repo prompt guidance chars | 3326 historical | 2294 current | -1032 (-31.0%) |
22
+ | Delivered repo-result bytes | 71,025 | 27,519 | -43,506 (-61.25%) |
23
+ | Tool calls | 6 | 8 | +2 |
24
+ | Continuation calls | 1 | 3 | +2 |
25
+ | Policy refusals | 0 | 0 | 0 |
26
+ | Full overrides | 0 | 0 | 0 |
27
+ | Critical facts recovered | 8 / 8 | 8 / 8 | unchanged |
28
+
29
+ The byte ratio is `0.3874551214`. The reduction is therefore not free: structure/AST recovery uses additional native cursor calls. Those calls must remain in the total task cost instead of being hidden behind an initial-output-only metric.
30
+
31
+ The policy function itself is not a material processing bottleneck in the current synthetic microbenchmark. After warm-up, seven runs of 100,000 policy applications measured roughly `0.255–0.338 µs/call`, with six of seven runs around `0.255–0.267 µs/call`. This is only policy CPU overhead, not end-to-end tool, filesystem, provider, or task latency.
32
+
33
+ ## Authorized live paired gate
34
+
35
+ The live gate was run with the explicitly authorized configured model `zai/glm-5.3`. Every Prompt Compact and Native Compact arm passed its behavioral assertions.
36
+
37
+ The first run exposed a real Native Compact recovery regression in `tool.architecture-first`: the model requested `repo_structure --max-files 50` twice in compact mode, received two `compact-limit-exceeded` refusals, retried with `20`, and also issued an unrelated `repo_architecture outputMode=full`. That arm used 10 tool calls versus 4 for Prompt Compact and 120,336 versus 64,731 parent tokens. This violated the P01-N acceptance rule against refusal/retry loops.
38
+
39
+ Native Compact model-facing argument guidance was then tightened without changing policy ceilings or refusal semantics: schema descriptions now publish the compact/full native limits and tell the model to correct the same rejected argument rather than broaden an unrelated tool. Deterministic policy tests remained green.
40
+
41
+ The authorized re-run removed the loop: all six arms passed with zero Native Compact refusals, zero full overrides, and zero retry-after-refusal events.
42
+
43
+ | Re-run aggregate | Prompt Compact | Native Compact | Delta |
44
+ | --- | ---: | ---: | ---: |
45
+ | Repo-result bytes | 1,286 | 983 | -303 (-23.6%) |
46
+ | Tool calls | 14 | 12 | -2 (-14.3%) |
47
+ | Parent tokens | 179,855 | 207,611 | +27,756 (+15.4%) |
48
+ | Elapsed time | 88.682 s | 96.219 s | +7.537 s (+8.5%) |
49
+ | Passed arms | 3 / 3 | 3 / 3 | unchanged |
50
+ | Refusals / full overrides / refusal retries | 0 / 0 / 0 | 0 / 0 / 0 | unchanged |
51
+
52
+ The per-case result is mixed rather than uniformly negative or positive. `tool.architecture-first` improved repo bytes (903 → 600), calls (8 → 5), and elapsed time (38.9 s → 30.9 s), but parent tokens still rose 67,019 → 86,298. `tool.semantic-repo-search` delivered the same 383 repo bytes while Native Compact used one extra tool call, about 10.2% more parent tokens, and about 22.8% more elapsed time. The known-file negative case preserved the required direct `Read` behavior.
53
+
54
+ These live numbers are model-run evidence, not a claim of deterministic causality: the same seeded case ordering produced materially different Prompt Compact call counts between runs. The correct conclusion is therefore that Native Compact is now behaviorally safe in this small gate, but not a proven total-cost win.
55
+
56
+ ## Residual result classes after repo Native Compact
57
+
58
+ P00 capture fixtures prove that native truncation does not make every other result small. Using the same synthetic 2,100-line fixture shape:
59
+
60
+ | Path | Delivered bytes | Truncated | Full-output handle |
61
+ | --- | ---: | --- | --- |
62
+ | Built-in `read` | 36,053 | yes | no |
63
+ | Built-in `bash` | 36,152 | yes | yes, temp |
64
+ | `ast_grep` | 36,194 | yes | yes, temp |
65
+
66
+ These are real residual classes relative to an 8 KiB prospective Gateway inline budget, but size alone is not a P02 requirement.
67
+
68
+ - Built-in `read` already has exact `offset`/`limit` continuation; the unresolved question is historical/stable recovery, not whether current bytes can be paged.
69
+ - Built-in `bash` already preserves a temp full-output path on successful truncation; the unresolved question is whether task quality or resume semantics require a durable permitted snapshot instead.
70
+ - `ast_grep` already preserves a complete temp output artifact when truncated and can be captured in its suite-owned wrapper before truncation; a new store must demonstrate incremental value over that path.
71
+ - `repo_*` broad output is already the main Native Compact target and should not be counted again as Gateway savings.
72
+
73
+ ## Authorized observe residual gate
74
+
75
+ The explicitly authorized `zai/glm-5.3` observe run used three newly-created synthetic sessions with `contextGateway.mode=observe`, `repoDiscovery.profile=native-compact`, and an 8 KiB observation budget. The persisted report contains only aggregate Context Gateway telemetry; raw result bodies, tool arguments, project paths, and archive references are not written to the report.
76
+
77
+ | Observed class | Result content bytes | Details bytes | Upstream-truncated | Over 8 KiB | Potential content bytes over budget |
78
+ | --- | ---: | ---: | ---: | ---: | ---: |
79
+ | Built-in `Read` / `code-read` | 52,153 | 52,280 | 1 | 1 | 43,961 |
80
+ | Built-in shell / `shell` | 52,211 | 52,368 | 1 | 1 | 44,019 |
81
+ | Real `ast_grep` / `ast-grep` | 52,040 | 52,531 | 1 | 1 | 43,848 |
82
+
83
+ The `Read` and `ast_grep` task-level assertions passed. The shell task produced the expected single successful, upstream-truncated shell result and then one unrelated small `other` error result, so its strict "one tool call only" task assertion failed. That extra result is kept visible rather than hidden, but it does not invalidate the measured shell residual class. The observe runner now reports `taskPassed` and `observationValid` separately so future evidence cannot confuse prompt compliance with telemetry validity.
84
+
85
+ The run also exposed a cheaper residual than a new store: current SDK truncation details retain `truncation.content`. Direct SDK contracts confirm that this field contains the large already-delivered prefix/tail again for built-in `read`, built-in `bash`, and the suite `ast_grep` wrapper. In the observe samples this makes `detailsBytes` roughly the same size as the delivered content. The checked OpenAI-completions serializer does not send tool-result details to the provider, but persisted session JSONL and in-memory downstream result metadata still carry that duplication. Current DCP sidecar persistence is explicitly not a decision input here: its serializer already strips output details, and DCP is scheduled for a separate redesign. Before P02, prefer a result-metadata normalization path that removes only redundant `truncation.content` while preserving structural truncation fields and generic downstream-observer behavior.
86
+
87
+ ## Why P02 does not start yet
88
+
89
+ P02 introduces durable sensitive-data storage, session/workspace authorization, immutable publication, hash validation, cross-process quota reservations, platform-specific atomicity/sync behavior, and later reader lifecycle obligations. The current evidence does not yet show enough residual task-level value to justify those costs.
90
+
91
+ The observe-residual requirement is satisfied for the measured synthetic `Read`, shell, and `ast_grep` sessions, not for all production workloads. The metadata-normalization follow-up is a separate **opt-in module disabled by default**. It removes `details.truncation.content` for the measured tool names when the SDK-shaped metadata and delivered prefix match. Unknown names and nonmatching shapes are skipped; a matching name/shape is not proof of the provenance of an arbitrary replacement tool, which remains a P01-R compatibility boundary. Existing contracts show more than 20 KiB of JSONL fixture reduction with unchanged visible content, structural truncation fields, and checked OpenAI-completions payload. This is serialized metadata savings, not demonstrated provider-token or heap savings. Gateway off/observe alone remain non-transforming; an explicitly enabled `observe + normalizer` arm intentionally changes details and must be labeled separately.
92
+
93
+ The saved guarded recovery report has three passing task assertions and sequences `Read → Read → Read`, `Bash → Read`, and `ast_grep → Read`. It predates R-A validator/report v2. Although the Bash/ast-grep harness was configured to delete the producer sources and the old validator had a real `path` versus `file_path` bug, the aggregate report does not retain the transient producer handle, exact Read arguments/result correlation, or native continuation hint required by v2. Therefore **all three historical recovery strategies are unknown under the R-A provenance contract**, including the old Read-offset strategy that v1 labeled PASS. The original report remains unchanged.
94
+
95
+ This supports investigating native recovery, not a universal conclusion that storage is unnecessary. A correct recovery can still incur expensive repeated delivery; three synthetic successes do not establish temporary-output durability, satisfactory first-result size, or total-cost improvement.
96
+
97
+ The following evidence is still required before reversing this No-Go:
98
+
99
+ 1. Establish a reproducible task and reliable provenance/oracle where the checked storeless path is insufficient: correctness, required historical lifetime, or unacceptable total delivery/cost despite successful recovery. A recovery failure is not a mandatory prerequisite, and a large first result alone is not sufficient evidence.
100
+ 2. Explain why cheaper native scope/range controls, metadata cleanup, existing permitted outputs, and retrieval of actually persisted session text do not satisfy that task. Do not assume the future plan-32 runtime already exists or that JSONL recovers bytes never persisted there.
101
+ 3. Propose a limited P02 adapter/lifetime scope and evidence that new snapshot semantics add value. Compare against the checked storeless control, count all recovery/latency, and report model variance. Unsupported paths are listed explicitly, not claimed as protected or used as measured savings.
102
+ 4. Record the decision and obtain separate implementation authorization. Only relevant R-A/R-B/R-C/R-G contracts are prerequisites; P02 does not require completing every optional storeless investigation or the DCP rewrite.
103
+
104
+ ## Decision
105
+
106
+ **P02 remains deferred; the P01-R storeless branch is complete.** Keep store-backed enforce disabled. Do not create the immutable store, artifact readers, quotas, catalogs, or retention machinery under a different name merely to advance the roadmap. The final P01-R release decision keeps Native Compact and the metadata normalizer explicit opt-in, test/build parsing observe-only, and the other measured surfaces passthrough/limited; see `context-gateway-p01r-rg-evidence.md`. None of that is a Go for P02.
107
+
108
+ Plan 32 owns the new DCP journal, session-recovery pagination, and removal of the sidecar/legacy machinery. It is independent of the completed P01-R branch. P07 remains only a conditional integration gate for a future claimed Hybrid profile. No new validator-v2 live recovery run, user-config change, or manual synchronization was required for the P01-R release scope.
109
+
110
+ ## Verification used for this review
111
+
112
+ - `PI_CONTEXT_GATEWAY_BENCHMARK_REPORT=1 bun test test/context-gateway/benchmark.test.ts` — 2 pass, 0 fail; report values above.
113
+ - `bun test test/repo-native-compact.test.ts test/context-gateway/benchmark.test.ts` after the refusal-loop guidance fix — 12 pass, 0 fail, 108 assertions.
114
+ - `npm run typecheck` in `external/pi-tools-suite` — pass.
115
+ - Policy-only synthetic microbenchmark: seven rounds × 100,000 calls after warm-up; results stated above.
116
+ - Authorized live reports: `test/evals/artifacts/p01n-2026-09-07T17-23-48-741Z/p01n-paired-report.json` (regression discovery) and `test/evals/artifacts/p01n-2026-09-07T17-30-21-479Z/p01n-paired-report.json` (post-fix gate).
117
+ - Authorized observe report: `test/evals/artifacts/context-gateway-observe-2026-09-07T18-00-44-288Z/context-gateway-observe-report.json`; three expected residual classes were observed without persisted raw bodies.
118
+ - Optional non-store normalizer: `src/truncation-metadata-normalizer/index.ts`, disabled by default. `test/context-gateway/metadata-normalization.test.ts` plus the SDK pipeline contract prove conservative matching, JSONL reduction, unchanged visible content, generic downstream-observer compatibility, and a byte-equivalent checked provider payload.
119
+ - Guarded native-recovery report: `test/evals/artifacts/context-gateway-recovery-2026-09-07T18-32-36-215Z/context-gateway-recovery-report.json`; preserved unchanged and treated as v2-provenance `unknown` for all three recovery strategies.
120
+ - R-A deterministic evidence: `docs/context-gateway-p01r-ra-evidence.md`; corpus/validator/report v2, transient exact-handle probe, source/run SHA-256 identity, negative fixtures and safe-report contract. Focused gate: 17 pass, 0 fail, 105 assertions; suite typecheck/diff-check pass; live runner remains fail-closed without an explicitly selected model.
121
+ - R-F lifecycle/UI evidence: `docs/context-gateway-p01r-rf-evidence.md`; Context Gateway lifecycle/result gate 49 pass, 0 fail, 337 assertions; root TUI/session 98/98; ACP lazy current-result 4/4 plus persisted-history/request-boundary 6/6 and typecheck; Desktop transcript/client 28/28 with zero check errors (two pre-existing accessibility warnings). No native-temp reader endpoint was added.
122
+ - R-G final storeless release evidence: `docs/context-gateway-p01r-rg-evidence.md`; external deterministic P01-R gate 144 pass, 0 fail, 1011 assertions, plus root schema/typecheck/diff hygiene. Storeless defaults remain conservative and no additional live model call was required for the accepted scope.
@@ -0,0 +1,133 @@
1
+ # Context Gateway P01-N Native Compact measurement
2
+
3
+ <!-- markdownlint-disable MD013 -->
4
+
5
+ > Status: P01-N paired/observe evidence is recorded; native-recovery provenance has explicit validation limitations. **P01-R storeless hardening is planned independently; P02 is deferred**, not the entire roadmap. No new implementation or live evaluation is performed by this documentation revision.
6
+ > Repository baseline: `daa1b06`, Pi SDK `0.85.1`, 7 September 2026.
7
+
8
+ ## What is measured
9
+
10
+ P01-N separates three effects that must not be attributed to one another:
11
+
12
+ 1. **Historical Baseline prompt** — the pre-trim repo tool guidance measured before `daa1b06`.
13
+ 2. **Prompt Compact** — current repo descriptions/snippets/guidelines, with the historical runtime behavior (`repoDiscovery.profile=baseline`).
14
+ 3. **Native Compact** — the same current prompts plus `repoDiscovery.profile=native-compact`.
15
+
16
+ The historical pre-trim Baseline prompt is not reconstructed by loading the entire previous extension commit in live evals. The previous commit differs in code outside repo prompt text, so doing that would confound the comparison. Historical prompt size is retained as a separate textual baseline; live paired evaluation compares current Prompt Compact against current Native Compact on the same codebase.
17
+
18
+ ## Deterministic offline paired corpus
19
+
20
+ Executable evidence: `test/context-gateway/benchmark.test.ts`.
21
+
22
+ The corpus exercises five repo paths with fixed critical facts:
23
+
24
+ - semantic search followed by one narrow `--include-content --max-files 1` call;
25
+ - structure with a native `--cursor` continuation;
26
+ - AST with a native `--cursor` continuation;
27
+ - signature-first explain;
28
+ - shallow deps.
29
+
30
+ It is intentionally synthetic. It proves runtime byte/cursor/refusal behavior and deterministic critical-fact recovery, not model judgment or production latency.
31
+
32
+ | Metric | Historical/Prompt runtime baseline | Native Compact | Delta |
33
+ | --- | ---: | ---: | ---: |
34
+ | Repo prompt guidance chars | 3326 historical | 2294 current | -1032 (-31.0%) |
35
+ | Delivered repo-result bytes | 71,025 | 27,519 | -43,506 (-61.25%) |
36
+ | Tool calls | 6 | 8 | +2 |
37
+ | Continuation calls | 1 | 3 | +2 |
38
+ | Native policy refusals | 0 | 0 | 0 |
39
+ | Full overrides | 0 | 0 | 0 |
40
+ | Critical facts recovered | 8 / 8 | 8 / 8 | unchanged |
41
+
42
+ The Native Compact byte ratio in this corpus is `0.3874551214` of baseline. The extra calls are the cost of using native continuations instead of receiving the broad structure/AST output in one response.
43
+
44
+ Prompt Compact is runtime-equivalent to Baseline in this offline corpus by construction: prompt text is not a model and therefore cannot change tool selection in a deterministic wrapper-only test. That distinction is explicit so prompt savings are not misreported as runtime shaping.
45
+
46
+ ## Runtime policy evidence
47
+
48
+ Executable evidence: `test/repo-native-compact.test.ts`.
49
+
50
+ - default profile remains `baseline`;
51
+ - Native Compact compact delivery is bounded to 400 lines / 12 KiB;
52
+ - explicit `outputMode=full` is bounded to 2000 lines / 50 KiB;
53
+ - `--flag=value` is normalized before validation;
54
+ - duplicate, unknown, malformed, conflicting, and over-limit flags are refused before `idx` executes;
55
+ - native result schema uses integer/min/max bounds and does not rely on `execute()` alone for basic type constraints;
56
+ - final UTF-8 output is bounded after execution as well as native flags before execution;
57
+ - refusal metadata contains only allowlisted policy fields, not query/argv/body.
58
+
59
+ `test/context-gateway/observe.test.ts` additionally proves that observe telemetry accepts only allowlisted Native Compact policy outcomes and does not copy query/body or forged policy data from unrelated tools.
60
+
61
+ ## Live paired gate
62
+
63
+ The eval harness now records result byte counts plus Native Compact refusals/full overrides/retry-after-refusal without recording result bodies. `test/evals/run-p01n-paired.ts` runs Prompt Compact and Native Compact in deterministic randomized order for the same model/case and reports:
64
+
65
+ - existing behavioral/tool-selection assertions;
66
+ - repo/tool-result bytes;
67
+ - parent/worker tokens;
68
+ - tool-call count;
69
+ - elapsed time;
70
+ - Native Compact refusals, full overrides, and retry-after-refusal count.
71
+
72
+ Run it only with an explicitly selected already-configured live model, for example:
73
+
74
+ ```sh
75
+ PI_TOOLS_SUITE_EVAL_MODELS='zai/glm-5.3' npm run evals:p01n
76
+ ```
77
+
78
+ The command refuses to run when `PI_TOOLS_SUITE_EVAL_MODELS` is unset. Live model evaluation is intentionally not folded into ordinary deterministic tests.
79
+
80
+ ### Authorized `zai/glm-5.3` result
81
+
82
+ The first authorized live run found a Native Compact refusal loop in `tool.architecture-first`: two over-limit compact `repo_structure --max-files 50` calls were refused and retried before the model settled on the compact limit. Model-facing schema/recovery guidance was tightened without changing the enforced ceilings, and deterministic policy tests stayed green.
83
+
84
+ The post-fix paired run passed every arm with zero refusals/full overrides/retry-after-refusal events. Aggregate results across the three paired cases were:
85
+
86
+ | Metric | Prompt Compact | Native Compact | Delta |
87
+ | --- | ---: | ---: | ---: |
88
+ | Repo-result bytes | 1,286 | 983 | -23.6% |
89
+ | Tool calls | 14 | 12 | -14.3% |
90
+ | Parent tokens | 179,855 | 207,611 | +15.4% |
91
+ | Elapsed time | 88.682 s | 96.219 s | +8.5% |
92
+ | Passed pairs | 3 / 3 | 3 / 3 | unchanged |
93
+ | Refusals / full / refusal retries | 0 / 0 / 0 | 0 / 0 / 0 | unchanged |
94
+
95
+ This closes the small live/task comparison gate, but it does not establish a universal token/latency effect: model behavior varied materially between repeated runs even with deterministic arm ordering. The evidence supports keeping Native Compact available and bounded, not claiming that it always lowers total task cost.
96
+
97
+ ## Authorized observe residual evidence
98
+
99
+ Three newly-created synthetic `zai/glm-5.3` sessions ran with `contextGateway.mode=observe`, Native Compact enabled for `repo_*`, and `maxResultBytes=8192`. The test-only recorder uses the same `ContextGatewayTelemetry` implementation as the extension and persists only its aggregate snapshot.
100
+
101
+ The expected residual classes were all observed as upstream-truncated and over budget: built-in `Read` delivered 52,153 content bytes, shell 52,211, and real `ast_grep` 52,040. Potential content bytes above the 8 KiB observation budget were 43,961, 44,019, and 43,848 respectively. No tool arguments or result bodies are stored in the observe report.
102
+
103
+ Metadata is also material: the corresponding `detailsBytes` were 52,280, 52,368, and 52,531. Direct installed-SDK contracts show why: truncation details include a `truncation.content` string duplicating the already-delivered truncated text. This duplication is now handled by a cheaper optional result-metadata normalizer before any durable store work; the module is disabled by default and activates only through the existing suite `modules` opt-in surface.
104
+
105
+ The current DCP sidecar is not used as evidence for this optimization. Its persistence format already compacts away tool output/details, and DCP is expected to be redesigned separately. The relevant surfaces here are Pi session JSONL, in-memory tool-result metadata, checked provider serialization, renderers, and generic downstream result observers.
106
+
107
+ The Read and ast-grep task assertions passed. The shell session made one extra unrelated tool call after the valid shell observation, so task compliance failed while the shell observation itself remained valid. The observe runner now keeps those statuses separate.
108
+
109
+ The follow-up `truncation-metadata-normalizer` is separate and disabled by default. It removes `details.truncation.content` only for the measured tool names when the SDK-shaped fields and delivered prefix match; unknown names and nonmatching metadata are skipped. Arbitrary tool overrides matching those names are not independently certified. In a persisted fixture the duplicate sentinel falls from two copies to one and JSONL shrinks by more than 20 KiB, preserving structural fields and visible content. The checked OpenAI-completions payload is exactly equal. These are serialized metadata savings, not measured provider-token or heap savings. Gateway off/observe alone remain byte-preserving; an explicitly enabled normalizer is a separate transforming arm. Renderer/final-pipeline/security/lifecycle checks were subsequently completed in P01-R R-B/R-F; the final storeless release scope and rollback are recorded in `context-gateway-p01r-rg-evidence.md`.
110
+
111
+ ## Guarded native recovery evidence
112
+
113
+ The next synthetic gate asked whether a fact hidden by a large first result can be recovered without a new durable Gateway source. The saved guarded report `test/evals/artifacts/context-gateway-recovery-2026-09-07T18-32-36-215Z/context-gateway-recovery-report.json` records:
114
+
115
+ | Case | Task assertion | Saved strategy assertion | Tool sequence | Aggregate parent tokens |
116
+ | --- | --- | --- | --- | ---: |
117
+ | Read offset | PASS | PASS (v1 validator) | Read → Read → Read | 213,135 |
118
+ | Bash temp output | PASS | FAIL | Bash → Read | 75,253 |
119
+ | ast_grep temp output | PASS | FAIL | ast_grep → Read | 81,861 |
120
+
121
+ The producer source fixtures were configured for deletion after the initial Bash/ast-grep result, and the v1 evaluator had a real `Read.file_path` versus `path` bug. R-A now replaces that logic with validator/report v2: exact producer call/result correlation, exact issued native handle, result-before-Read order, successful producer/Read outcomes, fact-bearing native source/read result, exact native offset hint for Read continuation, verified synthetic cleanup, and explicit task/observation/availability/strategy statuses. However, the saved v1 aggregate lacks the transient handle/Read args/probes needed by v2, so it cannot be replayed into a corrected live result. All three saved recovery strategies are therefore **unknown under v2 provenance**. The old report is not rewritten; a new v2 live run would require separate authorization.
122
+
123
+ R-A deterministic evidence is recorded in `docs/context-gateway-p01r-ra-evidence.md`. Focused offline validation passes 17 tests / 105 assertions and records source/harness/corpus/validator identities without raw paths/body in final reports.
124
+
125
+ The evidence is consistent with useful native recovery but does not prove universal storage redundancy, cheap recovery, or durable temporary files. The token figures include aggregate parent usage, not a direct measure of recovery-only input or money. First-result size, repeated delivery, metadata/JSONL size, provider traffic, and historical availability remain separate questions.
126
+
127
+ ## P02 decision status
128
+
129
+ **P02 remains deferred.** Existing evidence does not justify the new durable-store complexity. The independent P01-R storeless branch was subsequently completed: Native Compact and the metadata normalizer remain explicit opt-in features, test/build parsing remains observe-only, and the other measured surfaces remain passthrough/limited. Completion of P01-R does not unlock P02.
130
+
131
+ Likely residual classes remain only candidates, not proven P02 requirements: generic built-in `read` after its own truncation boundary, shell output where the native temp handle is insufficient for desired history semantics, arbitrary custom tools, and child browser/subagent artifacts. MCP remains unsupported in the current Pix adapter and therefore cannot be used as evidence for a Gateway adapter.
132
+
133
+ Revisit P02 only for a concrete task demonstrating incremental value from new snapshot/lifetime semantics over the checked storeless alternatives. Native recovery may be correct but still too costly; failure is not the sole qualifying condition, and high cost alone does not prove a durable store will help. Record the narrower scope, comparison, uncertainty, and separate implementation authorization. Plan 32 owns the new DCP/session-recovery implementation without sidecar or legacy; P07 is only a future Hybrid integration gate.
@@ -0,0 +1,111 @@
1
+ # Context Gateway P01-R / R-A evidence
2
+
3
+ <!-- markdownlint-disable MD013 -->
4
+
5
+ > Date: 7 September 2026.
6
+ > Repository HEAD during deterministic gate: `daa1b06`.
7
+ > Installed Pi SDK: `@earendil-works/pi-coding-agent` `0.85.1`.
8
+ > Scope: deterministic recovery provenance, run identity and native capability accounting only. No new live model run was performed for R-A.
9
+
10
+ ## Result
11
+
12
+ R-A's deterministic infrastructure is complete. Future recovery runs use corpus v2,
13
+ validator v2 and report v2. Exact producer-native-handle provenance exists only in
14
+ a disposable per-run probe file inside the synthetic fixture project. The final
15
+ report contains only status enums/booleans/counts, tool names and source-code/runtime
16
+ identity; it does not retain tool inputs, native paths, result bodies, artifact
17
+ contents or hidden-fact fingerprints.
18
+
19
+ This closes the validation flaw in the old runner, but it does **not** turn the
20
+ existing live report into a new successful run.
21
+
22
+ ## Deterministic gate
23
+
24
+ ```text
25
+ bun test test/evals/recovery-corpus.test.ts \
26
+ test/evals/recovery-run-identity.test.ts \
27
+ test/evals/recovery-validation.test.ts \
28
+ test/evals/recovery-report.test.ts \
29
+ test/evals/harness.test.ts
30
+ ```
31
+
32
+ Result: **17 pass, 0 fail, 105 assertions**.
33
+
34
+ Additional checks:
35
+
36
+ - `npm run typecheck` — pass.
37
+ - `git diff --check` — pass.
38
+ - `npm run evals:context-gateway-recovery` without `PI_TOOLS_SUITE_EVAL_MODELS` — fail-closed with exit `2`.
39
+
40
+ ## What validator v2 proves
41
+
42
+ For native temp-output recovery it requires one concrete producer call and matching
43
+ successful result, a producer-details `fullOutputPath`, a fact-bearing readable
44
+ native file, verified deletion of the synthetic source fixture, a later `Read`,
45
+ exact equality between that Read path (`path` or `file_path`) and the issued handle,
46
+ a successful Read result and the expected opaque fact in that Read result.
47
+
48
+ For native Read continuation it requires the same source path and the exact numeric
49
+ `Use offset=N to continue` hint from the preceding successful result. `offset > 1`
50
+ alone is no longer sufficient.
51
+
52
+ Negative fixtures reject:
53
+
54
+ - Read before the producer result;
55
+ - failed producer or failed recovery Read;
56
+ - unrelated temp paths;
57
+ - repeated producers;
58
+ - direct reads of the original producer source;
59
+ - full-output-looking text without a producer `details.fullOutputPath`;
60
+ - direct answer/guess without a recovery Read;
61
+ - unverified synthetic cleanup;
62
+ - wrong Read source or wrong continuation offset;
63
+ - unknown recovery case IDs.
64
+
65
+ ## Corpus and run identity
66
+
67
+ The opaque expected values live in `recovery-corpus.ts`, not in prompts. Tests prove
68
+ that each value exists in its intended synthetic source and does not appear in the
69
+ case prompt or generated suite config.
70
+
71
+ Future report identity records:
72
+
73
+ - report/harness/validator/corpus versions;
74
+ - Git HEAD plus dirty/not-dirty state;
75
+ - SHA-256 of `index.ts`, the tested package source tree (`src/`, `index.ts`, `package.json`), harness runner, recovery runner, validator and corpus;
76
+ - installed Pi SDK version;
77
+ - Node/Bun/platform/architecture;
78
+ - effective storeless config and non-interactive Pi flags;
79
+ - model/provider and deterministic arm order.
80
+
81
+ The identity uses relative logical names plus code hashes, not host absolute paths.
82
+ This means a dirty source tree is identified by its tested bytes rather than by HEAD
83
+ alone, and the user's watcher is not evidence of which source revision a run used.
84
+
85
+ ## Historical live report status
86
+
87
+ The saved guarded report at
88
+ `test/evals/artifacts/context-gateway-recovery-2026-09-07T18-32-36-215Z/context-gateway-recovery-report.json`
89
+ is retained verbatim. It predates the v2 transient provenance probe and safe identity.
90
+
91
+ Its task assertions/tool-name sequences remain historical observations, but exact
92
+ native recovery provenance is **unknown under v2** for all three cases:
93
+
94
+ | Case | Saved v1 task/strategy | R-A v2 interpretation |
95
+ | --- | --- | --- |
96
+ | `recovery.read-offset` | task PASS / strategy PASS | Unknown: saved report lacks the exact native offset hint and Read args/result chain needed by v2. |
97
+ | `recovery.bash-temp-output` | task PASS / strategy FAIL | Unknown: saved report lacks producer `fullOutputPath` and exact Read-handle correlation. |
98
+ | `recovery.ast-grep-temp-output` | task PASS / strategy FAIL | Unknown for the same reason. |
99
+
100
+ The earlier `path` versus `file_path` unit fix explains one validator bug, but is not
101
+ evidence that the old live Read used the exact producer-issued native handle. A new
102
+ v2 live run, if needed later, requires separate authorization and creates a new
103
+ timestamped report rather than rewriting the old one.
104
+
105
+ ## Capability boundary
106
+
107
+ The current native recovery/lifetime matrix is recorded in
108
+ `test/context-gateway/p00-capabilities.md`. It explicitly distinguishes current-file
109
+ Read continuation, native index cursor, temporary full-output handle, persisted
110
+ session text and a future durable snapshot. Error/timeout/abort and expiry edge cases
111
+ that R-A does not certify remain R-C work.