pi-smart-compact 9.7.1 → 10.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ARCHITECTURE.md +973 -372
- package/CHANGELOG.md +721 -0
- package/LICENSE +8 -0
- package/README.md +128 -640
- package/SECURITY.md +34 -12
- package/SUPPORT.md +26 -9
- package/assets/DejaVu-LICENSE.txt +187 -0
- package/assets/DejaVuSansMono.ttf +0 -0
- package/assets/README.md +26 -0
- package/assets/skills/context-management/SKILL.md +34 -0
- package/dist/app/anchor-cache.d.ts +36 -0
- package/dist/app/anchor-cache.d.ts.map +1 -0
- package/dist/app/artifact-storage.d.ts +47 -0
- package/dist/app/artifact-storage.d.ts.map +1 -0
- package/dist/app/background-preparation.d.ts +39 -0
- package/dist/app/background-preparation.d.ts.map +1 -0
- package/dist/app/compaction-commit-store.d.ts +5 -1
- package/dist/app/compaction-commit-store.d.ts.map +1 -1
- package/dist/app/context-evidence.d.ts +57 -0
- package/dist/app/context-evidence.d.ts.map +1 -0
- package/dist/app/context-guide.d.ts +3 -0
- package/dist/app/context-guide.d.ts.map +1 -0
- package/dist/app/context-operations.d.ts +106 -0
- package/dist/app/context-operations.d.ts.map +1 -0
- package/dist/app/effective-state.d.ts +23 -0
- package/dist/app/effective-state.d.ts.map +1 -0
- package/dist/app/global-settings-runtime.d.ts +3 -3
- package/dist/app/global-settings-runtime.d.ts.map +1 -1
- package/dist/app/hindsight-memory.d.ts +100 -0
- package/dist/app/hindsight-memory.d.ts.map +1 -0
- package/dist/app/host-cache-ledger.d.ts +68 -0
- package/dist/app/host-cache-ledger.d.ts.map +1 -0
- package/dist/app/lazy-tools.d.ts +36 -0
- package/dist/app/lazy-tools.d.ts.map +1 -0
- package/dist/app/memory-backend.d.ts +58 -0
- package/dist/app/memory-backend.d.ts.map +1 -0
- package/dist/app/mnemopi-memory.d.ts +13 -0
- package/dist/app/mnemopi-memory.d.ts.map +1 -0
- package/dist/app/mnemopi-protocol.d.ts +78 -0
- package/dist/app/mnemopi-protocol.d.ts.map +1 -0
- package/dist/app/mnemopi-worker.d.ts +2 -0
- package/dist/app/mnemopi-worker.d.ts.map +1 -0
- package/dist/app/model-feasibility.d.ts +20 -0
- package/dist/app/model-feasibility.d.ts.map +1 -0
- package/dist/app/native-compaction.d.ts +88 -0
- package/dist/app/native-compaction.d.ts.map +1 -0
- package/dist/app/native-continuity-bridge.d.ts.map +1 -1
- package/dist/app/navigation-data.d.ts +28 -0
- package/dist/app/navigation-data.d.ts.map +1 -0
- package/dist/app/navigation-types.d.ts +60 -0
- package/dist/app/navigation-types.d.ts.map +1 -0
- package/dist/app/pending-slot.d.ts +11 -1
- package/dist/app/pending-slot.d.ts.map +1 -1
- package/dist/app/preflight.d.ts.map +1 -1
- package/dist/app/register-context-tools.d.ts +16 -3
- package/dist/app/register-context-tools.d.ts.map +1 -1
- package/dist/app/register-navigation.d.ts +20 -0
- package/dist/app/register-navigation.d.ts.map +1 -0
- package/dist/app/register-smart-compact-command.d.ts +17 -2
- package/dist/app/register-smart-compact-command.d.ts.map +1 -1
- package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
- package/dist/app/register-smart-context-tool.d.ts +55 -0
- package/dist/app/register-smart-context-tool.d.ts.map +1 -0
- package/dist/app/run-context.d.ts +1 -0
- package/dist/app/run-context.d.ts.map +1 -1
- package/dist/app/run-smart-compact.d.ts +3 -3
- package/dist/app/run-smart-compact.d.ts.map +1 -1
- package/dist/app/session-handoff.d.ts +64 -0
- package/dist/app/session-handoff.d.ts.map +1 -0
- package/dist/app/session-lineage.d.ts +17 -0
- package/dist/app/session-lineage.d.ts.map +1 -0
- package/dist/app/session-run-lock.d.ts +0 -2
- package/dist/app/session-run-lock.d.ts.map +1 -1
- package/dist/app/settled-auto-trigger.d.ts +2 -0
- package/dist/app/settled-auto-trigger.d.ts.map +1 -1
- package/dist/app/smart-compact-input.d.ts +1 -1
- package/dist/app/smart-compact-input.d.ts.map +1 -1
- package/dist/app/smart-compact-policy.d.ts +1 -1
- package/dist/app/smart-compact-policy.d.ts.map +1 -1
- package/dist/app/steps/extract.d.ts +45 -1
- package/dist/app/steps/extract.d.ts.map +1 -1
- package/dist/app/steps/metrics.d.ts +1 -0
- package/dist/app/steps/metrics.d.ts.map +1 -1
- package/dist/app/steps/persist.d.ts.map +1 -1
- package/dist/app/steps/prepare.d.ts.map +1 -1
- package/dist/app/steps/recover.d.ts +9 -0
- package/dist/app/steps/recover.d.ts.map +1 -1
- package/dist/app/steps/synthesize.d.ts.map +1 -1
- package/dist/app/steps/tier.d.ts.map +1 -1
- package/dist/app/steps/verify.d.ts.map +1 -1
- package/dist/app/steps/visual.d.ts +4 -0
- package/dist/app/steps/visual.d.ts.map +1 -0
- package/dist/app/steps/window.d.ts.map +1 -1
- package/dist/app/tool-artifacts.d.ts +27 -0
- package/dist/app/tool-artifacts.d.ts.map +1 -0
- package/dist/app/visual-archive.d.ts +29 -0
- package/dist/app/visual-archive.d.ts.map +1 -0
- package/dist/constants.d.ts +96 -1
- package/dist/constants.d.ts.map +1 -1
- package/dist/domain/compaction-usage.d.ts +16 -0
- package/dist/domain/compaction-usage.d.ts.map +1 -0
- package/dist/domain/model-capacity.d.ts +12 -0
- package/dist/domain/model-capacity.d.ts.map +1 -0
- package/dist/domain/provider-evaluation.d.ts +7 -0
- package/dist/domain/provider-evaluation.d.ts.map +1 -1
- package/dist/domain/telemetry.d.ts +43 -2
- package/dist/domain/telemetry.d.ts.map +1 -1
- package/dist/domain/tool-semantics.d.ts +23 -0
- package/dist/domain/tool-semantics.d.ts.map +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15757 -6942
- package/dist/infra/ai-messages.d.ts +1 -1
- package/dist/infra/ai-messages.d.ts.map +1 -1
- package/dist/infra/context-graph.d.ts +38 -7
- package/dist/infra/context-graph.d.ts.map +1 -1
- package/dist/infra/fs.d.ts.map +1 -1
- package/dist/infra/hindsight-client.d.ts +73 -0
- package/dist/infra/hindsight-client.d.ts.map +1 -0
- package/dist/infra/hindsight-receipts.d.ts +68 -0
- package/dist/infra/hindsight-receipts.d.ts.map +1 -0
- package/dist/infra/llm-client.d.ts +26 -23
- package/dist/infra/llm-client.d.ts.map +1 -1
- package/dist/infra/memory-ref.d.ts +27 -0
- package/dist/infra/memory-ref.d.ts.map +1 -0
- package/dist/infra/native-protocol.d.ts +54 -0
- package/dist/infra/native-protocol.d.ts.map +1 -0
- package/dist/infra/optional-components.d.ts +15 -0
- package/dist/infra/optional-components.d.ts.map +1 -0
- package/dist/infra/paths.d.ts +2 -0
- package/dist/infra/paths.d.ts.map +1 -1
- package/dist/infra/services.d.ts +15 -5
- package/dist/infra/services.d.ts.map +1 -1
- package/dist/infra/visual-renderer.d.ts +16 -0
- package/dist/infra/visual-renderer.d.ts.map +1 -0
- package/dist/mnemopi-worker.js +213 -0
- package/dist/phases/explore.d.ts +12 -9
- package/dist/phases/explore.d.ts.map +1 -1
- package/dist/phases/synthesize.d.ts +18 -3
- package/dist/phases/synthesize.d.ts.map +1 -1
- package/dist/phases/verify.d.ts +5 -1
- package/dist/phases/verify.d.ts.map +1 -1
- package/dist/rtk.d.ts +7 -0
- package/dist/rtk.d.ts.map +1 -0
- package/dist/rtk.js +767 -0
- package/dist/types.d.ts +128 -4
- package/dist/types.d.ts.map +1 -1
- package/dist/ui/dashboard-format.d.ts +2 -1
- package/dist/ui/dashboard-format.d.ts.map +1 -1
- package/dist/ui/dashboard-insights.d.ts +9 -1
- package/dist/ui/dashboard-insights.d.ts.map +1 -1
- package/dist/ui/error-format.d.ts +7 -2
- package/dist/ui/error-format.d.ts.map +1 -1
- package/dist/ui/handoff-overlay.d.ts +26 -0
- package/dist/ui/handoff-overlay.d.ts.map +1 -0
- package/dist/ui/home-overlay.d.ts +54 -0
- package/dist/ui/home-overlay.d.ts.map +1 -0
- package/dist/ui/metrics-dashboard-overlay.d.ts.map +1 -1
- package/dist/ui/metrics-report.d.ts.map +1 -1
- package/dist/ui/navigation-overlay.d.ts +92 -0
- package/dist/ui/navigation-overlay.d.ts.map +1 -0
- package/dist/ui/overlays.d.ts +12 -2
- package/dist/ui/overlays.d.ts.map +1 -1
- package/dist/ui/profiles.d.ts +51 -0
- package/dist/ui/profiles.d.ts.map +1 -0
- package/dist/ui/settings-complex.d.ts +49 -3
- package/dist/ui/settings-complex.d.ts.map +1 -1
- package/dist/ui/settings-list.d.ts +28 -0
- package/dist/ui/settings-list.d.ts.map +1 -0
- package/dist/ui/settings-overlay.d.ts +13 -6
- package/dist/ui/settings-overlay.d.ts.map +1 -1
- package/dist/ui/storage-report.d.ts +4 -0
- package/dist/ui/storage-report.d.ts.map +1 -0
- package/dist/utils/backups.d.ts.map +1 -1
- package/dist/utils/cache.d.ts +6 -2
- package/dist/utils/cache.d.ts.map +1 -1
- package/dist/utils/config.d.ts +12 -0
- package/dist/utils/config.d.ts.map +1 -1
- package/dist/utils/helpers.d.ts.map +1 -1
- package/dist/utils/id-fingerprint.d.ts +3 -1
- package/dist/utils/id-fingerprint.d.ts.map +1 -1
- package/dist/utils/issues.d.ts +61 -0
- package/dist/utils/issues.d.ts.map +1 -0
- package/dist/utils/pruning.d.ts.map +1 -1
- package/dist/utils/session-log.d.ts +0 -2
- package/dist/utils/session-log.d.ts.map +1 -1
- package/dist/utils/state.d.ts +3 -1
- package/dist/utils/state.d.ts.map +1 -1
- package/dist/utils/tokens.d.ts +10 -2
- package/dist/utils/tokens.d.ts.map +1 -1
- package/docs/MIGRATING_TO_V8.md +7 -1
- package/docs/README.md +69 -0
- package/docs/RELEASE.md +173 -56
- package/docs/assets/banner.png +0 -0
- package/docs/assets/banner.svg +1158 -70
- package/docs/assets/pi-smart-compact.png +0 -0
- package/docs/assets/pi-smart-compact.svg +24 -0
- package/docs/configuration.md +637 -0
- package/docs/evaluation.md +408 -0
- package/docs/guide.md +860 -0
- package/docs/hindsight-memory.md +314 -0
- package/docs/identity.md +124 -0
- package/package.json +44 -11
- package/dist/provider-eval.js +0 -2122
- package/dist/provider-scenario-eval.js +0 -2900
- package/dist/telemetry-report.js +0 -1973
- package/docs/provider-evaluation-2026-08-06.md +0 -63
|
@@ -0,0 +1,408 @@
|
|
|
1
|
+
# Evaluation and evidence
|
|
2
|
+
|
|
3
|
+
How Pi Continuity (published as the `pi-smart-compact` package) is evaluated,
|
|
4
|
+
which commands exist, and what each kind of evidence can and cannot support.
|
|
5
|
+
|
|
6
|
+
This page is for maintainers and evaluators working from a **development
|
|
7
|
+
checkout**. The evaluation and report CLIs run directly from source with Bun;
|
|
8
|
+
no build is needed. The installed npm package contains only the extension, the
|
|
9
|
+
optional RTK entry, the Mnemopi worker and declarations in `dist/`, not these
|
|
10
|
+
tools.
|
|
11
|
+
|
|
12
|
+
Related pages: [user guide](./guide.md) · [configuration](./configuration.md) ·
|
|
13
|
+
[architecture](../ARCHITECTURE.md) · [release checklist](./RELEASE.md) ·
|
|
14
|
+
[Hindsight memory backend](./hindsight-memory.md).
|
|
15
|
+
|
|
16
|
+
**Contents:** [evidence classes](#offline-and-live-evidence) ·
|
|
17
|
+
[release gates](#deterministic-release-gates) ·
|
|
18
|
+
[provider routing](#provider-routing-evidence) ·
|
|
19
|
+
[paired task evaluation](#paired-continuation-and-memory-evaluation) ·
|
|
20
|
+
[telemetry and canary gates](#telemetry-and-canary-gates) ·
|
|
21
|
+
[replay estimates](#replay-estimates) ·
|
|
22
|
+
[pilots and dated reports](#pilots-and-dated-reports)
|
|
23
|
+
|
|
24
|
+
Prerequisites for a source checkout: Bun 1.4.2 (the `packageManager` pin),
|
|
25
|
+
Node >=22.19 with npm on `PATH` for the Node-host smokes in `release:audit`,
|
|
26
|
+
and `rg` (ripgrep) on `PATH` for offline `task-eval` arms, which set
|
|
27
|
+
`PI_OFFLINE=1` and refuse to let Pi download tools.
|
|
28
|
+
|
|
29
|
+
## Offline and live evidence
|
|
30
|
+
|
|
31
|
+
Every result belongs to exactly one evidence class. Do not promote a claim from
|
|
32
|
+
one class to another.
|
|
33
|
+
|
|
34
|
+
| Class | Examples | Can show | Cannot show |
|
|
35
|
+
| --- | --- | --- | --- |
|
|
36
|
+
| Deterministic checks | `release:check`, `gate`, `bench`, unit tests | Contracts, invariants, bounded hot paths, packed-install behavior | Model quality, real savings, provider behavior |
|
|
37
|
+
| Offline lifecycle runs | `task-eval` (default), `session-pilot`, `context-compat-pilot`, `native-host-pilot` (default) | Real Pi `AgentSession`/tool/storage lifecycle with scripted model transport; oracle plumbing | Autonomous model decisions, live token cost or billing |
|
|
38
|
+
| Opt-in live probes | `provider-eval:live`, `task-eval --live`, `visual-pilot --live`, `PSC_NATIVE_LIVE=1`, Hindsight live canary | One bounded sample on one date, model and account | Production quality, other models, invoice-level cost |
|
|
39
|
+
| Local telemetry | `provider-eval`, `telemetry-report`, dashboards | Aggregates over runs recorded on this machine | Anything about runs not recorded, or statistical confidence |
|
|
40
|
+
| Replay estimates | `replay-eval` | Estimated prompt tokens and catalog-priced deltas for recorded sessions under alternative trim policies | Real savings, provider cache behavior, billing |
|
|
41
|
+
|
|
42
|
+
Rules that apply everywhere:
|
|
43
|
+
|
|
44
|
+
- A green deterministic or offline result never implies live quality, real
|
|
45
|
+
token savings, or a `PROMOTE` decision.
|
|
46
|
+
- Live runs are never part of `release:check`. Each live run needs a fresh,
|
|
47
|
+
explicit approval of the model, request count and token/cost exposure before
|
|
48
|
+
it starts. No approval carries over from an earlier run or document.
|
|
49
|
+
- Input guard counts are local estimates. Output reservations and wire caps
|
|
50
|
+
are distinct from provider-reported usage; missing usage stays unknown.
|
|
51
|
+
- Subscription (OAuth) usage is quota, not pay-as-you-go spend, and is never
|
|
52
|
+
priced at API rates.
|
|
53
|
+
- The verifier checks coverage of deterministic facts, structure and grounded
|
|
54
|
+
claims. It is a regression signal, not proof of semantic truth or of
|
|
55
|
+
lossless preservation.
|
|
56
|
+
- Dated reports are historical snapshots. Their measurements are not re-run
|
|
57
|
+
when the code changes; see [dated reports](#pilots-and-dated-reports).
|
|
58
|
+
|
|
59
|
+
## Deterministic release gates
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
bun install --frozen-lockfile
|
|
63
|
+
bun run release:check # typecheck + tests + gate + bench + build + release:audit + compat:pi latest
|
|
64
|
+
bun run gate # adversarial parser/verify/tool/cache/budget/scrub/damage fixtures
|
|
65
|
+
bun run bench # standalone hot-path p95 regression gate
|
|
66
|
+
bun run compat:pi 0.87.1 # locked minimum host, isolated workspace
|
|
67
|
+
bun run compat:pi latest # latest Pi host, isolated workspace
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
`release:audit` packs the tarball, installs it in an isolated frozen
|
|
71
|
+
workspace, checks the manifest, peers and packed file list, registers the
|
|
72
|
+
extension and its tools under stock Node, exercises Node SQLite and the
|
|
73
|
+
optional Mnemopi worker on the user-installed `bun` component with no Bun on
|
|
74
|
+
`PATH`, and
|
|
75
|
+
runs the offline source CLIs (`provider-eval`, `telemetry-report`, and all four
|
|
76
|
+
offline `task-eval` arms) under a temporary `HOME`. Apart from package
|
|
77
|
+
installation and a local loopback tripwire, it makes no network or model
|
|
78
|
+
request. The exact release procedure is in
|
|
79
|
+
[the release checklist](./RELEASE.md).
|
|
80
|
+
|
|
81
|
+
A previous green run does not carry over: rerun the full `release:check` after
|
|
82
|
+
any change to the candidate.
|
|
83
|
+
|
|
84
|
+
Pull-request CI runs frozen install, `typecheck`, `bun test`, `gate`, `bench`,
|
|
85
|
+
`build` and `release:audit`. The latest-Pi compatibility job runs only on a
|
|
86
|
+
schedule or manual dispatch. A green CI badge therefore does not replace a
|
|
87
|
+
full local `release:check` on the exact candidate.
|
|
88
|
+
|
|
89
|
+
## Provider routing evidence
|
|
90
|
+
|
|
91
|
+
All stages use the selected Pi model by default. Routing is explicit,
|
|
92
|
+
independent of modes, and never inferred or changed automatically:
|
|
93
|
+
|
|
94
|
+
| Stage | Config key | Fallback when unset |
|
|
95
|
+
| --- | --- | --- |
|
|
96
|
+
| Explore / segmentation | `segmentationModel` | resolved summary model |
|
|
97
|
+
| Synthesis / assembly | `summaryModel` | explicitly selected model, otherwise the chat model |
|
|
98
|
+
| Verification repair | `verificationModel` | resolved summary model |
|
|
99
|
+
|
|
100
|
+
Every run records per-stage provider, model, reliability, latency and token
|
|
101
|
+
telemetry, with schema-versioned verifier quality. Failed dispatched calls keep
|
|
102
|
+
content-free categories (authentication, rate limit, timeout, and so on), even
|
|
103
|
+
when a deterministic fallback completed the run. Older records without
|
|
104
|
+
categories stay unclassified.
|
|
105
|
+
|
|
106
|
+
### Advisory matrix from local telemetry
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
bun run provider-eval # text report, --min-samples defaults to 5
|
|
110
|
+
bun run provider-eval --min-samples=10 --json
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Reads the local metrics log and groups routes by stage, context pressure and
|
|
114
|
+
tool density. Only an explicitly attributed pre-repair synthesis score counts
|
|
115
|
+
as route quality; a run's final verifier score is never copied to Explore or
|
|
116
|
+
Verify. Legacy rows contribute latency and reliability, not quality. A cell is
|
|
117
|
+
eligible for a recommendation only with at least `--min-samples` runs, at least
|
|
118
|
+
80% call reliability, at least 50% stage-local quality coverage and an average
|
|
119
|
+
attributed quality of at least 85; scores shrink toward neutral under low
|
|
120
|
+
confidence. The report is advisory only: it never edits configuration or
|
|
121
|
+
selects a model.
|
|
122
|
+
|
|
123
|
+
### Opt-in live scenario probe
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
# Paid API or subscription quota. Run only with explicit approval.
|
|
127
|
+
bun run provider-eval:live --live \
|
|
128
|
+
--models=provider/model-a,provider/model-b
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Refuses to run without both `--live` and `--models` (1 to 8 models). Each model
|
|
132
|
+
receives the same three bounded coding-continuity scenarios (`implementation`,
|
|
133
|
+
`debugging`, `continuity`) sequentially, with a 1,500-token output cap and a
|
|
134
|
+
60-second timeout per call, scored by the deterministic verifier. It reports
|
|
135
|
+
score, latency and reported usage. Apply a route manually, and only after
|
|
136
|
+
representative evidence; one probe is not that evidence. The dated
|
|
137
|
+
[2026-08-06 baseline](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/provider-evaluation-2026-08-06.md)
|
|
138
|
+
(repository only) is an example of this output, not a current ranking.
|
|
139
|
+
|
|
140
|
+
## Paired continuation and memory evaluation
|
|
141
|
+
|
|
142
|
+
`task-eval` runs the same synthetic coding task through four real stock-Pi
|
|
143
|
+
`AgentSession` arms:
|
|
144
|
+
|
|
145
|
+
| Arm | Meaning |
|
|
146
|
+
| --- | --- |
|
|
147
|
+
| `no-compaction` | Baseline; context only grows |
|
|
148
|
+
| `recoverable-hygiene` | Recoverable trimming and retrieval, no summary |
|
|
149
|
+
| `eesv` | Real verified compactions |
|
|
150
|
+
| `hybrid` | Hygiene plus verified compactions |
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
bun run task-eval --help
|
|
154
|
+
bun run task-eval --out=/tmp/psc-task-eval-new # offline, all arms
|
|
155
|
+
bun run task-eval --arms=eesv,hybrid --repeats=3 --out=/tmp/psc-task-eval-eesv
|
|
156
|
+
bun run task-eval --repeats=1 --json --out=/tmp/psc-task-eval-smoke
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Options (from `--help`): `--arms`, `--repeats=1..8` (default five rounds, probes
|
|
160
|
+
after rounds two and five), `--out` (absolute directory; defaults to
|
|
161
|
+
`./task-eval-reports/<timestamp>`), `--json`, and two offline-only
|
|
162
|
+
cost-accounting fixtures, `--cache-warming=off|streaming|idle` and
|
|
163
|
+
`--background-prep`. Every arm runs with `toolLoading: "eager"`, so the arms
|
|
164
|
+
differ only in hygiene/offload knobs, not in on-demand tool discovery. With
|
|
165
|
+
`--cache-warming=idle`, arms that have automatic cleanup on may commit a
|
|
166
|
+
break-even trim at an idle boundary; the rebuilt context ends Pi's warming for
|
|
167
|
+
that entry, so zero warm replays there is expected, while the `no-compaction`
|
|
168
|
+
baseline keeps its refreshes.
|
|
169
|
+
|
|
170
|
+
The default transport is scripted and offline: the arm sets `PI_OFFLINE=1` so
|
|
171
|
+
Pi never downloads tools, and it fails before the first round when `rg`
|
|
172
|
+
(ripgrep, used by Pi's grep tool) is not on PATH. Independent oracles execute the
|
|
173
|
+
changed store/server and test process, check preserved constraints, errors and
|
|
174
|
+
side effects, require the newest decision, test unknown and false-premise
|
|
175
|
+
answers, and verify archive retrieval plus actual saved-memory recall and use.
|
|
176
|
+
Every arm runs the same explicitly approved synthetic memory task in temporary
|
|
177
|
+
stores. Reports compare requests, reported usage and cache classes, tool
|
|
178
|
+
interactions, compactions, retrieval, latency, hygiene and preparation
|
|
179
|
+
measurements. A green offline report proves lifecycle and oracle behavior, not
|
|
180
|
+
autonomous model quality, real token savings, or a production promotion.
|
|
181
|
+
|
|
182
|
+
### Live mode
|
|
183
|
+
|
|
184
|
+
```bash
|
|
185
|
+
# Prepare once, before approving any provider spend. The empty context sends
|
|
186
|
+
# no checkout, credentials or host files to the builder. Pull/apt needs network.
|
|
187
|
+
EMPTY_CONTEXT=$(mktemp -d)
|
|
188
|
+
docker build --file scripts/task-eval.Dockerfile \
|
|
189
|
+
--tag pi-smart-compact-task-eval:runtime-1 "$EMPTY_CONTEXT"
|
|
190
|
+
rmdir "$EMPTY_CONTEXT"
|
|
191
|
+
|
|
192
|
+
# Only with fresh explicit approval. PRIVATE_HOME contains only the approved
|
|
193
|
+
# frozen credential/selected model; PRIVATE_TMPDIR is caller-owned and private.
|
|
194
|
+
# Freeze the local endpoint before replacing HOME; do not copy Docker config.
|
|
195
|
+
DOCKER_ENDPOINT=$(docker context inspect --format '{{.Endpoints.docker.Host}}')
|
|
196
|
+
env -i HOME="$PRIVATE_HOME" TMPDIR="$PRIVATE_TMPDIR" PATH="$PATH" LANG=en_US.UTF-8 \
|
|
197
|
+
DOCKER_HOST="$DOCKER_ENDPOINT" bun run task-eval --live --models=provider/model \
|
|
198
|
+
--main-max-tokens=4096 --context-window=200000 \
|
|
199
|
+
--budget-requests=N --budget-input-tokens=N --budget-output-tokens=N \
|
|
200
|
+
--out=/absolute/private/report-directory
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Live mode requires the selected model and three positive budgets.
|
|
204
|
+
`--summary-model` may select another model on the same approved provider.
|
|
205
|
+
`--main-max-tokens` defaults to 4096. The context window is native unless an
|
|
206
|
+
explicit `--context-window` fixture is supplied; a fixture cannot exceed the
|
|
207
|
+
model catalog. Record both the fixture and native window when comparing arms.
|
|
208
|
+
Offline warming/preparation fixtures are refused in live mode.
|
|
209
|
+
|
|
210
|
+
Only the selected provider and selected custom model definitions enter the
|
|
211
|
+
runtime. API keys may come from that provider's stored credential or a literal
|
|
212
|
+
`models.json` key; commands and environment templates must already be resolved.
|
|
213
|
+
The credential is held by a read-only in-memory store. OAuth refresh material
|
|
214
|
+
is removed, mutation is refused, and expiring access tokens fail closed. Never
|
|
215
|
+
copy an entire daily auth/models file or source an ambient `.env` for a pilot.
|
|
216
|
+
|
|
217
|
+
Live tools require a local Unix-socket Docker daemon running Linux containers
|
|
218
|
+
and the prebuilt image above. On macOS, use a VM-backed daemon. The image is
|
|
219
|
+
resolved to its immutable ID before execution; startup probes must pass before
|
|
220
|
+
a provider call. No native fallback exists: raw `KERN_PROCARGS2` defeated the
|
|
221
|
+
macOS sandbox used by the rejected canary.5 candidate, even with a restricted
|
|
222
|
+
sysctl allowlist.
|
|
223
|
+
|
|
224
|
+
Model bash/read/write/grep, fixture snapshots and oracle processes run in fresh
|
|
225
|
+
containers. Only the synthetic project and owned HOME are bound; the runtime
|
|
226
|
+
image is read-only. No host credentials, Docker socket or ambient environment
|
|
227
|
+
are mounted/passed. Network and PID namespaces isolate host/sibling processes.
|
|
228
|
+
Capabilities are dropped, privilege elevation is disabled, and each container
|
|
229
|
+
is bounded to 64 PIDs, 512 MiB and one CPU. Calls are serialized. Tools execute
|
|
230
|
+
on Linux even when the evaluator is on macOS; record the image ID and policy
|
|
231
|
+
hash with the evidence. Whole-arm latency includes container startup/teardown.
|
|
232
|
+
These are evaluation prerequisites, not extension runtime dependencies.
|
|
233
|
+
|
|
234
|
+
Each container is created before it is started, avoiding a cancellation race
|
|
235
|
+
that could start a late container. Normal exit, timeout, abort and disposal
|
|
236
|
+
remove its whole process tree, including detached children. File reads have a
|
|
237
|
+
30-second command deadline and an 8 MiB output ceiling; local Docker control
|
|
238
|
+
operations have a separate 15-second deadline. Normal completion/failure
|
|
239
|
+
removes owned arm and child scratch directories. The caller must remove its
|
|
240
|
+
frozen-credential HOME/TMPDIR on signals; SIGKILL or a failed daemon can leave
|
|
241
|
+
owned resources requiring manual cleanup. Never delete unrelated containers.
|
|
242
|
+
|
|
243
|
+
The SDK fetch guard allows only the selected origin, refuses redirects, and
|
|
244
|
+
reserves a request's full output cap before dispatch. If that cap does not fit,
|
|
245
|
+
it refuses rather than shortening the response. Input is a character-derived
|
|
246
|
+
estimate, not a billed-token hard limit. Reports snapshot each arm after all
|
|
247
|
+
usage accounting settles, including failed dispatched calls, main/summary
|
|
248
|
+
classes, HTTP status, provider input/output/cache fields and per-request
|
|
249
|
+
elapsed time. Anthropic input excludes its separate cache fields; missing
|
|
250
|
+
fields remain null. Whole-arm time also includes tools and local oracles.
|
|
251
|
+
`--json` writes the same report to stdout and `task-eval-report.json`.
|
|
252
|
+
|
|
253
|
+
ChatGPT/Codex is refused by default: `--accept-codex-soft-cap` explicitly
|
|
254
|
+
selects unbounded output and can never satisfy a hard output-token budget.
|
|
255
|
+
Neither an offline loopback smoke nor one live paired sample satisfies the
|
|
256
|
+
production promotion gates below.
|
|
257
|
+
|
|
258
|
+
## Telemetry and canary gates
|
|
259
|
+
|
|
260
|
+
Raw local JSONL stays available to the interactive dashboard. The aggregate
|
|
261
|
+
report contains no session or project IDs, prompts, summaries, paths or error
|
|
262
|
+
text:
|
|
263
|
+
|
|
264
|
+
```bash
|
|
265
|
+
bun run telemetry-report # --min-canary-runs defaults to 20 (minimum 5)
|
|
266
|
+
bun run telemetry-report --min-canary-runs=20 --json
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
Failures use a stable content-free taxonomy (cancelled, timeout, rate limit,
|
|
270
|
+
authentication, budget, output limit, provider, persistence, validation,
|
|
271
|
+
verification, yield, internal). Verification and yield failures keep only
|
|
272
|
+
content-free diagnostics.
|
|
273
|
+
|
|
274
|
+
### Cohorts
|
|
275
|
+
|
|
276
|
+
Set `telemetryChannel: "canary"` only on an externally selected canary
|
|
277
|
+
installation; the default is `stable`. Canary evidence is limited to schema-v2
|
|
278
|
+
runs of the version under evaluation. Entries without an explicit release
|
|
279
|
+
channel are reported as unattributed and excluded from both cohorts, never
|
|
280
|
+
treated as stable. Reports separate total, attempted and host-confirmed applied
|
|
281
|
+
runs. Dry runs, staged or discarded preparations and voluntary cancellations are
|
|
282
|
+
not applied evidence; cancellations are neutral, while real timeouts and
|
|
283
|
+
provider failures count.
|
|
284
|
+
|
|
285
|
+
### Decision rules
|
|
286
|
+
|
|
287
|
+
The report returns `ROLLBACK`, `HOLD` or `PROMOTE` (implemented in
|
|
288
|
+
`src/domain/telemetry.ts`). It never deploys, rolls back or edits
|
|
289
|
+
configuration; promotion authority stays with the release owner.
|
|
290
|
+
|
|
291
|
+
Rollback is evaluated once the canary has at least three attempted runs. Any
|
|
292
|
+
trigger returns `ROLLBACK`:
|
|
293
|
+
|
|
294
|
+
| Metric | Trigger |
|
|
295
|
+
| --- | --- |
|
|
296
|
+
| Failure rate | Canary above 5% absolute, or at least 5pp above stable |
|
|
297
|
+
| Verifier quality | Canary average below 85, or at least 5 points below stable |
|
|
298
|
+
| p95 duration | At least +50% versus stable (stable p95 at least 1 s) |
|
|
299
|
+
| Average tokens | At least +50% versus stable (stable average at least 1,000) |
|
|
300
|
+
| Fallback rate | At least +10pp versus stable |
|
|
301
|
+
| Damage rate | At least +10pp versus stable |
|
|
302
|
+
|
|
303
|
+
Without a trigger, `PROMOTE` requires all of the following; otherwise the
|
|
304
|
+
result is `HOLD` with the first missing reason:
|
|
305
|
+
|
|
306
|
+
- at least `--min-canary-runs` (default 20) host-applied canary runs;
|
|
307
|
+
- a stable baseline of at least `max(20, --min-canary-runs)` applied runs;
|
|
308
|
+
- at least 70% verifier-quality coverage in both cohorts;
|
|
309
|
+
- at least 70% run-correlated damage-observation coverage in both cohorts;
|
|
310
|
+
- canary average verifier quality of at least 85 and success of at least 95%;
|
|
311
|
+
- canary data confidence of at least 85.
|
|
312
|
+
|
|
313
|
+
Damage observations join their originating compaction by run ID and are
|
|
314
|
+
deduplicated per run. Missing observations are missing evidence, never clean
|
|
315
|
+
runs.
|
|
316
|
+
|
|
317
|
+
### Two confidence scores
|
|
318
|
+
|
|
319
|
+
Both are completeness heuristics, not statistical confidence:
|
|
320
|
+
|
|
321
|
+
| Score | Where | Components |
|
|
322
|
+
| --- | --- | --- |
|
|
323
|
+
| Canary data confidence | `telemetry-report` | canary sample 25, stable sample 15, canary quality coverage 20, canary damage coverage 20, stable damage coverage 20 |
|
|
324
|
+
| Dashboard Data Confidence | `/smart-compact dashboard` | recent sample 25, schema-v2 share 25, quality coverage 20, field completeness 20, freshness 10 (last complete run within 7 days) |
|
|
325
|
+
|
|
326
|
+
Legacy or incompatible evidence stays missing and lowers both. Dashboards also
|
|
327
|
+
show initial score, patch, LLM-repair and deterministic-floor provenance instead
|
|
328
|
+
of hiding repair behind the final score.
|
|
329
|
+
|
|
330
|
+
### Preparation and cost measurements
|
|
331
|
+
|
|
332
|
+
Completed background work that is discarded is recorded once with its reason
|
|
333
|
+
and cost, never as applied evidence; graceful shutdown waits for that record.
|
|
334
|
+
Reports show used and discarded preparation, time to ready, wait to use or
|
|
335
|
+
discard, reuse rate and discarded spend. Stage routes keep input, cache-read,
|
|
336
|
+
cache-write and output classes and mark estimated usage. These are measurements
|
|
337
|
+
only: savings floors, cooldowns, pressure gates and TTLs are unchanged by them.
|
|
338
|
+
|
|
339
|
+
### Replay estimates
|
|
340
|
+
|
|
341
|
+
`bun run replay-eval --sessions=<dir|file[,file…]> [--out=/abs/dir] [--json]
|
|
342
|
+
[--break-even=8,16,24,48] [--rebuild-min=16384] [--limit=N] [--since=DAYS]
|
|
343
|
+
[--progress]` replays recorded session files in memory (never written; a file
|
|
344
|
+
that changes while it is read, such as a live session, is skipped and counted;
|
|
345
|
+
`--since` keeps files modified in the last `DAYS` days; `--progress` prints one
|
|
346
|
+
line per file to stderr) and judges the automatic-trim timing constants
|
|
347
|
+
`AUTO_TRIM_BREAK_EVEN_REQUESTS` and `REBUILD_MIN_TOKENS`. Recorded automatic trims are removed first so every
|
|
348
|
+
policy starts from the same history. Policies:
|
|
349
|
+
|
|
350
|
+
- `none`: no automatic trim.
|
|
351
|
+
- `pressure`: the old rule; a ready batch commits at a turn boundary only when
|
|
352
|
+
the estimated prompt reaches 0.8 × the catalog context window. Live, the gate
|
|
353
|
+
is the configured start percentage of `min(window, maxContextTokens)` against
|
|
354
|
+
Pi's reported usage, which includes the system prompt and tool definitions, so
|
|
355
|
+
pressure fires later in replay than live.
|
|
356
|
+
- `timed-<N>`: the current rule with `N` in place of the break-even limit;
|
|
357
|
+
pressure commits, `N* ≤ N` commits (`break-even`), otherwise the batch is held
|
|
358
|
+
and applied at the first request after the previous request's cache lifetime
|
|
359
|
+
(`cold`). Planning, cooldown and protected prefixes use the extension's own
|
|
360
|
+
`planContextTrim`/`trimEntries`/`trimTokens`.
|
|
361
|
+
|
|
362
|
+
Cost model per request: the projected context is estimated per message; the
|
|
363
|
+
cached prefix is the longest run of identical projected messages shared with
|
|
364
|
+
the previous request (0 after the cache lifetime or a model switch);
|
|
365
|
+
`uncached = prompt − cached`; a rebuild is `uncached ≥ max(--rebuild-min,
|
|
366
|
+
0.5 × prompt)`; price = `cacheRead × cached + (cacheWrite, else input) ×
|
|
367
|
+
uncached` at catalog rates. System prompt, tool definitions and output are
|
|
368
|
+
identical across policies and excluded. The recorded baseline (usage and
|
|
369
|
+
`usage.cost.total`) is the only measured figure; subscription requests report
|
|
370
|
+
tokens only. `--json` writes `<out>/replay-eval.json` with session ids and
|
|
371
|
+
numbers, no message text or paths. Absolute estimates are not calibrated to
|
|
372
|
+
recorded usage (a first run over three Codex sessions estimated about 4× the
|
|
373
|
+
recorded `input + cacheRead`); compare policies by their Δ, never by the
|
|
374
|
+
absolute column.
|
|
375
|
+
|
|
376
|
+
## Pilots and dated reports
|
|
377
|
+
|
|
378
|
+
Pilot scripts are development tools with narrow purposes. Offline defaults make
|
|
379
|
+
no provider request.
|
|
380
|
+
|
|
381
|
+
| Command | Default | Purpose |
|
|
382
|
+
| --- | --- | --- |
|
|
383
|
+
| `bun scripts/session-pilot.ts` | Offline | Real `AgentSession` with scripted model transport, Pi Continuity only: tools, anchor and pivot with carryover, trimming/retrieval, rewind, compaction and reopen |
|
|
384
|
+
| `PSC_CLAUDE_OAUTH_EXTENSION=<pi-claude-oauth-adapter>/extensions/index.ts bun scripts/native-host-pilot.ts` | Offline fake provider | Provider-native compaction on stock Pi; the Anthropic OAuth route needs the standalone adapter (patched final-payload build for billing on nested requests). `PSC_NATIVE_ROUTES=codex-oauth,openai-api-key` runs without it and `PSC_PILOT_SHORT=1` shortens each route. `PSC_NATIVE_LIVE=1` sends real, ledger-capped requests and needs explicit approval |
|
|
385
|
+
| `bun run scripts/rtk-pilot.ts /absolute/path/to/rtk` | Local only | Synthetic RTK rewrite contract; characters, not provider tokens |
|
|
386
|
+
| `bun run scripts/visual-pilot.ts --model=provider/id` | Offline planning | Bitmap versus text evidence; `--live` authorizes at most 9 sequential requests and needs explicit approval |
|
|
387
|
+
| `PSC_HINDSIGHT_LIVE=1 … bun run test/hindsight-live.canary.ts` | Not run by `bun test` | Live Hindsight contract with a hard call budget; see [Hindsight memory](./hindsight-memory.md#tests) |
|
|
388
|
+
|
|
389
|
+
Dated reports record what was measured on their date, with the code and host
|
|
390
|
+
versions stated inside. They are kept for provenance and are not updated
|
|
391
|
+
retroactively; later findings are added as dated addenda. Current behavior is
|
|
392
|
+
described in the [guide](./guide.md), [configuration](./configuration.md) and
|
|
393
|
+
[architecture](../ARCHITECTURE.md). Reports live in the repository only; the
|
|
394
|
+
npm package does not include `docs/reports/` or `docs/findings/`. Pilot
|
|
395
|
+
reports have a machine-readable `.json` companion beside them.
|
|
396
|
+
|
|
397
|
+
| Report | Scope |
|
|
398
|
+
| --- | --- |
|
|
399
|
+
| [Provider evaluation baseline, 2026-08-06](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/provider-evaluation-2026-08-06.md) | Live three-scenario probe across five models; advisory; 2026-09-25 addendum on route-report token semantics |
|
|
400
|
+
| [Context hygiene and continuity, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/context-hygiene-2026-09-24.md) | Hygiene design and offline experiments |
|
|
401
|
+
| [Hindsight and provider-native compaction research, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/hindsight-native-compaction-research-2026-09-24.md) | Pre-implementation research plus later measured results |
|
|
402
|
+
| [Full AgentSession offline pilot, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/session-pilot-2026-09-24.md) | Scripted-transport lifecycle pilot; 2026-09-25 candidate and `9.8.0-canary.1` follow-ups |
|
|
403
|
+
| [Visual evidence pilot, 2026-09-24](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/reports/visual-pilot-2026-09-24.md) | Live synthetic bitmap-versus-text reading pilot on one model |
|
|
404
|
+
|
|
405
|
+
External review findings, one folder per reviewer, are indexed in
|
|
406
|
+
[`docs/findings/`](https://github.com/alpertarhan/pi-smart-compact/blob/main/docs/findings/README.md).
|
|
407
|
+
They are advisory analyses of a specific revision range, not release gates or
|
|
408
|
+
evidence of current behavior.
|