ds4-context-engine 0.3.5 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -264,6 +264,7 @@ The following example shows the main configuration groups. Omitted values use th
264
264
  "minimumOutputReserve": 8192,
265
265
  "preferredOutputReserve": 32768,
266
266
  "recentTailTokens": 64000,
267
+ "rescueImmediatePredecessor": true,
267
268
  "maxPinnedTokens": 16000,
268
269
  "maxMemoryTokens": 8000,
269
270
  "maxRetrievedHistoryTokens": 16000,
@@ -0,0 +1,93 @@
1
+ # ADR-062 — Cache-aware context planning
2
+
3
+ Status: accepted and implemented in the coordinated 0.3.7 release (opt-in, default off; see the [release record](../releases/0.3.7.md)).
4
+
5
+ ## Context
6
+
7
+ DS4 reduces provider input by selecting a bounded managed context. This is
8
+ economically sound when every input token costs the same or when the provider
9
+ offers no prompt cache. Providers with a large context window and a large
10
+ cache-miss/cache-hit price ratio (for example DeepSeek V4 Flash, roughly 31x
11
+ off-peak) invert the trade-off: a small but frequently re-planned prompt can
12
+ cost more than a larger stable one, because every re-plan invalidates the
13
+ provider prefix cache.
14
+
15
+ The 0.3.6 planner applies an automatic 64k recent-tail ceiling to all models
16
+ above 256k until the conversational tail exceeds the cap. When the cap is
17
+ exceeded, the oldest turn leaves and the shared prefix with the previous
18
+ request collapses: everything after the first divergence becomes a cache miss.
19
+ Whether this is more expensive than a wider tail depends on the workload
20
+ (requests per turn, turns per epoch, share of cached reads observed).
21
+
22
+ The runtime already records cache read/write shares per exact `provider/model`
23
+ (volatile calibration plus persisted manifests) and Pi exposes per-million
24
+ cost rates on the active model. The portable core never hardcodes prices.
25
+
26
+ ## Decision
27
+
28
+ - Add an optional economic cost profile to the portable `ModelDescriptor`
29
+ (`input`, `output`, `cacheRead`, `cacheWrite` per million), populated from
30
+ Pi model metadata in `snapshotModel`. Absent fields degrade cache-aware
31
+ planning to the previous behavior.
32
+ - Add `context.cacheAware` (default `off`), a policy object with:
33
+ - `mode`: `off` preserves the previous planner exactly; `auto` may extend
34
+ the recent tail when pricing and observed cache shares justify it;
35
+ - `minimumCacheSampleCount`, `minimumCacheReadShare`,
36
+ `minimumMissHitRatio`: hard gates before any extension is eligible;
37
+ - `minimumImprovementRatio`: relative cost improvement required before an
38
+ alternative plan is adopted (hysteresis);
39
+ - `maxTailBudgetShare`: the extended tail may never exceed this fraction of
40
+ the active input budget;
41
+ - `expectedRequestsPerTurn`, `expectedTurnsPerEpoch`, `stickinessEpochs`:
42
+ deterministic cost model horizon plus hysteresis before dropping an
43
+ adopted extended tail.
44
+ - In `auto` the runtime computes `context.cacheAware` decision and compares
45
+ two plan candidates (nominal tail vs extended tail) with a deterministic
46
+ epoch cost model:
47
+ - stable plans (original conversation fits their tail cap) pay one cold
48
+ transition per epoch and warm requests afterwards;
49
+ - sliding plans (conversation exceeds their cap) pay a cold request every
50
+ turn of the epoch.
51
+ The extended plan is adopted only when it is cheaper over the whole epoch,
52
+ after the minimum improvement margin.
53
+ - The planner accepts an optional `cacheAwareTailTokens` override that bypasses
54
+ the automatic context-window ceiling; it remains bounded by the active/hard
55
+ input budgets and atomic groups, so no hard limit, privacy, pin, current
56
+ request or atomicity guarantee is weakened.
57
+ - Manifest `planning.cacheAware` (optional, numbers only, never content)
58
+ records the decision, tail, miss/hit ratio, observed cache share, sample
59
+ count, estimated reusable prefix tokens, estimated request cost and the
60
+ winning candidate. `/context tokens` and `/context explain` surface the same
61
+ metadata. `context.excluded_oversized_turn` and previous guarantees are
62
+ unchanged.
63
+ - Add a deterministic synthetic prefix-cache simulator covering the report
64
+ scenarios (append-only under the cap, sliding past 64k, one response per
65
+ prompt, five tool cycles, wide vs sliding tails, model switch, pricing
66
+ comparison). No provider, no credentials, no CI network access.
67
+
68
+ ## Rejected alternatives
69
+
70
+ - Hardcoding provider prices: rejected, the core stays provider-independent.
71
+ - Raising the default tail ceiling globally: rejected, it changes behavior
72
+ for all models (including ones without cache discounts).
73
+ - Always extending the tail for eligible models: rejected, the epoch model
74
+ shows the nominal plan can still be cheaper on workloads where the tail
75
+ slides infrequently; the improvement margin keeps the switch conservative.
76
+
77
+ ## Compatibility and validation
78
+
79
+ Configuration is additive; `context.cacheAware.mode` defaults to `off`, so
80
+ `0.3.6` behavior (and manifest absence of the cacheAware block) is preserved.
81
+ Automatic behavior is opt-in and requires observed samples plus positive
82
+ pricing. The epoch cost is an estimate used only for the comparison; the
83
+ actual billing remains whatever the provider reports. The workaround override
84
+ (`modelAwareness.overrides` with a large `recentTailTokens`, zeroed retrieval
85
+ and disabled compaction) remains available and is unaffected; it is more
86
+ aggressive than `mode: "auto"` because it also removes retrieval/project
87
+ supplements and compaction, which `auto` deliberately preserves for quality.
88
+
89
+ Coverage: cache-policy unit tests (prefix, costs, decision gates), config
90
+ validation, planner override tests, the synthetic prefix-cache simulator, and
91
+ runtime integration tests for off/auto/no-discount/under-budget paths. A
92
+ real-provider DeepSeek A/B benchmark remains out of scope and voluntary
93
+ (protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
@@ -65,5 +65,6 @@ The initial decisions from the development plan are accepted:
65
65
  | [059](059-optional-anchored-editing.md) | Opt in to anchored edit expansion inside Pi's native mutation queue | Accepted |
66
66
  | [060](060-optional-portable-agent-tools.md) | Opt in to edit reports, adaptive reads/results and session-owned local jobs | Accepted |
67
67
  | [061](061-compaction-latency.md) | Bound compaction update calls, input budgets, concurrent segments and phase timings | Accepted |
68
+ | [062](062-cache-aware-context-planning.md) | Opt-in cache-aware tail planning using model pricing and observed cache shares | Accepted |
68
69
 
69
70
  Each decision will receive a dedicated record when implementation pressure introduces alternatives or consequences not already covered by the development plan.
@@ -0,0 +1,66 @@
1
+ # Benchmark A/B DeepSeek — cache-aware planning
2
+
3
+ Protocollo volontario, fuori CI, nessuna credenziale nel repo. Serve a decidere se
4
+ `context.cacheAware.mode` può passare da `auto` (opt-in) a **default** in una
5
+ release futura. Gate dichiarato in [ADR-062](ADR/062-cache-aware-context-planning.md).
6
+
7
+ ## Obiettivo
8
+
9
+ Misurare, sulla **stessa** conversazione reale, il costo provider per turno con:
10
+
11
+ - **Sessione A (baseline)**: comportamento 0.3.6 — `context.cacheAware` assente
12
+ (default `off`), tail automatica 64k, retrieval 16k, project 20k.
13
+ - **Sessione B (auto)**: `context.cacheAware.mode: "auto"` con i default di
14
+ policy (gates 0.5 cache-share, ratio 20, margine 10%).
15
+ - **Sessione C (workaround estremo, opzionale)**: `modelAwareness.overrides` con
16
+ `recentTailTokens: 500000`, `maxRetrievedHistoryTokens: 0`,
17
+ `maxProjectTokens: 0`, `compaction.enabled: false` — cache-first puro.
18
+
19
+ Le tre sessioni devono avere la **stessa sequenza di turni** (stesso carico di
20
+ lavoro: tool call, retrieval, project touches) e lo **stesso modello/provided
21
+ ID DeepSeek**.
22
+
23
+ ## Cosa registrare per ogni turno
24
+
25
+ Da `/context tokens` e `/context explain`:
26
+
27
+ - `recentTailTokens` scelti (nominal vs extended) e `cacheAware` decision/reasons;
28
+ - manifest `planning.cacheAware` (missHitRatio, cacheReadShare, sampleCount,
29
+ reusablePrefixTokens, estimatedCost, candidate);
30
+ - `ProviderCacheMetrics` reali dopo ogni richiesta: `inputTokens`,
31
+ `cacheReadTokens`, `cacheWriteTokens`, share read/write per provider/model.
32
+
33
+ Dal provider (costo effettivo):
34
+
35
+ - input non-cached, cache-read (hit), cache-write, output per richiesta;
36
+ - costo totale della sessione (per-turno e totale).
37
+
38
+ ## Metriche
39
+
40
+ 1. **Costo per turno** (media e totale sessione): A vs B vs C.
41
+ 2. **Cache-read share osservata** per config: conferma che la tail estesa
42
+ mantiene il prefisso stabile (share alta) vs sliding (share ~0 dopo 64k).
43
+ 3. **Qualità** (non solo $$$): recall retrieval, `Planner exclusions`,
44
+ current request/atomicità intatte, completamenti corretti nelle fasi tool.
45
+ Con retrieval azzerato (C) documentare i regressi di qualità.
46
+ 4. **Eventi di transizione**: quante volte il prefisso si è invalidato (turni
47
+ con cacheRead ≈ 0), confronto sliding vs stable.
48
+
49
+ ## Criterio di promozione a default
50
+
51
+ Promuovere `mode: "auto"` a default solo se, su ≥ 3 sessioni reali:
52
+
53
+ - costo medio per turno di B < A (margine ≥ il `minimumImprovementRatio`
54
+ configurato, di default 10%);
55
+ - nessun regresso di qualità misurabile (recall retrieval uguale o migliore;
56
+ current request/atomicità preservate);
57
+ - nessuna oscillazione plan (decisione che alterna nominal/esteso senza
58
+ motivo: verificare `stickinessEpochs` e diagnostica).
59
+
60
+ ## Esecuzione sicura
61
+
62
+ - Nessuna credenziale nel repo; il benchmark usa la sessione Pi normale con il
63
+ provider DeepSeek già autenticato.
64
+ - Non modificare la config attiva in modo permanente: copie di configurazione
65
+ per sessione, o valori temporanei poi ripristinati.
66
+ - Fuori CI: niente rete/credenziali nel test suite.
@@ -20,6 +20,7 @@ A Context Manifest explains the context visible at DS4's Pi `context` hook witho
20
20
  - artifact IDs, SHA-256, bytes, MIME, classification, exact source entry/tool IDs, error state, and before/after token estimates;
21
21
  - provider destination and allow-set names, selected classification counts, blocked/excluded/redacted counts, final provider-check count, and enforcement stage;
22
22
  - planner mode/version, original and selected counts, group counts, internal budgets, duration, and fallback reason;
23
+ - optional cache-aware planning decision: eligibility, tail extension, requested tail tokens, miss/hit price ratio, observed cache-read share, sample count, estimated reusable prefix tokens, estimated request cost in dollars, and winning candidate (`nominal`/`cache-aware`); numbers only, never content;
23
24
  - learned-ranking mode/status, feature/model versions, candidate count, aggregate disagreement/rank shift, duration, and generic static-fallback reason;
24
25
  - planner and policy versions;
25
26
  - deterministic SHA-256 over system prompt, active tools, and messages;
@@ -54,6 +55,8 @@ A manifest at or below 256 KiB is stored unchanged. Above that preferred bound,
54
55
 
55
56
  Each manifest transaction prunes at most 32 excess rows and 8 MiB of serialized payload; one individually oversized oldest row may exceed the byte limit to guarantee progress. Calibration pruning is independently limited to 32 rows per related profile write. This incrementally repairs an existing oversized database without adding a long startup write or extending SQLite lock duration with an unbounded purge. Deleted pages become reusable by SQLite; the database file may remain at its previous high-water size until explicit [offline maintenance](STORAGE_MAINTENANCE.md). No retention action edits canonical Pi JSONL or project files.
56
57
 
58
+ The `save()` result carries the derived `inventory` of the persisted projection, and the runtime exposes it through `RuntimeDiagnostics.persistedInventory`; `/context manifest` and `/context excluded` therefore report the truthful persisted completeness (`complete` or `excluded-rollup` with retained/total counts) instead of defaulting to `complete` when the in-memory manifest has no inventory attached.
59
+
57
60
  ## Reproducibility
58
61
 
59
62
  Object keys are normalized before hashing, so equivalent tool schemas with different key insertion order produce the same prompt hash. The estimator version is stored explicitly as `chars-v1`; planner/policy versions describe selection behavior. Golden tests protect manifest shape, model-profile resolution, token accounting, and hash stability.
@@ -12,7 +12,7 @@ The managed planner is synchronous, deterministic, provider-independent, and doe
12
12
  6. Merge groups linked by assistant tool calls and every matching tool result.
13
13
  7. Select the current request, labelled pin groups, and applicable allowed persistent pins as mandatory.
14
14
  8. Enforce `maxPinnedTokens`, then fit relevant allowed durable memory under `maxMemoryTokens`.
15
- 9. Walk older turns newest-first, stopping at the first group that would break the contiguous recent tail, target, or hard limit.
15
+ 9. Walk older turns newest-first, stopping at the first group that would break the contiguous recent tail, target, or hard limit. When the immediate-predecessor turn itself exceeds the recent-tail cap but still fits the target and hard budgets, it is kept verbatim (`context.rescueImmediatePredecessor`, default true) and the tail then closes.
16
16
  10. Privacy-filter and fit source-labelled historical retrieval groups under `maxRetrievedHistoryTokens`.
17
17
  11. Privacy-filter and fit hash-current project snippets under `maxProjectTokens`.
18
18
  12. Fit active allowed Pi compaction/branch summaries in the remaining summary and input budgets.
@@ -41,6 +41,8 @@ Artifact condensation occurs before atomic grouping. It preserves `toolCallId`,
41
41
 
42
42
  The retrieval engine produces independent synthetic user-role evidence groups. They are never mandatory: recent turns have priority 100, durable memory 90, retrieved history 85, project snippets 80, and active summaries 75. Each group is selected or excluded whole, carries its original Pi entry ID, and is represented as `retrieval` in the Context Manifest. If planner validation falls back, every synthetic evidence message is discarded and Pi receives its original `AgentMessage[]` unchanged.
43
43
 
44
+ Retrieval deduplicates against the entries the managed plan actually commits, not against Pi's native context: before retrieving, the runtime plans the context with mandatory supplements only, maps the committed messages back to Pi session entry IDs, and passes those IDs as the retrieval exclusion set. An entry that Pi still exposes but the planner excludes (for example a turn larger than the recent-tail cap) therefore remains retrievable and reappears as bounded `retrieval` evidence instead of being silently lost. The Context Manifest planning block records `rescuedImmediatePredecessor` and `oversizedTurnExclusions`; an oversized exclusion also emits a `context.excluded_oversized_turn` warning, and `/context explain` surfaces both counters.
45
+
44
46
  Evidence text is a JSON-quoted historical excerpt with an explicit data-only boundary. It is inserted immediately before the latest real user request, so the current task remains the final message and provider conversation order stays deterministic.
45
47
 
46
48
  ## Project snippets
@@ -84,6 +86,42 @@ M14 can queue the finalized manifest after planning when `quality.enabled` is tr
84
86
 
85
87
  M18 can evaluate bounded metadata-only features after privacy exclusion and before supplemental candidates enter category fitting. `shadow` keeps every static score/order authoritative and records aggregate disagreement only. `active` is accepted only for a compatible checksummed model carrying an eligible held-out promotion report. Privacy exclusions, mandatory pins/current turns, atomic groups and hard budgets cannot be overridden. See [`LEARNED_RANKING.md`](LEARNED_RANKING.md).
86
88
 
89
+ ## Cache-aware tail planning (0.3.7, opt-in)
90
+
91
+ With `context.cacheAware.mode = "auto"` the runtime may extend the recent tail
92
+ beyond the automatic context-window ceiling when the model pricing and the
93
+ observed cache shares justify it economically. The policy never hardcodes
94
+ prices: it reads the per-million rates exposed on the active Pi model and the
95
+ cache read/write shares already recorded for the exact `provider/model`.
96
+
97
+ Eligibility gates (defaults shown):
98
+
99
+ - `minimumCacheSampleCount: 3` observed calibration samples;
100
+ - `minimumCacheReadShare: 0.5` observed share;
101
+ - `minimumMissHitRatio: 20` cache-miss / cache-hit price ratio.
102
+
103
+ When eligible, the runtime compares two plan candidates with a deterministic
104
+ epoch cost model:
105
+
106
+ - the nominal plan (existing tail, sliding below 64k) pays a cold request on
107
+ every turn of the epoch because its prefix is invalidated by the slide;
108
+ - the extended plan (up to `maxTailBudgetShare` of the active input budget) is
109
+ stable when the conversation fits, so it pays one cold transition per epoch
110
+ and warm requests afterwards.
111
+
112
+ The extended plan is adopted only when it wins over the whole epoch
113
+ (`expectedRequestsPerTurn`, `expectedTurnsPerEpoch`) after the
114
+ `minimumImprovementRatio` margin. Once adopted it is kept (deliberate epoch)
115
+ until it loses `stickinessEpochs` consecutive comparisons, preventing
116
+ oscillation. The override remains bounded by the active and hard input
117
+ budgets and by atomic groups, so current request, pins, privacy and atomicity
118
+ guarantees are unchanged.
119
+
120
+ `context.cacheAware.mode` defaults to `off`, preserving the 0.3.6 behavior
121
+ exactly. Without prices, samples or a cache discount, the decision degrades to
122
+ the nominal plan automatically. See [ADR-062](ADR/062-cache-aware-context-planning.md)
123
+ for the model and rejected alternatives.
124
+
87
125
  ## Current limits
88
126
 
89
127
  The planner does not call a model inside the `context` hook. Model calibration uses only finalized provider usage and deterministic local statistics. Historical/project retrieval can opt into derived semantic candidates; learned supplemental reranking remains off by default and active mode is promotion-gated. Project symbol extraction is heuristic, artifact search is literal, and memory/pin creation is manual-first. Automatic memory extraction remains disabled; M10 supplies policy enforcement but not an automatic classifier or confirmation workflow. Provider-payload coverage targets Pi 0.84.3's supported serializers, and DS4 must load after any extension allowed to replace payloads when strict final ordering is required.
@@ -1,6 +1,6 @@
1
1
  # DS4 Context Engine 0.3.5
2
2
 
3
- Status: local release gates passed; commit, push and coordinated publication authorized. Validation-only CI, publication and registry verification are pending.
3
+ Status: published as stable on npm under `latest`; annotated tag `v0.3.5` points to validated release commit `8b1d5f4`.
4
4
 
5
5
  ## Added since 0.3.4
6
6
 
@@ -65,6 +65,22 @@ Candidate validation on Node.js `26.5.1`, from a sanitized release source with a
65
65
 
66
66
  Pre-existing local `allowScripts` additions and `.serena/` are excluded from release commits and public packages.
67
67
 
68
+ The final clean committed worktree repeated fresh `npm ci`, all **508 tests**, clean-consumer package verification and tarball inspection successfully. Validation-only CI run [`33970061602`](https://github.com/Alucard24/ds4-context-engine/actions/runs/33970061602) on `8b1d5f4` passed both Node `22.19.0` and `24.x`, including the full suite and package checks.
69
+
68
70
  ## Registry verification and release
69
71
 
70
- Pending successful local gates, validation-only CI, manual publication and exact registry verification. No release tag has been created yet.
72
+ Published manually as `alucard_24`, in dependency order, from reviewed tarballs built in the clean committed worktree:
73
+
74
+ - `ds4-context-core@0.3.5`: shasum `71105c666a8c88f1abfe5ba3efefe63fc32bba74`;
75
+ - `ds4-context-reference-adapter@0.3.5`: shasum `8eca29472ea4721606d947818eba0aa53a6c6f63`;
76
+ - `ds4-context-engine@0.3.5`: shasum `304b61c21e28f89e82edbee39880210c2c9dc868`.
77
+
78
+ `npm run registry:check -- 0.3.5` passed: fresh exact-version installation, matching adapter/core dependencies, public core exports, compiled reference conformance, packaged quality corpus, storage CLI usage probe and isolated offline Pi extension startup. Registry SHA-1 and SHA-512 integrity values match the local tarballs for all three packages.
79
+
80
+ The immediate first post-publication install returned `ETARGET` for the engine. Once exact-version registry metadata was available, verification was repeated with npm `prefer-online` and passed; no package was republished or tag moved.
81
+
82
+ Verified dist-tags for all three: `latest=0.3.5`, `beta=0.3.0-beta.3`, `alpha=0.3.0-alpha.5`, `rc=0.2.0-rc.1`.
83
+
84
+ Annotated tag `v0.3.5` targets `8b1d5f4`, the tested/published source. This post-publication evidence update is documentation-only.
85
+
86
+ GitHub Release: https://github.com/Alucard24/ds4-context-engine/releases/tag/v0.3.5
@@ -0,0 +1,53 @@
1
+ # DS4 Context Engine 0.3.6
2
+
3
+ Status: published release on 2026-09-05; tag `v0.3.6`.
4
+
5
+ This coordinated release fixes a managed-context selection defect: a single oversized atomic conversation turn could cause the planner to discard turns that are still essential to the current request, and its own retrieval second pass could not recover them.
6
+
7
+ ## Fixed
8
+
9
+ - **Immediate-predecessor rescue**: when the turn immediately preceding the current request exceeds the recent-tail cap but still fits the active target and hard budgets, the planner keeps it verbatim instead of closing the entire recent tail at that group. The tail closes after the rescue, so older turns keep their prior bounded behavior. Controlled by `context.rescueImmediatePredecessor` (default `true`).
10
+ - **Retrieval second chance**: retrieval now deduplicates against the entries the managed plan actually commits, not against Pi's native context. Before the retrieval pass, the runtime plans the context with mandatory supplements only, maps the committed messages back to Pi session entry IDs, and passes those IDs as the retrieval exclusion set. Entries the planner excludes (for example an oversized turn) can therefore reappear as bounded, source-labelled `retrieval` evidence instead of being silently lost.
11
+ - **Diagnostics**: a `context.excluded_oversized_turn` warning reports how many turn groups at or above the recent-tail cap were excluded, whether the immediate predecessor was rescued, and the active limits. The Context Manifest planning block records optional `rescuedImmediatePredecessor` and `oversizedTurnExclusions` counters, and `/context explain` surfaces both.
12
+
13
+ No validation, privacy, provenance, cancellation, fallback, canonical-history, or bounded-retention behavior was weakened. The runtime retry policy, compaction pipeline, and storage formats are unchanged.
14
+
15
+ ## Compatibility and persistence
16
+
17
+ The context manifest adds only optional metadata fields. `ds4-context-config-v1`, `runtime-adapter-v1`, `ds4-context-persistence-tool-v1`, `ds4-context-persistence-result-v1`, SQLite schema 15, and migration checksums are unchanged.
18
+
19
+ ## Package/version policy
20
+
21
+ The coordinated version is `0.3.6` for:
22
+
23
+ ```text
24
+ ds4-context-core
25
+ ds4-context-reference-adapter
26
+ ds4-context-engine
27
+ ```
28
+
29
+ Both adapters depend exactly on `ds4-context-core@0.3.6`. The packages were published manually under npm `latest`. GitHub Actions remains validation-only with OIDC and package-write permissions denied.
30
+
31
+ ## Validation evidence
32
+
33
+ Local candidate verification on Node.js `26.5.1`:
34
+
35
+ - `npm run check`: 81 files and 518 tests passed (including planner rescue cases, second-chance retrieval integration, and entry-id mapping).
36
+ - `npm run quality:compare`: candidate quality `0.9875` versus baseline `0.808156`.
37
+ - `npm run schema:context-persistence`: 1,266 bytes and 317 estimated tokens; below the 1,500 absolute and 320 relative limits.
38
+ - `npm run latency:check -- <exact ds4-context-core@0.1.2>`: passed with ratio `1.082064`, at or below `1.10`.
39
+ - `npm run pack:check`: verified core (235 files), reference adapter (7 files), and Pi adapter (87 files) in a clean consumer.
40
+ - `npm pack --dry-run --json` for all three packages: passed with the same bounded inventories.
41
+ - `git diff --check`: passed.
42
+ - Protected CI, compatibility golden, Pi fixture, migration, canonical Pin/Memory, and persistence-tool contract files: unchanged except the intentional `compatibility-0.2.0.json` default addition.
43
+
44
+ - `npm run registry:check -- 0.3.6`: passed against all three exact published versions; `latest` resolves to `0.3.6` for every package.
45
+
46
+ Exact registry verification passed before the annotated tag and GitHub release were created.
47
+
48
+ ## Documentation
49
+
50
+ - [`../CONTEXT_PLANNER.md`](../CONTEXT_PLANNER.md)
51
+ - [`../COMPACTION.md`](../COMPACTION.md)
52
+ - [`../RELEASING.md`](../RELEASING.md)
53
+ - [`0.3.5.md`](0.3.5.md)
@@ -0,0 +1,83 @@
1
+ # Release 0.3.7 — Cache-aware context planning (opt-in)
2
+
3
+ **Version analyzed:** DS4 Context Engine `0.3.7`
4
+ **Commit:** `a03009a4b14d6ab1842fa800076c4b300bd70d8f`
5
+ **Coordinated packages:** `ds4-context-core` 0.3.7, `ds4-context-reference-adapter` 0.3.7, `ds4-context-engine` 0.3.7
6
+
7
+ ## Summary
8
+
9
+ Adds an opt-in cache-aware planning policy driven by model pricing and
10
+ observed cache shares, plus a deterministic synthetic prefix-cache simulator.
11
+ The default behavior is unchanged: `context.cacheAware.mode` defaults to `off`,
12
+ the manifest does not include a `cacheAware` block, and the planner uses the
13
+ same tail caps as 0.3.6.
14
+
15
+ ## New configuration
16
+
17
+ `context.cacheAware` (object, default off):
18
+
19
+ | Field | Default | Meaning |
20
+ |---|---|---|
21
+ | `mode` | `off` | `off` preserves 0.3.6 behavior; `auto` may extend the recent tail. |
22
+ | `minimumCacheSampleCount` | `3` | Observed samples required before acting. |
23
+ | `minimumCacheReadShare` | `0.5` | Observed cache-read share required before acting. |
24
+ | `minimumMissHitRatio` | `20` | Minimum cache-miss / cache-hit price ratio. |
25
+ | `minimumImprovementRatio` | `0.1` | Relative improvement required to switch plan. |
26
+ | `maxTailBudgetShare` | `0.5` | Max fraction of the active input budget for the extended tail. |
27
+ | `expectedRequestsPerTurn` | `4` | Provider requests per user turn in the cost model. |
28
+ | `expectedTurnsPerEpoch` | `4` | User turns per planning epoch in the cost model. |
29
+ | `stickinessEpochs` | `2` | Consecutive epoch losses before dropping an adopted extended tail. |
30
+
31
+ ## Changes
32
+
33
+ - `ModelDescriptor` gains an optional `cost` profile (`input`, `output`,
34
+ `cacheRead`, `cacheWrite` per million), populated from Pi model metadata in
35
+ `snapshotModel`. No prices are hardcoded in the core.
36
+ - `planManagedContext` accepts an optional `cacheAwareTailTokens` override that
37
+ bypasses the automatic context-window ceiling while remaining bounded by the
38
+ active/hard input budgets and atomic groups.
39
+ - Runtime: when `mode: "auto"` and the eligibility gates pass, the runtime
40
+ compares the nominal plan with an extended-tail plan using a deterministic
41
+ epoch cost model (stable plans pay one cold transition per epoch; sliding
42
+ plans pay a cold request per turn) and adopts the extended plan only when it
43
+ wins after the minimum improvement margin.
44
+ - Manifest: optional `planning.cacheAware` (numbers only, never content) with
45
+ eligibility, tail tokens, miss/hit ratio, observed share, sample count,
46
+ estimated reusable prefix tokens, estimated request cost and winning
47
+ candidate; surfaced in `/context tokens` and `/context explain`.
48
+ - Core: `packages/core/src/planner/cache-policy.ts` with pure, deterministic
49
+ functions for common-prefix estimation, request cost and the tail decision.
50
+ - Tests: `cache-policy` unit tests, config validation, planner override tests,
51
+ the synthetic prefix-cache simulator (`cache-prefix-simulator`) and runtime
52
+ integration tests for off/auto/no-discount/under-budget paths.
53
+
54
+ ## Compatibility
55
+
56
+ - Additive configuration; absent fields use the documented defaults.
57
+ - `context.cacheAware.mode: "off"` reproduces the 0.3.6 behavior exactly.
58
+ - No new required database schema; manifests persist the optional block as-is.
59
+ - The existing guarantees (current request, atomicity, privacy, pins, hard
60
+ limits, fail-open, naive compaction path) are unchanged.
61
+ - Model metadata without cache pricing or observed samples degrades the
62
+ decision to the nominal plan.
63
+
64
+ ## Validation
65
+
66
+ - `npm run check` (excluding the known machine-load-dependent
67
+ `long-session` timeout flake): 83 files, 550 tests passed.
68
+ - Planner unit tests: 25 passed.
69
+ - Cache-policy unit tests: 16 passed.
70
+ - Prefix-cache simulator: 7 passed.
71
+ - Cache-aware runtime integration: 4 passed.
72
+ - Config catalog/loader validation: passed.
73
+
74
+ ## Known limits
75
+
76
+ - The epoch cost model is an estimate for plan comparison; actual billing is
77
+ whatever the provider reports.
78
+ - A real-provider DeepSeek A/B benchmark remains voluntary and out of CI
79
+ (protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
80
+ - The workaround override (large `recentTailTokens`, zeroed retrieval and
81
+ disabled compaction) remains available and is unaffected; it is more aggressive
82
+ than `mode: "auto"`, which deliberately preserves retrieval/project and
83
+ compaction for quality.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ds4-context-engine",
3
- "version": "0.3.5",
3
+ "version": "0.3.7",
4
4
  "description": "Non-destructive, provider-independent context management for Pi.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -62,7 +62,7 @@
62
62
  ]
63
63
  },
64
64
  "dependencies": {
65
- "ds4-context-core": "0.3.5"
65
+ "ds4-context-core": "0.3.7"
66
66
  },
67
67
  "peerDependencies": {
68
68
  "@earendil-works/pi-ai": "0.84.3",
@@ -296,6 +296,14 @@ function formatTokens(diagnostics: RuntimeDiagnostics): string {
296
296
  `Actual provider input: ${count(manifest?.actualInputTokens)}`,
297
297
  ` Uncached input: ${count(manifest?.providerUsage?.inputTokens)}`,
298
298
  ` Cache read / write: ${count(manifest?.providerUsage?.cacheReadTokens)} / ${count(manifest?.providerUsage?.cacheWriteTokens)}`,
299
+ ...(manifest?.planning?.cacheAware
300
+ ? [
301
+ `Cache-aware plan: ${manifest.planning.cacheAware.candidate ?? "n/a"}${manifest.planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
302
+ ` Reusable prefix: ${count(manifest.planning.cacheAware.reusablePrefixTokens)} estimated`,
303
+ ` Est. request cost: ${manifest.planning.cacheAware.estimatedCost === undefined ? "n/a" : `$${manifest.planning.cacheAware.estimatedCost.toFixed(6)}`}`,
304
+ ` Miss/hit price ratio: ${manifest.planning.cacheAware.missHitRatio === undefined ? "n/a" : manifest.planning.cacheAware.missHitRatio.toFixed(2)}`,
305
+ ]
306
+ : []),
299
307
  `Pi reported context: ${count(manifest?.piReportedContextTokens ?? observation?.reportedTokens)}`,
300
308
  `Model context window: ${count(budget?.contextWindow ?? manifest?.contextWindow)}`,
301
309
  `Output reserve: ${count(budget?.outputReserve ?? manifest?.outputReserve)}`,
@@ -316,7 +324,7 @@ function formatManifest(diagnostics: RuntimeDiagnostics): string {
316
324
  const manifest = diagnostics.lastManifest;
317
325
  if (!manifest) return "No Context Manifest has been built for this session yet.";
318
326
 
319
- const inventory = manifest.persistedInventory;
327
+ const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
320
328
  const kinds = new Map<string, { items: number; tokens: number }>();
321
329
  for (const item of manifest.included) {
322
330
  const aggregate = kinds.get(item.kind) ?? { items: 0, tokens: 0 };
@@ -369,7 +377,7 @@ function formatManifestItems(diagnostics: RuntimeDiagnostics, type: "included" |
369
377
  const manifest = diagnostics.lastManifest;
370
378
  if (!manifest) return "No Context Manifest has been built for this session yet.";
371
379
  const items = manifest[type];
372
- const inventory = manifest.persistedInventory;
380
+ const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
373
381
  return [
374
382
  `DS4 Context ${type === "included" ? "Included" : "Excluded"} Items`,
375
383
  "",
@@ -403,6 +411,29 @@ function formatExplain(diagnostics: RuntimeDiagnostics): string {
403
411
  `Message target: ${count(planning.messageTargetTokens)}`,
404
412
  `Message hard limit: ${count(planning.messageHardLimitTokens)}`,
405
413
  `Recent-tail limit: ${count(planning.recentTailTokenLimit)}`,
414
+ ...(planning.rescuedImmediatePredecessor
415
+ ? ["Rescued predecessor: yes (immediate previous turn kept beyond the recent-tail cap within the input budget)"]
416
+ : []),
417
+ ...(planning.oversizedTurnExclusions
418
+ ? [`Oversized turn excl: ${count(planning.oversizedTurnExclusions)} (turn group(s) at/above the recent-tail cap; recovered by retrieval only if it fits)`]
419
+ : []),
420
+ ...(planning.cacheAware
421
+ ? [
422
+ `Cache-aware plan: ${planning.cacheAware.candidate ?? "n/a"}${planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
423
+ ...(planning.cacheAware.missHitRatio !== undefined
424
+ ? [` Miss/hit ratio: ${planning.cacheAware.missHitRatio.toFixed(2)}`]
425
+ : []),
426
+ ...(planning.cacheAware.cacheReadShare !== undefined
427
+ ? [` Cache-read share: ${(planning.cacheAware.cacheReadShare * 100).toFixed(1)}% (${count(planning.cacheAware.sampleCount)} samples)`]
428
+ : []),
429
+ ...(planning.cacheAware.reusablePrefixTokens !== undefined
430
+ ? [` Reusable prefix: ${count(planning.cacheAware.reusablePrefixTokens)} tokens`]
431
+ : []),
432
+ ...(planning.cacheAware.estimatedCost !== undefined
433
+ ? [` Est. request cost: $${planning.cacheAware.estimatedCost.toFixed(6)}`]
434
+ : []),
435
+ ]
436
+ : []),
406
437
  `Selected groups: ${count(planning.selectedGroupCount)}`,
407
438
  `Excluded groups: ${count(planning.excludedGroupCount)}`,
408
439
  `Duration: ${planning.durationMs === undefined ? "n/a" : `${planning.durationMs.toFixed(1)} ms`}`,
@@ -72,7 +72,10 @@ import type {
72
72
  PrivacyManifest,
73
73
  ProviderUsageManifest,
74
74
  } from "ds4-context-core/manifest/context-manifest";
75
- import { HARD_PERSISTED_MANIFEST_BYTES } from "ds4-context-core/manifest/context-manifest-storage";
75
+ import {
76
+ HARD_PERSISTED_MANIFEST_BYTES,
77
+ type PersistedManifestInventory,
78
+ } from "ds4-context-core/manifest/context-manifest-storage";
76
79
  import {
77
80
  unavailableStorageDiagnostics,
78
81
  type StorageDiagnostics,
@@ -97,7 +100,14 @@ import {
97
100
  type PinScope,
98
101
  type ProjectMemorySource,
99
102
  } from "ds4-context-core/memory/memory-types";
100
- import { planManagedContext, type SupplementalContextMessage } from "ds4-context-core/planner/context-planner";
103
+ import { planManagedContext, type ManagedContextPlan, type SupplementalContextMessage } from "ds4-context-core/planner/context-planner";
104
+ import {
105
+ decideCacheAwareTail,
106
+ estimateReusablePrefixTokens,
107
+ estimateRequestCost,
108
+ type CacheAwarePlanDecision,
109
+ type CachePricing,
110
+ } from "ds4-context-core/planner/cache-policy";
101
111
  import {
102
112
  disabledContextQualityDiagnostics,
103
113
  evaluateManifestQuality,
@@ -151,6 +161,8 @@ import {
151
161
  buildPiObserverManifest,
152
162
  findExactPiMessageSourceIds,
153
163
  findPiPinnedMessageIndices,
164
+ findPiSourceEntryIds,
165
+ fingerprint,
154
166
  } from "../pi-adapter/context-observer.ts";
155
167
  import { projectSessionFileMutations } from "../pi-adapter/memory-adapter.ts";
156
168
  import { ProjectMemorySynchronizer } from "../pi-adapter/project-memory-sync.ts";
@@ -283,6 +295,10 @@ function numericUsage(value: unknown): number {
283
295
  : 0;
284
296
  }
285
297
 
298
+ function roundedCost(value: number): number {
299
+ return Math.round(value * 1_000_000) / 1_000_000;
300
+ }
301
+
286
302
  function providerUsageManifest(input: {
287
303
  inputTokens: number;
288
304
  cacheReadTokens: number;
@@ -373,6 +389,8 @@ export interface RuntimeDiagnostics {
373
389
  indexed?: SessionIndexStats;
374
390
  observation?: ContextObservation;
375
391
  lastManifest?: ContextManifest;
392
+ /** Inventory of the persisted projection of the last successfully stored manifest. */
393
+ persistedInventory?: PersistedManifestInventory;
376
394
  retrieval: RetrievalDiagnostics;
377
395
  project: ProjectKnowledgeDiagnostics;
378
396
  memory: MemoryDiagnostics;
@@ -410,6 +428,7 @@ export class Ds4ContextRuntime {
410
428
  private databasePath?: string;
411
429
  private observation?: ContextObservation;
412
430
  private lastManifest?: ContextManifest;
431
+ private lastPersistedInventory?: PersistedManifestInventory;
413
432
  private pendingManifestId?: string;
414
433
  private pendingManifestPersisted = false;
415
434
  private retrievalEngine?: HistoricalRetrievalEngine;
@@ -432,6 +451,18 @@ export class Ds4ContextRuntime {
432
451
  private lastContextProfileKey?: string;
433
452
  private readonly knownModelProfiles = new Set<string>();
434
453
  private readonly volatileCalibration = new Map<string, TokenCalibrationSample[]>();
454
+ /** Fingerprints of the messages sent in the last managed plan (empty when unknown). */
455
+ private lastPlanMessageHashes: string[] = [];
456
+ /** Per-message token estimates of the last managed plan, aligned with hashes. */
457
+ private lastPlanMessageTokens: number[] = [];
458
+ /** Cache-aware decision of the last plan; used for /context diagnostics. */
459
+ private lastCacheAwareDecision?: CacheAwarePlanDecision;
460
+ /** Provider/model key of the last plan; a switch invalidates the cached prefix. */
461
+ private lastPlanModelKey?: string;
462
+ /** Consecutive epochs in which the extended tail lost the comparison (hysteresis). */
463
+ private cacheAwareExtendedLossStreak = 0;
464
+ /** Extended-tail adoption state: while set, the extended tail is kept even if a single epoch comparison loses. */
465
+ private cacheAwareStickExtended = false;
435
466
  private lastMemoryMutationSignature?: string;
436
467
  private artifactManager?: ArtifactManager;
437
468
  private lastArtifacts: ArtifactDiagnostics = disabledArtifactDiagnostics();
@@ -480,6 +511,7 @@ export class Ds4ContextRuntime {
480
511
  this.pendingQuality.length = 0;
481
512
  this.lastIndexResult = undefined;
482
513
  this.lastManifest = undefined;
514
+ this.lastPersistedInventory = undefined;
483
515
  this.pendingManifestId = undefined;
484
516
  this.pendingManifestPersisted = false;
485
517
  this.retrievalEngine = undefined;
@@ -584,6 +616,7 @@ export class Ds4ContextRuntime {
584
616
  this.syncSessionIndex(ctx);
585
617
  this.lastManifest = this.database.manifests.getLatest(this.session.sessionId);
586
618
  if (this.lastManifest) {
619
+ this.lastPersistedInventory = this.lastManifest.persistedInventory;
587
620
  this.lastContextProfileKey = modelProfileKey(
588
621
  this.lastManifest.provider,
589
622
  this.lastManifest.model,
@@ -762,6 +795,79 @@ export class Ds4ContextRuntime {
762
795
  };
763
796
  }
764
797
 
798
+ /**
799
+ * Optional per-million pricing proxied from Pi model metadata; undefined
800
+ * values degrade cache-aware planning to the previous behavior.
801
+ */
802
+ private static cachePricing(cost: ModelDescriptor["cost"]): CachePricing | undefined {
803
+ if (cost === undefined) return undefined;
804
+ return {
805
+ inputPerMillion: cost.input,
806
+ cacheReadPerMillion: cost.cacheRead,
807
+ cacheWritePerMillion: cost.cacheWrite,
808
+ outputPerMillion: cost.output,
809
+ };
810
+ }
811
+
812
+ /**
813
+ * Estimated cost of keeping a plan for an epoch of `turnsPerEpoch` user
814
+ * turns with `requestsPerTurn` provider requests each, in dollars.
815
+ *
816
+ * Model (documented, deterministic):
817
+ * - STABLE plan (original conversation fits the tail cap): the first
818
+ * request of the epoch is cold (it reuses only the prefix shared with the
819
+ * previously sent plan); every following request is warm.
820
+ * - SLIDING plan (the conversation exceeds the tail cap): the prefix is
821
+ * invalidated by every new turn, so each turn of the epoch pays one cold
822
+ * request plus the remaining warm requests.
823
+ *
824
+ * Returns undefined when pricing is incomplete or the plan is not managed.
825
+ * Costs are metadata estimates only; no provider content is exposed.
826
+ */
827
+ private cacheAwareEpochCost(
828
+ plan: ManagedContextPlan<ContextEvent["messages"][number]>,
829
+ fixedTokens: number,
830
+ cost: NonNullable<ModelDescriptor["cost"]>,
831
+ requestsPerTurn: number,
832
+ turnsPerEpoch: number,
833
+ modelKey: string,
834
+ ): number | undefined {
835
+ const pricing = Ds4ContextRuntime.cachePricing(cost);
836
+ if (!pricing || plan.mode !== "managed") return undefined;
837
+ const hashes: string[] = [];
838
+ const tokens: number[] = [];
839
+ let total = fixedTokens;
840
+ for (const message of plan.messages) {
841
+ hashes.push(fingerprint(message));
842
+ const estimate = estimateMessagesTokens([message]);
843
+ tokens.push(estimate);
844
+ total += estimate;
845
+ }
846
+ const previousHashes = modelKey !== this.lastPlanModelKey ? [] : this.lastPlanMessageHashes;
847
+ const reusable = previousHashes.length > 0
848
+ ? estimateReusablePrefixTokens(previousHashes, hashes, tokens)
849
+ : 0;
850
+ const coldCost = estimateRequestCost({
851
+ totalInputTokens: total,
852
+ reusablePrefixTokens: reusable,
853
+ }, pricing);
854
+ const warmCost = estimateRequestCost({
855
+ totalInputTokens: total,
856
+ reusablePrefixTokens: total,
857
+ }, pricing);
858
+ if (coldCost.total === undefined || warmCost.total === undefined) return undefined;
859
+ const sliding = plan.planning.originalMessageTokens > plan.planning.recentTailTokenLimit;
860
+ if (sliding) {
861
+ // Each turn of the epoch invalidates the prefix: one cold request plus
862
+ // the remaining warm requests per turn.
863
+ const perTurn = coldCost.total + Math.max(0, requestsPerTurn - 1) * warmCost.total;
864
+ return roundedCost(perTurn * turnsPerEpoch);
865
+ }
866
+ // Stable plan: one cold transition for the epoch, then all warm.
867
+ const epochCost = coldCost.total + Math.max(0, requestsPerTurn * turnsPerEpoch - 1) * warmCost.total;
868
+ return roundedCost(epochCost);
869
+ }
870
+
765
871
  private resolveModelPolicy(model: ModelDescriptor): {
766
872
  awareness: ResolvedModelAwareness;
767
873
  budget: ContextBudget;
@@ -973,10 +1079,119 @@ export class Ds4ContextRuntime {
973
1079
  memoryTokens: 0,
974
1080
  };
975
1081
  if (this.memoryManager) this.lastMemory = this.memoryManager.diagnostics();
1082
+ const fixedTokens = baseline.composition.systemTokens + baseline.composition.toolTokens;
1083
+ const pinnedMessageIndices = findPiPinnedMessageIndices(effectiveEvent, ctx);
1084
+ const retrievalEnabled = this.config.retrieval.exact
1085
+ || this.config.retrieval.fts
1086
+ || this.config.retrieval.semantic;
1087
+ const dedupSupplementalMessages = [
1088
+ ...memorySelection.pins.map((evidence) => ({
1089
+ id: `pin:${evidence.item.id}`,
1090
+ message: evidence.message,
1091
+ kind: "pin" as const,
1092
+ sourceIds: [evidence.item.id],
1093
+ score: 950,
1094
+ reason: evidence.reason,
1095
+ })),
1096
+ ...memorySelection.memories.map((evidence) => ({
1097
+ id: `memory:${evidence.item.id}`,
1098
+ message: evidence.message,
1099
+ kind: "memory" as const,
1100
+ sourceIds: [evidence.item.id],
1101
+ score: 90 + Math.min(0.999999, Math.max(0, evidence.score) / 1_000),
1102
+ reason: evidence.reason,
1103
+ })),
1104
+ ] satisfies Array<SupplementalContextMessage<ContextEvent["messages"][number]>>;
1105
+ const nominalDedupPlan = planManagedContext({
1106
+ messages: effectiveEvent.messages,
1107
+ fixedTokens,
1108
+ budget,
1109
+ config: effectiveContextConfig,
1110
+ pinnedMessageIndices,
1111
+ supplementalMessages: dedupSupplementalMessages,
1112
+ });
1113
+ /**
1114
+ * Cache-aware candidate selection (opt-in, default off): compares the
1115
+ * estimated input cost of the nominal plan with an extended-tail plan
1116
+ * and switches only when the improvement beats the configured
1117
+ * hysteresis threshold. Costs are metadata estimates from model
1118
+ * pricing and message fingerprints; no provider content is exposed.
1119
+ */
1120
+ let cacheAwareTailTokens: number | undefined;
1121
+ let cacheAwareDecision: CacheAwarePlanDecision | undefined;
1122
+ if (effectiveContextConfig.cacheAware?.mode === "auto" && model?.cost && budget) {
1123
+ const decision = decideCacheAwareTail({
1124
+ config: effectiveContextConfig.cacheAware,
1125
+ pricing: Ds4ContextRuntime.cachePricing(model.cost),
1126
+ observedCacheReadShare: activeModel?.awareness.calibration.cache.sampleCount > 0
1127
+ ? activeModel.awareness.calibration.cache.cacheReadShare
1128
+ : undefined,
1129
+ sampleCount: activeModel?.awareness.calibration.cache.sampleCount ?? 0,
1130
+ nominalRecentTailTokens: activeModel?.awareness.limits.recentTailTokens
1131
+ ?? effectiveContextConfig.recentTailTokens,
1132
+ activeInputBudget: budget.activeInputBudget,
1133
+ });
1134
+ cacheAwareDecision = decision;
1135
+ if (decision.eligible && decision.tailExtended) {
1136
+ const extendedDedupPlan = planManagedContext({
1137
+ messages: effectiveEvent.messages,
1138
+ fixedTokens,
1139
+ budget,
1140
+ config: effectiveContextConfig,
1141
+ pinnedMessageIndices,
1142
+ supplementalMessages: dedupSupplementalMessages,
1143
+ cacheAwareTailTokens: decision.recentTailTokens,
1144
+ });
1145
+ const modelKey = modelProfileKey(model.provider, model.id);
1146
+ const nominalEpoch = this.cacheAwareEpochCost(nominalDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
1147
+ const extendedEpoch = this.cacheAwareEpochCost(extendedDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
1148
+ this.logger.debug("context.cache_aware_candidate", {
1149
+ eligible: decision.eligible,
1150
+ tailExtended: decision.tailExtended,
1151
+ recentTailTokens: decision.recentTailTokens,
1152
+ nominalEpoch,
1153
+ extendedEpoch,
1154
+ });
1155
+ const extendedWon = extendedEpoch !== undefined
1156
+ && nominalEpoch !== undefined
1157
+ && extendedEpoch < nominalEpoch * (1 - effectiveContextConfig.cacheAware.minimumImprovementRatio);
1158
+ // Hysteresis (deliberate epochs): once adopted, the extended tail is
1159
+ // kept while it loses at most a single epoch comparison; it is
1160
+ // dropped only after two consecutive losses, preventing
1161
+ // oscillation between plans. The streak resets on a win or model switch.
1162
+ if (extendedWon) {
1163
+ this.cacheAwareExtendedLossStreak = 0;
1164
+ this.cacheAwareStickExtended = true;
1165
+ } else if (this.cacheAwareStickExtended) {
1166
+ this.cacheAwareExtendedLossStreak += 1;
1167
+ if (this.cacheAwareExtendedLossStreak >= effectiveContextConfig.cacheAware.stickinessEpochs) {
1168
+ this.cacheAwareStickExtended = false;
1169
+ this.cacheAwareExtendedLossStreak = 0;
1170
+ }
1171
+ }
1172
+ if (this.cacheAwareStickExtended) {
1173
+ cacheAwareTailTokens = decision.recentTailTokens;
1174
+ }
1175
+ }
1176
+ }
1177
+ const retrievalActiveContextEntryIds = retrievalEnabled
1178
+ ? this.plannedContextEntryIds(ctx, cacheAwareTailTokens !== undefined
1179
+ ? planManagedContext({
1180
+ messages: effectiveEvent.messages,
1181
+ fixedTokens,
1182
+ budget,
1183
+ config: effectiveContextConfig,
1184
+ pinnedMessageIndices,
1185
+ supplementalMessages: dedupSupplementalMessages,
1186
+ cacheAwareTailTokens,
1187
+ })
1188
+ : nominalDedupPlan)
1189
+ : undefined;
976
1190
  const retrieval = this.retrieveHistory(
977
1191
  effectiveEvent,
978
1192
  ctx,
979
1193
  effectiveContextConfig.maxRetrievedHistoryTokens,
1194
+ retrievalActiveContextEntryIds,
980
1195
  );
981
1196
  const project = this.retrieveProjectKnowledge(
982
1197
  requestText,
@@ -1110,11 +1325,12 @@ export class Ds4ContextRuntime {
1110
1325
  });
1111
1326
  const plan = planManagedContext({
1112
1327
  messages: effectiveEvent.messages,
1113
- fixedTokens: baseline.composition.systemTokens + baseline.composition.toolTokens,
1328
+ fixedTokens,
1114
1329
  budget,
1115
1330
  config: effectiveContextConfig,
1116
- pinnedMessageIndices: findPiPinnedMessageIndices(effectiveEvent, ctx),
1331
+ pinnedMessageIndices,
1117
1332
  supplementalMessages: rankedSupplementalMessages,
1333
+ ...(cacheAwareTailTokens !== undefined ? { cacheAwareTailTokens } : {}),
1118
1334
  ...(ranking.diagnostics.status === "active"
1119
1335
  ? { supplementalSelectionOrder: ranking.ranked.map((candidate) => candidate.id) }
1120
1336
  : {}),
@@ -1175,6 +1391,75 @@ export class Ds4ContextRuntime {
1175
1391
  selected: plannerSelectedProject,
1176
1392
  };
1177
1393
  plan.planning.durationMs = Math.max(0, this.now() - planningStartedAt);
1394
+ /**
1395
+ * Cache-aware diagnostics are metadata-only (tokens, ratios, cost).
1396
+ * The reusable prefix is estimated against the previous managed plan;
1397
+ * if the strategy changed since the last plan, the estimate is
1398
+ * conservative (empty hashes) rather than optimistic. The block is
1399
+ * present only when the cache-aware policy is enabled (mode auto);
1400
+ * mode off leaves the manifest unchanged from 0.3.6.
1401
+ */
1402
+ if (plan.mode === "managed" && effectiveContextConfig.cacheAware?.mode === "auto") {
1403
+ const currentModelKey = model ? modelProfileKey(model.provider, model.id) : undefined;
1404
+ const previousHashes = currentModelKey !== undefined && this.lastPlanModelKey === currentModelKey
1405
+ ? this.lastPlanMessageHashes
1406
+ : [];
1407
+ const currentHashes: string[] = [];
1408
+ const currentTokens: number[] = [];
1409
+ for (const message of plan.messages) {
1410
+ currentHashes.push(fingerprint(message));
1411
+ currentTokens.push(estimateMessagesTokens([message]));
1412
+ }
1413
+ const reusablePrefixTokens = estimateReusablePrefixTokens(
1414
+ previousHashes,
1415
+ currentHashes,
1416
+ currentTokens,
1417
+ );
1418
+ const estimatedCost = model?.cost
1419
+ ? estimateRequestCost({
1420
+ totalInputTokens: plan.planning.fixedTokens
1421
+ + plan.selected.reduce((total, item) => total + item.tokens, 0),
1422
+ reusablePrefixTokens,
1423
+ }, Ds4ContextRuntime.cachePricing(model.cost) ?? {}).total
1424
+ : undefined;
1425
+ const candidate: "nominal" | "cache-aware" = cacheAwareTailTokens !== undefined
1426
+ ? "cache-aware"
1427
+ : "nominal";
1428
+ plan.planning.cacheAware = {
1429
+ eligible: cacheAwareDecision?.eligible ?? false,
1430
+ tailExtended: cacheAwareTailTokens !== undefined,
1431
+ ...(cacheAwareDecision?.recentTailTokens !== undefined
1432
+ ? { recentTailTokens: cacheAwareDecision.recentTailTokens }
1433
+ : {}),
1434
+ ...(cacheAwareDecision?.missHitRatio !== undefined
1435
+ ? { missHitRatio: cacheAwareDecision.missHitRatio }
1436
+ : {}),
1437
+ ...(cacheAwareDecision?.cacheReadShare !== undefined
1438
+ ? { cacheReadShare: cacheAwareDecision.cacheReadShare }
1439
+ : {}),
1440
+ ...(cacheAwareDecision?.sampleCount !== undefined
1441
+ ? { sampleCount: cacheAwareDecision.sampleCount }
1442
+ : {}),
1443
+ ...(reusablePrefixTokens > 0 ? { reusablePrefixTokens } : {}),
1444
+ ...(estimatedCost !== undefined ? { estimatedCost } : {}),
1445
+ candidate,
1446
+ };
1447
+ this.lastPlanMessageHashes = currentHashes;
1448
+ this.lastPlanMessageTokens = currentTokens;
1449
+ this.lastCacheAwareDecision = cacheAwareDecision;
1450
+ this.lastPlanModelKey = currentModelKey;
1451
+ }
1452
+ const oversizedTurnExclusions = plan.planning.oversizedTurnExclusions ?? 0;
1453
+ if (plan.mode === "managed" && oversizedTurnExclusions > 0) {
1454
+ this.logger.warn("context.excluded_oversized_turn", {
1455
+ oversizedTurnCount: oversizedTurnExclusions,
1456
+ rescuedImmediatePredecessor: plan.planning.rescuedImmediatePredecessor ?? false,
1457
+ recentTailTokenLimit: plan.planning.recentTailTokenLimit,
1458
+ messageTargetTokens: plan.planning.messageTargetTokens,
1459
+ selectedGroupCount: plan.planning.selectedGroupCount,
1460
+ excludedGroupCount: plan.planning.excludedGroupCount,
1461
+ });
1462
+ }
1178
1463
  const plannedEvent: ContextEvent = { type: "context", messages: plan.messages };
1179
1464
  const selectedArtifactReferences = plan.mode === "managed"
1180
1465
  ? plan.selected.flatMap((metadata) => {
@@ -1362,6 +1647,9 @@ export class Ds4ContextRuntime {
1362
1647
  }
1363
1648
  try {
1364
1649
  const result = this.database.manifests.save(manifest);
1650
+ if (result.status === "stored") {
1651
+ this.lastPersistedInventory = result.inventory;
1652
+ }
1365
1653
  if (result.status === "skipped-oversize") {
1366
1654
  this.logger.warn("context.manifest_persistence_skipped", {
1367
1655
  category: "oversize",
@@ -2758,10 +3046,36 @@ export class Ds4ContextRuntime {
2758
3046
  }
2759
3047
  }
2760
3048
 
3049
+ private plannedContextEntryIds(
3050
+ ctx: ExtensionContext,
3051
+ plan: ManagedContextPlan<ContextEvent["messages"][number]>,
3052
+ ): Set<string> {
3053
+ if (plan.mode === "fallback") {
3054
+ return new Set(ctx.sessionManager.buildContextEntries().map((entry) => entry.id));
3055
+ }
3056
+ const syntheticIndices = new Set(
3057
+ [...plan.selected, ...plan.excluded]
3058
+ .filter((metadata) =>
3059
+ metadata.kind === "memory"
3060
+ || (metadata.kind === "pin" && metadata.sourceId !== undefined)
3061
+ )
3062
+ .map((metadata) => metadata.originalIndex),
3063
+ );
3064
+ const selectedIndices = new Set(plan.selected.map((metadata) => metadata.originalIndex));
3065
+ const sources = findPiSourceEntryIds(plan.originalMessages, ctx, syntheticIndices);
3066
+ const entryIds = new Set<string>();
3067
+ for (const index of selectedIndices) {
3068
+ const sourceId = sources[index];
3069
+ if (sourceId) entryIds.add(sourceId);
3070
+ }
3071
+ return entryIds;
3072
+ }
3073
+
2761
3074
  private retrieveHistory(
2762
3075
  event: ContextEvent,
2763
3076
  ctx: ExtensionContext,
2764
3077
  maxTokens = this.config.context.maxRetrievedHistoryTokens,
3078
+ activeContextEntryIds?: ReadonlySet<string>,
2765
3079
  ): RetrievalDiagnostics {
2766
3080
  const requestText = currentRequestText(event.messages);
2767
3081
  if (!this.retrievalEngine || !this.session?.sessionFile || !requestText) {
@@ -2775,7 +3089,8 @@ export class Ds4ContextRuntime {
2775
3089
  sessionId: this.session.sessionId,
2776
3090
  requestText,
2777
3091
  activeBranchEntryIds: new Set(ctx.sessionManager.getBranch().map((entry) => entry.id)),
2778
- activeContextEntryIds: new Set(ctx.sessionManager.buildContextEntries().map((entry) => entry.id)),
3092
+ activeContextEntryIds: activeContextEntryIds
3093
+ ?? new Set(ctx.sessionManager.buildContextEntries().map((entry) => entry.id)),
2779
3094
  exact: this.config.retrieval.exact,
2780
3095
  fts: this.config.retrieval.fts,
2781
3096
  semantic: this.config.retrieval.semantic,
@@ -2965,6 +3280,7 @@ export class Ds4ContextRuntime {
2965
3280
  ...(indexed ? { indexed } : {}),
2966
3281
  ...(this.observation ? { observation: this.observation } : {}),
2967
3282
  ...(this.lastManifest ? { lastManifest: this.lastManifest } : {}),
3283
+ ...(this.lastPersistedInventory ? { persistedInventory: this.lastPersistedInventory } : {}),
2968
3284
  retrieval: this.lastRetrieval,
2969
3285
  project: this.lastProject,
2970
3286
  memory: this.lastMemory,
@@ -87,7 +87,7 @@ function sourceKind(entry: SessionEntry, role?: string): ContextManifestItemKind
87
87
  return "history";
88
88
  }
89
89
 
90
- function fingerprint(message: unknown): string {
90
+ export function fingerprint(message: unknown): string {
91
91
  return sha256(stableStringify(message));
92
92
  }
93
93
 
@@ -244,6 +244,22 @@ export function findExactPiMessageSourceIds(
244
244
  });
245
245
  }
246
246
 
247
+ /**
248
+ * Maps managed-plan messages back to Pi session entry ids. Synthetic evidence
249
+ * indices are skipped so role/order fallbacks stay aligned with real messages.
250
+ */
251
+ export function findPiSourceEntryIds(
252
+ messages: readonly unknown[],
253
+ ctx: Pick<ExtensionContext, "sessionManager">,
254
+ syntheticIndices: ReadonlySet<number> = new Set(),
255
+ ): Array<string | undefined> {
256
+ return mapMessageSources(
257
+ messages,
258
+ sourceCandidates(ctx.sessionManager.buildContextEntries()),
259
+ syntheticIndices,
260
+ ).map((source) => source.sourceId);
261
+ }
262
+
247
263
  export function findPiPinnedMessageIndices(event: ContextEvent, ctx: ExtensionContext): number[] {
248
264
  const candidates = sourceCandidates(ctx.sessionManager.buildContextEntries());
249
265
  const sources = mapMessageSources(event.messages, candidates);
@@ -35,5 +35,15 @@ export function snapshotModel(ctx: Pick<ExtensionContext, "model">): ModelDescri
35
35
  maxTokens: ctx.model.maxTokens,
36
36
  reasoning: ctx.model.reasoning,
37
37
  input: ctx.model.input,
38
+ ...(ctx.model.cost && typeof ctx.model.cost.input === "number"
39
+ ? {
40
+ cost: {
41
+ input: ctx.model.cost.input,
42
+ output: ctx.model.cost.output,
43
+ cacheRead: ctx.model.cost.cacheRead,
44
+ cacheWrite: ctx.model.cost.cacheWrite,
45
+ },
46
+ }
47
+ : {}),
38
48
  };
39
49
  }
@@ -1,4 +1,4 @@
1
- export const EXTENSION_VERSION = "0.3.5";
1
+ export const EXTENSION_VERSION = "0.3.7";
2
2
  export const SUPPORTED_PI_VERSION = "0.84.3";
3
3
  export const OBSERVER_PLANNER_VERSION = "observer-model-aware-v1";
4
4
  export const PLANNER_VERSION = "managed-learned-ranking-v1";