ds4-context-engine 0.3.6 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,93 @@
1
+ # ADR-062 — Cache-aware context planning
2
+
3
+ Status: accepted and implemented in the coordinated 0.3.7 release (opt-in, default off; see the [release record](../releases/0.3.7.md)).
4
+
5
+ ## Context
6
+
7
+ DS4 reduces provider input by selecting a bounded managed context. This is
8
+ economically sound when every input token costs the same or when the provider
9
+ offers no prompt cache. Providers with a large context window and a large
10
+ cache-miss/cache-hit price ratio (for example DeepSeek V4 Flash, roughly 31x
11
+ off-peak) invert the trade-off: a small but frequently re-planned prompt can
12
+ cost more than a larger stable one, because every re-plan invalidates the
13
+ provider prefix cache.
14
+
15
+ The 0.3.6 planner applies an automatic 64k recent-tail ceiling to all models
16
+ above 256k until the conversational tail exceeds the cap. When the cap is
17
+ exceeded, the oldest turn leaves and the shared prefix with the previous
18
+ request collapses: everything after the first divergence becomes a cache miss.
19
+ Whether this is more expensive than a wider tail depends on the workload
20
+ (requests per turn, turns per epoch, share of cached reads observed).
21
+
22
+ The runtime already records cache read/write shares per exact `provider/model`
23
+ (volatile calibration plus persisted manifests) and Pi exposes per-million
24
+ cost rates on the active model. The portable core never hardcodes prices.
25
+
26
+ ## Decision
27
+
28
+ - Add an optional economic cost profile to the portable `ModelDescriptor`
29
+ (`input`, `output`, `cacheRead`, `cacheWrite` per million), populated from
30
+ Pi model metadata in `snapshotModel`. Absent fields degrade cache-aware
31
+ planning to the previous behavior.
32
+ - Add `context.cacheAware` (default `off`), a policy object with:
33
+ - `mode`: `off` preserves the previous planner exactly; `auto` may extend
34
+ the recent tail when pricing and observed cache shares justify it;
35
+ - `minimumCacheSampleCount`, `minimumCacheReadShare`,
36
+ `minimumMissHitRatio`: hard gates before any extension is eligible;
37
+ - `minimumImprovementRatio`: relative cost improvement required before an
38
+ alternative plan is adopted (hysteresis);
39
+ - `maxTailBudgetShare`: the extended tail may never exceed this fraction of
40
+ the active input budget;
41
+ - `expectedRequestsPerTurn`, `expectedTurnsPerEpoch`, `stickinessEpochs`:
42
+ deterministic cost model horizon plus hysteresis before dropping an
43
+ adopted extended tail.
44
+ - In `auto` the runtime computes `context.cacheAware` decision and compares
45
+ two plan candidates (nominal tail vs extended tail) with a deterministic
46
+ epoch cost model:
47
+ - stable plans (original conversation fits their tail cap) pay one cold
48
+ transition per epoch and warm requests afterwards;
49
+ - sliding plans (conversation exceeds their cap) pay a cold request every
50
+ turn of the epoch.
51
+ The extended plan is adopted only when it is cheaper over the whole epoch,
52
+ after the minimum improvement margin.
53
+ - The planner accepts an optional `cacheAwareTailTokens` override that bypasses
54
+ the automatic context-window ceiling; it remains bounded by the active/hard
55
+ input budgets and atomic groups, so no hard limit, privacy, pin, current
56
+ request or atomicity guarantee is weakened.
57
+ - Manifest `planning.cacheAware` (optional, numbers only, never content)
58
+ records the decision, tail, miss/hit ratio, observed cache share, sample
59
+ count, estimated reusable prefix tokens, estimated request cost and the
60
+ winning candidate. `/context tokens` and `/context explain` surface the same
61
+ metadata. `context.excluded_oversized_turn` and previous guarantees are
62
+ unchanged.
63
+ - Add a deterministic synthetic prefix-cache simulator covering the report
64
+ scenarios (append-only under the cap, sliding past 64k, one response per
65
+ prompt, five tool cycles, wide vs sliding tails, model switch, pricing
66
+ comparison). No provider, no credentials, no CI network access.
67
+
68
+ ## Rejected alternatives
69
+
70
+ - Hardcoding provider prices: rejected, the core stays provider-independent.
71
+ - Raising the default tail ceiling globally: rejected, it changes behavior
72
+ for all models (including ones without cache discounts).
73
+ - Always extending the tail for eligible models: rejected, the epoch model
74
+ shows the nominal plan can still be cheaper on workloads where the tail
75
+ slides infrequently; the improvement margin keeps the switch conservative.
76
+
77
+ ## Compatibility and validation
78
+
79
+ Configuration is additive; `context.cacheAware.mode` defaults to `off`, so
80
+ `0.3.6` behavior (and manifest absence of the cacheAware block) is preserved.
81
+ Automatic behavior is opt-in and requires observed samples plus positive
82
+ pricing. The epoch cost is an estimate used only for the comparison; the
83
+ actual billing remains whatever the provider reports. The workaround override
84
+ (`modelAwareness.overrides` with a large `recentTailTokens`, zeroed retrieval
85
+ and disabled compaction) remains available and is unaffected; it is more
86
+ aggressive than `mode: "auto"` because it also removes retrieval/project
87
+ supplements and compaction, which `auto` deliberately preserves for quality.
88
+
89
+ Coverage: cache-policy unit tests (prefix, costs, decision gates), config
90
+ validation, planner override tests, the synthetic prefix-cache simulator, and
91
+ runtime integration tests for off/auto/no-discount/under-budget paths. A
92
+ real-provider DeepSeek A/B benchmark remains out of scope and voluntary
93
+ (protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
@@ -65,5 +65,6 @@ The initial decisions from the development plan are accepted:
65
65
  | [059](059-optional-anchored-editing.md) | Opt in to anchored edit expansion inside Pi's native mutation queue | Accepted |
66
66
  | [060](060-optional-portable-agent-tools.md) | Opt in to edit reports, adaptive reads/results and session-owned local jobs | Accepted |
67
67
  | [061](061-compaction-latency.md) | Bound compaction update calls, input budgets, concurrent segments and phase timings | Accepted |
68
+ | [062](062-cache-aware-context-planning.md) | Opt-in cache-aware tail planning using model pricing and observed cache shares | Accepted |
68
69
 
69
70
  Each decision will receive a dedicated record when implementation pressure introduces alternatives or consequences not already covered by the development plan.
@@ -0,0 +1,66 @@
1
+ # Benchmark A/B DeepSeek — cache-aware planning
2
+
3
+ Protocollo volontario, fuori CI, nessuna credenziale nel repo. Serve a decidere se
4
+ `context.cacheAware.mode` può passare da `auto` (opt-in) a **default** in una
5
+ release futura. Gate dichiarato in [ADR-062](ADR/062-cache-aware-context-planning.md).
6
+
7
+ ## Obiettivo
8
+
9
+ Misurare, sulla **stessa** conversazione reale, il costo provider per turno con:
10
+
11
+ - **Sessione A (baseline)**: comportamento 0.3.6 — `context.cacheAware` assente
12
+ (default `off`), tail automatica 64k, retrieval 16k, project 20k.
13
+ - **Sessione B (auto)**: `context.cacheAware.mode: "auto"` con i default di
14
+ policy (gates 0.5 cache-share, ratio 20, margine 10%).
15
+ - **Sessione C (workaround estremo, opzionale)**: `modelAwareness.overrides` con
16
+ `recentTailTokens: 500000`, `maxRetrievedHistoryTokens: 0`,
17
+ `maxProjectTokens: 0`, `compaction.enabled: false` — cache-first puro.
18
+
19
+ Le tre sessioni devono avere la **stessa sequenza di turni** (stesso carico di
20
+ lavoro: tool call, retrieval, project touches) e lo **stesso modello/provided
21
+ ID DeepSeek**.
22
+
23
+ ## Cosa registrare per ogni turno
24
+
25
+ Da `/context tokens` e `/context explain`:
26
+
27
+ - `recentTailTokens` scelti (nominal vs extended) e `cacheAware` decision/reasons;
28
+ - manifest `planning.cacheAware` (missHitRatio, cacheReadShare, sampleCount,
29
+ reusablePrefixTokens, estimatedCost, candidate);
30
+ - `ProviderCacheMetrics` reali dopo ogni richiesta: `inputTokens`,
31
+ `cacheReadTokens`, `cacheWriteTokens`, share read/write per provider/model.
32
+
33
+ Dal provider (costo effettivo):
34
+
35
+ - input non-cached, cache-read (hit), cache-write, output per richiesta;
36
+ - costo totale della sessione (per-turno e totale).
37
+
38
+ ## Metriche
39
+
40
+ 1. **Costo per turno** (media e totale sessione): A vs B vs C.
41
+ 2. **Cache-read share osservata** per config: conferma che la tail estesa
42
+ mantiene il prefisso stabile (share alta) vs sliding (share ~0 dopo 64k).
43
+ 3. **Qualità** (non solo $$$): recall retrieval, `Planner exclusions`,
44
+ current request/atomicità intatte, completamenti corretti nelle fasi tool.
45
+ Con retrieval azzerato (C) documentare i regressi di qualità.
46
+ 4. **Eventi di transizione**: quante volte il prefisso si è invalidato (turni
47
+ con cacheRead ≈ 0), confronto sliding vs stable.
48
+
49
+ ## Criterio di promozione a default
50
+
51
+ Promuovere `mode: "auto"` a default solo se, su ≥ 3 sessioni reali:
52
+
53
+ - costo medio per turno di B < A (margine ≥ il `minimumImprovementRatio`
54
+ configurato, di default 10%);
55
+ - nessun regresso di qualità misurabile (recall retrieval uguale o migliore;
56
+ current request/atomicità preservate);
57
+ - nessuna oscillazione plan (decisione che alterna nominal/esteso senza
58
+ motivo: verificare `stickinessEpochs` e diagnostica).
59
+
60
+ ## Esecuzione sicura
61
+
62
+ - Nessuna credenziale nel repo; il benchmark usa la sessione Pi normale con il
63
+ provider DeepSeek già autenticato.
64
+ - Non modificare la config attiva in modo permanente: copie di configurazione
65
+ per sessione, o valori temporanei poi ripristinati.
66
+ - Fuori CI: niente rete/credenziali nel test suite.
@@ -20,6 +20,7 @@ A Context Manifest explains the context visible at DS4's Pi `context` hook witho
20
20
  - artifact IDs, SHA-256, bytes, MIME, classification, exact source entry/tool IDs, error state, and before/after token estimates;
21
21
  - provider destination and allow-set names, selected classification counts, blocked/excluded/redacted counts, final provider-check count, and enforcement stage;
22
22
  - planner mode/version, original and selected counts, group counts, internal budgets, duration, and fallback reason;
23
+ - optional cache-aware planning decision: eligibility, tail extension, requested tail tokens, miss/hit price ratio, observed cache-read share, sample count, estimated reusable prefix tokens, estimated request cost in dollars, and winning candidate (`nominal`/`cache-aware`); numbers only, never content;
23
24
  - learned-ranking mode/status, feature/model versions, candidate count, aggregate disagreement/rank shift, duration, and generic static-fallback reason;
24
25
  - planner and policy versions;
25
26
  - deterministic SHA-256 over system prompt, active tools, and messages;
@@ -54,6 +55,8 @@ A manifest at or below 256 KiB is stored unchanged. Above that preferred bound,
54
55
 
55
56
  Each manifest transaction prunes at most 32 excess rows and 8 MiB of serialized payload; one individually oversized oldest row may exceed the byte limit to guarantee progress. Calibration pruning is independently limited to 32 rows per related profile write. This incrementally repairs an existing oversized database without adding a long startup write or extending SQLite lock duration with an unbounded purge. Deleted pages become reusable by SQLite; the database file may remain at its previous high-water size until explicit [offline maintenance](STORAGE_MAINTENANCE.md). No retention action edits canonical Pi JSONL or project files.
56
57
 
58
+ The `save()` result carries the derived `inventory` of the persisted projection, and the runtime exposes it through `RuntimeDiagnostics.persistedInventory`; `/context manifest` and `/context excluded` therefore report the truthful persisted completeness (`complete` or `excluded-rollup` with retained/total counts) instead of defaulting to `complete` when the in-memory manifest has no inventory attached.
59
+
57
60
  ## Reproducibility
58
61
 
59
62
  Object keys are normalized before hashing, so equivalent tool schemas with different key insertion order produce the same prompt hash. The estimator version is stored explicitly as `chars-v1`; planner/policy versions describe selection behavior. Golden tests protect manifest shape, model-profile resolution, token accounting, and hash stability.
@@ -86,6 +86,42 @@ M14 can queue the finalized manifest after planning when `quality.enabled` is tr
86
86
 
87
87
  M18 can evaluate bounded metadata-only features after privacy exclusion and before supplemental candidates enter category fitting. `shadow` keeps every static score/order authoritative and records aggregate disagreement only. `active` is accepted only for a compatible checksummed model carrying an eligible held-out promotion report. Privacy exclusions, mandatory pins/current turns, atomic groups and hard budgets cannot be overridden. See [`LEARNED_RANKING.md`](LEARNED_RANKING.md).
88
88
 
89
+ ## Cache-aware tail planning (0.3.7, opt-in)
90
+
91
+ With `context.cacheAware.mode = "auto"` the runtime may extend the recent tail
92
+ beyond the automatic context-window ceiling when the model pricing and the
93
+ observed cache shares justify it economically. The policy never hardcodes
94
+ prices: it reads the per-million rates exposed on the active Pi model and the
95
+ cache read/write shares already recorded for the exact `provider/model`.
96
+
97
+ Eligibility gates (defaults shown):
98
+
99
+ - `minimumCacheSampleCount: 3` observed calibration samples;
100
+ - `minimumCacheReadShare: 0.5` observed share;
101
+ - `minimumMissHitRatio: 20` cache-miss / cache-hit price ratio.
102
+
103
+ When eligible, the runtime compares two plan candidates with a deterministic
104
+ epoch cost model:
105
+
106
+ - the nominal plan (existing tail, sliding below 64k) pays a cold request on
107
+ every turn of the epoch because its prefix is invalidated by the slide;
108
+ - the extended plan (up to `maxTailBudgetShare` of the active input budget) is
109
+ stable when the conversation fits, so it pays one cold transition per epoch
110
+ and warm requests afterwards.
111
+
112
+ The extended plan is adopted only when it wins over the whole epoch
113
+ (`expectedRequestsPerTurn`, `expectedTurnsPerEpoch`) after the
114
+ `minimumImprovementRatio` margin. Once adopted it is kept (deliberate epoch)
115
+ until it loses `stickinessEpochs` consecutive comparisons, preventing
116
+ oscillation. The override remains bounded by the active and hard input
117
+ budgets and by atomic groups, so current request, pins, privacy and atomicity
118
+ guarantees are unchanged.
119
+
120
+ `context.cacheAware.mode` defaults to `off`, preserving the 0.3.6 behavior
121
+ exactly. Without prices, samples or a cache discount, the decision degrades to
122
+ the nominal plan automatically. See [ADR-062](ADR/062-cache-aware-context-planning.md)
123
+ for the model and rejected alternatives.
124
+
89
125
  ## Current limits
90
126
 
91
127
  The planner does not call a model inside the `context` hook. Model calibration uses only finalized provider usage and deterministic local statistics. Historical/project retrieval can opt into derived semantic candidates; learned supplemental reranking remains off by default and active mode is promotion-gated. Project symbol extraction is heuristic, artifact search is literal, and memory/pin creation is manual-first. Automatic memory extraction remains disabled; M10 supplies policy enforcement but not an automatic classifier or confirmation workflow. Provider-payload coverage targets Pi 0.84.3's supported serializers, and DS4 must load after any extension allowed to replace payloads when strict final ordering is required.
@@ -0,0 +1,83 @@
1
+ # Release 0.3.7 — Cache-aware context planning (opt-in)
2
+
3
+ **Version analyzed:** DS4 Context Engine `0.3.7`
4
+ **Commit:** `a03009a4b14d6ab1842fa800076c4b300bd70d8f`
5
+ **Coordinated packages:** `ds4-context-core` 0.3.7, `ds4-context-reference-adapter` 0.3.7, `ds4-context-engine` 0.3.7
6
+
7
+ ## Summary
8
+
9
+ Adds an opt-in cache-aware planning policy driven by model pricing and
10
+ observed cache shares, plus a deterministic synthetic prefix-cache simulator.
11
+ The default behavior is unchanged: `context.cacheAware.mode` defaults to `off`,
12
+ the manifest does not include a `cacheAware` block, and the planner uses the
13
+ same tail caps as 0.3.6.
14
+
15
+ ## New configuration
16
+
17
+ `context.cacheAware` (object, default off):
18
+
19
+ | Field | Default | Meaning |
20
+ |---|---|---|
21
+ | `mode` | `off` | `off` preserves 0.3.6 behavior; `auto` may extend the recent tail. |
22
+ | `minimumCacheSampleCount` | `3` | Observed samples required before acting. |
23
+ | `minimumCacheReadShare` | `0.5` | Observed cache-read share required before acting. |
24
+ | `minimumMissHitRatio` | `20` | Minimum cache-miss / cache-hit price ratio. |
25
+ | `minimumImprovementRatio` | `0.1` | Relative improvement required to switch plan. |
26
+ | `maxTailBudgetShare` | `0.5` | Max fraction of the active input budget for the extended tail. |
27
+ | `expectedRequestsPerTurn` | `4` | Provider requests per user turn in the cost model. |
28
+ | `expectedTurnsPerEpoch` | `4` | User turns per planning epoch in the cost model. |
29
+ | `stickinessEpochs` | `2` | Consecutive epoch losses before dropping an adopted extended tail. |
30
+
31
+ ## Changes
32
+
33
+ - `ModelDescriptor` gains an optional `cost` profile (`input`, `output`,
34
+ `cacheRead`, `cacheWrite` per million), populated from Pi model metadata in
35
+ `snapshotModel`. No prices are hardcoded in the core.
36
+ - `planManagedContext` accepts an optional `cacheAwareTailTokens` override that
37
+ bypasses the automatic context-window ceiling while remaining bounded by the
38
+ active/hard input budgets and atomic groups.
39
+ - Runtime: when `mode: "auto"` and the eligibility gates pass, the runtime
40
+ compares the nominal plan with an extended-tail plan using a deterministic
41
+ epoch cost model (stable plans pay one cold transition per epoch; sliding
42
+ plans pay a cold request per turn) and adopts the extended plan only when it
43
+ wins after the minimum improvement margin.
44
+ - Manifest: optional `planning.cacheAware` (numbers only, never content) with
45
+ eligibility, tail tokens, miss/hit ratio, observed share, sample count,
46
+ estimated reusable prefix tokens, estimated request cost and winning
47
+ candidate; surfaced in `/context tokens` and `/context explain`.
48
+ - Core: `packages/core/src/planner/cache-policy.ts` with pure, deterministic
49
+ functions for common-prefix estimation, request cost and the tail decision.
50
+ - Tests: `cache-policy` unit tests, config validation, planner override tests,
51
+ the synthetic prefix-cache simulator (`cache-prefix-simulator`) and runtime
52
+ integration tests for off/auto/no-discount/under-budget paths.
53
+
54
+ ## Compatibility
55
+
56
+ - Additive configuration; absent fields use the documented defaults.
57
+ - `context.cacheAware.mode: "off"` reproduces the 0.3.6 behavior exactly.
58
+ - No new required database schema; manifests persist the optional block as-is.
59
+ - The existing guarantees (current request, atomicity, privacy, pins, hard
60
+ limits, fail-open, naive compaction path) are unchanged.
61
+ - Model metadata without cache pricing or observed samples degrades the
62
+ decision to the nominal plan.
63
+
64
+ ## Validation
65
+
66
+ - `npm run check` (excluding the known machine-load-dependent
67
+ `long-session` timeout flake): 83 files, 550 tests passed.
68
+ - Planner unit tests: 25 passed.
69
+ - Cache-policy unit tests: 16 passed.
70
+ - Prefix-cache simulator: 7 passed.
71
+ - Cache-aware runtime integration: 4 passed.
72
+ - Config catalog/loader validation: passed.
73
+
74
+ ## Known limits
75
+
76
+ - The epoch cost model is an estimate for plan comparison; actual billing is
77
+ whatever the provider reports.
78
+ - A real-provider DeepSeek A/B benchmark remains voluntary and out of CI
79
+ (protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
80
+ - The workaround override (large `recentTailTokens`, zeroed retrieval and
81
+ disabled compaction) remains available and is unaffected; it is more aggressive
82
+ than `mode: "auto"`, which deliberately preserves retrieval/project and
83
+ compaction for quality.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ds4-context-engine",
3
- "version": "0.3.6",
3
+ "version": "0.3.7",
4
4
  "description": "Non-destructive, provider-independent context management for Pi.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -62,7 +62,7 @@
62
62
  ]
63
63
  },
64
64
  "dependencies": {
65
- "ds4-context-core": "0.3.6"
65
+ "ds4-context-core": "0.3.7"
66
66
  },
67
67
  "peerDependencies": {
68
68
  "@earendil-works/pi-ai": "0.84.3",
@@ -296,6 +296,14 @@ function formatTokens(diagnostics: RuntimeDiagnostics): string {
296
296
  `Actual provider input: ${count(manifest?.actualInputTokens)}`,
297
297
  ` Uncached input: ${count(manifest?.providerUsage?.inputTokens)}`,
298
298
  ` Cache read / write: ${count(manifest?.providerUsage?.cacheReadTokens)} / ${count(manifest?.providerUsage?.cacheWriteTokens)}`,
299
+ ...(manifest?.planning?.cacheAware
300
+ ? [
301
+ `Cache-aware plan: ${manifest.planning.cacheAware.candidate ?? "n/a"}${manifest.planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
302
+ ` Reusable prefix: ${count(manifest.planning.cacheAware.reusablePrefixTokens)} estimated`,
303
+ ` Est. request cost: ${manifest.planning.cacheAware.estimatedCost === undefined ? "n/a" : `$${manifest.planning.cacheAware.estimatedCost.toFixed(6)}`}`,
304
+ ` Miss/hit price ratio: ${manifest.planning.cacheAware.missHitRatio === undefined ? "n/a" : manifest.planning.cacheAware.missHitRatio.toFixed(2)}`,
305
+ ]
306
+ : []),
299
307
  `Pi reported context: ${count(manifest?.piReportedContextTokens ?? observation?.reportedTokens)}`,
300
308
  `Model context window: ${count(budget?.contextWindow ?? manifest?.contextWindow)}`,
301
309
  `Output reserve: ${count(budget?.outputReserve ?? manifest?.outputReserve)}`,
@@ -316,7 +324,7 @@ function formatManifest(diagnostics: RuntimeDiagnostics): string {
316
324
  const manifest = diagnostics.lastManifest;
317
325
  if (!manifest) return "No Context Manifest has been built for this session yet.";
318
326
 
319
- const inventory = manifest.persistedInventory;
327
+ const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
320
328
  const kinds = new Map<string, { items: number; tokens: number }>();
321
329
  for (const item of manifest.included) {
322
330
  const aggregate = kinds.get(item.kind) ?? { items: 0, tokens: 0 };
@@ -369,7 +377,7 @@ function formatManifestItems(diagnostics: RuntimeDiagnostics, type: "included" |
369
377
  const manifest = diagnostics.lastManifest;
370
378
  if (!manifest) return "No Context Manifest has been built for this session yet.";
371
379
  const items = manifest[type];
372
- const inventory = manifest.persistedInventory;
380
+ const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
373
381
  return [
374
382
  `DS4 Context ${type === "included" ? "Included" : "Excluded"} Items`,
375
383
  "",
@@ -409,6 +417,23 @@ function formatExplain(diagnostics: RuntimeDiagnostics): string {
409
417
  ...(planning.oversizedTurnExclusions
410
418
  ? [`Oversized turn excl: ${count(planning.oversizedTurnExclusions)} (turn group(s) at/above the recent-tail cap; recovered by retrieval only if it fits)`]
411
419
  : []),
420
+ ...(planning.cacheAware
421
+ ? [
422
+ `Cache-aware plan: ${planning.cacheAware.candidate ?? "n/a"}${planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
423
+ ...(planning.cacheAware.missHitRatio !== undefined
424
+ ? [` Miss/hit ratio: ${planning.cacheAware.missHitRatio.toFixed(2)}`]
425
+ : []),
426
+ ...(planning.cacheAware.cacheReadShare !== undefined
427
+ ? [` Cache-read share: ${(planning.cacheAware.cacheReadShare * 100).toFixed(1)}% (${count(planning.cacheAware.sampleCount)} samples)`]
428
+ : []),
429
+ ...(planning.cacheAware.reusablePrefixTokens !== undefined
430
+ ? [` Reusable prefix: ${count(planning.cacheAware.reusablePrefixTokens)} tokens`]
431
+ : []),
432
+ ...(planning.cacheAware.estimatedCost !== undefined
433
+ ? [` Est. request cost: $${planning.cacheAware.estimatedCost.toFixed(6)}`]
434
+ : []),
435
+ ]
436
+ : []),
412
437
  `Selected groups: ${count(planning.selectedGroupCount)}`,
413
438
  `Excluded groups: ${count(planning.excludedGroupCount)}`,
414
439
  `Duration: ${planning.durationMs === undefined ? "n/a" : `${planning.durationMs.toFixed(1)} ms`}`,
@@ -72,7 +72,10 @@ import type {
72
72
  PrivacyManifest,
73
73
  ProviderUsageManifest,
74
74
  } from "ds4-context-core/manifest/context-manifest";
75
- import { HARD_PERSISTED_MANIFEST_BYTES } from "ds4-context-core/manifest/context-manifest-storage";
75
+ import {
76
+ HARD_PERSISTED_MANIFEST_BYTES,
77
+ type PersistedManifestInventory,
78
+ } from "ds4-context-core/manifest/context-manifest-storage";
76
79
  import {
77
80
  unavailableStorageDiagnostics,
78
81
  type StorageDiagnostics,
@@ -98,6 +101,13 @@ import {
98
101
  type ProjectMemorySource,
99
102
  } from "ds4-context-core/memory/memory-types";
100
103
  import { planManagedContext, type ManagedContextPlan, type SupplementalContextMessage } from "ds4-context-core/planner/context-planner";
104
+ import {
105
+ decideCacheAwareTail,
106
+ estimateReusablePrefixTokens,
107
+ estimateRequestCost,
108
+ type CacheAwarePlanDecision,
109
+ type CachePricing,
110
+ } from "ds4-context-core/planner/cache-policy";
101
111
  import {
102
112
  disabledContextQualityDiagnostics,
103
113
  evaluateManifestQuality,
@@ -152,6 +162,7 @@ import {
152
162
  findExactPiMessageSourceIds,
153
163
  findPiPinnedMessageIndices,
154
164
  findPiSourceEntryIds,
165
+ fingerprint,
155
166
  } from "../pi-adapter/context-observer.ts";
156
167
  import { projectSessionFileMutations } from "../pi-adapter/memory-adapter.ts";
157
168
  import { ProjectMemorySynchronizer } from "../pi-adapter/project-memory-sync.ts";
@@ -284,6 +295,10 @@ function numericUsage(value: unknown): number {
284
295
  : 0;
285
296
  }
286
297
 
298
+ function roundedCost(value: number): number {
299
+ return Math.round(value * 1_000_000) / 1_000_000;
300
+ }
301
+
287
302
  function providerUsageManifest(input: {
288
303
  inputTokens: number;
289
304
  cacheReadTokens: number;
@@ -374,6 +389,8 @@ export interface RuntimeDiagnostics {
374
389
  indexed?: SessionIndexStats;
375
390
  observation?: ContextObservation;
376
391
  lastManifest?: ContextManifest;
392
+ /** Inventory of the persisted projection of the last successfully stored manifest. */
393
+ persistedInventory?: PersistedManifestInventory;
377
394
  retrieval: RetrievalDiagnostics;
378
395
  project: ProjectKnowledgeDiagnostics;
379
396
  memory: MemoryDiagnostics;
@@ -411,6 +428,7 @@ export class Ds4ContextRuntime {
411
428
  private databasePath?: string;
412
429
  private observation?: ContextObservation;
413
430
  private lastManifest?: ContextManifest;
431
+ private lastPersistedInventory?: PersistedManifestInventory;
414
432
  private pendingManifestId?: string;
415
433
  private pendingManifestPersisted = false;
416
434
  private retrievalEngine?: HistoricalRetrievalEngine;
@@ -433,6 +451,18 @@ export class Ds4ContextRuntime {
433
451
  private lastContextProfileKey?: string;
434
452
  private readonly knownModelProfiles = new Set<string>();
435
453
  private readonly volatileCalibration = new Map<string, TokenCalibrationSample[]>();
454
+ /** Fingerprints of the messages sent in the last managed plan (empty when unknown). */
455
+ private lastPlanMessageHashes: string[] = [];
456
+ /** Per-message token estimates of the last managed plan, aligned with hashes. */
457
+ private lastPlanMessageTokens: number[] = [];
458
+ /** Cache-aware decision of the last plan; used for /context diagnostics. */
459
+ private lastCacheAwareDecision?: CacheAwarePlanDecision;
460
+ /** Provider/model key of the last plan; a switch invalidates the cached prefix. */
461
+ private lastPlanModelKey?: string;
462
+ /** Consecutive epochs in which the extended tail lost the comparison (hysteresis). */
463
+ private cacheAwareExtendedLossStreak = 0;
464
+ /** Extended-tail adoption state: while set, the extended tail is kept even if a single epoch comparison loses. */
465
+ private cacheAwareStickExtended = false;
436
466
  private lastMemoryMutationSignature?: string;
437
467
  private artifactManager?: ArtifactManager;
438
468
  private lastArtifacts: ArtifactDiagnostics = disabledArtifactDiagnostics();
@@ -481,6 +511,7 @@ export class Ds4ContextRuntime {
481
511
  this.pendingQuality.length = 0;
482
512
  this.lastIndexResult = undefined;
483
513
  this.lastManifest = undefined;
514
+ this.lastPersistedInventory = undefined;
484
515
  this.pendingManifestId = undefined;
485
516
  this.pendingManifestPersisted = false;
486
517
  this.retrievalEngine = undefined;
@@ -585,6 +616,7 @@ export class Ds4ContextRuntime {
585
616
  this.syncSessionIndex(ctx);
586
617
  this.lastManifest = this.database.manifests.getLatest(this.session.sessionId);
587
618
  if (this.lastManifest) {
619
+ this.lastPersistedInventory = this.lastManifest.persistedInventory;
588
620
  this.lastContextProfileKey = modelProfileKey(
589
621
  this.lastManifest.provider,
590
622
  this.lastManifest.model,
@@ -763,6 +795,79 @@ export class Ds4ContextRuntime {
763
795
  };
764
796
  }
765
797
 
798
+ /**
799
+ * Optional per-million pricing proxied from Pi model metadata; undefined
800
+ * values degrade cache-aware planning to the previous behavior.
801
+ */
802
+ private static cachePricing(cost: ModelDescriptor["cost"]): CachePricing | undefined {
803
+ if (cost === undefined) return undefined;
804
+ return {
805
+ inputPerMillion: cost.input,
806
+ cacheReadPerMillion: cost.cacheRead,
807
+ cacheWritePerMillion: cost.cacheWrite,
808
+ outputPerMillion: cost.output,
809
+ };
810
+ }
811
+
812
+ /**
813
+ * Estimated cost of keeping a plan for an epoch of `turnsPerEpoch` user
814
+ * turns with `requestsPerTurn` provider requests each, in dollars.
815
+ *
816
+ * Model (documented, deterministic):
817
+ * - STABLE plan (original conversation fits the tail cap): the first
818
+ * request of the epoch is cold (it reuses only the prefix shared with the
819
+ * previously sent plan); every following request is warm.
820
+ * - SLIDING plan (the conversation exceeds the tail cap): the prefix is
821
+ * invalidated by every new turn, so each turn of the epoch pays one cold
822
+ * request plus the remaining warm requests.
823
+ *
824
+ * Returns undefined when pricing is incomplete or the plan is not managed.
825
+ * Costs are metadata estimates only; no provider content is exposed.
826
+ */
827
+ private cacheAwareEpochCost(
828
+ plan: ManagedContextPlan<ContextEvent["messages"][number]>,
829
+ fixedTokens: number,
830
+ cost: NonNullable<ModelDescriptor["cost"]>,
831
+ requestsPerTurn: number,
832
+ turnsPerEpoch: number,
833
+ modelKey: string,
834
+ ): number | undefined {
835
+ const pricing = Ds4ContextRuntime.cachePricing(cost);
836
+ if (!pricing || plan.mode !== "managed") return undefined;
837
+ const hashes: string[] = [];
838
+ const tokens: number[] = [];
839
+ let total = fixedTokens;
840
+ for (const message of plan.messages) {
841
+ hashes.push(fingerprint(message));
842
+ const estimate = estimateMessagesTokens([message]);
843
+ tokens.push(estimate);
844
+ total += estimate;
845
+ }
846
+ const previousHashes = modelKey !== this.lastPlanModelKey ? [] : this.lastPlanMessageHashes;
847
+ const reusable = previousHashes.length > 0
848
+ ? estimateReusablePrefixTokens(previousHashes, hashes, tokens)
849
+ : 0;
850
+ const coldCost = estimateRequestCost({
851
+ totalInputTokens: total,
852
+ reusablePrefixTokens: reusable,
853
+ }, pricing);
854
+ const warmCost = estimateRequestCost({
855
+ totalInputTokens: total,
856
+ reusablePrefixTokens: total,
857
+ }, pricing);
858
+ if (coldCost.total === undefined || warmCost.total === undefined) return undefined;
859
+ const sliding = plan.planning.originalMessageTokens > plan.planning.recentTailTokenLimit;
860
+ if (sliding) {
861
+ // Each turn of the epoch invalidates the prefix: one cold request plus
862
+ // the remaining warm requests per turn.
863
+ const perTurn = coldCost.total + Math.max(0, requestsPerTurn - 1) * warmCost.total;
864
+ return roundedCost(perTurn * turnsPerEpoch);
865
+ }
866
+ // Stable plan: one cold transition for the epoch, then all warm.
867
+ const epochCost = coldCost.total + Math.max(0, requestsPerTurn * turnsPerEpoch - 1) * warmCost.total;
868
+ return roundedCost(epochCost);
869
+ }
870
+
766
871
  private resolveModelPolicy(model: ModelDescriptor): {
767
872
  awareness: ResolvedModelAwareness;
768
873
  budget: ContextBudget;
@@ -979,32 +1084,108 @@ export class Ds4ContextRuntime {
979
1084
  const retrievalEnabled = this.config.retrieval.exact
980
1085
  || this.config.retrieval.fts
981
1086
  || this.config.retrieval.semantic;
982
- const retrievalActiveContextEntryIds = retrievalEnabled
983
- ? this.plannedContextEntryIds(ctx, planManagedContext({
1087
+ const dedupSupplementalMessages = [
1088
+ ...memorySelection.pins.map((evidence) => ({
1089
+ id: `pin:${evidence.item.id}`,
1090
+ message: evidence.message,
1091
+ kind: "pin" as const,
1092
+ sourceIds: [evidence.item.id],
1093
+ score: 950,
1094
+ reason: evidence.reason,
1095
+ })),
1096
+ ...memorySelection.memories.map((evidence) => ({
1097
+ id: `memory:${evidence.item.id}`,
1098
+ message: evidence.message,
1099
+ kind: "memory" as const,
1100
+ sourceIds: [evidence.item.id],
1101
+ score: 90 + Math.min(0.999999, Math.max(0, evidence.score) / 1_000),
1102
+ reason: evidence.reason,
1103
+ })),
1104
+ ] satisfies Array<SupplementalContextMessage<ContextEvent["messages"][number]>>;
1105
+ const nominalDedupPlan = planManagedContext({
1106
+ messages: effectiveEvent.messages,
1107
+ fixedTokens,
1108
+ budget,
1109
+ config: effectiveContextConfig,
1110
+ pinnedMessageIndices,
1111
+ supplementalMessages: dedupSupplementalMessages,
1112
+ });
1113
+ /**
1114
+ * Cache-aware candidate selection (opt-in, default off): compares the
1115
+ * estimated input cost of the nominal plan with an extended-tail plan
1116
+ * and switches only when the improvement beats the configured
1117
+ * hysteresis threshold. Costs are metadata estimates from model
1118
+ * pricing and message fingerprints; no provider content is exposed.
1119
+ */
1120
+ let cacheAwareTailTokens: number | undefined;
1121
+ let cacheAwareDecision: CacheAwarePlanDecision | undefined;
1122
+ if (effectiveContextConfig.cacheAware?.mode === "auto" && model?.cost && budget) {
1123
+ const decision = decideCacheAwareTail({
1124
+ config: effectiveContextConfig.cacheAware,
1125
+ pricing: Ds4ContextRuntime.cachePricing(model.cost),
1126
+ observedCacheReadShare: activeModel?.awareness.calibration.cache.sampleCount > 0
1127
+ ? activeModel.awareness.calibration.cache.cacheReadShare
1128
+ : undefined,
1129
+ sampleCount: activeModel?.awareness.calibration.cache.sampleCount ?? 0,
1130
+ nominalRecentTailTokens: activeModel?.awareness.limits.recentTailTokens
1131
+ ?? effectiveContextConfig.recentTailTokens,
1132
+ activeInputBudget: budget.activeInputBudget,
1133
+ });
1134
+ cacheAwareDecision = decision;
1135
+ if (decision.eligible && decision.tailExtended) {
1136
+ const extendedDedupPlan = planManagedContext({
984
1137
  messages: effectiveEvent.messages,
985
1138
  fixedTokens,
986
1139
  budget,
987
1140
  config: effectiveContextConfig,
988
1141
  pinnedMessageIndices,
989
- supplementalMessages: [
990
- ...memorySelection.pins.map((evidence) => ({
991
- id: `pin:${evidence.item.id}`,
992
- message: evidence.message,
993
- kind: "pin" as const,
994
- sourceIds: [evidence.item.id],
995
- score: 950,
996
- reason: evidence.reason,
997
- })),
998
- ...memorySelection.memories.map((evidence) => ({
999
- id: `memory:${evidence.item.id}`,
1000
- message: evidence.message,
1001
- kind: "memory" as const,
1002
- sourceIds: [evidence.item.id],
1003
- score: 90 + Math.min(0.999999, Math.max(0, evidence.score) / 1_000),
1004
- reason: evidence.reason,
1005
- })),
1006
- ],
1007
- }))
1142
+ supplementalMessages: dedupSupplementalMessages,
1143
+ cacheAwareTailTokens: decision.recentTailTokens,
1144
+ });
1145
+ const modelKey = modelProfileKey(model.provider, model.id);
1146
+ const nominalEpoch = this.cacheAwareEpochCost(nominalDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
1147
+ const extendedEpoch = this.cacheAwareEpochCost(extendedDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
1148
+ this.logger.debug("context.cache_aware_candidate", {
1149
+ eligible: decision.eligible,
1150
+ tailExtended: decision.tailExtended,
1151
+ recentTailTokens: decision.recentTailTokens,
1152
+ nominalEpoch,
1153
+ extendedEpoch,
1154
+ });
1155
+ const extendedWon = extendedEpoch !== undefined
1156
+ && nominalEpoch !== undefined
1157
+ && extendedEpoch < nominalEpoch * (1 - effectiveContextConfig.cacheAware.minimumImprovementRatio);
1158
+ // Hysteresis (deliberate epochs): once adopted, the extended tail is
1159
+ // kept while it loses at most a single epoch comparison; it is
1160
+ // dropped only after two consecutive losses, preventing
1161
+ // oscillation between plans. The streak resets on a win or model switch.
1162
+ if (extendedWon) {
1163
+ this.cacheAwareExtendedLossStreak = 0;
1164
+ this.cacheAwareStickExtended = true;
1165
+ } else if (this.cacheAwareStickExtended) {
1166
+ this.cacheAwareExtendedLossStreak += 1;
1167
+ if (this.cacheAwareExtendedLossStreak >= effectiveContextConfig.cacheAware.stickinessEpochs) {
1168
+ this.cacheAwareStickExtended = false;
1169
+ this.cacheAwareExtendedLossStreak = 0;
1170
+ }
1171
+ }
1172
+ if (this.cacheAwareStickExtended) {
1173
+ cacheAwareTailTokens = decision.recentTailTokens;
1174
+ }
1175
+ }
1176
+ }
1177
+ const retrievalActiveContextEntryIds = retrievalEnabled
1178
+ ? this.plannedContextEntryIds(ctx, cacheAwareTailTokens !== undefined
1179
+ ? planManagedContext({
1180
+ messages: effectiveEvent.messages,
1181
+ fixedTokens,
1182
+ budget,
1183
+ config: effectiveContextConfig,
1184
+ pinnedMessageIndices,
1185
+ supplementalMessages: dedupSupplementalMessages,
1186
+ cacheAwareTailTokens,
1187
+ })
1188
+ : nominalDedupPlan)
1008
1189
  : undefined;
1009
1190
  const retrieval = this.retrieveHistory(
1010
1191
  effectiveEvent,
@@ -1149,6 +1330,7 @@ export class Ds4ContextRuntime {
1149
1330
  config: effectiveContextConfig,
1150
1331
  pinnedMessageIndices,
1151
1332
  supplementalMessages: rankedSupplementalMessages,
1333
+ ...(cacheAwareTailTokens !== undefined ? { cacheAwareTailTokens } : {}),
1152
1334
  ...(ranking.diagnostics.status === "active"
1153
1335
  ? { supplementalSelectionOrder: ranking.ranked.map((candidate) => candidate.id) }
1154
1336
  : {}),
@@ -1209,6 +1391,64 @@ export class Ds4ContextRuntime {
1209
1391
  selected: plannerSelectedProject,
1210
1392
  };
1211
1393
  plan.planning.durationMs = Math.max(0, this.now() - planningStartedAt);
1394
+ /**
1395
+ * Cache-aware diagnostics are metadata-only (tokens, ratios, cost).
1396
+ * The reusable prefix is estimated against the previous managed plan;
1397
+ * if the strategy changed since the last plan, the estimate is
1398
+ * conservative (empty hashes) rather than optimistic. The block is
1399
+ * present only when the cache-aware policy is enabled (mode auto);
1400
+ * mode off leaves the manifest unchanged from 0.3.6.
1401
+ */
1402
+ if (plan.mode === "managed" && effectiveContextConfig.cacheAware?.mode === "auto") {
1403
+ const currentModelKey = model ? modelProfileKey(model.provider, model.id) : undefined;
1404
+ const previousHashes = currentModelKey !== undefined && this.lastPlanModelKey === currentModelKey
1405
+ ? this.lastPlanMessageHashes
1406
+ : [];
1407
+ const currentHashes: string[] = [];
1408
+ const currentTokens: number[] = [];
1409
+ for (const message of plan.messages) {
1410
+ currentHashes.push(fingerprint(message));
1411
+ currentTokens.push(estimateMessagesTokens([message]));
1412
+ }
1413
+ const reusablePrefixTokens = estimateReusablePrefixTokens(
1414
+ previousHashes,
1415
+ currentHashes,
1416
+ currentTokens,
1417
+ );
1418
+ const estimatedCost = model?.cost
1419
+ ? estimateRequestCost({
1420
+ totalInputTokens: plan.planning.fixedTokens
1421
+ + plan.selected.reduce((total, item) => total + item.tokens, 0),
1422
+ reusablePrefixTokens,
1423
+ }, Ds4ContextRuntime.cachePricing(model.cost) ?? {}).total
1424
+ : undefined;
1425
+ const candidate: "nominal" | "cache-aware" = cacheAwareTailTokens !== undefined
1426
+ ? "cache-aware"
1427
+ : "nominal";
1428
+ plan.planning.cacheAware = {
1429
+ eligible: cacheAwareDecision?.eligible ?? false,
1430
+ tailExtended: cacheAwareTailTokens !== undefined,
1431
+ ...(cacheAwareDecision?.recentTailTokens !== undefined
1432
+ ? { recentTailTokens: cacheAwareDecision.recentTailTokens }
1433
+ : {}),
1434
+ ...(cacheAwareDecision?.missHitRatio !== undefined
1435
+ ? { missHitRatio: cacheAwareDecision.missHitRatio }
1436
+ : {}),
1437
+ ...(cacheAwareDecision?.cacheReadShare !== undefined
1438
+ ? { cacheReadShare: cacheAwareDecision.cacheReadShare }
1439
+ : {}),
1440
+ ...(cacheAwareDecision?.sampleCount !== undefined
1441
+ ? { sampleCount: cacheAwareDecision.sampleCount }
1442
+ : {}),
1443
+ ...(reusablePrefixTokens > 0 ? { reusablePrefixTokens } : {}),
1444
+ ...(estimatedCost !== undefined ? { estimatedCost } : {}),
1445
+ candidate,
1446
+ };
1447
+ this.lastPlanMessageHashes = currentHashes;
1448
+ this.lastPlanMessageTokens = currentTokens;
1449
+ this.lastCacheAwareDecision = cacheAwareDecision;
1450
+ this.lastPlanModelKey = currentModelKey;
1451
+ }
1212
1452
  const oversizedTurnExclusions = plan.planning.oversizedTurnExclusions ?? 0;
1213
1453
  if (plan.mode === "managed" && oversizedTurnExclusions > 0) {
1214
1454
  this.logger.warn("context.excluded_oversized_turn", {
@@ -1407,6 +1647,9 @@ export class Ds4ContextRuntime {
1407
1647
  }
1408
1648
  try {
1409
1649
  const result = this.database.manifests.save(manifest);
1650
+ if (result.status === "stored") {
1651
+ this.lastPersistedInventory = result.inventory;
1652
+ }
1410
1653
  if (result.status === "skipped-oversize") {
1411
1654
  this.logger.warn("context.manifest_persistence_skipped", {
1412
1655
  category: "oversize",
@@ -3037,6 +3280,7 @@ export class Ds4ContextRuntime {
3037
3280
  ...(indexed ? { indexed } : {}),
3038
3281
  ...(this.observation ? { observation: this.observation } : {}),
3039
3282
  ...(this.lastManifest ? { lastManifest: this.lastManifest } : {}),
3283
+ ...(this.lastPersistedInventory ? { persistedInventory: this.lastPersistedInventory } : {}),
3040
3284
  retrieval: this.lastRetrieval,
3041
3285
  project: this.lastProject,
3042
3286
  memory: this.lastMemory,
@@ -87,7 +87,7 @@ function sourceKind(entry: SessionEntry, role?: string): ContextManifestItemKind
87
87
  return "history";
88
88
  }
89
89
 
90
- function fingerprint(message: unknown): string {
90
+ export function fingerprint(message: unknown): string {
91
91
  return sha256(stableStringify(message));
92
92
  }
93
93
 
@@ -35,5 +35,15 @@ export function snapshotModel(ctx: Pick<ExtensionContext, "model">): ModelDescri
35
35
  maxTokens: ctx.model.maxTokens,
36
36
  reasoning: ctx.model.reasoning,
37
37
  input: ctx.model.input,
38
+ ...(ctx.model.cost && typeof ctx.model.cost.input === "number"
39
+ ? {
40
+ cost: {
41
+ input: ctx.model.cost.input,
42
+ output: ctx.model.cost.output,
43
+ cacheRead: ctx.model.cost.cacheRead,
44
+ cacheWrite: ctx.model.cost.cacheWrite,
45
+ },
46
+ }
47
+ : {}),
38
48
  };
39
49
  }
@@ -1,4 +1,4 @@
1
- export const EXTENSION_VERSION = "0.3.6";
1
+ export const EXTENSION_VERSION = "0.3.7";
2
2
  export const SUPPORTED_PI_VERSION = "0.84.3";
3
3
  export const OBSERVER_PLANNER_VERSION = "observer-model-aware-v1";
4
4
  export const PLANNER_VERSION = "managed-learned-ranking-v1";