ds4-context-engine 0.3.6 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/ADR/062-cache-aware-context-planning.md +93 -0
- package/docs/ADR/README.md +1 -0
- package/docs/CACHE_AWARE_BENCHMARK.md +66 -0
- package/docs/CONTEXT_MANIFEST.md +3 -0
- package/docs/CONTEXT_PLANNER.md +36 -0
- package/docs/releases/0.3.7.md +83 -0
- package/package.json +2 -2
- package/src/extension/commands.ts +27 -2
- package/src/extension/runtime.ts +266 -22
- package/src/pi-adapter/context-observer.ts +1 -1
- package/src/pi-adapter/session-reader.ts +10 -0
- package/src/pi-adapter/version.ts +1 -1
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
# ADR-062 — Cache-aware context planning
|
|
2
|
+
|
|
3
|
+
Status: accepted and implemented in the coordinated 0.3.7 release (opt-in, default off; see the [release record](../releases/0.3.7.md)).
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
|
|
7
|
+
DS4 reduces provider input by selecting a bounded managed context. This is
|
|
8
|
+
economically sound when every input token costs the same or when the provider
|
|
9
|
+
offers no prompt cache. Providers with a large context window and a large
|
|
10
|
+
cache-miss/cache-hit price ratio (for example DeepSeek V4 Flash, roughly 31x
|
|
11
|
+
off-peak) invert the trade-off: a small but frequently re-planned prompt can
|
|
12
|
+
cost more than a larger stable one, because every re-plan invalidates the
|
|
13
|
+
provider prefix cache.
|
|
14
|
+
|
|
15
|
+
The 0.3.6 planner applies an automatic 64k recent-tail ceiling to all models
|
|
16
|
+
above 256k until the conversational tail exceeds the cap. When the cap is
|
|
17
|
+
exceeded, the oldest turn leaves and the shared prefix with the previous
|
|
18
|
+
request collapses: everything after the first divergence becomes a cache miss.
|
|
19
|
+
Whether this is more expensive than a wider tail depends on the workload
|
|
20
|
+
(requests per turn, turns per epoch, share of cached reads observed).
|
|
21
|
+
|
|
22
|
+
The runtime already records cache read/write shares per exact `provider/model`
|
|
23
|
+
(volatile calibration plus persisted manifests) and Pi exposes per-million
|
|
24
|
+
cost rates on the active model. The portable core never hardcodes prices.
|
|
25
|
+
|
|
26
|
+
## Decision
|
|
27
|
+
|
|
28
|
+
- Add an optional economic cost profile to the portable `ModelDescriptor`
|
|
29
|
+
(`input`, `output`, `cacheRead`, `cacheWrite` per million), populated from
|
|
30
|
+
Pi model metadata in `snapshotModel`. Absent fields degrade cache-aware
|
|
31
|
+
planning to the previous behavior.
|
|
32
|
+
- Add `context.cacheAware` (default `off`), a policy object with:
|
|
33
|
+
- `mode`: `off` preserves the previous planner exactly; `auto` may extend
|
|
34
|
+
the recent tail when pricing and observed cache shares justify it;
|
|
35
|
+
- `minimumCacheSampleCount`, `minimumCacheReadShare`,
|
|
36
|
+
`minimumMissHitRatio`: hard gates before any extension is eligible;
|
|
37
|
+
- `minimumImprovementRatio`: relative cost improvement required before an
|
|
38
|
+
alternative plan is adopted (hysteresis);
|
|
39
|
+
- `maxTailBudgetShare`: the extended tail may never exceed this fraction of
|
|
40
|
+
the active input budget;
|
|
41
|
+
- `expectedRequestsPerTurn`, `expectedTurnsPerEpoch`, `stickinessEpochs`:
|
|
42
|
+
deterministic cost model horizon plus hysteresis before dropping an
|
|
43
|
+
adopted extended tail.
|
|
44
|
+
- In `auto` the runtime computes `context.cacheAware` decision and compares
|
|
45
|
+
two plan candidates (nominal tail vs extended tail) with a deterministic
|
|
46
|
+
epoch cost model:
|
|
47
|
+
- stable plans (original conversation fits their tail cap) pay one cold
|
|
48
|
+
transition per epoch and warm requests afterwards;
|
|
49
|
+
- sliding plans (conversation exceeds their cap) pay a cold request every
|
|
50
|
+
turn of the epoch.
|
|
51
|
+
The extended plan is adopted only when it is cheaper over the whole epoch,
|
|
52
|
+
after the minimum improvement margin.
|
|
53
|
+
- The planner accepts an optional `cacheAwareTailTokens` override that bypasses
|
|
54
|
+
the automatic context-window ceiling; it remains bounded by the active/hard
|
|
55
|
+
input budgets and atomic groups, so no hard limit, privacy, pin, current
|
|
56
|
+
request or atomicity guarantee is weakened.
|
|
57
|
+
- Manifest `planning.cacheAware` (optional, numbers only, never content)
|
|
58
|
+
records the decision, tail, miss/hit ratio, observed cache share, sample
|
|
59
|
+
count, estimated reusable prefix tokens, estimated request cost and the
|
|
60
|
+
winning candidate. `/context tokens` and `/context explain` surface the same
|
|
61
|
+
metadata. `context.excluded_oversized_turn` and previous guarantees are
|
|
62
|
+
unchanged.
|
|
63
|
+
- Add a deterministic synthetic prefix-cache simulator covering the report
|
|
64
|
+
scenarios (append-only under the cap, sliding past 64k, one response per
|
|
65
|
+
prompt, five tool cycles, wide vs sliding tails, model switch, pricing
|
|
66
|
+
comparison). No provider, no credentials, no CI network access.
|
|
67
|
+
|
|
68
|
+
## Rejected alternatives
|
|
69
|
+
|
|
70
|
+
- Hardcoding provider prices: rejected, the core stays provider-independent.
|
|
71
|
+
- Raising the default tail ceiling globally: rejected, it changes behavior
|
|
72
|
+
for all models (including ones without cache discounts).
|
|
73
|
+
- Always extending the tail for eligible models: rejected, the epoch model
|
|
74
|
+
shows the nominal plan can still be cheaper on workloads where the tail
|
|
75
|
+
slides infrequently; the improvement margin keeps the switch conservative.
|
|
76
|
+
|
|
77
|
+
## Compatibility and validation
|
|
78
|
+
|
|
79
|
+
Configuration is additive; `context.cacheAware.mode` defaults to `off`, so
|
|
80
|
+
`0.3.6` behavior (and manifest absence of the cacheAware block) is preserved.
|
|
81
|
+
Automatic behavior is opt-in and requires observed samples plus positive
|
|
82
|
+
pricing. The epoch cost is an estimate used only for the comparison; the
|
|
83
|
+
actual billing remains whatever the provider reports. The workaround override
|
|
84
|
+
(`modelAwareness.overrides` with a large `recentTailTokens`, zeroed retrieval
|
|
85
|
+
and disabled compaction) remains available and is unaffected; it is more
|
|
86
|
+
aggressive than `mode: "auto"` because it also removes retrieval/project
|
|
87
|
+
supplements and compaction, which `auto` deliberately preserves for quality.
|
|
88
|
+
|
|
89
|
+
Coverage: cache-policy unit tests (prefix, costs, decision gates), config
|
|
90
|
+
validation, planner override tests, the synthetic prefix-cache simulator, and
|
|
91
|
+
runtime integration tests for off/auto/no-discount/under-budget paths. A
|
|
92
|
+
real-provider DeepSeek A/B benchmark remains out of scope and voluntary
|
|
93
|
+
(protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
|
package/docs/ADR/README.md
CHANGED
|
@@ -65,5 +65,6 @@ The initial decisions from the development plan are accepted:
|
|
|
65
65
|
| [059](059-optional-anchored-editing.md) | Opt in to anchored edit expansion inside Pi's native mutation queue | Accepted |
|
|
66
66
|
| [060](060-optional-portable-agent-tools.md) | Opt in to edit reports, adaptive reads/results and session-owned local jobs | Accepted |
|
|
67
67
|
| [061](061-compaction-latency.md) | Bound compaction update calls, input budgets, concurrent segments and phase timings | Accepted |
|
|
68
|
+
| [062](062-cache-aware-context-planning.md) | Opt-in cache-aware tail planning using model pricing and observed cache shares | Accepted |
|
|
68
69
|
|
|
69
70
|
Each decision will receive a dedicated record when implementation pressure introduces alternatives or consequences not already covered by the development plan.
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Benchmark A/B DeepSeek — cache-aware planning
|
|
2
|
+
|
|
3
|
+
Protocollo volontario, fuori CI, nessuna credenziale nel repo. Serve a decidere se
|
|
4
|
+
`context.cacheAware.mode` può passare da `auto` (opt-in) a **default** in una
|
|
5
|
+
release futura. Gate dichiarato in [ADR-062](ADR/062-cache-aware-context-planning.md).
|
|
6
|
+
|
|
7
|
+
## Obiettivo
|
|
8
|
+
|
|
9
|
+
Misurare, sulla **stessa** conversazione reale, il costo provider per turno con:
|
|
10
|
+
|
|
11
|
+
- **Sessione A (baseline)**: comportamento 0.3.6 — `context.cacheAware` assente
|
|
12
|
+
(default `off`), tail automatica 64k, retrieval 16k, project 20k.
|
|
13
|
+
- **Sessione B (auto)**: `context.cacheAware.mode: "auto"` con i default di
|
|
14
|
+
policy (gates 0.5 cache-share, ratio 20, margine 10%).
|
|
15
|
+
- **Sessione C (workaround estremo, opzionale)**: `modelAwareness.overrides` con
|
|
16
|
+
`recentTailTokens: 500000`, `maxRetrievedHistoryTokens: 0`,
|
|
17
|
+
`maxProjectTokens: 0`, `compaction.enabled: false` — cache-first puro.
|
|
18
|
+
|
|
19
|
+
Le tre sessioni devono avere la **stessa sequenza di turni** (stesso carico di
|
|
20
|
+
lavoro: tool call, retrieval, project touches) e lo **stesso modello/provided
|
|
21
|
+
ID DeepSeek**.
|
|
22
|
+
|
|
23
|
+
## Cosa registrare per ogni turno
|
|
24
|
+
|
|
25
|
+
Da `/context tokens` e `/context explain`:
|
|
26
|
+
|
|
27
|
+
- `recentTailTokens` scelti (nominal vs extended) e `cacheAware` decision/reasons;
|
|
28
|
+
- manifest `planning.cacheAware` (missHitRatio, cacheReadShare, sampleCount,
|
|
29
|
+
reusablePrefixTokens, estimatedCost, candidate);
|
|
30
|
+
- `ProviderCacheMetrics` reali dopo ogni richiesta: `inputTokens`,
|
|
31
|
+
`cacheReadTokens`, `cacheWriteTokens`, share read/write per provider/model.
|
|
32
|
+
|
|
33
|
+
Dal provider (costo effettivo):
|
|
34
|
+
|
|
35
|
+
- input non-cached, cache-read (hit), cache-write, output per richiesta;
|
|
36
|
+
- costo totale della sessione (per-turno e totale).
|
|
37
|
+
|
|
38
|
+
## Metriche
|
|
39
|
+
|
|
40
|
+
1. **Costo per turno** (media e totale sessione): A vs B vs C.
|
|
41
|
+
2. **Cache-read share osservata** per config: conferma che la tail estesa
|
|
42
|
+
mantiene il prefisso stabile (share alta) vs sliding (share ~0 dopo 64k).
|
|
43
|
+
3. **Qualità** (non solo $$$): recall retrieval, `Planner exclusions`,
|
|
44
|
+
current request/atomicità intatte, completamenti corretti nelle fasi tool.
|
|
45
|
+
Con retrieval azzerato (C) documentare i regressi di qualità.
|
|
46
|
+
4. **Eventi di transizione**: quante volte il prefisso si è invalidato (turni
|
|
47
|
+
con cacheRead ≈ 0), confronto sliding vs stable.
|
|
48
|
+
|
|
49
|
+
## Criterio di promozione a default
|
|
50
|
+
|
|
51
|
+
Promuovere `mode: "auto"` a default solo se, su ≥ 3 sessioni reali:
|
|
52
|
+
|
|
53
|
+
- costo medio per turno di B < A (margine ≥ il `minimumImprovementRatio`
|
|
54
|
+
configurato, di default 10%);
|
|
55
|
+
- nessun regresso di qualità misurabile (recall retrieval uguale o migliore;
|
|
56
|
+
current request/atomicità preservate);
|
|
57
|
+
- nessuna oscillazione plan (decisione che alterna nominal/esteso senza
|
|
58
|
+
motivo: verificare `stickinessEpochs` e diagnostica).
|
|
59
|
+
|
|
60
|
+
## Esecuzione sicura
|
|
61
|
+
|
|
62
|
+
- Nessuna credenziale nel repo; il benchmark usa la sessione Pi normale con il
|
|
63
|
+
provider DeepSeek già autenticato.
|
|
64
|
+
- Non modificare la config attiva in modo permanente: copie di configurazione
|
|
65
|
+
per sessione, o valori temporanei poi ripristinati.
|
|
66
|
+
- Fuori CI: niente rete/credenziali nel test suite.
|
package/docs/CONTEXT_MANIFEST.md
CHANGED
|
@@ -20,6 +20,7 @@ A Context Manifest explains the context visible at DS4's Pi `context` hook witho
|
|
|
20
20
|
- artifact IDs, SHA-256, bytes, MIME, classification, exact source entry/tool IDs, error state, and before/after token estimates;
|
|
21
21
|
- provider destination and allow-set names, selected classification counts, blocked/excluded/redacted counts, final provider-check count, and enforcement stage;
|
|
22
22
|
- planner mode/version, original and selected counts, group counts, internal budgets, duration, and fallback reason;
|
|
23
|
+
- optional cache-aware planning decision: eligibility, tail extension, requested tail tokens, miss/hit price ratio, observed cache-read share, sample count, estimated reusable prefix tokens, estimated request cost in dollars, and winning candidate (`nominal`/`cache-aware`); numbers only, never content;
|
|
23
24
|
- learned-ranking mode/status, feature/model versions, candidate count, aggregate disagreement/rank shift, duration, and generic static-fallback reason;
|
|
24
25
|
- planner and policy versions;
|
|
25
26
|
- deterministic SHA-256 over system prompt, active tools, and messages;
|
|
@@ -54,6 +55,8 @@ A manifest at or below 256 KiB is stored unchanged. Above that preferred bound,
|
|
|
54
55
|
|
|
55
56
|
Each manifest transaction prunes at most 32 excess rows and 8 MiB of serialized payload; one individually oversized oldest row may exceed the byte limit to guarantee progress. Calibration pruning is independently limited to 32 rows per related profile write. This incrementally repairs an existing oversized database without adding a long startup write or extending SQLite lock duration with an unbounded purge. Deleted pages become reusable by SQLite; the database file may remain at its previous high-water size until explicit [offline maintenance](STORAGE_MAINTENANCE.md). No retention action edits canonical Pi JSONL or project files.
|
|
56
57
|
|
|
58
|
+
The `save()` result carries the derived `inventory` of the persisted projection, and the runtime exposes it through `RuntimeDiagnostics.persistedInventory`; `/context manifest` and `/context excluded` therefore report the truthful persisted completeness (`complete` or `excluded-rollup` with retained/total counts) instead of defaulting to `complete` when the in-memory manifest has no inventory attached.
|
|
59
|
+
|
|
57
60
|
## Reproducibility
|
|
58
61
|
|
|
59
62
|
Object keys are normalized before hashing, so equivalent tool schemas with different key insertion order produce the same prompt hash. The estimator version is stored explicitly as `chars-v1`; planner/policy versions describe selection behavior. Golden tests protect manifest shape, model-profile resolution, token accounting, and hash stability.
|
package/docs/CONTEXT_PLANNER.md
CHANGED
|
@@ -86,6 +86,42 @@ M14 can queue the finalized manifest after planning when `quality.enabled` is tr
|
|
|
86
86
|
|
|
87
87
|
M18 can evaluate bounded metadata-only features after privacy exclusion and before supplemental candidates enter category fitting. `shadow` keeps every static score/order authoritative and records aggregate disagreement only. `active` is accepted only for a compatible checksummed model carrying an eligible held-out promotion report. Privacy exclusions, mandatory pins/current turns, atomic groups and hard budgets cannot be overridden. See [`LEARNED_RANKING.md`](LEARNED_RANKING.md).
|
|
88
88
|
|
|
89
|
+
## Cache-aware tail planning (0.3.7, opt-in)
|
|
90
|
+
|
|
91
|
+
With `context.cacheAware.mode = "auto"` the runtime may extend the recent tail
|
|
92
|
+
beyond the automatic context-window ceiling when the model pricing and the
|
|
93
|
+
observed cache shares justify it economically. The policy never hardcodes
|
|
94
|
+
prices: it reads the per-million rates exposed on the active Pi model and the
|
|
95
|
+
cache read/write shares already recorded for the exact `provider/model`.
|
|
96
|
+
|
|
97
|
+
Eligibility gates (defaults shown):
|
|
98
|
+
|
|
99
|
+
- `minimumCacheSampleCount: 3` observed calibration samples;
|
|
100
|
+
- `minimumCacheReadShare: 0.5` observed share;
|
|
101
|
+
- `minimumMissHitRatio: 20` cache-miss / cache-hit price ratio.
|
|
102
|
+
|
|
103
|
+
When eligible, the runtime compares two plan candidates with a deterministic
|
|
104
|
+
epoch cost model:
|
|
105
|
+
|
|
106
|
+
- the nominal plan (existing tail, sliding below 64k) pays a cold request on
|
|
107
|
+
every turn of the epoch because its prefix is invalidated by the slide;
|
|
108
|
+
- the extended plan (up to `maxTailBudgetShare` of the active input budget) is
|
|
109
|
+
stable when the conversation fits, so it pays one cold transition per epoch
|
|
110
|
+
and warm requests afterwards.
|
|
111
|
+
|
|
112
|
+
The extended plan is adopted only when it wins over the whole epoch
|
|
113
|
+
(`expectedRequestsPerTurn`, `expectedTurnsPerEpoch`) after the
|
|
114
|
+
`minimumImprovementRatio` margin. Once adopted it is kept (deliberate epoch)
|
|
115
|
+
until it loses `stickinessEpochs` consecutive comparisons, preventing
|
|
116
|
+
oscillation. The override remains bounded by the active and hard input
|
|
117
|
+
budgets and by atomic groups, so current request, pins, privacy and atomicity
|
|
118
|
+
guarantees are unchanged.
|
|
119
|
+
|
|
120
|
+
`context.cacheAware.mode` defaults to `off`, preserving the 0.3.6 behavior
|
|
121
|
+
exactly. Without prices, samples or a cache discount, the decision degrades to
|
|
122
|
+
the nominal plan automatically. See [ADR-062](ADR/062-cache-aware-context-planning.md)
|
|
123
|
+
for the model and rejected alternatives.
|
|
124
|
+
|
|
89
125
|
## Current limits
|
|
90
126
|
|
|
91
127
|
The planner does not call a model inside the `context` hook. Model calibration uses only finalized provider usage and deterministic local statistics. Historical/project retrieval can opt into derived semantic candidates; learned supplemental reranking remains off by default and active mode is promotion-gated. Project symbol extraction is heuristic, artifact search is literal, and memory/pin creation is manual-first. Automatic memory extraction remains disabled; M10 supplies policy enforcement but not an automatic classifier or confirmation workflow. Provider-payload coverage targets Pi 0.84.3's supported serializers, and DS4 must load after any extension allowed to replace payloads when strict final ordering is required.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Release 0.3.7 — Cache-aware context planning (opt-in)
|
|
2
|
+
|
|
3
|
+
**Version analyzed:** DS4 Context Engine `0.3.7`
|
|
4
|
+
**Commit:** `a03009a4b14d6ab1842fa800076c4b300bd70d8f`
|
|
5
|
+
**Coordinated packages:** `ds4-context-core` 0.3.7, `ds4-context-reference-adapter` 0.3.7, `ds4-context-engine` 0.3.7
|
|
6
|
+
|
|
7
|
+
## Summary
|
|
8
|
+
|
|
9
|
+
Adds an opt-in cache-aware planning policy driven by model pricing and
|
|
10
|
+
observed cache shares, plus a deterministic synthetic prefix-cache simulator.
|
|
11
|
+
The default behavior is unchanged: `context.cacheAware.mode` defaults to `off`,
|
|
12
|
+
the manifest does not include a `cacheAware` block, and the planner uses the
|
|
13
|
+
same tail caps as 0.3.6.
|
|
14
|
+
|
|
15
|
+
## New configuration
|
|
16
|
+
|
|
17
|
+
`context.cacheAware` (object, default off):
|
|
18
|
+
|
|
19
|
+
| Field | Default | Meaning |
|
|
20
|
+
|---|---|---|
|
|
21
|
+
| `mode` | `off` | `off` preserves 0.3.6 behavior; `auto` may extend the recent tail. |
|
|
22
|
+
| `minimumCacheSampleCount` | `3` | Observed samples required before acting. |
|
|
23
|
+
| `minimumCacheReadShare` | `0.5` | Observed cache-read share required before acting. |
|
|
24
|
+
| `minimumMissHitRatio` | `20` | Minimum cache-miss / cache-hit price ratio. |
|
|
25
|
+
| `minimumImprovementRatio` | `0.1` | Relative improvement required to switch plan. |
|
|
26
|
+
| `maxTailBudgetShare` | `0.5` | Max fraction of the active input budget for the extended tail. |
|
|
27
|
+
| `expectedRequestsPerTurn` | `4` | Provider requests per user turn in the cost model. |
|
|
28
|
+
| `expectedTurnsPerEpoch` | `4` | User turns per planning epoch in the cost model. |
|
|
29
|
+
| `stickinessEpochs` | `2` | Consecutive epoch losses before dropping an adopted extended tail. |
|
|
30
|
+
|
|
31
|
+
## Changes
|
|
32
|
+
|
|
33
|
+
- `ModelDescriptor` gains an optional `cost` profile (`input`, `output`,
|
|
34
|
+
`cacheRead`, `cacheWrite` per million), populated from Pi model metadata in
|
|
35
|
+
`snapshotModel`. No prices are hardcoded in the core.
|
|
36
|
+
- `planManagedContext` accepts an optional `cacheAwareTailTokens` override that
|
|
37
|
+
bypasses the automatic context-window ceiling while remaining bounded by the
|
|
38
|
+
active/hard input budgets and atomic groups.
|
|
39
|
+
- Runtime: when `mode: "auto"` and the eligibility gates pass, the runtime
|
|
40
|
+
compares the nominal plan with an extended-tail plan using a deterministic
|
|
41
|
+
epoch cost model (stable plans pay one cold transition per epoch; sliding
|
|
42
|
+
plans pay a cold request per turn) and adopts the extended plan only when it
|
|
43
|
+
wins after the minimum improvement margin.
|
|
44
|
+
- Manifest: optional `planning.cacheAware` (numbers only, never content) with
|
|
45
|
+
eligibility, tail tokens, miss/hit ratio, observed share, sample count,
|
|
46
|
+
estimated reusable prefix tokens, estimated request cost and winning
|
|
47
|
+
candidate; surfaced in `/context tokens` and `/context explain`.
|
|
48
|
+
- Core: `packages/core/src/planner/cache-policy.ts` with pure, deterministic
|
|
49
|
+
functions for common-prefix estimation, request cost and the tail decision.
|
|
50
|
+
- Tests: `cache-policy` unit tests, config validation, planner override tests,
|
|
51
|
+
the synthetic prefix-cache simulator (`cache-prefix-simulator`) and runtime
|
|
52
|
+
integration tests for off/auto/no-discount/under-budget paths.
|
|
53
|
+
|
|
54
|
+
## Compatibility
|
|
55
|
+
|
|
56
|
+
- Additive configuration; absent fields use the documented defaults.
|
|
57
|
+
- `context.cacheAware.mode: "off"` reproduces the 0.3.6 behavior exactly.
|
|
58
|
+
- No new required database schema; manifests persist the optional block as-is.
|
|
59
|
+
- The existing guarantees (current request, atomicity, privacy, pins, hard
|
|
60
|
+
limits, fail-open, naive compaction path) are unchanged.
|
|
61
|
+
- Model metadata without cache pricing or observed samples degrades the
|
|
62
|
+
decision to the nominal plan.
|
|
63
|
+
|
|
64
|
+
## Validation
|
|
65
|
+
|
|
66
|
+
- `npm run check` (excluding the known machine-load-dependent
|
|
67
|
+
`long-session` timeout flake): 83 files, 550 tests passed.
|
|
68
|
+
- Planner unit tests: 25 passed.
|
|
69
|
+
- Cache-policy unit tests: 16 passed.
|
|
70
|
+
- Prefix-cache simulator: 7 passed.
|
|
71
|
+
- Cache-aware runtime integration: 4 passed.
|
|
72
|
+
- Config catalog/loader validation: passed.
|
|
73
|
+
|
|
74
|
+
## Known limits
|
|
75
|
+
|
|
76
|
+
- The epoch cost model is an estimate for plan comparison; actual billing is
|
|
77
|
+
whatever the provider reports.
|
|
78
|
+
- A real-provider DeepSeek A/B benchmark remains voluntary and out of CI
|
|
79
|
+
(protocol: [CACHE_AWARE_BENCHMARK.md](../CACHE_AWARE_BENCHMARK.md)).
|
|
80
|
+
- The workaround override (large `recentTailTokens`, zeroed retrieval and
|
|
81
|
+
disabled compaction) remains available and is unaffected; it is more aggressive
|
|
82
|
+
than `mode: "auto"`, which deliberately preserves retrieval/project and
|
|
83
|
+
compaction for quality.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ds4-context-engine",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.7",
|
|
4
4
|
"description": "Non-destructive, provider-independent context management for Pi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -62,7 +62,7 @@
|
|
|
62
62
|
]
|
|
63
63
|
},
|
|
64
64
|
"dependencies": {
|
|
65
|
-
"ds4-context-core": "0.3.
|
|
65
|
+
"ds4-context-core": "0.3.7"
|
|
66
66
|
},
|
|
67
67
|
"peerDependencies": {
|
|
68
68
|
"@earendil-works/pi-ai": "0.84.3",
|
|
@@ -296,6 +296,14 @@ function formatTokens(diagnostics: RuntimeDiagnostics): string {
|
|
|
296
296
|
`Actual provider input: ${count(manifest?.actualInputTokens)}`,
|
|
297
297
|
` Uncached input: ${count(manifest?.providerUsage?.inputTokens)}`,
|
|
298
298
|
` Cache read / write: ${count(manifest?.providerUsage?.cacheReadTokens)} / ${count(manifest?.providerUsage?.cacheWriteTokens)}`,
|
|
299
|
+
...(manifest?.planning?.cacheAware
|
|
300
|
+
? [
|
|
301
|
+
`Cache-aware plan: ${manifest.planning.cacheAware.candidate ?? "n/a"}${manifest.planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
|
|
302
|
+
` Reusable prefix: ${count(manifest.planning.cacheAware.reusablePrefixTokens)} estimated`,
|
|
303
|
+
` Est. request cost: ${manifest.planning.cacheAware.estimatedCost === undefined ? "n/a" : `$${manifest.planning.cacheAware.estimatedCost.toFixed(6)}`}`,
|
|
304
|
+
` Miss/hit price ratio: ${manifest.planning.cacheAware.missHitRatio === undefined ? "n/a" : manifest.planning.cacheAware.missHitRatio.toFixed(2)}`,
|
|
305
|
+
]
|
|
306
|
+
: []),
|
|
299
307
|
`Pi reported context: ${count(manifest?.piReportedContextTokens ?? observation?.reportedTokens)}`,
|
|
300
308
|
`Model context window: ${count(budget?.contextWindow ?? manifest?.contextWindow)}`,
|
|
301
309
|
`Output reserve: ${count(budget?.outputReserve ?? manifest?.outputReserve)}`,
|
|
@@ -316,7 +324,7 @@ function formatManifest(diagnostics: RuntimeDiagnostics): string {
|
|
|
316
324
|
const manifest = diagnostics.lastManifest;
|
|
317
325
|
if (!manifest) return "No Context Manifest has been built for this session yet.";
|
|
318
326
|
|
|
319
|
-
const inventory = manifest.persistedInventory;
|
|
327
|
+
const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
|
|
320
328
|
const kinds = new Map<string, { items: number; tokens: number }>();
|
|
321
329
|
for (const item of manifest.included) {
|
|
322
330
|
const aggregate = kinds.get(item.kind) ?? { items: 0, tokens: 0 };
|
|
@@ -369,7 +377,7 @@ function formatManifestItems(diagnostics: RuntimeDiagnostics, type: "included" |
|
|
|
369
377
|
const manifest = diagnostics.lastManifest;
|
|
370
378
|
if (!manifest) return "No Context Manifest has been built for this session yet.";
|
|
371
379
|
const items = manifest[type];
|
|
372
|
-
const inventory = manifest.persistedInventory;
|
|
380
|
+
const inventory = manifest.persistedInventory ?? diagnostics.persistedInventory;
|
|
373
381
|
return [
|
|
374
382
|
`DS4 Context ${type === "included" ? "Included" : "Excluded"} Items`,
|
|
375
383
|
"",
|
|
@@ -409,6 +417,23 @@ function formatExplain(diagnostics: RuntimeDiagnostics): string {
|
|
|
409
417
|
...(planning.oversizedTurnExclusions
|
|
410
418
|
? [`Oversized turn excl: ${count(planning.oversizedTurnExclusions)} (turn group(s) at/above the recent-tail cap; recovered by retrieval only if it fits)`]
|
|
411
419
|
: []),
|
|
420
|
+
...(planning.cacheAware
|
|
421
|
+
? [
|
|
422
|
+
`Cache-aware plan: ${planning.cacheAware.candidate ?? "n/a"}${planning.cacheAware.tailExtended ? " (tail extended)" : ""}`,
|
|
423
|
+
...(planning.cacheAware.missHitRatio !== undefined
|
|
424
|
+
? [` Miss/hit ratio: ${planning.cacheAware.missHitRatio.toFixed(2)}`]
|
|
425
|
+
: []),
|
|
426
|
+
...(planning.cacheAware.cacheReadShare !== undefined
|
|
427
|
+
? [` Cache-read share: ${(planning.cacheAware.cacheReadShare * 100).toFixed(1)}% (${count(planning.cacheAware.sampleCount)} samples)`]
|
|
428
|
+
: []),
|
|
429
|
+
...(planning.cacheAware.reusablePrefixTokens !== undefined
|
|
430
|
+
? [` Reusable prefix: ${count(planning.cacheAware.reusablePrefixTokens)} tokens`]
|
|
431
|
+
: []),
|
|
432
|
+
...(planning.cacheAware.estimatedCost !== undefined
|
|
433
|
+
? [` Est. request cost: $${planning.cacheAware.estimatedCost.toFixed(6)}`]
|
|
434
|
+
: []),
|
|
435
|
+
]
|
|
436
|
+
: []),
|
|
412
437
|
`Selected groups: ${count(planning.selectedGroupCount)}`,
|
|
413
438
|
`Excluded groups: ${count(planning.excludedGroupCount)}`,
|
|
414
439
|
`Duration: ${planning.durationMs === undefined ? "n/a" : `${planning.durationMs.toFixed(1)} ms`}`,
|
package/src/extension/runtime.ts
CHANGED
|
@@ -72,7 +72,10 @@ import type {
|
|
|
72
72
|
PrivacyManifest,
|
|
73
73
|
ProviderUsageManifest,
|
|
74
74
|
} from "ds4-context-core/manifest/context-manifest";
|
|
75
|
-
import {
|
|
75
|
+
import {
|
|
76
|
+
HARD_PERSISTED_MANIFEST_BYTES,
|
|
77
|
+
type PersistedManifestInventory,
|
|
78
|
+
} from "ds4-context-core/manifest/context-manifest-storage";
|
|
76
79
|
import {
|
|
77
80
|
unavailableStorageDiagnostics,
|
|
78
81
|
type StorageDiagnostics,
|
|
@@ -98,6 +101,13 @@ import {
|
|
|
98
101
|
type ProjectMemorySource,
|
|
99
102
|
} from "ds4-context-core/memory/memory-types";
|
|
100
103
|
import { planManagedContext, type ManagedContextPlan, type SupplementalContextMessage } from "ds4-context-core/planner/context-planner";
|
|
104
|
+
import {
|
|
105
|
+
decideCacheAwareTail,
|
|
106
|
+
estimateReusablePrefixTokens,
|
|
107
|
+
estimateRequestCost,
|
|
108
|
+
type CacheAwarePlanDecision,
|
|
109
|
+
type CachePricing,
|
|
110
|
+
} from "ds4-context-core/planner/cache-policy";
|
|
101
111
|
import {
|
|
102
112
|
disabledContextQualityDiagnostics,
|
|
103
113
|
evaluateManifestQuality,
|
|
@@ -152,6 +162,7 @@ import {
|
|
|
152
162
|
findExactPiMessageSourceIds,
|
|
153
163
|
findPiPinnedMessageIndices,
|
|
154
164
|
findPiSourceEntryIds,
|
|
165
|
+
fingerprint,
|
|
155
166
|
} from "../pi-adapter/context-observer.ts";
|
|
156
167
|
import { projectSessionFileMutations } from "../pi-adapter/memory-adapter.ts";
|
|
157
168
|
import { ProjectMemorySynchronizer } from "../pi-adapter/project-memory-sync.ts";
|
|
@@ -284,6 +295,10 @@ function numericUsage(value: unknown): number {
|
|
|
284
295
|
: 0;
|
|
285
296
|
}
|
|
286
297
|
|
|
298
|
+
function roundedCost(value: number): number {
|
|
299
|
+
return Math.round(value * 1_000_000) / 1_000_000;
|
|
300
|
+
}
|
|
301
|
+
|
|
287
302
|
function providerUsageManifest(input: {
|
|
288
303
|
inputTokens: number;
|
|
289
304
|
cacheReadTokens: number;
|
|
@@ -374,6 +389,8 @@ export interface RuntimeDiagnostics {
|
|
|
374
389
|
indexed?: SessionIndexStats;
|
|
375
390
|
observation?: ContextObservation;
|
|
376
391
|
lastManifest?: ContextManifest;
|
|
392
|
+
/** Inventory of the persisted projection of the last successfully stored manifest. */
|
|
393
|
+
persistedInventory?: PersistedManifestInventory;
|
|
377
394
|
retrieval: RetrievalDiagnostics;
|
|
378
395
|
project: ProjectKnowledgeDiagnostics;
|
|
379
396
|
memory: MemoryDiagnostics;
|
|
@@ -411,6 +428,7 @@ export class Ds4ContextRuntime {
|
|
|
411
428
|
private databasePath?: string;
|
|
412
429
|
private observation?: ContextObservation;
|
|
413
430
|
private lastManifest?: ContextManifest;
|
|
431
|
+
private lastPersistedInventory?: PersistedManifestInventory;
|
|
414
432
|
private pendingManifestId?: string;
|
|
415
433
|
private pendingManifestPersisted = false;
|
|
416
434
|
private retrievalEngine?: HistoricalRetrievalEngine;
|
|
@@ -433,6 +451,18 @@ export class Ds4ContextRuntime {
|
|
|
433
451
|
private lastContextProfileKey?: string;
|
|
434
452
|
private readonly knownModelProfiles = new Set<string>();
|
|
435
453
|
private readonly volatileCalibration = new Map<string, TokenCalibrationSample[]>();
|
|
454
|
+
/** Fingerprints of the messages sent in the last managed plan (empty when unknown). */
|
|
455
|
+
private lastPlanMessageHashes: string[] = [];
|
|
456
|
+
/** Per-message token estimates of the last managed plan, aligned with hashes. */
|
|
457
|
+
private lastPlanMessageTokens: number[] = [];
|
|
458
|
+
/** Cache-aware decision of the last plan; used for /context diagnostics. */
|
|
459
|
+
private lastCacheAwareDecision?: CacheAwarePlanDecision;
|
|
460
|
+
/** Provider/model key of the last plan; a switch invalidates the cached prefix. */
|
|
461
|
+
private lastPlanModelKey?: string;
|
|
462
|
+
/** Consecutive epochs in which the extended tail lost the comparison (hysteresis). */
|
|
463
|
+
private cacheAwareExtendedLossStreak = 0;
|
|
464
|
+
/** Extended-tail adoption state: while set, the extended tail is kept even if a single epoch comparison loses. */
|
|
465
|
+
private cacheAwareStickExtended = false;
|
|
436
466
|
private lastMemoryMutationSignature?: string;
|
|
437
467
|
private artifactManager?: ArtifactManager;
|
|
438
468
|
private lastArtifacts: ArtifactDiagnostics = disabledArtifactDiagnostics();
|
|
@@ -481,6 +511,7 @@ export class Ds4ContextRuntime {
|
|
|
481
511
|
this.pendingQuality.length = 0;
|
|
482
512
|
this.lastIndexResult = undefined;
|
|
483
513
|
this.lastManifest = undefined;
|
|
514
|
+
this.lastPersistedInventory = undefined;
|
|
484
515
|
this.pendingManifestId = undefined;
|
|
485
516
|
this.pendingManifestPersisted = false;
|
|
486
517
|
this.retrievalEngine = undefined;
|
|
@@ -585,6 +616,7 @@ export class Ds4ContextRuntime {
|
|
|
585
616
|
this.syncSessionIndex(ctx);
|
|
586
617
|
this.lastManifest = this.database.manifests.getLatest(this.session.sessionId);
|
|
587
618
|
if (this.lastManifest) {
|
|
619
|
+
this.lastPersistedInventory = this.lastManifest.persistedInventory;
|
|
588
620
|
this.lastContextProfileKey = modelProfileKey(
|
|
589
621
|
this.lastManifest.provider,
|
|
590
622
|
this.lastManifest.model,
|
|
@@ -763,6 +795,79 @@ export class Ds4ContextRuntime {
|
|
|
763
795
|
};
|
|
764
796
|
}
|
|
765
797
|
|
|
798
|
+
/**
|
|
799
|
+
* Optional per-million pricing proxied from Pi model metadata; undefined
|
|
800
|
+
* values degrade cache-aware planning to the previous behavior.
|
|
801
|
+
*/
|
|
802
|
+
private static cachePricing(cost: ModelDescriptor["cost"]): CachePricing | undefined {
|
|
803
|
+
if (cost === undefined) return undefined;
|
|
804
|
+
return {
|
|
805
|
+
inputPerMillion: cost.input,
|
|
806
|
+
cacheReadPerMillion: cost.cacheRead,
|
|
807
|
+
cacheWritePerMillion: cost.cacheWrite,
|
|
808
|
+
outputPerMillion: cost.output,
|
|
809
|
+
};
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
/**
|
|
813
|
+
* Estimated cost of keeping a plan for an epoch of `turnsPerEpoch` user
|
|
814
|
+
* turns with `requestsPerTurn` provider requests each, in dollars.
|
|
815
|
+
*
|
|
816
|
+
* Model (documented, deterministic):
|
|
817
|
+
* - STABLE plan (original conversation fits the tail cap): the first
|
|
818
|
+
* request of the epoch is cold (it reuses only the prefix shared with the
|
|
819
|
+
* previously sent plan); every following request is warm.
|
|
820
|
+
* - SLIDING plan (the conversation exceeds the tail cap): the prefix is
|
|
821
|
+
* invalidated by every new turn, so each turn of the epoch pays one cold
|
|
822
|
+
* request plus the remaining warm requests.
|
|
823
|
+
*
|
|
824
|
+
* Returns undefined when pricing is incomplete or the plan is not managed.
|
|
825
|
+
* Costs are metadata estimates only; no provider content is exposed.
|
|
826
|
+
*/
|
|
827
|
+
private cacheAwareEpochCost(
|
|
828
|
+
plan: ManagedContextPlan<ContextEvent["messages"][number]>,
|
|
829
|
+
fixedTokens: number,
|
|
830
|
+
cost: NonNullable<ModelDescriptor["cost"]>,
|
|
831
|
+
requestsPerTurn: number,
|
|
832
|
+
turnsPerEpoch: number,
|
|
833
|
+
modelKey: string,
|
|
834
|
+
): number | undefined {
|
|
835
|
+
const pricing = Ds4ContextRuntime.cachePricing(cost);
|
|
836
|
+
if (!pricing || plan.mode !== "managed") return undefined;
|
|
837
|
+
const hashes: string[] = [];
|
|
838
|
+
const tokens: number[] = [];
|
|
839
|
+
let total = fixedTokens;
|
|
840
|
+
for (const message of plan.messages) {
|
|
841
|
+
hashes.push(fingerprint(message));
|
|
842
|
+
const estimate = estimateMessagesTokens([message]);
|
|
843
|
+
tokens.push(estimate);
|
|
844
|
+
total += estimate;
|
|
845
|
+
}
|
|
846
|
+
const previousHashes = modelKey !== this.lastPlanModelKey ? [] : this.lastPlanMessageHashes;
|
|
847
|
+
const reusable = previousHashes.length > 0
|
|
848
|
+
? estimateReusablePrefixTokens(previousHashes, hashes, tokens)
|
|
849
|
+
: 0;
|
|
850
|
+
const coldCost = estimateRequestCost({
|
|
851
|
+
totalInputTokens: total,
|
|
852
|
+
reusablePrefixTokens: reusable,
|
|
853
|
+
}, pricing);
|
|
854
|
+
const warmCost = estimateRequestCost({
|
|
855
|
+
totalInputTokens: total,
|
|
856
|
+
reusablePrefixTokens: total,
|
|
857
|
+
}, pricing);
|
|
858
|
+
if (coldCost.total === undefined || warmCost.total === undefined) return undefined;
|
|
859
|
+
const sliding = plan.planning.originalMessageTokens > plan.planning.recentTailTokenLimit;
|
|
860
|
+
if (sliding) {
|
|
861
|
+
// Each turn of the epoch invalidates the prefix: one cold request plus
|
|
862
|
+
// the remaining warm requests per turn.
|
|
863
|
+
const perTurn = coldCost.total + Math.max(0, requestsPerTurn - 1) * warmCost.total;
|
|
864
|
+
return roundedCost(perTurn * turnsPerEpoch);
|
|
865
|
+
}
|
|
866
|
+
// Stable plan: one cold transition for the epoch, then all warm.
|
|
867
|
+
const epochCost = coldCost.total + Math.max(0, requestsPerTurn * turnsPerEpoch - 1) * warmCost.total;
|
|
868
|
+
return roundedCost(epochCost);
|
|
869
|
+
}
|
|
870
|
+
|
|
766
871
|
private resolveModelPolicy(model: ModelDescriptor): {
|
|
767
872
|
awareness: ResolvedModelAwareness;
|
|
768
873
|
budget: ContextBudget;
|
|
@@ -979,32 +1084,108 @@ export class Ds4ContextRuntime {
|
|
|
979
1084
|
const retrievalEnabled = this.config.retrieval.exact
|
|
980
1085
|
|| this.config.retrieval.fts
|
|
981
1086
|
|| this.config.retrieval.semantic;
|
|
982
|
-
const
|
|
983
|
-
|
|
1087
|
+
const dedupSupplementalMessages = [
|
|
1088
|
+
...memorySelection.pins.map((evidence) => ({
|
|
1089
|
+
id: `pin:${evidence.item.id}`,
|
|
1090
|
+
message: evidence.message,
|
|
1091
|
+
kind: "pin" as const,
|
|
1092
|
+
sourceIds: [evidence.item.id],
|
|
1093
|
+
score: 950,
|
|
1094
|
+
reason: evidence.reason,
|
|
1095
|
+
})),
|
|
1096
|
+
...memorySelection.memories.map((evidence) => ({
|
|
1097
|
+
id: `memory:${evidence.item.id}`,
|
|
1098
|
+
message: evidence.message,
|
|
1099
|
+
kind: "memory" as const,
|
|
1100
|
+
sourceIds: [evidence.item.id],
|
|
1101
|
+
score: 90 + Math.min(0.999999, Math.max(0, evidence.score) / 1_000),
|
|
1102
|
+
reason: evidence.reason,
|
|
1103
|
+
})),
|
|
1104
|
+
] satisfies Array<SupplementalContextMessage<ContextEvent["messages"][number]>>;
|
|
1105
|
+
const nominalDedupPlan = planManagedContext({
|
|
1106
|
+
messages: effectiveEvent.messages,
|
|
1107
|
+
fixedTokens,
|
|
1108
|
+
budget,
|
|
1109
|
+
config: effectiveContextConfig,
|
|
1110
|
+
pinnedMessageIndices,
|
|
1111
|
+
supplementalMessages: dedupSupplementalMessages,
|
|
1112
|
+
});
|
|
1113
|
+
/**
|
|
1114
|
+
* Cache-aware candidate selection (opt-in, default off): compares the
|
|
1115
|
+
* estimated input cost of the nominal plan with an extended-tail plan
|
|
1116
|
+
* and switches only when the improvement beats the configured
|
|
1117
|
+
* hysteresis threshold. Costs are metadata estimates from model
|
|
1118
|
+
* pricing and message fingerprints; no provider content is exposed.
|
|
1119
|
+
*/
|
|
1120
|
+
let cacheAwareTailTokens: number | undefined;
|
|
1121
|
+
let cacheAwareDecision: CacheAwarePlanDecision | undefined;
|
|
1122
|
+
if (effectiveContextConfig.cacheAware?.mode === "auto" && model?.cost && budget) {
|
|
1123
|
+
const decision = decideCacheAwareTail({
|
|
1124
|
+
config: effectiveContextConfig.cacheAware,
|
|
1125
|
+
pricing: Ds4ContextRuntime.cachePricing(model.cost),
|
|
1126
|
+
observedCacheReadShare: activeModel?.awareness.calibration.cache.sampleCount > 0
|
|
1127
|
+
? activeModel.awareness.calibration.cache.cacheReadShare
|
|
1128
|
+
: undefined,
|
|
1129
|
+
sampleCount: activeModel?.awareness.calibration.cache.sampleCount ?? 0,
|
|
1130
|
+
nominalRecentTailTokens: activeModel?.awareness.limits.recentTailTokens
|
|
1131
|
+
?? effectiveContextConfig.recentTailTokens,
|
|
1132
|
+
activeInputBudget: budget.activeInputBudget,
|
|
1133
|
+
});
|
|
1134
|
+
cacheAwareDecision = decision;
|
|
1135
|
+
if (decision.eligible && decision.tailExtended) {
|
|
1136
|
+
const extendedDedupPlan = planManagedContext({
|
|
984
1137
|
messages: effectiveEvent.messages,
|
|
985
1138
|
fixedTokens,
|
|
986
1139
|
budget,
|
|
987
1140
|
config: effectiveContextConfig,
|
|
988
1141
|
pinnedMessageIndices,
|
|
989
|
-
supplementalMessages:
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1142
|
+
supplementalMessages: dedupSupplementalMessages,
|
|
1143
|
+
cacheAwareTailTokens: decision.recentTailTokens,
|
|
1144
|
+
});
|
|
1145
|
+
const modelKey = modelProfileKey(model.provider, model.id);
|
|
1146
|
+
const nominalEpoch = this.cacheAwareEpochCost(nominalDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
|
|
1147
|
+
const extendedEpoch = this.cacheAwareEpochCost(extendedDedupPlan, fixedTokens, model.cost, effectiveContextConfig.cacheAware.expectedRequestsPerTurn, effectiveContextConfig.cacheAware.expectedTurnsPerEpoch, modelKey);
|
|
1148
|
+
this.logger.debug("context.cache_aware_candidate", {
|
|
1149
|
+
eligible: decision.eligible,
|
|
1150
|
+
tailExtended: decision.tailExtended,
|
|
1151
|
+
recentTailTokens: decision.recentTailTokens,
|
|
1152
|
+
nominalEpoch,
|
|
1153
|
+
extendedEpoch,
|
|
1154
|
+
});
|
|
1155
|
+
const extendedWon = extendedEpoch !== undefined
|
|
1156
|
+
&& nominalEpoch !== undefined
|
|
1157
|
+
&& extendedEpoch < nominalEpoch * (1 - effectiveContextConfig.cacheAware.minimumImprovementRatio);
|
|
1158
|
+
// Hysteresis (deliberate epochs): once adopted, the extended tail is
|
|
1159
|
+
// kept while it loses at most a single epoch comparison; it is
|
|
1160
|
+
// dropped only after two consecutive losses, preventing
|
|
1161
|
+
// oscillation between plans. The streak resets on a win or model switch.
|
|
1162
|
+
if (extendedWon) {
|
|
1163
|
+
this.cacheAwareExtendedLossStreak = 0;
|
|
1164
|
+
this.cacheAwareStickExtended = true;
|
|
1165
|
+
} else if (this.cacheAwareStickExtended) {
|
|
1166
|
+
this.cacheAwareExtendedLossStreak += 1;
|
|
1167
|
+
if (this.cacheAwareExtendedLossStreak >= effectiveContextConfig.cacheAware.stickinessEpochs) {
|
|
1168
|
+
this.cacheAwareStickExtended = false;
|
|
1169
|
+
this.cacheAwareExtendedLossStreak = 0;
|
|
1170
|
+
}
|
|
1171
|
+
}
|
|
1172
|
+
if (this.cacheAwareStickExtended) {
|
|
1173
|
+
cacheAwareTailTokens = decision.recentTailTokens;
|
|
1174
|
+
}
|
|
1175
|
+
}
|
|
1176
|
+
}
|
|
1177
|
+
const retrievalActiveContextEntryIds = retrievalEnabled
|
|
1178
|
+
? this.plannedContextEntryIds(ctx, cacheAwareTailTokens !== undefined
|
|
1179
|
+
? planManagedContext({
|
|
1180
|
+
messages: effectiveEvent.messages,
|
|
1181
|
+
fixedTokens,
|
|
1182
|
+
budget,
|
|
1183
|
+
config: effectiveContextConfig,
|
|
1184
|
+
pinnedMessageIndices,
|
|
1185
|
+
supplementalMessages: dedupSupplementalMessages,
|
|
1186
|
+
cacheAwareTailTokens,
|
|
1187
|
+
})
|
|
1188
|
+
: nominalDedupPlan)
|
|
1008
1189
|
: undefined;
|
|
1009
1190
|
const retrieval = this.retrieveHistory(
|
|
1010
1191
|
effectiveEvent,
|
|
@@ -1149,6 +1330,7 @@ export class Ds4ContextRuntime {
|
|
|
1149
1330
|
config: effectiveContextConfig,
|
|
1150
1331
|
pinnedMessageIndices,
|
|
1151
1332
|
supplementalMessages: rankedSupplementalMessages,
|
|
1333
|
+
...(cacheAwareTailTokens !== undefined ? { cacheAwareTailTokens } : {}),
|
|
1152
1334
|
...(ranking.diagnostics.status === "active"
|
|
1153
1335
|
? { supplementalSelectionOrder: ranking.ranked.map((candidate) => candidate.id) }
|
|
1154
1336
|
: {}),
|
|
@@ -1209,6 +1391,64 @@ export class Ds4ContextRuntime {
|
|
|
1209
1391
|
selected: plannerSelectedProject,
|
|
1210
1392
|
};
|
|
1211
1393
|
plan.planning.durationMs = Math.max(0, this.now() - planningStartedAt);
|
|
1394
|
+
/**
|
|
1395
|
+
* Cache-aware diagnostics are metadata-only (tokens, ratios, cost).
|
|
1396
|
+
* The reusable prefix is estimated against the previous managed plan;
|
|
1397
|
+
* if the strategy changed since the last plan, the estimate is
|
|
1398
|
+
* conservative (empty hashes) rather than optimistic. The block is
|
|
1399
|
+
* present only when the cache-aware policy is enabled (mode auto);
|
|
1400
|
+
* mode off leaves the manifest unchanged from 0.3.6.
|
|
1401
|
+
*/
|
|
1402
|
+
if (plan.mode === "managed" && effectiveContextConfig.cacheAware?.mode === "auto") {
|
|
1403
|
+
const currentModelKey = model ? modelProfileKey(model.provider, model.id) : undefined;
|
|
1404
|
+
const previousHashes = currentModelKey !== undefined && this.lastPlanModelKey === currentModelKey
|
|
1405
|
+
? this.lastPlanMessageHashes
|
|
1406
|
+
: [];
|
|
1407
|
+
const currentHashes: string[] = [];
|
|
1408
|
+
const currentTokens: number[] = [];
|
|
1409
|
+
for (const message of plan.messages) {
|
|
1410
|
+
currentHashes.push(fingerprint(message));
|
|
1411
|
+
currentTokens.push(estimateMessagesTokens([message]));
|
|
1412
|
+
}
|
|
1413
|
+
const reusablePrefixTokens = estimateReusablePrefixTokens(
|
|
1414
|
+
previousHashes,
|
|
1415
|
+
currentHashes,
|
|
1416
|
+
currentTokens,
|
|
1417
|
+
);
|
|
1418
|
+
const estimatedCost = model?.cost
|
|
1419
|
+
? estimateRequestCost({
|
|
1420
|
+
totalInputTokens: plan.planning.fixedTokens
|
|
1421
|
+
+ plan.selected.reduce((total, item) => total + item.tokens, 0),
|
|
1422
|
+
reusablePrefixTokens,
|
|
1423
|
+
}, Ds4ContextRuntime.cachePricing(model.cost) ?? {}).total
|
|
1424
|
+
: undefined;
|
|
1425
|
+
const candidate: "nominal" | "cache-aware" = cacheAwareTailTokens !== undefined
|
|
1426
|
+
? "cache-aware"
|
|
1427
|
+
: "nominal";
|
|
1428
|
+
plan.planning.cacheAware = {
|
|
1429
|
+
eligible: cacheAwareDecision?.eligible ?? false,
|
|
1430
|
+
tailExtended: cacheAwareTailTokens !== undefined,
|
|
1431
|
+
...(cacheAwareDecision?.recentTailTokens !== undefined
|
|
1432
|
+
? { recentTailTokens: cacheAwareDecision.recentTailTokens }
|
|
1433
|
+
: {}),
|
|
1434
|
+
...(cacheAwareDecision?.missHitRatio !== undefined
|
|
1435
|
+
? { missHitRatio: cacheAwareDecision.missHitRatio }
|
|
1436
|
+
: {}),
|
|
1437
|
+
...(cacheAwareDecision?.cacheReadShare !== undefined
|
|
1438
|
+
? { cacheReadShare: cacheAwareDecision.cacheReadShare }
|
|
1439
|
+
: {}),
|
|
1440
|
+
...(cacheAwareDecision?.sampleCount !== undefined
|
|
1441
|
+
? { sampleCount: cacheAwareDecision.sampleCount }
|
|
1442
|
+
: {}),
|
|
1443
|
+
...(reusablePrefixTokens > 0 ? { reusablePrefixTokens } : {}),
|
|
1444
|
+
...(estimatedCost !== undefined ? { estimatedCost } : {}),
|
|
1445
|
+
candidate,
|
|
1446
|
+
};
|
|
1447
|
+
this.lastPlanMessageHashes = currentHashes;
|
|
1448
|
+
this.lastPlanMessageTokens = currentTokens;
|
|
1449
|
+
this.lastCacheAwareDecision = cacheAwareDecision;
|
|
1450
|
+
this.lastPlanModelKey = currentModelKey;
|
|
1451
|
+
}
|
|
1212
1452
|
const oversizedTurnExclusions = plan.planning.oversizedTurnExclusions ?? 0;
|
|
1213
1453
|
if (plan.mode === "managed" && oversizedTurnExclusions > 0) {
|
|
1214
1454
|
this.logger.warn("context.excluded_oversized_turn", {
|
|
@@ -1407,6 +1647,9 @@ export class Ds4ContextRuntime {
|
|
|
1407
1647
|
}
|
|
1408
1648
|
try {
|
|
1409
1649
|
const result = this.database.manifests.save(manifest);
|
|
1650
|
+
if (result.status === "stored") {
|
|
1651
|
+
this.lastPersistedInventory = result.inventory;
|
|
1652
|
+
}
|
|
1410
1653
|
if (result.status === "skipped-oversize") {
|
|
1411
1654
|
this.logger.warn("context.manifest_persistence_skipped", {
|
|
1412
1655
|
category: "oversize",
|
|
@@ -3037,6 +3280,7 @@ export class Ds4ContextRuntime {
|
|
|
3037
3280
|
...(indexed ? { indexed } : {}),
|
|
3038
3281
|
...(this.observation ? { observation: this.observation } : {}),
|
|
3039
3282
|
...(this.lastManifest ? { lastManifest: this.lastManifest } : {}),
|
|
3283
|
+
...(this.lastPersistedInventory ? { persistedInventory: this.lastPersistedInventory } : {}),
|
|
3040
3284
|
retrieval: this.lastRetrieval,
|
|
3041
3285
|
project: this.lastProject,
|
|
3042
3286
|
memory: this.lastMemory,
|
|
@@ -87,7 +87,7 @@ function sourceKind(entry: SessionEntry, role?: string): ContextManifestItemKind
|
|
|
87
87
|
return "history";
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
-
function fingerprint(message: unknown): string {
|
|
90
|
+
export function fingerprint(message: unknown): string {
|
|
91
91
|
return sha256(stableStringify(message));
|
|
92
92
|
}
|
|
93
93
|
|
|
@@ -35,5 +35,15 @@ export function snapshotModel(ctx: Pick<ExtensionContext, "model">): ModelDescri
|
|
|
35
35
|
maxTokens: ctx.model.maxTokens,
|
|
36
36
|
reasoning: ctx.model.reasoning,
|
|
37
37
|
input: ctx.model.input,
|
|
38
|
+
...(ctx.model.cost && typeof ctx.model.cost.input === "number"
|
|
39
|
+
? {
|
|
40
|
+
cost: {
|
|
41
|
+
input: ctx.model.cost.input,
|
|
42
|
+
output: ctx.model.cost.output,
|
|
43
|
+
cacheRead: ctx.model.cost.cacheRead,
|
|
44
|
+
cacheWrite: ctx.model.cost.cacheWrite,
|
|
45
|
+
},
|
|
46
|
+
}
|
|
47
|
+
: {}),
|
|
38
48
|
};
|
|
39
49
|
}
|