opencode-cache-engine 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +76 -13
- package/docs/cache-policy-inventory.md +81 -8
- package/package.json +1 -1
- package/src/cache-policy-core.mjs +101 -25
- package/test/cache-engine.test.mjs +192 -18
package/README.md
CHANGED
|
@@ -18,15 +18,30 @@ plugin under `~/.config/opencode/plugins/`. For released installs where
|
|
|
18
18
|
reproducibility matters, pin an exact package version rather than relying on
|
|
19
19
|
`@latest` resolution or a moving cache entry; see [Installation](#installation).
|
|
20
20
|
|
|
21
|
+
Quick Installation (TUI):
|
|
22
|
+
``` text
|
|
23
|
+
opencode plugin opencode-cache-engine
|
|
24
|
+
```
|
|
25
|
+
|
|
21
26
|
`CacheEngine` is an OpenCode plugin designed for long-running agent sessions where prompt-cache efficiency affects both latency and cost. It keeps the harness conservative for providers whose cache behavior is already automatic, while applying provider-specific optimizations where the provider exposes useful cache controls or where prompt structure can be safely improved.
|
|
22
27
|
|
|
23
28
|
The plugin currently has four cache-policy families:
|
|
24
29
|
|
|
25
30
|
* **DeepSeek** — passive cache observability; request structure is preserved.
|
|
26
|
-
* **GPT-5.6** — documented cache-key/options metadata, with prompt text
|
|
31
|
+
* **GPT-5.6 and later** — documented cache-key/options metadata, with prompt text
|
|
32
|
+
unchanged. GPT-6 and future 5.6+/6+/7+ versions resolve through the same
|
|
33
|
+
documented boundary.
|
|
27
34
|
* **GLM-5.3** — narrow, content-preserving `<env>` relocation and diagnostics.
|
|
28
35
|
* **MiMo-V2.6** — narrow, content-preserving `<env>` relocation and diagnostics.
|
|
29
36
|
|
|
37
|
+
Family classification is not hard-coded in the runtime. A pure policy registry
|
|
38
|
+
and resolver in `src/cache-policy-core.mjs` returns a structured result
|
|
39
|
+
(`creator`, `family`, `baseline`, `overlays`, `transport`, `matchType`,
|
|
40
|
+
`matchReason`), and the hooks gate their behavior on that result. The registry is
|
|
41
|
+
the single runtime source of policy classification. The first-party research
|
|
42
|
+
behind each registry entry is recorded in
|
|
43
|
+
[docs/cache-policy-inventory.md](docs/cache-policy-inventory.md).
|
|
44
|
+
|
|
30
45
|
For both MiMo-V2.6 and GLM-5.3, CacheEngine adds its deterministic
|
|
31
46
|
`x-session-id` request header only when OpenCode identifies the actual provider
|
|
32
47
|
as `openrouter`. It does not add that OpenRouter-specific header for
|
|
@@ -44,12 +59,12 @@ The plugin operates at the OpenCode harness level rather than implementing a pro
|
|
|
44
59
|
|
|
45
60
|
It:
|
|
46
61
|
|
|
47
|
-
1.
|
|
48
|
-
2. Applies only the
|
|
62
|
+
1. Resolves the model/provider policy through the registry resolver.
|
|
63
|
+
2. Applies only the mutations registered for that policy.
|
|
49
64
|
3. Observes system-prompt and tool-definition stability.
|
|
50
65
|
4. Records provider-reported cache token usage.
|
|
51
66
|
5. Adds a deterministic compaction continuation block.
|
|
52
|
-
6. Applies GPT-5.6 cache-control metadata.
|
|
67
|
+
6. Applies GPT-5.6-and-later cache-control metadata.
|
|
53
68
|
7. Applies the GLM-5.3 and MiMo-V2.6 volatile-environment relocation.
|
|
54
69
|
8. Records diagnostics that help determine whether prompt-shape changes correlate with cache behavior.
|
|
55
70
|
9. Records MiMo/GLM affinity outcomes and provider-identity changes.
|
|
@@ -59,7 +74,7 @@ The plugin deliberately avoids pretending that a local hash is proof of a provid
|
|
|
59
74
|
|
|
60
75
|
# Provider behavior
|
|
61
76
|
|
|
62
|
-
## DeepSeek V4
|
|
77
|
+
## DeepSeek V4 and later
|
|
63
78
|
|
|
64
79
|
### Policy: passive
|
|
65
80
|
|
|
@@ -78,6 +93,13 @@ The DeepSeek branch exists primarily to preserve a stable harness while providin
|
|
|
78
93
|
|
|
79
94
|
This is intentional. The implementation describes DeepSeek as a passive policy whose purpose is to preserve the existing high-cache-rate behavior rather than introduce new request mutations.
|
|
80
95
|
|
|
96
|
+
Since v0.4.3 this is formalized as the documented **"DeepSeek V4 and later"**
|
|
97
|
+
family. Canonical ids (`deepseek-flash`, `deepseek-v4-pro`) and the accepted
|
|
98
|
+
`deepseek-v4-flash` aliases resolve to this passive baseline, and pre-V4 or
|
|
99
|
+
unknown future `*deepseek*` ids fall back to the same passive baseline. No cache
|
|
100
|
+
key, cache-control field, prompt rewrite, or OpenRouter affinity is ever added
|
|
101
|
+
for DeepSeek.
|
|
102
|
+
|
|
81
103
|
The plugin still observes:
|
|
82
104
|
|
|
83
105
|
* system-prompt shape
|
|
@@ -115,7 +137,7 @@ DeepSeek:
|
|
|
115
137
|
|
|
116
138
|
### Policy: active cache control
|
|
117
139
|
|
|
118
|
-
GPT-5.6 is the only
|
|
140
|
+
GPT-5.6 and later is the only policy family that actively injects cache-control request metadata.
|
|
119
141
|
|
|
120
142
|
The plugin adds:
|
|
121
143
|
|
|
@@ -404,8 +426,8 @@ and are reported diagnostically; the message content is left untouched.
|
|
|
404
426
|
|
|
405
427
|
| Policy family | Detection | Prompt text changed? | Cache metadata changed? | OpenRouter affinity header | Primary cache signal |
|
|
406
428
|
| ------------- | --------- | ------------------- | ----------------------- | -------------------------- | -------------------- |
|
|
407
|
-
| DeepSeek | `deepseek` | No | No | None | provider `cache.read` / `cache.write` |
|
|
408
|
-
| GPT-5.6 | `gpt
|
|
429
|
+
| DeepSeek | `deepseek` (V4-and-later family + passive fallback) | No | No | None | provider `cache.read` / `cache.write` |
|
|
430
|
+
| GPT-5.6 and later | version boundary `gpt-<major>[.<minor>] ≥ 5.6` on OpenAI-ish endpoints (includes GPT-6) | No | Yes: `prompt_cache_key` + options | None | provider cache tokens |
|
|
409
431
|
| GLM-5.3 | `glm-5.3*` | Yes, narrowly (`<env>` tail) | No provider cache key | `x-session-id` on OpenRouter only | provider cache tokens (GLM ratio) |
|
|
410
432
|
| MiMo-V2.6 | Flash / Pro only | Yes, narrowly (`<env>` tail) | No: implicit caching only | `x-session-id` on OpenRouter only | `cached_tokens / prompt_tokens` |
|
|
411
433
|
|
|
@@ -923,7 +945,7 @@ The model detector recognizes:
|
|
|
923
945
|
* GLM-5.3 variants
|
|
924
946
|
* MiMo-V2.6 Flash and Pro (`xiaomi/mimo-v2.6-flash`, `mimo-v2.6-pro`, ...)
|
|
925
947
|
|
|
926
|
-
GPT-5.6 has an additional OpenAI/Azure-context check so a string containing `gpt-5.6` does not automatically cause GPT-specific fields to be sent to an unrelated endpoint.
|
|
948
|
+
The GPT-5.6-and-later family has an additional OpenAI/Azure-context check, so a string containing a qualifying GPT version (for example `gpt-5.6` or `gpt-6`) does not automatically cause GPT-specific fields to be sent to an unrelated endpoint.
|
|
927
949
|
|
|
928
950
|
MiMo detection targets exactly Flash and Pro: it excludes `mimo-v2.5`,
|
|
929
951
|
`mimo-v2.5-pro`, and `mimo-v2.6-pro-ultraspeed`.
|
|
@@ -953,7 +975,8 @@ For that reason, a stable provider route is preferable when your goal is to meas
|
|
|
953
975
|
|
|
954
976
|
# Architecture
|
|
955
977
|
|
|
956
|
-
The implementation is split
|
|
978
|
+
The implementation is split across a hook entry point, a pure logic core, and a
|
|
979
|
+
pure policy registry.
|
|
957
980
|
|
|
958
981
|
## `cache-engine.ts`
|
|
959
982
|
|
|
@@ -989,7 +1012,7 @@ This contains dependency-light pure logic.
|
|
|
989
1012
|
|
|
990
1013
|
It owns:
|
|
991
1014
|
|
|
992
|
-
*
|
|
1015
|
+
* the legacy `detectPolicy()` compatibility wrapper (delegating to the registry)
|
|
993
1016
|
* configuration parsing
|
|
994
1017
|
* hashing
|
|
995
1018
|
* canonicalization
|
|
@@ -1006,6 +1029,42 @@ Keeping these functions in plain JavaScript allows the logic to be tested indepe
|
|
|
1006
1029
|
|
|
1007
1030
|
---
|
|
1008
1031
|
|
|
1032
|
+
## `cache-policy-core.mjs`
|
|
1033
|
+
|
|
1034
|
+
This is the pure policy registry and resolver. It separates cache policy from
|
|
1035
|
+
request mutation:
|
|
1036
|
+
|
|
1037
|
+
* creator / family classification
|
|
1038
|
+
* baseline cache-policy descriptors (documented facts)
|
|
1039
|
+
* model-specific overlays (for example GLM/MiMo `<env>` relocation)
|
|
1040
|
+
* transport capabilities (for example OpenRouter `x-session-id` affinity)
|
|
1041
|
+
* explicit, inventory-traceable inheritance (`inheritsFrom`)
|
|
1042
|
+
* safe neutral fallback for unknown or future models
|
|
1043
|
+
|
|
1044
|
+
`resolvePolicy(model)` returns `creator`, `family`, `baseline`, `overlays`,
|
|
1045
|
+
`transport`, `matchType`, and `matchReason`. `resolveRuntimePolicy(model)`
|
|
1046
|
+
returns the runtime-facing descriptor the hooks consume: the legacy policy
|
|
1047
|
+
string plus explicit capability flags.
|
|
1048
|
+
|
|
1049
|
+
Only registry entries marked `legacy` enable runtime behavior; documented but
|
|
1050
|
+
non-legacy entries (for example `mimo-v2.6-pro-ultraspeed`) and all unknown
|
|
1051
|
+
models resolve to a neutral runtime. A newer or unknown model therefore never
|
|
1052
|
+
inherits a current model's mutation unless the registry explicitly registers it.
|
|
1053
|
+
The GPT family is a documented exception in the sense that its boundary is
|
|
1054
|
+
version-based (`GPT-5.6 and later`), so GPT-6 and future 5.6+/6+/7+ versions are
|
|
1055
|
+
covered by the registered boundary rather than by an exact-model list.
|
|
1056
|
+
|
|
1057
|
+
Transport is kept separate from cache policy: OpenRouter affinity is a transport
|
|
1058
|
+
capability, not part of a creator's cache semantics. Overlays are also explicit,
|
|
1059
|
+
so being classified into a family does not by itself enable a prompt
|
|
1060
|
+
transformation.
|
|
1061
|
+
|
|
1062
|
+
The module is pure: no network calls and no runtime documentation lookups. The
|
|
1063
|
+
legacy `detectPolicy()` in `cache-engine-core.mjs` remains a thin compatibility
|
|
1064
|
+
wrapper over the resolver's legacy path.
|
|
1065
|
+
|
|
1066
|
+
---
|
|
1067
|
+
|
|
1009
1068
|
## Tests
|
|
1010
1069
|
|
|
1011
1070
|
The repository's test suite validates the provider-independent and provider-specific logic.
|
|
@@ -1013,6 +1072,7 @@ The repository's test suite validates the provider-independent and provider-spec
|
|
|
1013
1072
|
Coverage includes:
|
|
1014
1073
|
|
|
1015
1074
|
* model detection
|
|
1075
|
+
* policy registry resolution and runtime-policy equivalence
|
|
1016
1076
|
* GPT cache-key stability
|
|
1017
1077
|
* GPT cache-option defaults
|
|
1018
1078
|
* protection against overwriting existing cache options
|
|
@@ -1176,11 +1236,14 @@ opencode-cache-engine/
|
|
|
1176
1236
|
├── src/
|
|
1177
1237
|
│ ├── cache-engine.ts
|
|
1178
1238
|
│ ├── cache-engine-core.mjs
|
|
1239
|
+
│ ├── cache-policy-core.mjs
|
|
1179
1240
|
│ └── tui.mjs
|
|
1180
1241
|
├── test/
|
|
1181
1242
|
│ └── cache-engine.test.mjs
|
|
1182
1243
|
├── examples/
|
|
1183
1244
|
│ └── cache-engine.json
|
|
1245
|
+
├── docs/
|
|
1246
|
+
│ └── cache-policy-inventory.md
|
|
1184
1247
|
├── package.json
|
|
1185
1248
|
├── README.md
|
|
1186
1249
|
└── LICENSE
|
|
@@ -1219,7 +1282,7 @@ release, use:
|
|
|
1219
1282
|
```json
|
|
1220
1283
|
{
|
|
1221
1284
|
"plugin": [
|
|
1222
|
-
"opencode-cache-engine@0.
|
|
1285
|
+
"opencode-cache-engine@0.4.1"
|
|
1223
1286
|
]
|
|
1224
1287
|
}
|
|
1225
1288
|
```
|
|
@@ -1286,7 +1349,7 @@ A prefix change is a diagnostic signal, not automatic proof of a cache miss.
|
|
|
1286
1349
|
|
|
1287
1350
|
## GPT-5.6 cache options are missing
|
|
1288
1351
|
|
|
1289
|
-
Verify that the model is
|
|
1352
|
+
Verify that the model is within the documented GPT-5.6-and-later boundary (for example `gpt-5.6-*` or `gpt-6-*`) and that the endpoint is recognized as OpenAI/Azure-compatible.
|
|
1290
1353
|
|
|
1291
1354
|
The detector intentionally rejects ambiguous OpenAI-compatible providers rather than guessing.
|
|
1292
1355
|
|
|
@@ -32,22 +32,47 @@ Caching behavior is never inferred from pricing alone, from one SDK's type
|
|
|
32
32
|
declarations, or from a third-party blog when first-party documentation exists.
|
|
33
33
|
"unknown — first-party docs insufficient" is used instead of a guess.
|
|
34
34
|
|
|
35
|
+
## Runtime integration
|
|
36
|
+
|
|
37
|
+
This inventory is the research input for the policy registry in
|
|
38
|
+
`src/cache-policy-core.mjs`. As of **v0.4.1** the runtime hooks consume
|
|
39
|
+
`resolveRuntimePolicy(model)` and gate behavior on the registry's explicit
|
|
40
|
+
capabilities, so the "CacheEngine current treatment" column below describes
|
|
41
|
+
resolver-driven behavior.
|
|
42
|
+
|
|
43
|
+
Two guarantees follow from that migration:
|
|
44
|
+
|
|
45
|
+
- `resolveRuntimePolicy(model).policy` equals the legacy `detectPolicy(model)`
|
|
46
|
+
string, so telemetry and gating are unchanged for every model supported in
|
|
47
|
+
v0.3.6.
|
|
48
|
+
- Registry entries marked non-`legacy` (for example
|
|
49
|
+
`mimo-v2.6-pro-ultraspeed`) and all unknown models resolve to a neutral
|
|
50
|
+
runtime, so no documented-but-unwired model gains a current model's mutation.
|
|
51
|
+
|
|
52
|
+
The runtime reads the registry at classification time only; there are no network
|
|
53
|
+
calls and no runtime documentation lookups.
|
|
54
|
+
|
|
35
55
|
## CacheEngine current treatment (baseline for the matrix)
|
|
36
56
|
|
|
37
|
-
Source: `src/cache-
|
|
38
|
-
`src/cache-engine.ts`
|
|
57
|
+
Source: the pure registry/resolver (`src/cache-policy-core.mjs`) and the hook
|
|
58
|
+
entry (`src/cache-engine.ts`), as wired in v0.4.1. The behavior described here is
|
|
59
|
+
the same as at the pre-resolver revision `19b87f2`; only the classification
|
|
60
|
+
source changed.
|
|
39
61
|
|
|
40
62
|
| Family | Detection (verbatim) | Current treatment | Affinity header |
|
|
41
63
|
| --- | --- | --- | --- |
|
|
42
64
|
| DeepSeek | `/deepseek/i` on `${apiID} ${modelID}` or `providerID` | Passive; no mutation | none |
|
|
43
|
-
| GPT-5.6 |
|
|
65
|
+
| GPT-5.6 | version boundary `gpt-<major>[.<minor>] ≥ 5.6` on slug **and** `isOpenAIish` (provider `openai`/`azure`, slug `openai/`/`azure/`, or npm `@ai-sdk/openai`/`@ai-sdk/azure`); since v0.4.2 covers GPT-6 and later | Inject missing `promptCacheKey` + `promptCacheOptions` (`implicit`, `30m`) | none |
|
|
44
66
|
| GLM-5.3 | `/glm-5\.3(?![\d.])/i` on slug | Relocate identifiable `<env>` block to system tail | `x-session-id` only when `providerID === "openrouter"` |
|
|
45
67
|
| MiMo-V2.6 | `/mimo-v2\.6-(flash\|pro)(?![\w-])/i` on slug | Relocate identifiable `<env>` block to system tail; provider-change telemetry | `x-session-id` only when `providerID === "openrouter"` |
|
|
46
68
|
| Neutral | everything else | Byte-untouched | none |
|
|
47
69
|
|
|
48
70
|
Detection consequences worth stating explicitly:
|
|
49
71
|
|
|
50
|
-
- `gpt-6
|
|
72
|
+
- `gpt-6` / `gpt-6-*` (and any future 5.6+/6+/7+ version) is matched by the
|
|
73
|
+
documented GPT-5.6-and-later boundary → GPT policy. **[O]** (v0.4.2)
|
|
74
|
+
- `gpt-5.5`, `gpt-5.2`, `gpt-4o`, and the malformed `gpt-5.60` are **not**
|
|
75
|
+
matched → neutral. **[O]**
|
|
51
76
|
- `deepseek-v5` (or any future `*deepseek*` id) matches the passive DeepSeek
|
|
52
77
|
branch because the regex is a bare substring test. **[O]**
|
|
53
78
|
- `mimo-v2.6-pro-ultraspeed` is **not** matched: the `(?![\w-])` lookahead fails
|
|
@@ -95,8 +120,10 @@ catch-all "earlier models". [D]
|
|
|
95
120
|
baseline: it injects only `promptCacheKey` + `promptCacheOptions{mode:implicit,
|
|
96
121
|
ttl:"30m"}`, preserves runtime-supplied values, and never sets context/output
|
|
97
122
|
limits. It does not use explicit breakpoints or `prewarm`, which is a subset of
|
|
98
|
-
the documented capability.
|
|
99
|
-
|
|
123
|
+
the documented capability. Since v0.4.2 the policy resolves by the documented
|
|
124
|
+
"GPT-5.6 and later" boundary, so GPT-6 (astra/sol/luna) receives the same
|
|
125
|
+
baseline; no GPT-6 cache-control exception is documented (OpenAI *Prompt
|
|
126
|
+
caching* guide, re-verified 2026-09-27), and none is coded.
|
|
100
127
|
|
|
101
128
|
---
|
|
102
129
|
|
|
@@ -254,11 +281,11 @@ made here); **hold** = do not inherit without first-party evidence.
|
|
|
254
281
|
| Creator | Model / example pattern | Cache policy (documented) | CacheEngine current treatment | Recommended family inheritance | Recommended exact-model exception | Confidence | Source | Verified |
|
|
255
282
|
| --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
|
256
283
|
| OpenAI | `gpt-5.6`, `gpt-5.6-*` (sol/terra/luna/cyber) | "GPT-5.6 and later": implicit default, optional explicit breakpoints, min 1,024, TTL 30m, write 1.25×/read 0.1× | GPT policy: inject `promptCacheKey` + `promptCacheOptions{implicit,30m}` | keep | none documented; CacheEngine's subset is valid | High (docs) / Medium (treatment) | OpenAI *Prompt caching*; *GPT-5.6 Sol* | 2026-09-26 |
|
|
257
|
-
| OpenAI | GPT-6 / current later GPT family: `gpt-6`, `gpt-6-*` (astra/sol/luna) | Inherits the GPT-5.6-and-later policy | **
|
|
284
|
+
| OpenAI | GPT-6 / current later GPT family: `gpt-6`, `gpt-6-*` (astra/sol/luna) | Inherits the GPT-5.6-and-later policy | **GPT policy** via the GPT-5.6-and-later boundary (v0.4.2) | **keep** — covered by the boundary predicate; no exact-model entry needed | none documented | High (docs) / High (treatment) | OpenAI *Using GPT-6*; *GPT-6 Astra*; *Prompt caching* | 2026-09-27 |
|
|
258
285
|
| OpenAI | pre-5.6 negative controls: `gpt-5.5`, `gpt-5.4`, `gpt-5.2`, `gpt-5.1`, `gpt-5`, `gpt-4.1`, `gpt-4o` | Implicit only; different min-length class; `in_memory`/`24h` retention; `prompt_cache_key` for routing | neutral | **hold** — do not inherit 5.6 policy | n/a | High | OpenAI *Prompt caching*; *Pricing* | 2026-09-26 |
|
|
259
286
|
| DeepSeek | V4: `deepseek-v4-pro`, legacy `deepseek-v4-flash` | Provider-wide automatic disk cache; implicit; prefix-unit matching; hit/miss token fields | Passive (no mutation) | keep | none | High | DeepSeek *Context Caching*; *Models & Pricing* | 2026-09-26 |
|
|
260
287
|
| DeepSeek | V4.1 / current V4-family: `deepseek-flash` (MODEL VERSION "DeepSeek-V4.1-Flash") | Same provider-wide automatic policy; cache-hit pricing listed for both current models | Passive | keep (creator/family baseline) | none documented | High | DeepSeek *Models & Pricing*; *news260910* | 2026-09-26 |
|
|
261
|
-
| DeepSeek | future-looking V4+ identifiers: `deepseek-v4.1`, `deepseek-v4`, `deepseek-v5` | Not documented as request ids (`deepseek-v4.1`/`deepseek-v4` invalid or version-string only) | Passive via
|
|
288
|
+
| DeepSeek | future-looking V4+ identifiers: `deepseek-v4.1`, `deepseek-v4`, `deepseek-v5` | Not documented as request ids (`deepseek-v4.1`/`deepseek-v4` invalid or version-string only) | Passive via the V4-and-later family predicate or the safe creator fallback (v0.4.3); no mutation | keep passive; treat as unknown-friendly | none | Medium (detection) / Low (future ids) | DeepSeek *Models & Pricing*; *Chat Completions API* | 2026-09-27 |
|
|
262
289
|
| DeepSeek | pre-V4 negative controls: `deepseek-chat`, `deepseek-reasoner` | Retired names (retired 2026-07-24); no separate V4+ cache policy claimed | Passive | hold | n/a | High | DeepSeek *Change Log*; *news260424* | 2026-09-26 |
|
|
263
290
|
| Z.AI | GLM 5.3: `glm-5.3`, `glm-5.3-flash`, `glm-5.3-flashx` | Implicit automatic caching; `cached_tokens`; stable-prompt-first guidance; no documented min/TTL/key | GLM policy: `<env>` relocation; OpenRouter `x-session-id` | keep (env relocation is an exact overlay, not a Z.AI control) | env relocation is the overlay; keep scoped to GLM-5.3 | Medium | Z.AI *Context Caching*; *Chat Completion*; *Pricing* | 2026-09-26 |
|
|
264
291
|
| Z.AI | current later GLM generations (documented): none newer than 5.3; newest below is `glm-5.2`/`glm-5.1`/`glm-5`/`glm-4.7` | Same implicit mechanism documented service-wide; cached-input price per model | neutral (only `glm-5.3` matched) | **hold** — no doc says 5.3 overlay extends upward; none newer documented | n/a | High (no later gens documented) | Z.AI *New Released*; *Pricing* | 2026-09-26 |
|
|
@@ -311,3 +338,49 @@ a newer model inherits an older policy. They must not be resolved by guessing.
|
|
|
311
338
|
- The existing test suite was run only to confirm the repository remains green
|
|
312
339
|
(116/116).
|
|
313
340
|
- This release contains research and documentation only.
|
|
341
|
+
|
|
342
|
+
### Follow-up: v0.4.1 runtime integration
|
|
343
|
+
|
|
344
|
+
- v0.4.0 added the registry (`src/cache-policy-core.mjs`) without wiring it.
|
|
345
|
+
- v0.4.1 wired the runtime to `resolveRuntimePolicy()` and preserved behavior:
|
|
346
|
+
every model supported in v0.3.6 keeps its prior treatment, and non-legacy or
|
|
347
|
+
unknown models remain neutral.
|
|
348
|
+
- The v0.4.0 research statements above are unchanged; only the runtime now reads
|
|
349
|
+
this registry as its single source of policy classification.
|
|
350
|
+
|
|
351
|
+
### Follow-up: v0.4.2 GPT-5.6-and-later boundary
|
|
352
|
+
|
|
353
|
+
- OpenAI's documented "GPT-5.6 and later" boundary was re-verified against the
|
|
354
|
+
first-party *Prompt caching* guide and *Using GPT-6* guide on **2026-09-27**.
|
|
355
|
+
GPT-6 (astra/sol/luna) is documented in the same cache regime with no
|
|
356
|
+
cache-control exception.
|
|
357
|
+
- v0.4.2 replaces the exact `gpt-5.6` string match with a version-boundary
|
|
358
|
+
predicate (`gpt-<major>[.<minor>] ≥ 5.6`), still gated on the OpenAI-ish
|
|
359
|
+
provider check. GPT-6 and future 5.6+/6+/7+ models therefore need no
|
|
360
|
+
exact-model registry entry, while `gpt-5.5` and earlier and malformed ids such
|
|
361
|
+
as `gpt-5.60` stay neutral.
|
|
362
|
+
- The only OpenAI cache controls injected remain `promptCacheKey` +
|
|
363
|
+
`promptCacheOptions{implicit,30m}`; no breakpoint or prewarm behavior was
|
|
364
|
+
added. DeepSeek, GLM, MiMo, and OpenRouter affinity behavior are unchanged.
|
|
365
|
+
|
|
366
|
+
### Follow-up: v0.4.3 DeepSeek V4-and-later passive coverage
|
|
367
|
+
|
|
368
|
+
- DeepSeek first-party docs were re-verified on **2026-09-27**. Confirmed:
|
|
369
|
+
context caching is provider-wide and passive — no `prompt_cache_key`, flag, or
|
|
370
|
+
breakpoint exists, and Anthropic-style `cache_control` is documented as
|
|
371
|
+
**ignored**. Only `user_id` is cache-relevant (KVCache isolation). Usage fields
|
|
372
|
+
are `prompt_cache_hit_tokens`, `prompt_cache_miss_tokens`, and
|
|
373
|
+
`prompt_tokens_details.cached_tokens`.
|
|
374
|
+
- Canonical request ids are `deepseek-flash` (= DeepSeek-V4.1-Flash) and
|
|
375
|
+
`deepseek-v4-pro`; `deepseek-v4-flash`/`deepseek-v4-flash-vision-exp` are
|
|
376
|
+
accepted retired aliases, and `deepseek-chat`/`deepseek-reasoner` are
|
|
377
|
+
discontinued. First-party docs publish **no** generational naming rule, and the
|
|
378
|
+
V4.1 codename id `deepseek-flash` carries no version token.
|
|
379
|
+
- v0.4.3 adds a passive `deepseek.v4-plus` family entry: canonical ids match by
|
|
380
|
+
exact id, `deepseek-v<major≥4>` version tokens match by predicate, and
|
|
381
|
+
pre-V4 / retired / unknown future `*deepseek*` ids fall through to the passive
|
|
382
|
+
creator baseline. No cache-control field, cache key, prompt rewrite, or
|
|
383
|
+
OpenRouter affinity is introduced for DeepSeek.
|
|
384
|
+
- Evidence caveat: first-party pages conflict on whether `deepseek-v4-pro` still
|
|
385
|
+
routes as a distinct model in late 2026; this does not affect the passive
|
|
386
|
+
policy, which carries no mutation either way.
|
package/package.json
CHANGED
|
@@ -56,6 +56,52 @@ export function isOpenAIish(s) {
|
|
|
56
56
|
return false
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
+
// The documented OpenAI cache-policy boundary is the generation phrase
|
|
60
|
+
// "GPT-5.6 and later" (docs/cache-policy-inventory.md §1; OpenAI *Prompt
|
|
61
|
+
// caching* guide, re-verified 2026-09-27). This matcher expresses that boundary
|
|
62
|
+
// by version rather than by an exact-model string, so future 5.6+/6+/7+ models
|
|
63
|
+
// need no registry entry:
|
|
64
|
+
// - major > 5 -> in family
|
|
65
|
+
// - major === 5 && minor >= 6 -> in family
|
|
66
|
+
// - everything else -> out
|
|
67
|
+
// The token must be followed by a non-digit/non-dot boundary, so malformed ids
|
|
68
|
+
// such as "gpt-5.60" and "gpt-5.6.1" do not match (same guard as pre-v0.4.2).
|
|
69
|
+
// OpenAI minor versions are single-digit, so a multi-digit minor is treated as
|
|
70
|
+
// malformed rather than as a higher version.
|
|
71
|
+
export function isGpt56OrLater(slug) {
|
|
72
|
+
const text = String(slug ?? "").toLowerCase()
|
|
73
|
+
const re = /gpt-(\d{1,3})(?:\.(\d))?(?![\d.])/g
|
|
74
|
+
let m
|
|
75
|
+
while ((m = re.exec(text)) !== null) {
|
|
76
|
+
const major = Number(m[1])
|
|
77
|
+
const minor = m[2] === undefined ? 0 : Number(m[2])
|
|
78
|
+
if (major > 5) return true
|
|
79
|
+
if (major === 5 && minor >= 6) return true
|
|
80
|
+
}
|
|
81
|
+
return false
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// DeepSeek V4-and-later coverage (docs/cache-policy-inventory.md §2; 2026-09-27
|
|
85
|
+
// first-party re-check). DeepSeek caching is provider-wide and passive: there is
|
|
86
|
+
// no cache key, flag, or breakpoint, and Anthropic-style `cache_control` is
|
|
87
|
+
// documented as ignored. This predicate therefore only classifies a version
|
|
88
|
+
// token (`deepseek-v<major>[.<minor>]` with major >= 4) into the passive family;
|
|
89
|
+
// it grants no mutation. Because the baseline is passive, matching an unknown
|
|
90
|
+
// future `deepseek-v5+` id is safe by construction.
|
|
91
|
+
//
|
|
92
|
+
// First-party docs do NOT publish a generational naming rule, and the current
|
|
93
|
+
// V4.1 codename id `deepseek-flash` carries no version token, so it is covered
|
|
94
|
+
// by explicit exact ids rather than by this predicate.
|
|
95
|
+
export function isDeepseekV4OrLater(slug) {
|
|
96
|
+
const text = String(slug ?? "").toLowerCase()
|
|
97
|
+
const re = /deepseek-v(\d+)(?:\.(\d+))?(?![\d.])/g
|
|
98
|
+
let m
|
|
99
|
+
while ((m = re.exec(text)) !== null) {
|
|
100
|
+
if (Number(m[1]) >= 4) return true
|
|
101
|
+
}
|
|
102
|
+
return false
|
|
103
|
+
}
|
|
104
|
+
|
|
59
105
|
// Candidate ids for exact/alias lookup. Includes the raw apiID/modelID, the
|
|
60
106
|
// lower-cased forms, and a single stripped transport/vendor prefix
|
|
61
107
|
// (e.g. "openai/gpt-5.6-luna" -> "gpt-5.6-luna", "xiaomi/mimo-v2.6-flash" ->
|
|
@@ -251,33 +297,32 @@ const rt = (policy, overrides = {}) => ({
|
|
|
251
297
|
|
|
252
298
|
export const POLICY_REGISTRY = [
|
|
253
299
|
{
|
|
254
|
-
|
|
300
|
+
// v0.4.2: one documented GPT-5.6-and-later family, matched by the version
|
|
301
|
+
// boundary rather than an exact model string. GPT-6 (astra/sol/luna) is
|
|
302
|
+
// documented in the same regime with no cache-control exception, so it
|
|
303
|
+
// inherits this baseline and overlay. Future 5.6+/6+/7+ models resolve here
|
|
304
|
+
// without a new registry entry.
|
|
305
|
+
id: "openai.gpt-5.6-plus",
|
|
255
306
|
creator: "openai",
|
|
256
307
|
family: "gpt-5.6",
|
|
257
308
|
kind: "family",
|
|
258
|
-
|
|
309
|
+
predicate: isGpt56OrLater,
|
|
259
310
|
requiresOpenAIish: true,
|
|
260
|
-
exactIds: [
|
|
311
|
+
exactIds: [
|
|
312
|
+
"gpt-5.6-sol",
|
|
313
|
+
"gpt-5.6-terra",
|
|
314
|
+
"gpt-5.6-luna",
|
|
315
|
+
"gpt-5.6-cyber",
|
|
316
|
+
"gpt-6-astra",
|
|
317
|
+
"gpt-6-sol",
|
|
318
|
+
"gpt-6-luna",
|
|
319
|
+
],
|
|
261
320
|
baseline: "openai.gpt56.cache",
|
|
262
321
|
overlays: ["gpt56.prompt-cache-options"],
|
|
263
322
|
legacy: true,
|
|
264
323
|
runtime: rt("gpt56", { gptCacheMetadata: true }),
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
{
|
|
268
|
-
id: "openai.gpt-6",
|
|
269
|
-
creator: "openai",
|
|
270
|
-
family: "gpt-6",
|
|
271
|
-
kind: "family",
|
|
272
|
-
pattern: /gpt-6(?![\d.])/i,
|
|
273
|
-
requiresOpenAIish: true,
|
|
274
|
-
exactIds: ["gpt-6-astra", "gpt-6-sol", "gpt-6-luna"],
|
|
275
|
-
baseline: "openai.gpt56.cache",
|
|
276
|
-
inheritsFrom: "gpt-5.6",
|
|
277
|
-
overlays: [],
|
|
278
|
-
legacy: false,
|
|
279
|
-
runtime: rt("neutral"),
|
|
280
|
-
note: "Documented inheritance of the GPT-5.6-and-later baseline. No CacheEngine overlay is registered for gpt-6 yet, so the runtime stays neutral.",
|
|
324
|
+
boundary: "GPT-5.6 and later",
|
|
325
|
+
note: "Documented boundary 'GPT-5.6 and later' (OpenAI Prompt caching guide, re-verified 2026-09-27) includes GPT-6 with no documented cache-control exception. No explicit breakpoint or prewarm behavior is registered.",
|
|
281
326
|
inventoryRef: "§1 OpenAI",
|
|
282
327
|
},
|
|
283
328
|
{
|
|
@@ -333,6 +378,28 @@ export const POLICY_REGISTRY = [
|
|
|
333
378
|
inventoryRef: "§4 Xiaomi MiMo",
|
|
334
379
|
},
|
|
335
380
|
{
|
|
381
|
+
// v0.4.3: formalize the documented "DeepSeek V4 and later" family. Coverage
|
|
382
|
+
// is passive (no mutation, no overlays). Version ids inherit via the
|
|
383
|
+
// predicate; the V4.1 codename id `deepseek-flash` has no version token and
|
|
384
|
+
// is matched by exact id. Pre-V4 and unknown future ids fall through to the
|
|
385
|
+
// passive creator baseline below, so nothing speculative is ever applied.
|
|
386
|
+
id: "deepseek.v4-plus",
|
|
387
|
+
creator: "deepseek",
|
|
388
|
+
family: "deepseek",
|
|
389
|
+
kind: "family",
|
|
390
|
+
predicate: isDeepseekV4OrLater,
|
|
391
|
+
exactIds: ["deepseek-flash", "deepseek-v4-pro"],
|
|
392
|
+
baseline: "deepseek.kv-cache",
|
|
393
|
+
overlays: [],
|
|
394
|
+
legacy: true,
|
|
395
|
+
runtime: rt("deepseek"),
|
|
396
|
+
boundary: "DeepSeek V4 and later",
|
|
397
|
+
note: "DeepSeek caching is provider-wide and passive (no cache key, flag, breakpoint, or cache-control; Anthropic `cache_control` is ignored). Verified 2026-09-27. Canonical current ids: `deepseek-flash` (V4.1-Flash) and `deepseek-v4-pro`; `deepseek-v4-flash`/`deepseek-v4-flash-vision-exp` are accepted retired aliases.",
|
|
398
|
+
inventoryRef: "§2 DeepSeek",
|
|
399
|
+
},
|
|
400
|
+
{
|
|
401
|
+
// Safe passive fallback for any other `*deepseek*` id (pre-V4, retired, or
|
|
402
|
+
// unknown future models) so DeepSeek always fails safe to observation only.
|
|
336
403
|
id: "deepseek.baseline",
|
|
337
404
|
creator: "deepseek",
|
|
338
405
|
family: "deepseek",
|
|
@@ -387,6 +454,12 @@ function overlaysFor(ids) {
|
|
|
387
454
|
return (ids ?? []).map((id) => OVERLAYS[id]).filter(Boolean)
|
|
388
455
|
}
|
|
389
456
|
|
|
457
|
+
// A family entry matches by regex `pattern` or by a pure `predicate(slug)`.
|
|
458
|
+
function familyMatches(entry, slug) {
|
|
459
|
+
if (typeof entry.predicate === "function") return entry.predicate(slug)
|
|
460
|
+
return entry.pattern ? entry.pattern.test(slug) : false
|
|
461
|
+
}
|
|
462
|
+
|
|
390
463
|
// Only legacy entries carry runtime capabilities. A non-legacy entry (gpt-6,
|
|
391
464
|
// Pro UltraSpeed) resolves for information but stays neutral at runtime.
|
|
392
465
|
function runtimeForEntry(entry) {
|
|
@@ -454,17 +527,20 @@ export function resolvePolicy(model) {
|
|
|
454
527
|
return resultFromFamily(alias.family, alias.creator, "exact", reason, id, alias.inventoryRef, alias.status ?? null, transport, alias.legacy)
|
|
455
528
|
}
|
|
456
529
|
|
|
457
|
-
// 2. Exact model ids (documented models).
|
|
530
|
+
// 2. Exact model ids (documented models). The entry's context gate still
|
|
531
|
+
// applies, so an exact OpenAI id on a non-OpenAI endpoint is never guessed.
|
|
458
532
|
for (const entry of POLICY_REGISTRY) {
|
|
459
533
|
if (!entry.exactIds || entry.exactIds.length === 0) continue
|
|
460
534
|
const hit = ids.find((id) => entry.exactIds.includes(id))
|
|
461
|
-
if (hit)
|
|
535
|
+
if (!hit) continue
|
|
536
|
+
if (entry.requiresOpenAIish && !isOpenAIish(s)) continue
|
|
537
|
+
return resultFromEntry(entry, "exact", `exact-id:${hit}`, hit, transport)
|
|
462
538
|
}
|
|
463
539
|
|
|
464
|
-
// 3. Model family / range
|
|
540
|
+
// 3. Model family / range matchers (regex pattern or version predicate).
|
|
465
541
|
for (const entry of POLICY_REGISTRY) {
|
|
466
|
-
if (entry.kind !== "family"
|
|
467
|
-
if (!entry
|
|
542
|
+
if (entry.kind !== "family") continue
|
|
543
|
+
if (!familyMatches(entry, s.slug)) continue
|
|
468
544
|
if (entry.requiresOpenAIish && !isOpenAIish(s)) continue
|
|
469
545
|
return resultFromEntry(entry, "family", `family-pattern:${entry.id}`, null, transport)
|
|
470
546
|
}
|
|
@@ -498,7 +574,7 @@ export function resolveLegacyFamily(model) {
|
|
|
498
574
|
for (const entry of POLICY_REGISTRY) {
|
|
499
575
|
if (!entry.legacy) continue
|
|
500
576
|
if (entry.kind === "family") {
|
|
501
|
-
if (!entry
|
|
577
|
+
if (!familyMatches(entry, s.slug)) continue
|
|
502
578
|
if (entry.requiresOpenAIish && !isOpenAIish(s)) continue
|
|
503
579
|
return entry.family
|
|
504
580
|
}
|
|
@@ -53,6 +53,8 @@ import {
|
|
|
53
53
|
MODEL_ALIASES,
|
|
54
54
|
OVERLAYS,
|
|
55
55
|
POLICY_REGISTRY,
|
|
56
|
+
isDeepseekV4OrLater,
|
|
57
|
+
isGpt56OrLater,
|
|
56
58
|
resolveLegacyFamily,
|
|
57
59
|
resolvePolicy,
|
|
58
60
|
resolveRuntimePolicy,
|
|
@@ -1359,26 +1361,27 @@ test("resolvePolicy: GPT-5.6 exact + inventory aliases resolve to the gpt-5.6 fa
|
|
|
1359
1361
|
assert.equal(orVariant.matchType, "exact")
|
|
1360
1362
|
})
|
|
1361
1363
|
|
|
1362
|
-
test("
|
|
1364
|
+
test("v0.4.2: GPT-6 resolves through the documented GPT-5.6-and-later boundary", () => {
|
|
1363
1365
|
const r = resolvePolicy(M("openrouter", "openai/gpt-6-luna"))
|
|
1364
1366
|
assert.equal(r.creator, "openai")
|
|
1365
|
-
assert.equal(r.family, "gpt-6")
|
|
1367
|
+
assert.equal(r.family, "gpt-5.6")
|
|
1366
1368
|
assert.equal(baseId(r), "openai.gpt56.cache")
|
|
1367
|
-
|
|
1368
|
-
assert.
|
|
1369
|
+
// GPT-6 is documented in the same cache regime, so it gets the same overlay.
|
|
1370
|
+
assert.deepEqual(overlayIds(r), ["gpt56.prompt-cache-options"])
|
|
1371
|
+
assert.equal(resolvePolicy(M("openai", "gpt-6-astra")).family, "gpt-5.6")
|
|
1369
1372
|
|
|
1370
|
-
//
|
|
1371
|
-
const
|
|
1372
|
-
assert.equal(
|
|
1373
|
-
assert.
|
|
1374
|
-
|
|
1375
|
-
assert.deepEqual(gpt6.overlays, [])
|
|
1373
|
+
// The boundary is version-based, not an exact-model list.
|
|
1374
|
+
const entry = POLICY_REGISTRY.find((e) => e.id === "openai.gpt-5.6-plus")
|
|
1375
|
+
assert.equal(entry.boundary, "GPT-5.6 and later")
|
|
1376
|
+
assert.equal(typeof entry.predicate, "function")
|
|
1377
|
+
assert.ok(entry.inventoryRef)
|
|
1376
1378
|
})
|
|
1377
1379
|
|
|
1378
|
-
test("
|
|
1379
|
-
|
|
1380
|
-
assert.equal(detectPolicy(M("openai", "gpt-6
|
|
1381
|
-
assert.equal(
|
|
1380
|
+
test("v0.4.2: the legacy detectPolicy wrapper follows the same boundary", () => {
|
|
1381
|
+
assert.equal(detectPolicy(M("openai", "gpt-6-astra")), POLICY_GPT56)
|
|
1382
|
+
assert.equal(detectPolicy(M("openai", "gpt-5.6")), POLICY_GPT56)
|
|
1383
|
+
assert.equal(detectPolicy(M("openai", "gpt-5.5")), POLICY_NEUTRAL)
|
|
1384
|
+
assert.equal(detectPolicy(M("openai", "gpt-5.60")), POLICY_NEUTRAL)
|
|
1382
1385
|
})
|
|
1383
1386
|
|
|
1384
1387
|
test("resolvePolicy: pre-5.6 GPT negative controls are neutral with no overlays", () => {
|
|
@@ -1390,11 +1393,12 @@ test("resolvePolicy: pre-5.6 GPT negative controls are neutral with no overlays"
|
|
|
1390
1393
|
}
|
|
1391
1394
|
})
|
|
1392
1395
|
|
|
1393
|
-
test("resolvePolicy: DeepSeek V4 / V4.1 resolve to the
|
|
1396
|
+
test("resolvePolicy: DeepSeek V4 / V4.1 resolve to the passive baseline (no overlay)", () => {
|
|
1394
1397
|
const v4 = resolvePolicy(M("deepseek", "deepseek-v4-pro"))
|
|
1395
1398
|
assert.equal(v4.creator, "deepseek")
|
|
1396
1399
|
assert.equal(v4.family, "deepseek")
|
|
1397
|
-
|
|
1400
|
+
// v0.4.3: deepseek-v4-pro is a documented canonical id, so it matches exactly.
|
|
1401
|
+
assert.equal(v4.matchType, "exact")
|
|
1398
1402
|
assert.equal(baseId(v4), "deepseek.kv-cache")
|
|
1399
1403
|
assert.deepEqual(overlayIds(v4), [])
|
|
1400
1404
|
|
|
@@ -1632,11 +1636,18 @@ async function runPolicyMigrationProbe() {
|
|
|
1632
1636
|
const CASES = [
|
|
1633
1637
|
{ name: "gpt-5.6", model: { providerID: "openai", id: "gpt-5.6", api: { id: "gpt-5.6", npm: "@ai-sdk/openai" } }, expect: { policy: "gpt56", env: false, gpt: true, header: false } },
|
|
1634
1638
|
{ name: "gpt-5.6-openrouter", model: { providerID: "openrouter", id: "openai/gpt-5.6-sol", api: { id: "openai/gpt-5.6-sol" } }, expect: { policy: "gpt56", env: false, gpt: true, header: false } },
|
|
1635
|
-
{ name: "gpt-6-astra", model: { providerID: "openai", id: "gpt-6-astra", api: { id: "gpt-6-astra", npm: "@ai-sdk/openai" } }, expect: { policy: "
|
|
1639
|
+
{ name: "gpt-6-astra", model: { providerID: "openai", id: "gpt-6-astra", api: { id: "gpt-6-astra", npm: "@ai-sdk/openai" } }, expect: { policy: "gpt56", env: false, gpt: true, header: false } },
|
|
1640
|
+
{ name: "gpt-6-openrouter", model: { providerID: "openrouter", id: "openai/gpt-6-luna", api: { id: "openai/gpt-6-luna" } }, expect: { policy: "gpt56", env: false, gpt: true, header: false } },
|
|
1641
|
+
{ name: "gpt-6-openai-compatible", model: { providerID: "openai-compatible", id: "gpt-6-astra", api: { id: "gpt-6-astra" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1642
|
+
{ name: "gpt-5.60-malformed", model: { providerID: "openai", id: "gpt-5.60", api: { id: "gpt-5.60", npm: "@ai-sdk/openai" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1636
1643
|
{ name: "gpt-5.5", model: { providerID: "openai", id: "gpt-5.5", api: { id: "gpt-5.5", npm: "@ai-sdk/openai" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1637
1644
|
{ name: "gpt-daybreak-alias", model: { providerID: "openai", id: "gpt-daybreak-blue-latest", api: { id: "gpt-daybreak-blue-latest", npm: "@ai-sdk/openai" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1638
1645
|
{ name: "deepseek-v4-pro", model: { providerID: "deepseek", id: "deepseek-v4-pro", api: { id: "deepseek-v4-pro" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1639
1646
|
{ name: "deepseek-flash", model: { providerID: "deepseek", id: "deepseek-flash", api: { id: "deepseek-flash" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1647
|
+
{ name: "deepseek-v5-future", model: { providerID: "deepseek", id: "deepseek-v5", api: { id: "deepseek-v5" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1648
|
+
{ name: "deepseek-v3-pre", model: { providerID: "deepseek", id: "deepseek-v3", api: { id: "deepseek-v3" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1649
|
+
{ name: "deepseek-openrouter", model: { providerID: "openrouter", id: "deepseek/deepseek-v4-pro", api: { id: "deepseek/deepseek-v4-pro" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1650
|
+
{ name: "deepseek-gateway", model: { providerID: "acme-gateway", id: "my-deepseek-mirror", api: { id: "my-deepseek-mirror" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1640
1651
|
{ name: "glm-5.3-direct", model: { providerID: "zai", id: "glm-5.3", api: { id: "glm-5.3" } }, expect: { policy: "glm53", env: true, gpt: false, header: false } },
|
|
1641
1652
|
{ name: "glm-5.3-openrouter", model: { providerID: "openrouter", id: "z-ai/glm-5.3-flash", api: { id: "z-ai/glm-5.3-flash" } }, expect: { policy: "glm53", env: true, gpt: false, header: true } },
|
|
1642
1653
|
{ name: "glm-5.2", model: { providerID: "zai", id: "glm-5.2", api: { id: "glm-5.2" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
@@ -1717,9 +1728,10 @@ test("v0.4.1: GPT-5.6 keeps promptCacheOptions implicit/30m through the resolver
|
|
|
1717
1728
|
test("v0.4.1: future-looking and unknown models gain no mutation", async () => {
|
|
1718
1729
|
const { results } = await policyMigrationResults()
|
|
1719
1730
|
const noMutation = [
|
|
1720
|
-
"gpt-6-astra",
|
|
1721
1731
|
"gpt-daybreak-alias",
|
|
1722
1732
|
"gpt-5.5",
|
|
1733
|
+
"gpt-6-openai-compatible",
|
|
1734
|
+
"gpt-5.60-malformed",
|
|
1723
1735
|
"glm-5.2",
|
|
1724
1736
|
"mimo-v2.6-pro-ultraspeed",
|
|
1725
1737
|
"mimo-v2.5",
|
|
@@ -1741,3 +1753,165 @@ test("v0.4.1: non-OpenRouter models never receive the OpenRouter header", async
|
|
|
1741
1753
|
assert.equal(r.affinityHeaderAttached, false, `${name}: no affinity header off OpenRouter`)
|
|
1742
1754
|
}
|
|
1743
1755
|
})
|
|
1756
|
+
|
|
1757
|
+
// ===========================================================================
|
|
1758
|
+
// v0.4.2 GPT-5.6-and-later boundary
|
|
1759
|
+
//
|
|
1760
|
+
// Source: OpenAI *Prompt caching* guide, "GPT-5.6 and later" generation
|
|
1761
|
+
// boundary, re-verified 2026-09-27 (docs/cache-policy-inventory.md §1). GPT-6
|
|
1762
|
+
// is documented in the same regime with no cache-control exception.
|
|
1763
|
+
// ===========================================================================
|
|
1764
|
+
|
|
1765
|
+
test("v0.4.2: isGpt56OrLater matches the documented boundary by version", () => {
|
|
1766
|
+
const inFamily = [
|
|
1767
|
+
"gpt-5.6",
|
|
1768
|
+
"gpt-5.6-luna",
|
|
1769
|
+
"openai/gpt-5.6-sol",
|
|
1770
|
+
"gpt-6",
|
|
1771
|
+
"gpt-6-astra",
|
|
1772
|
+
"gpt-6-sol",
|
|
1773
|
+
"gpt-6-luna",
|
|
1774
|
+
"gpt-5.7",
|
|
1775
|
+
"gpt-7",
|
|
1776
|
+
"gpt-6.1",
|
|
1777
|
+
]
|
|
1778
|
+
for (const id of inFamily) assert.equal(isGpt56OrLater(id), true, `${id} is in family`)
|
|
1779
|
+
const outOfFamily = [
|
|
1780
|
+
"gpt-5.5",
|
|
1781
|
+
"gpt-5.4",
|
|
1782
|
+
"gpt-5.2",
|
|
1783
|
+
"gpt-5.1",
|
|
1784
|
+
"gpt-5",
|
|
1785
|
+
"gpt-4.1",
|
|
1786
|
+
"gpt-4o",
|
|
1787
|
+
"gpt-5.60",
|
|
1788
|
+
"gpt-5.6.1",
|
|
1789
|
+
"gpt-4",
|
|
1790
|
+
"claude-sonnet-4-5",
|
|
1791
|
+
]
|
|
1792
|
+
for (const id of outOfFamily) assert.equal(isGpt56OrLater(id), false, `${id} is out of family`)
|
|
1793
|
+
})
|
|
1794
|
+
|
|
1795
|
+
test("v0.4.2: GPT boundary respects OpenAI/provider gating", () => {
|
|
1796
|
+
assert.equal(resolveRuntimePolicy(M("openai", "gpt-5.6")).gptCacheMetadata, true)
|
|
1797
|
+
assert.equal(resolveRuntimePolicy(M("openai", "gpt-6-astra")).gptCacheMetadata, true)
|
|
1798
|
+
assert.equal(resolveRuntimePolicy(M("azure", "gpt-6-sol")).gptCacheMetadata, true)
|
|
1799
|
+
assert.equal(resolveRuntimePolicy(M("openrouter", "openai/gpt-6-luna")).gptCacheMetadata, true)
|
|
1800
|
+
assert.equal(resolveRuntimePolicy(M("openai-compatible", "gpt-6-astra")).gptCacheMetadata, false)
|
|
1801
|
+
assert.equal(resolveRuntimePolicy(M("llama.cpp", "gpt-5.6")).gptCacheMetadata, false)
|
|
1802
|
+
assert.equal(resolveRuntimePolicy(M("openai", "gpt-5.5")).gptCacheMetadata, false)
|
|
1803
|
+
assert.equal(resolveRuntimePolicy(M("openai", "gpt-5.60")).gptCacheMetadata, false)
|
|
1804
|
+
})
|
|
1805
|
+
|
|
1806
|
+
test("v0.4.2: covered later GPT requests receive the same documented baseline at runtime", async () => {
|
|
1807
|
+
const { results } = await policyMigrationResults()
|
|
1808
|
+
const gpt6 = results.find((r) => r.name === "gpt-6-astra")
|
|
1809
|
+
assert.equal(gpt6.runtimePolicy, "gpt56")
|
|
1810
|
+
assert.equal(gpt6.gptOptionInjected, true)
|
|
1811
|
+
assert.deepEqual(gpt6.gptOptions, { mode: "implicit", ttl: "30m" })
|
|
1812
|
+
// Same baseline as GPT-5.6, no new mechanism.
|
|
1813
|
+
const gpt56 = results.find((r) => r.name === "gpt-5.6")
|
|
1814
|
+
assert.deepEqual(gpt6.gptOptions, gpt56.gptOptions)
|
|
1815
|
+
// No prompt transformation or affinity was introduced for gpt-6.
|
|
1816
|
+
assert.equal(gpt6.systemRelocated, false)
|
|
1817
|
+
assert.equal(gpt6.affinityHeaderAttached, false)
|
|
1818
|
+
})
|
|
1819
|
+
|
|
1820
|
+
test("v0.4.2: pre-5.6 and out-of-family GPT ids get no GPT options", async () => {
|
|
1821
|
+
const { results } = await policyMigrationResults()
|
|
1822
|
+
for (const name of ["gpt-5.5", "gpt-5.60-malformed", "gpt-6-openai-compatible"]) {
|
|
1823
|
+
const r = results.find((x) => x.name === name)
|
|
1824
|
+
assert.equal(r.gptOptionInjected, false, `${name}: no GPT options`)
|
|
1825
|
+
assert.equal(r.runtimePolicy, "neutral", `${name}: neutral runtime`)
|
|
1826
|
+
}
|
|
1827
|
+
})
|
|
1828
|
+
|
|
1829
|
+
// ===========================================================================
|
|
1830
|
+
// v0.4.3 DeepSeek V4-and-later passive coverage
|
|
1831
|
+
//
|
|
1832
|
+
// Source: DeepSeek first-party docs re-verified 2026-09-27
|
|
1833
|
+
// (docs/cache-policy-inventory.md §2). Caching is provider-wide and passive:
|
|
1834
|
+
// no cache key/flag/breakpoint; Anthropic `cache_control` is ignored.
|
|
1835
|
+
// ===========================================================================
|
|
1836
|
+
|
|
1837
|
+
test("v0.4.3: isDeepseekV4OrLater matches V4+ version tokens only", () => {
|
|
1838
|
+
const inFamily = ["deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4.1", "deepseek-v4.5-pro", "deepseek-v5", "deepseek-v10", "deepseek/deepseek-v4-pro"]
|
|
1839
|
+
for (const id of inFamily) assert.equal(isDeepseekV4OrLater(id), true, `${id} is V4+`)
|
|
1840
|
+
const outOfFamily = ["deepseek-v3", "deepseek-v2", "deepseek-chat", "deepseek-reasoner", "deepseek-coder", "deepseek-flash", "my-deepseek-mirror", ""]
|
|
1841
|
+
for (const id of outOfFamily) assert.equal(isDeepseekV4OrLater(id), false, `${id} is not a V4+ version token`)
|
|
1842
|
+
})
|
|
1843
|
+
|
|
1844
|
+
test("v0.4.3: canonical V4 ids and aliases resolve to the passive DeepSeek family", () => {
|
|
1845
|
+
for (const id of ["deepseek-flash", "deepseek-v4-pro"]) {
|
|
1846
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1847
|
+
assert.equal(r.creator, "deepseek")
|
|
1848
|
+
assert.equal(r.family, "deepseek")
|
|
1849
|
+
assert.equal(baseId(r), "deepseek.kv-cache")
|
|
1850
|
+
assert.deepEqual(overlayIds(r), [])
|
|
1851
|
+
const rt = resolveRuntimePolicy(M("deepseek", id))
|
|
1852
|
+
assert.equal(rt.policy, "deepseek")
|
|
1853
|
+
assert.equal(rt.gptCacheMetadata, false)
|
|
1854
|
+
assert.equal(rt.envRelocation, null)
|
|
1855
|
+
assert.equal(rt.cacheRatio, null)
|
|
1856
|
+
assert.equal(rt.openRouterAffinity, false)
|
|
1857
|
+
}
|
|
1858
|
+
// v0.4.0 alias handling is preserved.
|
|
1859
|
+
const alias = resolvePolicy(M("deepseek", "deepseek-v4-flash"))
|
|
1860
|
+
assert.equal(alias.family, "deepseek")
|
|
1861
|
+
assert.ok(alias.matchReason.startsWith("alias:deepseek-v4-flash"))
|
|
1862
|
+
assert.equal(resolveRuntimePolicy(M("deepseek", "deepseek-v4-flash")).policy, "deepseek")
|
|
1863
|
+
assert.equal(resolvePolicy(M("deepseek", "deepseek-chat")).family, "deepseek")
|
|
1864
|
+
})
|
|
1865
|
+
|
|
1866
|
+
test("v0.4.3: later and unknown DeepSeek ids stay passive (no speculative mutation)", () => {
|
|
1867
|
+
for (const id of ["deepseek-v5", "deepseek-v6.2", "deepseek-v4.1-pro", "deepseek-nova"]) {
|
|
1868
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1869
|
+
assert.equal(r.creator, "deepseek")
|
|
1870
|
+
assert.equal(r.family, "deepseek")
|
|
1871
|
+
assert.deepEqual(overlayIds(r), [], `${id}: no overlay`)
|
|
1872
|
+
const rt = resolveRuntimePolicy(M("deepseek", id))
|
|
1873
|
+
assert.equal(rt.policy, "deepseek", `${id}: passive policy`)
|
|
1874
|
+
assert.equal(rt.gptCacheMetadata, false)
|
|
1875
|
+
assert.equal(rt.envRelocation, null)
|
|
1876
|
+
assert.equal(rt.cacheRatio, null)
|
|
1877
|
+
assert.equal(rt.openRouterAffinity, false)
|
|
1878
|
+
}
|
|
1879
|
+
})
|
|
1880
|
+
|
|
1881
|
+
test("v0.4.3: pre-V4 DeepSeek ids remain passive and outside the V4+ family entry", () => {
|
|
1882
|
+
for (const id of ["deepseek-v3", "deepseek-v2", "deepseek-coder"]) {
|
|
1883
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1884
|
+
assert.equal(r.family, "deepseek")
|
|
1885
|
+
// handled by the safe creator fallback, not the V4+ family predicate
|
|
1886
|
+
assert.equal(r.matchType, "creator", `${id}: creator fallback`)
|
|
1887
|
+
assert.equal(resolveRuntimePolicy(M("deepseek", id)).policy, "deepseek")
|
|
1888
|
+
}
|
|
1889
|
+
})
|
|
1890
|
+
|
|
1891
|
+
test("v0.4.3: DeepSeek keeps the generic read/write ratio and no family-specific fields", () => {
|
|
1892
|
+
const rt = resolveRuntimePolicy(M("deepseek", "deepseek-v4-pro"))
|
|
1893
|
+
assert.equal(rt.cacheRatio, null) // generic read/(read+write) accounting is used
|
|
1894
|
+
assert.equal(hitRatePct(30, 70), 30)
|
|
1895
|
+
})
|
|
1896
|
+
|
|
1897
|
+
test("v0.4.3: DeepSeek never receives OpenRouter affinity or GPT/GLM/MiMo fields", async () => {
|
|
1898
|
+
const { results } = await policyMigrationResults()
|
|
1899
|
+
const deepseekCases = [
|
|
1900
|
+
"deepseek-v4-pro",
|
|
1901
|
+
"deepseek-flash",
|
|
1902
|
+
"deepseek-v5-future",
|
|
1903
|
+
"deepseek-v3-pre",
|
|
1904
|
+
"deepseek-openrouter",
|
|
1905
|
+
"deepseek-gateway",
|
|
1906
|
+
]
|
|
1907
|
+
for (const name of deepseekCases) {
|
|
1908
|
+
const r = results.find((x) => x.name === name)
|
|
1909
|
+
assert.ok(r, `${name} present`)
|
|
1910
|
+
assert.equal(r.runtimePolicy, "deepseek", `${name}: passive policy`)
|
|
1911
|
+
assert.equal(r.detectPolicy, "deepseek", `${name}: detectPolicy agrees`)
|
|
1912
|
+
assert.equal(r.gptOptionInjected, false, `${name}: no GPT options leak`)
|
|
1913
|
+
assert.equal(r.systemRelocated, false, `${name}: no GLM/MiMo env relocation`)
|
|
1914
|
+
assert.equal(r.affinityHeaderAttached, false, `${name}: no OpenRouter affinity`)
|
|
1915
|
+
assert.equal(r.existingHeadersPreserved, true, `${name}: headers preserved`)
|
|
1916
|
+
}
|
|
1917
|
+
})
|