opencode-cache-engine 0.4.2 → 0.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/docs/cache-policy-inventory.md +45 -2
- package/package.json +1 -1
- package/src/cache-policy-core.mjs +85 -0
- package/test/cache-engine.test.mjs +193 -2
package/README.md
CHANGED
|
@@ -31,7 +31,7 @@ The plugin currently has four cache-policy families:
|
|
|
31
31
|
* **GPT-5.6 and later** — documented cache-key/options metadata, with prompt text
|
|
32
32
|
unchanged. GPT-6 and future 5.6+/6+/7+ versions resolve through the same
|
|
33
33
|
documented boundary.
|
|
34
|
-
* **GLM-5.3** — narrow, content-preserving `<env>` relocation
|
|
34
|
+
* **GLM-5.3 and later** — GLM implicit-cache baseline and diagnostics; GLM-5.3 additionally uses a narrow, content-preserving `<env>` relocation overlay.
|
|
35
35
|
* **MiMo-V2.6** — narrow, content-preserving `<env>` relocation and diagnostics.
|
|
36
36
|
|
|
37
37
|
Family classification is not hard-coded in the runtime. A pure policy registry
|
|
@@ -42,7 +42,7 @@ the single runtime source of policy classification. The first-party research
|
|
|
42
42
|
behind each registry entry is recorded in
|
|
43
43
|
[docs/cache-policy-inventory.md](docs/cache-policy-inventory.md).
|
|
44
44
|
|
|
45
|
-
For
|
|
45
|
+
For MiMo-V2.6 and the GLM-5.3-and-later family, CacheEngine adds its deterministic
|
|
46
46
|
`x-session-id` request header only when OpenCode identifies the actual provider
|
|
47
47
|
as `openrouter`. It does not add that OpenRouter-specific header for
|
|
48
48
|
non-OpenRouter providers; direct provider endpoints retain their provider-native
|
|
@@ -74,7 +74,7 @@ The plugin deliberately avoids pretending that a local hash is proof of a provid
|
|
|
74
74
|
|
|
75
75
|
# Provider behavior
|
|
76
76
|
|
|
77
|
-
## DeepSeek V4
|
|
77
|
+
## DeepSeek V4 and later
|
|
78
78
|
|
|
79
79
|
### Policy: passive
|
|
80
80
|
|
|
@@ -93,6 +93,13 @@ The DeepSeek branch exists primarily to preserve a stable harness while providin
|
|
|
93
93
|
|
|
94
94
|
This is intentional. The implementation describes DeepSeek as a passive policy whose purpose is to preserve the existing high-cache-rate behavior rather than introduce new request mutations.
|
|
95
95
|
|
|
96
|
+
Since v0.4.3 this is formalized as the documented **"DeepSeek V4 and later"**
|
|
97
|
+
family. Canonical ids (`deepseek-flash`, `deepseek-v4-pro`) and the accepted
|
|
98
|
+
`deepseek-v4-flash` aliases resolve to this passive baseline, and pre-V4 or
|
|
99
|
+
unknown future `*deepseek*` ids fall back to the same passive baseline. No cache
|
|
100
|
+
key, cache-control field, prompt rewrite, or OpenRouter affinity is ever added
|
|
101
|
+
for DeepSeek.
|
|
102
|
+
|
|
96
103
|
The plugin still observes:
|
|
97
104
|
|
|
98
105
|
* system-prompt shape
|
|
@@ -198,6 +205,12 @@ GLM-5.3 and MiMo-V2.6 use the only prompt-text transformation in the current
|
|
|
198
205
|
plugin: a narrow, content-preserving relocation of the identifiable `<env>`
|
|
199
206
|
block for the eligible model family.
|
|
200
207
|
|
|
208
|
+
Since v0.4.4 the GLM **family baseline** and the GLM-5.3 **overlay** are
|
|
209
|
+
separate. A resolved GLM-5.3-and-later model inherits the implicit-cache baseline
|
|
210
|
+
and the non-mutating GLM diagnostics/transport, but the `<env>` relocation below
|
|
211
|
+
is a GLM-5.3-specific overlay and is **not** inherited by a newer GLM merely
|
|
212
|
+
because its version number is higher.
|
|
213
|
+
|
|
201
214
|
The plugin identifies OpenCode's volatile `<env>` section and moves it to the **tail of the system prompt**.
|
|
202
215
|
|
|
203
216
|
Conceptually:
|
|
@@ -419,9 +432,9 @@ and are reported diagnostically; the message content is left untouched.
|
|
|
419
432
|
|
|
420
433
|
| Policy family | Detection | Prompt text changed? | Cache metadata changed? | OpenRouter affinity header | Primary cache signal |
|
|
421
434
|
| ------------- | --------- | ------------------- | ----------------------- | -------------------------- | -------------------- |
|
|
422
|
-
| DeepSeek | `deepseek` | No | No | None | provider `cache.read` / `cache.write` |
|
|
435
|
+
| DeepSeek | `deepseek` (V4-and-later family + passive fallback) | No | No | None | provider `cache.read` / `cache.write` |
|
|
423
436
|
| GPT-5.6 and later | version boundary `gpt-<major>[.<minor>] ≥ 5.6` on OpenAI-ish endpoints (includes GPT-6) | No | Yes: `prompt_cache_key` + options | None | provider cache tokens |
|
|
424
|
-
| GLM-5.3 | `glm-5.3
|
|
437
|
+
| GLM-5.3 and later | `glm-5.3+` | Yes, narrowly (`<env>` tail) on GLM-5.3 only | No provider cache key | `x-session-id` on OpenRouter only | provider cache tokens (GLM ratio) |
|
|
425
438
|
| MiMo-V2.6 | Flash / Pro only | Yes, narrowly (`<env>` tail) | No: implicit caching only | `x-session-id` on OpenRouter only | `cached_tokens / prompt_tokens` |
|
|
426
439
|
|
|
427
440
|
`x-session-id` is an HTTP affinity header, not a provider cache key or
|
|
@@ -1385,7 +1398,7 @@ not matched.
|
|
|
1385
1398
|
|
|
1386
1399
|
## OpenRouter affinity header is not added
|
|
1387
1400
|
|
|
1388
|
-
CacheEngine adds its `x-session-id` only for a detected MiMo-V2.6 or GLM-5.3
|
|
1401
|
+
CacheEngine adds its `x-session-id` only for a detected MiMo-V2.6 or GLM-5.3-and-later
|
|
1389
1402
|
request when the actual OpenCode `providerID` is exactly `openrouter`. A direct
|
|
1390
1403
|
provider route or missing provider identity is bypassed. If a case-insensitive
|
|
1391
1404
|
`x-session-id` is already present in model or plugin headers, it is preserved
|
|
@@ -285,10 +285,10 @@ made here); **hold** = do not inherit without first-party evidence.
|
|
|
285
285
|
| OpenAI | pre-5.6 negative controls: `gpt-5.5`, `gpt-5.4`, `gpt-5.2`, `gpt-5.1`, `gpt-5`, `gpt-4.1`, `gpt-4o` | Implicit only; different min-length class; `in_memory`/`24h` retention; `prompt_cache_key` for routing | neutral | **hold** — do not inherit 5.6 policy | n/a | High | OpenAI *Prompt caching*; *Pricing* | 2026-09-26 |
|
|
286
286
|
| DeepSeek | V4: `deepseek-v4-pro`, legacy `deepseek-v4-flash` | Provider-wide automatic disk cache; implicit; prefix-unit matching; hit/miss token fields | Passive (no mutation) | keep | none | High | DeepSeek *Context Caching*; *Models & Pricing* | 2026-09-26 |
|
|
287
287
|
| DeepSeek | V4.1 / current V4-family: `deepseek-flash` (MODEL VERSION "DeepSeek-V4.1-Flash") | Same provider-wide automatic policy; cache-hit pricing listed for both current models | Passive | keep (creator/family baseline) | none documented | High | DeepSeek *Models & Pricing*; *news260910* | 2026-09-26 |
|
|
288
|
-
| DeepSeek | future-looking V4+ identifiers: `deepseek-v4.1`, `deepseek-v4`, `deepseek-v5` | Not documented as request ids (`deepseek-v4.1`/`deepseek-v4` invalid or version-string only) | Passive via
|
|
288
|
+
| DeepSeek | future-looking V4+ identifiers: `deepseek-v4.1`, `deepseek-v4`, `deepseek-v5` | Not documented as request ids (`deepseek-v4.1`/`deepseek-v4` invalid or version-string only) | Passive via the V4-and-later family predicate or the safe creator fallback (v0.4.3); no mutation | keep passive; treat as unknown-friendly | none | Medium (detection) / Low (future ids) | DeepSeek *Models & Pricing*; *Chat Completions API* | 2026-09-27 |
|
|
289
289
|
| DeepSeek | pre-V4 negative controls: `deepseek-chat`, `deepseek-reasoner` | Retired names (retired 2026-07-24); no separate V4+ cache policy claimed | Passive | hold | n/a | High | DeepSeek *Change Log*; *news260424* | 2026-09-26 |
|
|
290
290
|
| Z.AI | GLM 5.3: `glm-5.3`, `glm-5.3-flash`, `glm-5.3-flashx` | Implicit automatic caching; `cached_tokens`; stable-prompt-first guidance; no documented min/TTL/key | GLM policy: `<env>` relocation; OpenRouter `x-session-id` | keep (env relocation is an exact overlay, not a Z.AI control) | env relocation is the overlay; keep scoped to GLM-5.3 | Medium | Z.AI *Context Caching*; *Chat Completion*; *Pricing* | 2026-09-26 |
|
|
291
|
-
| Z.AI | current later GLM generations (documented): none newer than 5.3; newest below is `glm-5.2`/`glm-5.1`/`glm-5`/`glm-4.7` | Same implicit mechanism documented service-wide; cached-input price per model |
|
|
291
|
+
| Z.AI | current later GLM generations (documented): none newer than 5.3; newest below is `glm-5.2`/`glm-5.1`/`glm-5`/`glm-4.7` | Same implicit mechanism documented service-wide; cached-input price per model | GLM-5.3 family baseline via the 5.3-and-later boundary; **no** `<env>` overlay (v0.4.4) | **baseline only** — overlay stays 5.3-explicit | none documented | High (no later gens documented) | Z.AI *New Released*; *Pricing* | 2026-09-27 |
|
|
292
292
|
| Z.AI | 5.2 and earlier negative controls: `glm-5.2`, `glm-5.1`, `glm-5`, `glm-4.7`, `glm-4.6`, `glm-4.5`, `glm-4-32b-*` | Cacheable (except `glm-4-32b-0414-128k`), different cached-input pricing; no cache-semantics difference documented | neutral | hold | n/a | High | Z.AI *Pricing*; *Chat Completion* (enum) | 2026-09-26 |
|
|
293
293
|
| Xiaomi | MiMo V2.6 Flash: `mimo-v2.6-flash` (OR `xiaomi/mimo-v2.6-flash`) | Provider-managed implicit caching; `cached_tokens`; no documented min/TTL/key/prefix rules | MiMo policy: `<env>` relocation; provider-change telemetry; OpenRouter `x-session-id` | keep scoped as exact-model overlay | env relocation unsupported by docs → treat as overlay | Medium | MiMo *Models*; *Pricing*; *openai-api* | 2026-09-26 |
|
|
294
294
|
| Xiaomi | MiMo V2.6 Pro: `mimo-v2.6-pro` (OR `xiaomi/mimo-v2.6-pro`) | Same documented per-model implicit caching; per-model pricing | MiMo policy (same as Flash) | keep | none documented | Medium | MiMo *Models*; *Pricing* | 2026-09-26 |
|
|
@@ -362,3 +362,46 @@ a newer model inherits an older policy. They must not be resolved by guessing.
|
|
|
362
362
|
- The only OpenAI cache controls injected remain `promptCacheKey` +
|
|
363
363
|
`promptCacheOptions{implicit,30m}`; no breakpoint or prewarm behavior was
|
|
364
364
|
added. DeepSeek, GLM, MiMo, and OpenRouter affinity behavior are unchanged.
|
|
365
|
+
|
|
366
|
+
### Follow-up: v0.4.3 DeepSeek V4-and-later passive coverage
|
|
367
|
+
|
|
368
|
+
- DeepSeek first-party docs were re-verified on **2026-09-27**. Confirmed:
|
|
369
|
+
context caching is provider-wide and passive — no `prompt_cache_key`, flag, or
|
|
370
|
+
breakpoint exists, and Anthropic-style `cache_control` is documented as
|
|
371
|
+
**ignored**. Only `user_id` is cache-relevant (KVCache isolation). Usage fields
|
|
372
|
+
are `prompt_cache_hit_tokens`, `prompt_cache_miss_tokens`, and
|
|
373
|
+
`prompt_tokens_details.cached_tokens`.
|
|
374
|
+
- Canonical request ids are `deepseek-flash` (= DeepSeek-V4.1-Flash) and
|
|
375
|
+
`deepseek-v4-pro`; `deepseek-v4-flash`/`deepseek-v4-flash-vision-exp` are
|
|
376
|
+
accepted retired aliases, and `deepseek-chat`/`deepseek-reasoner` are
|
|
377
|
+
discontinued. First-party docs publish **no** generational naming rule, and the
|
|
378
|
+
V4.1 codename id `deepseek-flash` carries no version token.
|
|
379
|
+
- v0.4.3 adds a passive `deepseek.v4-plus` family entry: canonical ids match by
|
|
380
|
+
exact id, `deepseek-v<major≥4>` version tokens match by predicate, and
|
|
381
|
+
pre-V4 / retired / unknown future `*deepseek*` ids fall through to the passive
|
|
382
|
+
creator baseline. No cache-control field, cache key, prompt rewrite, or
|
|
383
|
+
OpenRouter affinity is introduced for DeepSeek.
|
|
384
|
+
- Evidence caveat: first-party pages conflict on whether `deepseek-v4-pro` still
|
|
385
|
+
routes as a distinct model in late 2026; this does not affect the passive
|
|
386
|
+
policy, which carries no mutation either way.
|
|
387
|
+
|
|
388
|
+
### Follow-up: v0.4.4 GLM-5.3-and-later baseline vs GLM-5.3 overlay
|
|
389
|
+
|
|
390
|
+
- Z.AI docs were re-verified on **2026-09-27**. GLM-5.3 is the newest documented
|
|
391
|
+
text generation (no `glm-5.4`/`glm-6`); GLM-5.3 is explicitly "the same base
|
|
392
|
+
model as GLM-5.2" with post-training differences. Caching is implicit with no
|
|
393
|
+
`cache_control`/`prompt_cache_key`/breakpoint/TTL field; `cached_tokens` is the
|
|
394
|
+
reported field. Z.AI publishes **no** generational-inheritance rule.
|
|
395
|
+
- The `<env>` relocation has **no first-party basis** — it is a CacheEngine
|
|
396
|
+
implementation overlay. v0.4.4 therefore separates the concepts:
|
|
397
|
+
- **Family baseline** (`zai.glm-5.3-plus`, predicate `glm >= 5.3`): implicit
|
|
398
|
+
caching plus the non-mutating GLM diagnostics/transport (thinking-integrity
|
|
399
|
+
telemetry, GLM cache ratio, provider-change telemetry, OpenRouter
|
|
400
|
+
`x-session-id`). No prompt rewrite.
|
|
401
|
+
- **GLM-5.3 overlay**: the `<env>` relocation stays registered only on the
|
|
402
|
+
GLM-5.3 family entry, so a later GLM does **not** inherit it.
|
|
403
|
+
- GLM-5.2 and earlier remain neutral. No new GLM cache-control field is added,
|
|
404
|
+
and GLM-5.3 behavior is unchanged.
|
|
405
|
+
- Evidence caveat: because no later GLM generation is documented, the
|
|
406
|
+
"GLM-5.3 and later" boundary is a CacheEngine inference about a passive
|
|
407
|
+
baseline, not a Z.AI contract; it is safe because it introduces no mutation.
|
package/package.json
CHANGED
|
@@ -81,6 +81,46 @@ export function isGpt56OrLater(slug) {
|
|
|
81
81
|
return false
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
// DeepSeek V4-and-later coverage (docs/cache-policy-inventory.md §2; 2026-09-27
|
|
85
|
+
// first-party re-check). DeepSeek caching is provider-wide and passive: there is
|
|
86
|
+
// no cache key, flag, or breakpoint, and Anthropic-style `cache_control` is
|
|
87
|
+
// documented as ignored. This predicate therefore only classifies a version
|
|
88
|
+
// token (`deepseek-v<major>[.<minor>]` with major >= 4) into the passive family;
|
|
89
|
+
// it grants no mutation. Because the baseline is passive, matching an unknown
|
|
90
|
+
// future `deepseek-v5+` id is safe by construction.
|
|
91
|
+
//
|
|
92
|
+
// First-party docs do NOT publish a generational naming rule, and the current
|
|
93
|
+
// V4.1 codename id `deepseek-flash` carries no version token, so it is covered
|
|
94
|
+
// by explicit exact ids rather than by this predicate.
|
|
95
|
+
export function isDeepseekV4OrLater(slug) {
|
|
96
|
+
const text = String(slug ?? "").toLowerCase()
|
|
97
|
+
const re = /deepseek-v(\d+)(?:\.(\d+))?(?![\d.])/g
|
|
98
|
+
let m
|
|
99
|
+
while ((m = re.exec(text)) !== null) {
|
|
100
|
+
if (Number(m[1]) >= 4) return true
|
|
101
|
+
}
|
|
102
|
+
return false
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// GLM-5.3-and-later family baseline (docs/cache-policy-inventory.md §3; Z.AI
|
|
106
|
+
// docs re-verified 2026-09-27). Z.AI publishes no generational-inheritance rule
|
|
107
|
+
// and no explicit cache control, so this is a CacheEngine inference about a
|
|
108
|
+
// passive/implicit baseline; it is safe because it grants no mutation. The
|
|
109
|
+
// GLM-5.3 `<env>` relocation is a separate, explicit overlay and is NOT granted
|
|
110
|
+
// by this predicate. GLM-5.2 and earlier stay outside this family.
|
|
111
|
+
export function isGlm53OrLater(slug) {
|
|
112
|
+
const text = String(slug ?? "").toLowerCase()
|
|
113
|
+
const re = /glm-(\d{1,3})(?:\.(\d))?(?![\d.])/g
|
|
114
|
+
let m
|
|
115
|
+
while ((m = re.exec(text)) !== null) {
|
|
116
|
+
const major = Number(m[1])
|
|
117
|
+
const minor = m[2] === undefined ? 0 : Number(m[2])
|
|
118
|
+
if (major > 5) return true
|
|
119
|
+
if (major === 5 && minor >= 3) return true
|
|
120
|
+
}
|
|
121
|
+
return false
|
|
122
|
+
}
|
|
123
|
+
|
|
84
124
|
// Candidate ids for exact/alias lookup. Includes the raw apiID/modelID, the
|
|
85
125
|
// lower-cased forms, and a single stripped transport/vendor prefix
|
|
86
126
|
// (e.g. "openai/gpt-5.6-luna" -> "gpt-5.6-luna", "xiaomi/mimo-v2.6-flash" ->
|
|
@@ -323,6 +363,29 @@ export const POLICY_REGISTRY = [
|
|
|
323
363
|
}),
|
|
324
364
|
inventoryRef: "§3 Z.AI GLM",
|
|
325
365
|
},
|
|
366
|
+
{
|
|
367
|
+
// v0.4.4: "GLM-5.3 and later" family baseline. A future 5.3+ model inherits
|
|
368
|
+
// the implicit-cache baseline and its non-mutating diagnostics/transport,
|
|
369
|
+
// but NOT the GLM-5.3-specific `<env>` relocation overlay (`overlays: []`
|
|
370
|
+
// and no `envRelocation` capability). GLM-5.2 and earlier stay neutral.
|
|
371
|
+
id: "zai.glm-5.3-plus",
|
|
372
|
+
creator: "z.ai",
|
|
373
|
+
family: "glm-5.3",
|
|
374
|
+
kind: "family",
|
|
375
|
+
predicate: isGlm53OrLater,
|
|
376
|
+
baseline: "zai.implicit-cache",
|
|
377
|
+
overlays: [],
|
|
378
|
+
legacy: true,
|
|
379
|
+
runtime: rt("glm53", {
|
|
380
|
+
thinkingIntegrity: true,
|
|
381
|
+
cacheRatio: "glm",
|
|
382
|
+
providerChange: "glm",
|
|
383
|
+
openRouterAffinity: true,
|
|
384
|
+
}),
|
|
385
|
+
boundary: "GLM-5.3 and later",
|
|
386
|
+
note: "Z.AI publishes no generational-inheritance rule and no explicit cache control. The baseline is implicit caching; the `<env>` relocation is a GLM-5.3-only CacheEngine overlay and is intentionally not inherited. No cache-control field is invented.",
|
|
387
|
+
inventoryRef: "§3 Z.AI GLM",
|
|
388
|
+
},
|
|
326
389
|
{
|
|
327
390
|
id: "xiaomi.mimo-v2.6",
|
|
328
391
|
creator: "xiaomi",
|
|
@@ -357,6 +420,28 @@ export const POLICY_REGISTRY = [
|
|
|
357
420
|
inventoryRef: "§4 Xiaomi MiMo",
|
|
358
421
|
},
|
|
359
422
|
{
|
|
423
|
+
// v0.4.3: formalize the documented "DeepSeek V4 and later" family. Coverage
|
|
424
|
+
// is passive (no mutation, no overlays). Version ids inherit via the
|
|
425
|
+
// predicate; the V4.1 codename id `deepseek-flash` has no version token and
|
|
426
|
+
// is matched by exact id. Pre-V4 and unknown future ids fall through to the
|
|
427
|
+
// passive creator baseline below, so nothing speculative is ever applied.
|
|
428
|
+
id: "deepseek.v4-plus",
|
|
429
|
+
creator: "deepseek",
|
|
430
|
+
family: "deepseek",
|
|
431
|
+
kind: "family",
|
|
432
|
+
predicate: isDeepseekV4OrLater,
|
|
433
|
+
exactIds: ["deepseek-flash", "deepseek-v4-pro"],
|
|
434
|
+
baseline: "deepseek.kv-cache",
|
|
435
|
+
overlays: [],
|
|
436
|
+
legacy: true,
|
|
437
|
+
runtime: rt("deepseek"),
|
|
438
|
+
boundary: "DeepSeek V4 and later",
|
|
439
|
+
note: "DeepSeek caching is provider-wide and passive (no cache key, flag, breakpoint, or cache-control; Anthropic `cache_control` is ignored). Verified 2026-09-27. Canonical current ids: `deepseek-flash` (V4.1-Flash) and `deepseek-v4-pro`; `deepseek-v4-flash`/`deepseek-v4-flash-vision-exp` are accepted retired aliases.",
|
|
440
|
+
inventoryRef: "§2 DeepSeek",
|
|
441
|
+
},
|
|
442
|
+
{
|
|
443
|
+
// Safe passive fallback for any other `*deepseek*` id (pre-V4, retired, or
|
|
444
|
+
// unknown future models) so DeepSeek always fails safe to observation only.
|
|
360
445
|
id: "deepseek.baseline",
|
|
361
446
|
creator: "deepseek",
|
|
362
447
|
family: "deepseek",
|
|
@@ -53,6 +53,8 @@ import {
|
|
|
53
53
|
MODEL_ALIASES,
|
|
54
54
|
OVERLAYS,
|
|
55
55
|
POLICY_REGISTRY,
|
|
56
|
+
isDeepseekV4OrLater,
|
|
57
|
+
isGlm53OrLater,
|
|
56
58
|
isGpt56OrLater,
|
|
57
59
|
resolveLegacyFamily,
|
|
58
60
|
resolvePolicy,
|
|
@@ -1392,11 +1394,12 @@ test("resolvePolicy: pre-5.6 GPT negative controls are neutral with no overlays"
|
|
|
1392
1394
|
}
|
|
1393
1395
|
})
|
|
1394
1396
|
|
|
1395
|
-
test("resolvePolicy: DeepSeek V4 / V4.1 resolve to the
|
|
1397
|
+
test("resolvePolicy: DeepSeek V4 / V4.1 resolve to the passive baseline (no overlay)", () => {
|
|
1396
1398
|
const v4 = resolvePolicy(M("deepseek", "deepseek-v4-pro"))
|
|
1397
1399
|
assert.equal(v4.creator, "deepseek")
|
|
1398
1400
|
assert.equal(v4.family, "deepseek")
|
|
1399
|
-
|
|
1401
|
+
// v0.4.3: deepseek-v4-pro is a documented canonical id, so it matches exactly.
|
|
1402
|
+
assert.equal(v4.matchType, "exact")
|
|
1400
1403
|
assert.equal(baseId(v4), "deepseek.kv-cache")
|
|
1401
1404
|
assert.deepEqual(overlayIds(v4), [])
|
|
1402
1405
|
|
|
@@ -1642,8 +1645,14 @@ async function runPolicyMigrationProbe() {
|
|
|
1642
1645
|
{ name: "gpt-daybreak-alias", model: { providerID: "openai", id: "gpt-daybreak-blue-latest", api: { id: "gpt-daybreak-blue-latest", npm: "@ai-sdk/openai" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1643
1646
|
{ name: "deepseek-v4-pro", model: { providerID: "deepseek", id: "deepseek-v4-pro", api: { id: "deepseek-v4-pro" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1644
1647
|
{ name: "deepseek-flash", model: { providerID: "deepseek", id: "deepseek-flash", api: { id: "deepseek-flash" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1648
|
+
{ name: "deepseek-v5-future", model: { providerID: "deepseek", id: "deepseek-v5", api: { id: "deepseek-v5" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1649
|
+
{ name: "deepseek-v3-pre", model: { providerID: "deepseek", id: "deepseek-v3", api: { id: "deepseek-v3" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1650
|
+
{ name: "deepseek-openrouter", model: { providerID: "openrouter", id: "deepseek/deepseek-v4-pro", api: { id: "deepseek/deepseek-v4-pro" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1651
|
+
{ name: "deepseek-gateway", model: { providerID: "acme-gateway", id: "my-deepseek-mirror", api: { id: "my-deepseek-mirror" } }, expect: { policy: "deepseek", env: false, gpt: false, header: false } },
|
|
1645
1652
|
{ name: "glm-5.3-direct", model: { providerID: "zai", id: "glm-5.3", api: { id: "glm-5.3" } }, expect: { policy: "glm53", env: true, gpt: false, header: false } },
|
|
1646
1653
|
{ name: "glm-5.3-openrouter", model: { providerID: "openrouter", id: "z-ai/glm-5.3-flash", api: { id: "z-ai/glm-5.3-flash" } }, expect: { policy: "glm53", env: true, gpt: false, header: true } },
|
|
1654
|
+
{ name: "glm-5.4-direct", model: { providerID: "zai", id: "glm-5.4", api: { id: "glm-5.4" } }, expect: { policy: "glm53", env: false, gpt: false, header: false } },
|
|
1655
|
+
{ name: "glm-5.4-openrouter", model: { providerID: "openrouter", id: "z-ai/glm-5.4", api: { id: "z-ai/glm-5.4" } }, expect: { policy: "glm53", env: false, gpt: false, header: true } },
|
|
1647
1656
|
{ name: "glm-5.2", model: { providerID: "zai", id: "glm-5.2", api: { id: "glm-5.2" } }, expect: { policy: "neutral", env: false, gpt: false, header: false } },
|
|
1648
1657
|
{ name: "mimo-v2.6-flash-direct", model: { providerID: "xiaomi", id: "mimo-v2.6-flash", api: { id: "mimo-v2.6-flash" } }, expect: { policy: "mimo26", env: true, gpt: false, header: false } },
|
|
1649
1658
|
{ name: "mimo-v2.6-pro-openrouter", model: { providerID: "openrouter", id: "xiaomi/mimo-v2.6-pro", api: { id: "xiaomi/mimo-v2.6-pro" } }, expect: { policy: "mimo26", env: true, gpt: false, header: true } },
|
|
@@ -1819,3 +1828,185 @@ test("v0.4.2: pre-5.6 and out-of-family GPT ids get no GPT options", async () =>
|
|
|
1819
1828
|
assert.equal(r.runtimePolicy, "neutral", `${name}: neutral runtime`)
|
|
1820
1829
|
}
|
|
1821
1830
|
})
|
|
1831
|
+
|
|
1832
|
+
// ===========================================================================
|
|
1833
|
+
// v0.4.3 DeepSeek V4-and-later passive coverage
|
|
1834
|
+
//
|
|
1835
|
+
// Source: DeepSeek first-party docs re-verified 2026-09-27
|
|
1836
|
+
// (docs/cache-policy-inventory.md §2). Caching is provider-wide and passive:
|
|
1837
|
+
// no cache key/flag/breakpoint; Anthropic `cache_control` is ignored.
|
|
1838
|
+
// ===========================================================================
|
|
1839
|
+
|
|
1840
|
+
test("v0.4.3: isDeepseekV4OrLater matches V4+ version tokens only", () => {
|
|
1841
|
+
const inFamily = ["deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4.1", "deepseek-v4.5-pro", "deepseek-v5", "deepseek-v10", "deepseek/deepseek-v4-pro"]
|
|
1842
|
+
for (const id of inFamily) assert.equal(isDeepseekV4OrLater(id), true, `${id} is V4+`)
|
|
1843
|
+
const outOfFamily = ["deepseek-v3", "deepseek-v2", "deepseek-chat", "deepseek-reasoner", "deepseek-coder", "deepseek-flash", "my-deepseek-mirror", ""]
|
|
1844
|
+
for (const id of outOfFamily) assert.equal(isDeepseekV4OrLater(id), false, `${id} is not a V4+ version token`)
|
|
1845
|
+
})
|
|
1846
|
+
|
|
1847
|
+
test("v0.4.3: canonical V4 ids and aliases resolve to the passive DeepSeek family", () => {
|
|
1848
|
+
for (const id of ["deepseek-flash", "deepseek-v4-pro"]) {
|
|
1849
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1850
|
+
assert.equal(r.creator, "deepseek")
|
|
1851
|
+
assert.equal(r.family, "deepseek")
|
|
1852
|
+
assert.equal(baseId(r), "deepseek.kv-cache")
|
|
1853
|
+
assert.deepEqual(overlayIds(r), [])
|
|
1854
|
+
const rt = resolveRuntimePolicy(M("deepseek", id))
|
|
1855
|
+
assert.equal(rt.policy, "deepseek")
|
|
1856
|
+
assert.equal(rt.gptCacheMetadata, false)
|
|
1857
|
+
assert.equal(rt.envRelocation, null)
|
|
1858
|
+
assert.equal(rt.cacheRatio, null)
|
|
1859
|
+
assert.equal(rt.openRouterAffinity, false)
|
|
1860
|
+
}
|
|
1861
|
+
// v0.4.0 alias handling is preserved.
|
|
1862
|
+
const alias = resolvePolicy(M("deepseek", "deepseek-v4-flash"))
|
|
1863
|
+
assert.equal(alias.family, "deepseek")
|
|
1864
|
+
assert.ok(alias.matchReason.startsWith("alias:deepseek-v4-flash"))
|
|
1865
|
+
assert.equal(resolveRuntimePolicy(M("deepseek", "deepseek-v4-flash")).policy, "deepseek")
|
|
1866
|
+
assert.equal(resolvePolicy(M("deepseek", "deepseek-chat")).family, "deepseek")
|
|
1867
|
+
})
|
|
1868
|
+
|
|
1869
|
+
test("v0.4.3: later and unknown DeepSeek ids stay passive (no speculative mutation)", () => {
|
|
1870
|
+
for (const id of ["deepseek-v5", "deepseek-v6.2", "deepseek-v4.1-pro", "deepseek-nova"]) {
|
|
1871
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1872
|
+
assert.equal(r.creator, "deepseek")
|
|
1873
|
+
assert.equal(r.family, "deepseek")
|
|
1874
|
+
assert.deepEqual(overlayIds(r), [], `${id}: no overlay`)
|
|
1875
|
+
const rt = resolveRuntimePolicy(M("deepseek", id))
|
|
1876
|
+
assert.equal(rt.policy, "deepseek", `${id}: passive policy`)
|
|
1877
|
+
assert.equal(rt.gptCacheMetadata, false)
|
|
1878
|
+
assert.equal(rt.envRelocation, null)
|
|
1879
|
+
assert.equal(rt.cacheRatio, null)
|
|
1880
|
+
assert.equal(rt.openRouterAffinity, false)
|
|
1881
|
+
}
|
|
1882
|
+
})
|
|
1883
|
+
|
|
1884
|
+
test("v0.4.3: pre-V4 DeepSeek ids remain passive and outside the V4+ family entry", () => {
|
|
1885
|
+
for (const id of ["deepseek-v3", "deepseek-v2", "deepseek-coder"]) {
|
|
1886
|
+
const r = resolvePolicy(M("deepseek", id))
|
|
1887
|
+
assert.equal(r.family, "deepseek")
|
|
1888
|
+
// handled by the safe creator fallback, not the V4+ family predicate
|
|
1889
|
+
assert.equal(r.matchType, "creator", `${id}: creator fallback`)
|
|
1890
|
+
assert.equal(resolveRuntimePolicy(M("deepseek", id)).policy, "deepseek")
|
|
1891
|
+
}
|
|
1892
|
+
})
|
|
1893
|
+
|
|
1894
|
+
test("v0.4.3: DeepSeek keeps the generic read/write ratio and no family-specific fields", () => {
|
|
1895
|
+
const rt = resolveRuntimePolicy(M("deepseek", "deepseek-v4-pro"))
|
|
1896
|
+
assert.equal(rt.cacheRatio, null) // generic read/(read+write) accounting is used
|
|
1897
|
+
assert.equal(hitRatePct(30, 70), 30)
|
|
1898
|
+
})
|
|
1899
|
+
|
|
1900
|
+
test("v0.4.3: DeepSeek never receives OpenRouter affinity or GPT/GLM/MiMo fields", async () => {
|
|
1901
|
+
const { results } = await policyMigrationResults()
|
|
1902
|
+
const deepseekCases = [
|
|
1903
|
+
"deepseek-v4-pro",
|
|
1904
|
+
"deepseek-flash",
|
|
1905
|
+
"deepseek-v5-future",
|
|
1906
|
+
"deepseek-v3-pre",
|
|
1907
|
+
"deepseek-openrouter",
|
|
1908
|
+
"deepseek-gateway",
|
|
1909
|
+
]
|
|
1910
|
+
for (const name of deepseekCases) {
|
|
1911
|
+
const r = results.find((x) => x.name === name)
|
|
1912
|
+
assert.ok(r, `${name} present`)
|
|
1913
|
+
assert.equal(r.runtimePolicy, "deepseek", `${name}: passive policy`)
|
|
1914
|
+
assert.equal(r.detectPolicy, "deepseek", `${name}: detectPolicy agrees`)
|
|
1915
|
+
assert.equal(r.gptOptionInjected, false, `${name}: no GPT options leak`)
|
|
1916
|
+
assert.equal(r.systemRelocated, false, `${name}: no GLM/MiMo env relocation`)
|
|
1917
|
+
assert.equal(r.affinityHeaderAttached, false, `${name}: no OpenRouter affinity`)
|
|
1918
|
+
assert.equal(r.existingHeadersPreserved, true, `${name}: headers preserved`)
|
|
1919
|
+
}
|
|
1920
|
+
})
|
|
1921
|
+
|
|
1922
|
+
// ===========================================================================
|
|
1923
|
+
// v0.4.4 GLM-5.3-and-later family baseline vs the GLM-5.3-specific overlay
|
|
1924
|
+
//
|
|
1925
|
+
// Source: Z.AI docs re-verified 2026-09-27 (docs/cache-policy-inventory.md §3).
|
|
1926
|
+
// Z.AI documents implicit caching (no cache control) and publishes no
|
|
1927
|
+
// generational-inheritance rule; the `<env>` relocation is a CacheEngine overlay
|
|
1928
|
+
// with no first-party basis. GLM-5.2 and earlier stay neutral.
|
|
1929
|
+
// ===========================================================================
|
|
1930
|
+
|
|
1931
|
+
test("v0.4.4: isGlm53OrLater matches GLM-5.3+ version tokens only", () => {
|
|
1932
|
+
const inFamily = ["glm-5.3", "glm-5.3-flash", "glm-5.3-flashx", "glm-5.4", "glm-5.9", "glm-6", "z-ai/glm-5.3-flash"]
|
|
1933
|
+
for (const id of inFamily) assert.equal(isGlm53OrLater(id), true, `${id} is 5.3+`)
|
|
1934
|
+
const outOfFamily = ["glm-5.2", "glm-5.1", "glm-5", "glm-4.7", "glm-4.6", "glm-4.5", "glm-4.5-air", "glm-4-32b-0414-128k", ""]
|
|
1935
|
+
for (const id of outOfFamily) assert.equal(isGlm53OrLater(id), false, `${id} is pre-5.3`)
|
|
1936
|
+
})
|
|
1937
|
+
|
|
1938
|
+
test("v0.4.4: GLM-5.3 keeps its overlay; a later GLM gets the baseline only", () => {
|
|
1939
|
+
const g53 = resolvePolicy(M("zai", "glm-5.3"))
|
|
1940
|
+
assert.equal(g53.family, "glm-5.3")
|
|
1941
|
+
assert.equal(baseId(g53), "zai.implicit-cache")
|
|
1942
|
+
assert.deepEqual(overlayIds(g53), ["glm53.env-relocation"])
|
|
1943
|
+
|
|
1944
|
+
const g54 = resolvePolicy(M("zai", "glm-5.4"))
|
|
1945
|
+
assert.equal(g54.creator, "z.ai")
|
|
1946
|
+
assert.equal(g54.family, "glm-5.3")
|
|
1947
|
+
assert.equal(baseId(g54), "zai.implicit-cache") // same family baseline
|
|
1948
|
+
assert.deepEqual(overlayIds(g54), []) // overlay is NOT inherited
|
|
1949
|
+
|
|
1950
|
+
// Runtime capability separation: baseline diagnostics/transport yes, prompt rewrite no.
|
|
1951
|
+
const c53 = resolveRuntimePolicy(M("zai", "glm-5.3"))
|
|
1952
|
+
const c54 = resolveRuntimePolicy(M("zai", "glm-5.4"))
|
|
1953
|
+
assert.equal(c54.policy, "glm53")
|
|
1954
|
+
assert.equal(c54.envRelocation, null)
|
|
1955
|
+
assert.equal(c54.thinkingIntegrity, true)
|
|
1956
|
+
assert.equal(c54.cacheRatio, "glm")
|
|
1957
|
+
assert.equal(c54.providerChange, "glm")
|
|
1958
|
+
assert.equal(c54.openRouterAffinity, true)
|
|
1959
|
+
assert.equal(c53.envRelocation, "glm")
|
|
1960
|
+
|
|
1961
|
+
// Boundary metadata is traceable and the overlay is registered separately.
|
|
1962
|
+
const plus = POLICY_REGISTRY.find((e) => e.id === "zai.glm-5.3-plus")
|
|
1963
|
+
assert.equal(plus.boundary, "GLM-5.3 and later")
|
|
1964
|
+
assert.deepEqual(plus.overlays, [])
|
|
1965
|
+
assert.ok(plus.inventoryRef)
|
|
1966
|
+
})
|
|
1967
|
+
|
|
1968
|
+
test("v0.4.4: GLM-5.2 and earlier stay neutral", () => {
|
|
1969
|
+
for (const id of ["glm-5.2", "glm-5.1", "glm-5", "glm-4.7", "glm-4.6", "glm-4.5"]) {
|
|
1970
|
+
const r = resolvePolicy(M("zai", id))
|
|
1971
|
+
assert.equal(r.family, "neutral", `${id} neutral`)
|
|
1972
|
+
assert.deepEqual(overlayIds(r), [])
|
|
1973
|
+
assert.equal(resolveRuntimePolicy(M("zai", id)).policy, "neutral")
|
|
1974
|
+
}
|
|
1975
|
+
})
|
|
1976
|
+
|
|
1977
|
+
test("v0.4.4: legacy detectPolicy follows the GLM-5.3-and-later boundary", () => {
|
|
1978
|
+
assert.equal(detectPolicy(M("zai", "glm-5.3-flash")), POLICY_GLM53)
|
|
1979
|
+
assert.equal(detectPolicy(M("zai", "glm-5.4")), POLICY_GLM53)
|
|
1980
|
+
assert.equal(detectPolicy(M("zai", "glm-5.2")), POLICY_NEUTRAL)
|
|
1981
|
+
})
|
|
1982
|
+
|
|
1983
|
+
test("v0.4.4: <env> absent leaves system content unchanged", () => {
|
|
1984
|
+
const plain = "Stable instructions only.\nNo environment block here."
|
|
1985
|
+
const r = relocateVolatileEnvBlock(plain)
|
|
1986
|
+
assert.equal(r.changed, false)
|
|
1987
|
+
assert.equal(r.text, plain)
|
|
1988
|
+
})
|
|
1989
|
+
|
|
1990
|
+
test("v0.4.4: later GLM inherits the baseline but never the <env> rewrite (runtime)", async () => {
|
|
1991
|
+
const { results } = await policyMigrationResults()
|
|
1992
|
+
const g53 = results.find((r) => r.name === "glm-5.3-direct")
|
|
1993
|
+
const g54 = results.find((r) => r.name === "glm-5.4-direct")
|
|
1994
|
+
assert.ok(g53 && g54)
|
|
1995
|
+
// Identical system text with a valid <env> block present in both cases.
|
|
1996
|
+
assert.equal(g53.systemRelocated, true) // 5.3 overlay fires
|
|
1997
|
+
assert.equal(g54.systemRelocated, false) // later GLM does NOT inherit it
|
|
1998
|
+
assert.equal(g54.runtimePolicy, "glm53") // but the family baseline applies
|
|
1999
|
+
assert.equal(g54.gptOptionInjected, false)
|
|
2000
|
+
assert.equal(g54.affinityHeaderAttached, false)
|
|
2001
|
+
})
|
|
2002
|
+
|
|
2003
|
+
test("v0.4.4: GLM OpenRouter affinity is transport-gated for later GLM too", async () => {
|
|
2004
|
+
const { results } = await policyMigrationResults()
|
|
2005
|
+
const or = results.find((r) => r.name === "glm-5.4-openrouter")
|
|
2006
|
+
const direct = results.find((r) => r.name === "glm-5.4-direct")
|
|
2007
|
+
assert.equal(or.affinityHeaderAttached, true)
|
|
2008
|
+
assert.equal(or.systemRelocated, false)
|
|
2009
|
+
assert.equal(direct.affinityHeaderAttached, false)
|
|
2010
|
+
assert.equal(or.existingHeadersPreserved, true)
|
|
2011
|
+
assert.equal(direct.existingHeadersPreserved, true)
|
|
2012
|
+
})
|