opencode-cache-engine 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +1032 -0
- package/examples/cache-engine.json +25 -0
- package/opencode-cache-engine-0.1.0.tgz +0 -0
- package/package.json +33 -0
- package/src/cache-engine-core.mjs +728 -0
- package/src/cache-engine.ts +753 -0
- package/test/cache-engine.test.mjs +765 -0
|
@@ -0,0 +1,728 @@
|
|
|
1
|
+
// cache-engine-core.mjs
|
|
2
|
+
//
|
|
3
|
+
// Pure, dependency-light logic for the cache-engine plugin. Kept in plain JS so
|
|
4
|
+
// the unit tests (cache-engine.test.mjs) can `import` it under Node without a
|
|
5
|
+
// TypeScript compiler. The plugin entry (cache-engine.ts) imports this module.
|
|
6
|
+
//
|
|
7
|
+
// This module is PROVIDER-AWARE: it classifies a model into a cache-policy
|
|
8
|
+
// family (deepseek | gpt56 | glm53 | neutral) and exposes small pure helpers for
|
|
9
|
+
// each family's strategy. The plugin entry (cache-engine.ts) remains the only
|
|
10
|
+
// place that touches OpenCode hooks; every decision here is testable in Node.
|
|
11
|
+
//
|
|
12
|
+
// Terminology note: these functions deal with the *observed* system/tool
|
|
13
|
+
// prefix shape. An observed change means the request's prefix bytes changed; it
|
|
14
|
+
// is NOT proof that the provider's cache key changed or that a cache miss
|
|
15
|
+
// occurred. Provider-reported cache token counts are the only authoritative
|
|
16
|
+
// signal; local hashes are diagnostics.
|
|
17
|
+
|
|
18
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs"
|
|
19
|
+
import { createHash } from "node:crypto"
|
|
20
|
+
import { homedir } from "node:os"
|
|
21
|
+
import { dirname, join } from "node:path"
|
|
22
|
+
|
|
23
|
+
export const CONFIG_FILENAME = "cache-engine.json"
|
|
24
|
+
export const DEFAULT_CONFIG_PATH = join(homedir(), ".config/opencode", CONFIG_FILENAME)
|
|
25
|
+
export const DEFAULT_METRICS_FILE = join(homedir(), ".cache/opencode/cache-metrics.jsonl")
|
|
26
|
+
|
|
27
|
+
export const DIGEST_TEMPLATE = `## Session digest (cache-stable continuation block)
|
|
28
|
+
- Goal:
|
|
29
|
+
- Decisions made:
|
|
30
|
+
- Pending:
|
|
31
|
+
- Active files:
|
|
32
|
+
`
|
|
33
|
+
|
|
34
|
+
// Cache-policy families. "neutral" preserves stock behavior (no mutation).
|
|
35
|
+
export const POLICY_DEEPSEEK = "deepseek"
|
|
36
|
+
export const POLICY_GPT56 = "gpt56"
|
|
37
|
+
export const POLICY_GLM53 = "glm53"
|
|
38
|
+
export const POLICY_NEUTRAL = "neutral"
|
|
39
|
+
|
|
40
|
+
export const GPT56_DEFAULT_TTL = "30m"
|
|
41
|
+
export const GPT56_DEFAULT_MODE = "implicit"
|
|
42
|
+
|
|
43
|
+
// ---------------------------------------------------------------------------
|
|
44
|
+
// Configuration
|
|
45
|
+
// ---------------------------------------------------------------------------
|
|
46
|
+
|
|
47
|
+
function defaultPolicies() {
|
|
48
|
+
return {
|
|
49
|
+
deepseek: { enabled: true },
|
|
50
|
+
gpt56: {
|
|
51
|
+
enabled: true,
|
|
52
|
+
promptCacheKey: true,
|
|
53
|
+
// Disabled by default: this OpenCode runtime does not expose reliable
|
|
54
|
+
// fork lineage (session.fork copies messages without setting parent_id),
|
|
55
|
+
// so cross-fork cache-root inheritance cannot be applied safely. The code
|
|
56
|
+
// path remains and can be re-enabled if the runtime gains proper lineage.
|
|
57
|
+
cacheRootKey: false,
|
|
58
|
+
compactionCacheIsolation: true,
|
|
59
|
+
reasoningEffortDiagnostics: true,
|
|
60
|
+
mode: GPT56_DEFAULT_MODE,
|
|
61
|
+
ttl: GPT56_DEFAULT_TTL,
|
|
62
|
+
},
|
|
63
|
+
glm53: {
|
|
64
|
+
enabled: true,
|
|
65
|
+
stabilizeSystem: true,
|
|
66
|
+
preserveThinkingIntegrity: true,
|
|
67
|
+
},
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export function defaultConfig() {
|
|
72
|
+
return {
|
|
73
|
+
enabled: true,
|
|
74
|
+
metricsFile: DEFAULT_METRICS_FILE,
|
|
75
|
+
compactTemplate: true,
|
|
76
|
+
logPrefixChanges: true,
|
|
77
|
+
policies: defaultPolicies(),
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export function expandHome(p) {
|
|
82
|
+
if (p === "~") return homedir()
|
|
83
|
+
if (typeof p === "string" && p.startsWith("~/")) return join(homedir(), p.slice(2))
|
|
84
|
+
return p
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const boolOr = (v, fb) => (typeof v === "boolean" ? v : fb)
|
|
88
|
+
|
|
89
|
+
function parsePolicy(rawPolicy, defaults) {
|
|
90
|
+
const out = { ...defaults }
|
|
91
|
+
if (!rawPolicy || typeof rawPolicy !== "object") return out
|
|
92
|
+
for (const k of Object.keys(defaults)) {
|
|
93
|
+
if (typeof defaults[k] === "boolean") out[k] = boolOr(rawPolicy[k], defaults[k])
|
|
94
|
+
}
|
|
95
|
+
// gpt56 mode/ttl are strings with validated values
|
|
96
|
+
if (defaults.mode !== undefined) {
|
|
97
|
+
const mode = typeof rawPolicy.mode === "string" ? rawPolicy.mode : defaults.mode
|
|
98
|
+
out.mode = mode === "implicit" || mode === "explicit" ? mode : defaults.mode
|
|
99
|
+
}
|
|
100
|
+
if (defaults.ttl !== undefined) {
|
|
101
|
+
out.ttl = typeof rawPolicy.ttl === "string" && rawPolicy.ttl.length > 0 ? rawPolicy.ttl : defaults.ttl
|
|
102
|
+
}
|
|
103
|
+
return out
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// Env override is applied first, then the file may override. Invalid input
|
|
107
|
+
// falls back to defaults. Returns a fresh object; never mutates callers.
|
|
108
|
+
export function parseConfig(raw, env) {
|
|
109
|
+
const cfg = defaultConfig()
|
|
110
|
+
const e = env || {}
|
|
111
|
+
if (typeof e.CACHE_ENGINE_METRICS_FILE === "string" && e.CACHE_ENGINE_METRICS_FILE.length > 0) {
|
|
112
|
+
cfg.metricsFile = expandHome(e.CACHE_ENGINE_METRICS_FILE)
|
|
113
|
+
}
|
|
114
|
+
if (raw && typeof raw === "object") {
|
|
115
|
+
if (typeof raw.enabled === "boolean") cfg.enabled = raw.enabled
|
|
116
|
+
if (typeof raw.metricsFile === "string" && raw.metricsFile.length > 0) cfg.metricsFile = expandHome(raw.metricsFile)
|
|
117
|
+
if (typeof raw.compactTemplate === "boolean") cfg.compactTemplate = raw.compactTemplate
|
|
118
|
+
if (typeof raw.logPrefixChanges === "boolean") cfg.logPrefixChanges = raw.logPrefixChanges
|
|
119
|
+
if (raw.policies && typeof raw.policies === "object") {
|
|
120
|
+
const d = defaultPolicies()
|
|
121
|
+
for (const fam of ["deepseek", "gpt56", "glm53"]) {
|
|
122
|
+
if (raw.policies[fam]) cfg.policies[fam] = parsePolicy(raw.policies[fam], d[fam])
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
return cfg
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export function loadConfig({ configPath = DEFAULT_CONFIG_PATH, env } = {}) {
|
|
130
|
+
try {
|
|
131
|
+
if (existsSync(configPath)) {
|
|
132
|
+
return parseConfig(JSON.parse(readFileSync(configPath, "utf8")), env)
|
|
133
|
+
}
|
|
134
|
+
} catch {
|
|
135
|
+
// malformed config -> defaults
|
|
136
|
+
}
|
|
137
|
+
return parseConfig(undefined, env)
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// Ensure the metrics file's parent directory exists. Best-effort; never throws.
|
|
141
|
+
export function ensureMetricsDir(metricsFile) {
|
|
142
|
+
try {
|
|
143
|
+
mkdirSync(dirname(metricsFile), { recursive: true })
|
|
144
|
+
} catch {
|
|
145
|
+
/* best-effort */
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// ---------------------------------------------------------------------------
|
|
150
|
+
// Metrics writer (best-effort, must never throw)
|
|
151
|
+
// ---------------------------------------------------------------------------
|
|
152
|
+
|
|
153
|
+
export function createRecorder(metricsFile) {
|
|
154
|
+
return {
|
|
155
|
+
record(line) {
|
|
156
|
+
try {
|
|
157
|
+
appendFileSync(metricsFile, JSON.stringify(line) + "\n")
|
|
158
|
+
} catch {
|
|
159
|
+
// Telemetry is best-effort; a failed write must not break a request.
|
|
160
|
+
}
|
|
161
|
+
},
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// ---------------------------------------------------------------------------
|
|
166
|
+
// Hashing / canonicalization
|
|
167
|
+
// ---------------------------------------------------------------------------
|
|
168
|
+
|
|
169
|
+
export function shorthash(s) {
|
|
170
|
+
return createHash("sha256").update(String(s ?? "")).digest("hex").slice(0, 16)
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// Deterministic stringification. Object keys are sorted so object/insertion
|
|
174
|
+
// ordering can never create a false change. Non-serializable values (functions,
|
|
175
|
+
// symbols, undefined) are normalized to stable markers rather than omitted, so
|
|
176
|
+
// the output is total and reproducible.
|
|
177
|
+
export function canonicalStringify(value) {
|
|
178
|
+
if (value === null) return "null"
|
|
179
|
+
const t = typeof value
|
|
180
|
+
if (t === "string") return JSON.stringify(value)
|
|
181
|
+
if (t === "number") return Number.isFinite(value) ? String(value) : '"__nonfinite__"'
|
|
182
|
+
if (t === "boolean") return String(value)
|
|
183
|
+
if (t === "bigint") return String(value)
|
|
184
|
+
if (t === "undefined") return '"__undefined__"'
|
|
185
|
+
if (t === "function" || t === "symbol") return '"__nonserializable__"'
|
|
186
|
+
if (Array.isArray(value)) return `[${value.map(canonicalStringify).join(",")}]`
|
|
187
|
+
if (t === "object") {
|
|
188
|
+
const obj = value
|
|
189
|
+
const keys = Object.keys(obj).sort()
|
|
190
|
+
return `{${keys.map((k) => `${JSON.stringify(k)}:${canonicalStringify(obj[k])}`).join(",")}}`
|
|
191
|
+
}
|
|
192
|
+
return '"__unknown__"'
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// ---------------------------------------------------------------------------
|
|
196
|
+
// Model detection -> cache-policy family
|
|
197
|
+
// ---------------------------------------------------------------------------
|
|
198
|
+
|
|
199
|
+
// Normalize a model-like object into a searchable haystack. Accepts both the
|
|
200
|
+
// full OpenCode Model ({providerID, id, api:{id,npm}, name}) and slim test
|
|
201
|
+
// objects ({providerID, modelID/apiID}).
|
|
202
|
+
function modelSignals(model) {
|
|
203
|
+
const m = model && typeof model === "object" ? model : {}
|
|
204
|
+
const api = m.api && typeof m.api === "object" ? m.api : {}
|
|
205
|
+
const providerID = String(m.providerID ?? m.provider ?? "")
|
|
206
|
+
const apiID = String(m.modelID ?? api.id ?? m.id ?? m.apiID ?? "")
|
|
207
|
+
const modelID = String(m.id ?? "")
|
|
208
|
+
const npm = String(api.npm ?? m.npm ?? "")
|
|
209
|
+
const name = String(m.name ?? "")
|
|
210
|
+
const slug = `${apiID} ${modelID}`.trim()
|
|
211
|
+
return { providerID, apiID, modelID, npm, name, slug: slug.toLowerCase() }
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// OpenAI-ish context is required before we apply GPT-5.6 options, so we never
|
|
215
|
+
// send GPT-5.6-only fields to a non-OpenAI endpoint merely because a model
|
|
216
|
+
// string contains "gpt-5.6". A slug that explicitly starts with openai/ or
|
|
217
|
+
// azure/ (typical for openrouter/azure/openai-compatible routes) also counts
|
|
218
|
+
// because the upstream IS OpenAI. A bare openai-compatible provider with no
|
|
219
|
+
// such slug does NOT count: we must not guess.
|
|
220
|
+
function isOpenAIish(s) {
|
|
221
|
+
const { providerID, slug, npm } = s
|
|
222
|
+
const p = providerID.toLowerCase()
|
|
223
|
+
if (p === "openai" || p === "azure") return true
|
|
224
|
+
if (slug.startsWith("openai/") || slug.startsWith("azure/")) return true
|
|
225
|
+
if (/@ai-sdk\/openai|@ai-sdk\/azure/.test(npm)) return true
|
|
226
|
+
return false
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// The gpt-5.6 family, tolerating OpenCode's variants: gpt-5.6, gpt-5.6-luna,
|
|
230
|
+
// gpt-5.6-luna:flex, gpt-5.6-luna-pro:flex, gpt-5.6-<anything>.
|
|
231
|
+
// A trailing digit guard avoids matching hypothetical "gpt-5.60" etc.
|
|
232
|
+
const GPT56_RE = /gpt-5\.6(?![\d.])/i
|
|
233
|
+
// GLM 5.3 family only (not glm-4.x / glm-4.6 etc).
|
|
234
|
+
const GLM53_RE = /glm-5\.3(?![\d.])/i
|
|
235
|
+
const DEEPSEEK_RE = /deepseek/i
|
|
236
|
+
|
|
237
|
+
// Pure classifier. Returns one of the POLICY_* keys. `model` may be a full
|
|
238
|
+
// OpenCode Model, or {providerID, modelID|apiID|id}.
|
|
239
|
+
export function detectPolicy(model) {
|
|
240
|
+
if (!model || typeof model !== "object") return POLICY_NEUTRAL
|
|
241
|
+
const s = modelSignals(model)
|
|
242
|
+
if (!s.slug) return POLICY_NEUTRAL
|
|
243
|
+
if (GPT56_RE.test(s.slug) && isOpenAIish(s)) return POLICY_GPT56
|
|
244
|
+
if (GLM53_RE.test(s.slug)) return POLICY_GLM53
|
|
245
|
+
if (DEEPSEEK_RE.test(s.slug) || DEEPSEEK_RE.test(s.providerID)) return POLICY_DEEPSEEK
|
|
246
|
+
return POLICY_NEUTRAL
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
export function policyEnabled(cfg, family) {
|
|
250
|
+
const pol = cfg?.policies?.[family]
|
|
251
|
+
if (!pol) return false
|
|
252
|
+
return pol.enabled !== false
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// ---------------------------------------------------------------------------
|
|
256
|
+
// GPT-5.6 cache options
|
|
257
|
+
// ---------------------------------------------------------------------------
|
|
258
|
+
|
|
259
|
+
// Build the delta to merge into the request's provider options for a GPT-5.6
|
|
260
|
+
// model. Conservative: `implicit` mode + a stable session-derived key, only
|
|
261
|
+
// added when the fields are not already present (so we never fight the runtime
|
|
262
|
+
// or an explicit provider config). No explicit breakpoints by default.
|
|
263
|
+
// Returns {} when nothing should change.
|
|
264
|
+
//
|
|
265
|
+
// `existingOptions` is the outgoing options record (output.options in the
|
|
266
|
+
// chat.params hook). We never overwrite what is already there.
|
|
267
|
+
export function gptCacheOptionsDelta(existingOptions, { key, mode = GPT56_DEFAULT_MODE, ttl = GPT56_DEFAULT_TTL } = {}) {
|
|
268
|
+
const delta = {}
|
|
269
|
+
if (!existingOptions || typeof existingOptions !== "object") return delta
|
|
270
|
+
if (typeof key === "string" && key.length > 0 && existingOptions.promptCacheKey === undefined) {
|
|
271
|
+
delta.promptCacheKey = key
|
|
272
|
+
}
|
|
273
|
+
const existingMode = existingOptions.promptCacheOptions
|
|
274
|
+
if (existingMode === undefined || existingMode === null) {
|
|
275
|
+
const m = mode === "explicit" ? "explicit" : "implicit"
|
|
276
|
+
delta.promptCacheOptions = { mode: m, ttl: typeof ttl === "string" && ttl.length > 0 ? ttl : GPT56_DEFAULT_TTL }
|
|
277
|
+
}
|
|
278
|
+
return delta
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// ---------------------------------------------------------------------------
|
|
282
|
+
// GLM-5.3 system stabilization
|
|
283
|
+
// ---------------------------------------------------------------------------
|
|
284
|
+
|
|
285
|
+
// The OpenCode system string begins with the agent prompt, then an env block
|
|
286
|
+
// ("You are powered by the model named ... Today's date: ... </env>") whose
|
|
287
|
+
// only per-day volatile byte is the date line. For GLM-5.3 we relocate that
|
|
288
|
+
// whole identifiable env block to the END of the system string so a daily date
|
|
289
|
+
// change only invalidates the tail of the prompt, leaving the long stable
|
|
290
|
+
// prefix intact. Content is preserved byte-for-byte (only position changes).
|
|
291
|
+
//
|
|
292
|
+
// Returns { text, changed }. When the block cannot be identified unambiguously,
|
|
293
|
+
// returns the input unchanged (changed:false). This is a content-preserving
|
|
294
|
+
// reorder of clearly volatile metadata only -- it never reorders arbitrary
|
|
295
|
+
// instructions. This function is ONLY applied when the caller has already
|
|
296
|
+
// classified the model as GLM-5.3.
|
|
297
|
+
export function relocateVolatileEnvBlock(text) {
|
|
298
|
+
if (typeof text !== "string") return { text, changed: false }
|
|
299
|
+
const START = "You are powered by the model named "
|
|
300
|
+
const startIdx = text.indexOf(START)
|
|
301
|
+
if (startIdx < 0) return { text, changed: false }
|
|
302
|
+
const endMarker = "</env>"
|
|
303
|
+
const endIdx = text.indexOf(endMarker, startIdx)
|
|
304
|
+
if (endIdx < 0) return { text, changed: false }
|
|
305
|
+
const blockEnd = endIdx + endMarker.length
|
|
306
|
+
const block = text.slice(startIdx, blockEnd)
|
|
307
|
+
const rest = text.slice(0, startIdx) + text.slice(blockEnd)
|
|
308
|
+
if (rest.trim().length === 0) return { text, changed: false }
|
|
309
|
+
const sep = rest.endsWith("\n") ? "" : "\n"
|
|
310
|
+
const out = rest + sep + block
|
|
311
|
+
return { text: out, changed: out !== text }
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
// ---------------------------------------------------------------------------
|
|
315
|
+
// Prefix shape diagnostics (decomposition)
|
|
316
|
+
//
|
|
317
|
+
// The system prefix is a single long string. To answer "did the STABLE prefix
|
|
318
|
+
// change?" rather than "did the whole system prompt change?", we compare the
|
|
319
|
+
// current system text against the session's first-seen baseline and split at
|
|
320
|
+
// the longest common byte prefix: everything up to the first difference is the
|
|
321
|
+
// stable prefix; the tail is the volatile suffix (e.g. the relocated env block
|
|
322
|
+
// with its daily date).
|
|
323
|
+
// ---------------------------------------------------------------------------
|
|
324
|
+
|
|
325
|
+
export function commonPrefixLength(a, b) {
|
|
326
|
+
if (typeof a !== "string" || typeof b !== "string") return 0
|
|
327
|
+
const n = Math.min(a.length, b.length)
|
|
328
|
+
let i = 0
|
|
329
|
+
while (i < n && a.charCodeAt(i) === b.charCodeAt(i)) i++
|
|
330
|
+
return i
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
// Returns hash fields for a system observation given the baseline text.
|
|
334
|
+
export function systemShapeHashes(baseline, current) {
|
|
335
|
+
const fullSystemHash = shorthash(current)
|
|
336
|
+
const common = commonPrefixLength(baseline, current)
|
|
337
|
+
const stableSystemPrefixHash = shorthash(current.slice(0, common))
|
|
338
|
+
const volatile = current.slice(common)
|
|
339
|
+
const volatileSystemSuffixHash = volatile.length > 0 ? shorthash(volatile) : null
|
|
340
|
+
return { fullSystemHash, stableSystemPrefixHash, volatileSystemSuffixHash }
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
// ---------------------------------------------------------------------------
|
|
344
|
+
// Tool fingerprints
|
|
345
|
+
// ---------------------------------------------------------------------------
|
|
346
|
+
|
|
347
|
+
// Keep only the model-visible fields of a tool definition. This intentionally
|
|
348
|
+
// drops runtime-only state (object identity, function refs, timestamps,
|
|
349
|
+
// arbitrary metadata) that is not part of what the model sees.
|
|
350
|
+
export function normalizeTool(tool) {
|
|
351
|
+
const t = tool || {}
|
|
352
|
+
const params = t.parameters === undefined || t.parameters === null ? null : t.parameters
|
|
353
|
+
return {
|
|
354
|
+
id: typeof t.id === "string" ? t.id : String(t.id ?? ""),
|
|
355
|
+
description: typeof t.description === "string" ? t.description : "",
|
|
356
|
+
parameters: params,
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const canonicalTool = (t) => canonicalStringify(normalizeTool(t))
|
|
361
|
+
|
|
362
|
+
// SEMANTIC fingerprint: order-INSENSITIVE (sorted). Detects meaningful tool
|
|
363
|
+
// definition changes regardless of ordering. Returns null on unusable input.
|
|
364
|
+
export function toolFingerprint(tools) {
|
|
365
|
+
if (!Array.isArray(tools)) return null
|
|
366
|
+
const parts = tools.map(canonicalTool).sort()
|
|
367
|
+
return shorthash(parts.join("\u0000"))
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
// WIRE fingerprint: order-SENSITIVE (registry order, i.e. the closest
|
|
371
|
+
// deterministic pre-wire representation available to a plugin). The provider
|
|
372
|
+
// caches what is actually sent; OpenCode additionally sorts tools
|
|
373
|
+
// alphabetically before sending, so registry order is NOT byte-equal to the
|
|
374
|
+
// wire. We document that and fingerprint the closest representation rather than
|
|
375
|
+
// pretending it is exact.
|
|
376
|
+
export function toolWireFingerprint(tools) {
|
|
377
|
+
if (!Array.isArray(tools)) return null
|
|
378
|
+
const parts = tools.map(canonicalTool)
|
|
379
|
+
return shorthash(parts.join("\u0000"))
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
// ---------------------------------------------------------------------------
|
|
383
|
+
// Prefix shape comparison
|
|
384
|
+
// ---------------------------------------------------------------------------
|
|
385
|
+
|
|
386
|
+
// Compare a previous observed shape against the current one. A dimension is
|
|
387
|
+
// only reported as changed when BOTH sides carry a real (non-null) hash; an
|
|
388
|
+
// unknown dimension is never treated as a change. Returns the array of changed
|
|
389
|
+
// dimensions ("system", "tools") which is empty when nothing changed.
|
|
390
|
+
// System change = the full system hash changed. Tools change = the semantic
|
|
391
|
+
// OR the wire tool fingerprint changed (both non-null on each side).
|
|
392
|
+
export function shapeDiff(prev, cur) {
|
|
393
|
+
const changed = []
|
|
394
|
+
if (!prev || !cur) return changed
|
|
395
|
+
const full = (v) => v.fullSystemHash ?? v.systemHash
|
|
396
|
+
if (full(prev) != null && full(cur) != null && full(prev) !== full(cur)) changed.push("system")
|
|
397
|
+
const semPrev = prev.semanticToolsHash ?? prev.toolsHash
|
|
398
|
+
const semCur = cur.semanticToolsHash ?? cur.toolsHash
|
|
399
|
+
if (semPrev != null && semCur != null && semPrev !== semCur) changed.push("tools")
|
|
400
|
+
else {
|
|
401
|
+
// wire-only reorder: semantic equal but registry order changed
|
|
402
|
+
const wPrev = prev.wireToolsHash
|
|
403
|
+
const wCur = cur.wireToolsHash
|
|
404
|
+
if (wPrev != null && wCur != null && wPrev !== wCur) changed.push("tools")
|
|
405
|
+
}
|
|
406
|
+
return changed
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
// Granular per-field diff. Returns the list of field names (in `fields`) whose
|
|
410
|
+
// value differs between prev and cur while BOTH sides are non-null. Used for
|
|
411
|
+
// fine-grained telemetry (stable prefix vs volatile suffix vs wire tools).
|
|
412
|
+
export function shapeFieldDiffs(prev, cur, fields) {
|
|
413
|
+
const out = []
|
|
414
|
+
if (!prev || !cur) return out
|
|
415
|
+
for (const f of fields || []) {
|
|
416
|
+
if (prev[f] != null && cur[f] != null && prev[f] !== cur[f]) out.push(f)
|
|
417
|
+
}
|
|
418
|
+
return out
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
// ---------------------------------------------------------------------------
|
|
422
|
+
// Usage aggregation
|
|
423
|
+
// ---------------------------------------------------------------------------
|
|
424
|
+
|
|
425
|
+
// cacheHitRate = cache.read / (cache.read + cache.write). Returns null when
|
|
426
|
+
// there is no denominator (no read/write tokens observed).
|
|
427
|
+
export function hitRatePct(read, write) {
|
|
428
|
+
const denom = read + write
|
|
429
|
+
if (denom <= 0) return null
|
|
430
|
+
return Math.round((100 * read) / denom)
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
// GLM-5.3 hit ratio vs TOTAL prompt tokens: cached / (read + write + input).
|
|
434
|
+
// Returns null when the denominator is unknown/zero.
|
|
435
|
+
export function glmHitRatio(read, write, input) {
|
|
436
|
+
const denom = read + write + input
|
|
437
|
+
if (denom <= 0 || !Number.isFinite(read)) return null
|
|
438
|
+
return Math.round((100 * read) / denom)
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// Decide whether a `usage` record should be emitted for an aggregation sample.
|
|
442
|
+
// We must not fabricate a zero-valued cache event merely because the session
|
|
443
|
+
// became idle: a sample only counts when at least one assistant message with
|
|
444
|
+
// cache token data was aggregated.
|
|
445
|
+
export function shouldAggregate(count, read, write) {
|
|
446
|
+
return count > 0 && (read > 0 || write > 0)
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
// ---------------------------------------------------------------------------
|
|
450
|
+
// Message scanning / cursor
|
|
451
|
+
//
|
|
452
|
+
// `client.session.messages` returns messages newest-first (confirmed against
|
|
453
|
+
// the runtime). We track a single stable boundary: `lastProcessedMessageID`.
|
|
454
|
+
// Everything NEWER than the boundary is unprocessed; scanning stops as soon as
|
|
455
|
+
// the boundary is reached, so repeated `session.idle` events never double-count
|
|
456
|
+
// historical messages and no unbounded per-message Set is required.
|
|
457
|
+
// ---------------------------------------------------------------------------
|
|
458
|
+
|
|
459
|
+
function reasoningHashesFor(m) {
|
|
460
|
+
const parts = m && m.parts
|
|
461
|
+
if (!Array.isArray(parts)) return []
|
|
462
|
+
const hashes = []
|
|
463
|
+
for (const p of parts) {
|
|
464
|
+
if (p && p.type === "reasoning" && typeof p.text === "string" && p.text.length > 0) {
|
|
465
|
+
hashes.push(shorthash(p.text))
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
return hashes
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
// page: Array<{ info: { id, role, tokens }, parts }>, newest-first.
|
|
472
|
+
// startCursor: lastProcessedMessageID or null (first aggregation).
|
|
473
|
+
// Returns sums for the unprocessed segment plus scan bookkeeping.
|
|
474
|
+
export function scanPage(page, startCursor) {
|
|
475
|
+
let read = 0
|
|
476
|
+
let write = 0
|
|
477
|
+
let input = 0
|
|
478
|
+
let count = 0
|
|
479
|
+
let reachedStart = false
|
|
480
|
+
const seenIds = []
|
|
481
|
+
const reasoning = []
|
|
482
|
+
for (const m of page || []) {
|
|
483
|
+
const info = m && m.info
|
|
484
|
+
const id = info && info.id
|
|
485
|
+
if (typeof id !== "string" || id.length === 0) continue
|
|
486
|
+
if (startCursor != null && id === startCursor) {
|
|
487
|
+
reachedStart = true
|
|
488
|
+
break
|
|
489
|
+
}
|
|
490
|
+
seenIds.push(id)
|
|
491
|
+
if (info.role !== "assistant") continue
|
|
492
|
+
const t = info.tokens
|
|
493
|
+
if (t) {
|
|
494
|
+
read += t.cache && Number.isFinite(t.cache.read) ? t.cache.read : 0
|
|
495
|
+
write += t.cache && Number.isFinite(t.cache.write) ? t.cache.write : 0
|
|
496
|
+
input += Number.isFinite(t.input) ? t.input : 0
|
|
497
|
+
count += 1
|
|
498
|
+
const rh = reasoningHashesFor(m)
|
|
499
|
+
if (rh.length > 0) reasoning.push({ id, hashes: rh })
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
return { read, write, input, count, reachedStart, seenIds, reasoning }
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
// Compute the new `lastProcessedMessageID` after scanning.
|
|
506
|
+
//
|
|
507
|
+
// Messages append at the TOP of a newest-first list, so the correct boundary is
|
|
508
|
+
// the NEWEST message that has been processed (the first element of the first
|
|
509
|
+
// scanned page): the next aggregation scans down from the top and stops as soon
|
|
510
|
+
// as it reaches that boundary. When the tail of the session was reached without
|
|
511
|
+
// ever hitting the boundary (e.g. the old boundary was pruned), we still know
|
|
512
|
+
// every message on the first scanned page was new and processed, so the newest
|
|
513
|
+
// of those becomes the new boundary.
|
|
514
|
+
export function nextProcessedCursor(firstPage, startCursor) {
|
|
515
|
+
if (!Array.isArray(firstPage) || firstPage.length === 0) return startCursor
|
|
516
|
+
const newest = firstPage[0]
|
|
517
|
+
const id = newest && newest.info && newest.info.id
|
|
518
|
+
return typeof id === "string" && id.length > 0 ? id : startCursor
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
// ---------------------------------------------------------------------------
|
|
522
|
+
// Reasoning-integrity diagnostics (GLM-5.3 preserved thinking)
|
|
523
|
+
//
|
|
524
|
+
// INSTRUMENTATION ONLY. The plugin never rewrites, deletes, reorders, or
|
|
525
|
+
// deduplicates reasoning content. GLM preserved thinking is cache-friendly
|
|
526
|
+
// ONLY when the previous reasoning_content is replayed complete, unmodified,
|
|
527
|
+
// and in original order; duplicated or reordered reasoning can explode context
|
|
528
|
+
// and destroy cache efficiency. We detect such anomalies and record counts
|
|
529
|
+
// (never content).
|
|
530
|
+
// ---------------------------------------------------------------------------
|
|
531
|
+
|
|
532
|
+
// currentSeq: reasoning-part hashes of the newest assistant message, in order.
|
|
533
|
+
// lastSeq: reasoning-part hashes of the previous assistant message.
|
|
534
|
+
// seen: Set/Map of reasoning hashes seen in EARLIER messages.
|
|
535
|
+
// Returns counts; reordered is only meaningful when the same multiset of blocks
|
|
536
|
+
// reappears in a different order (a faithful replay that got shuffled).
|
|
537
|
+
// modified is only meaningful when a same-cardinality block sequence partially
|
|
538
|
+
// overlaps the previous one (a replay in which some block content was swapped).
|
|
539
|
+
export function detectReasoningIssues(currentSeq, lastSeq, seen) {
|
|
540
|
+
const out = { withinDuplicates: 0, crossDuplicates: 0, reordered: false, modified: false }
|
|
541
|
+
if (!Array.isArray(currentSeq) || currentSeq.length === 0) return out
|
|
542
|
+
|
|
543
|
+
// duplicated identical reasoning objects within one message
|
|
544
|
+
out.withinDuplicates = currentSeq.length - new Set(currentSeq).size
|
|
545
|
+
|
|
546
|
+
// blocks that were already seen in earlier messages (duplicate replay)
|
|
547
|
+
const firstIdx = new Map()
|
|
548
|
+
currentSeq.forEach((h, i) => {
|
|
549
|
+
if (!firstIdx.has(h)) firstIdx.set(h, i)
|
|
550
|
+
})
|
|
551
|
+
let cross = 0
|
|
552
|
+
for (const [h, i] of firstIdx) {
|
|
553
|
+
if (seen && seen.has && seen.has(h)) cross += 1
|
|
554
|
+
}
|
|
555
|
+
out.crossDuplicates = cross
|
|
556
|
+
|
|
557
|
+
// reordered historical reasoning: same multiset as the previous message but a
|
|
558
|
+
// different sequence (only comparable when both are non-empty and equal-sized)
|
|
559
|
+
if (Array.isArray(lastSeq) && lastSeq.length > 0 && lastSeq.length === currentSeq.length) {
|
|
560
|
+
const sameSet = lastSeq.every((h) => firstIdx.has(h))
|
|
561
|
+
const sameOrder = lastSeq.every((h, i) => h === currentSeq[i])
|
|
562
|
+
if (sameSet && !sameOrder) out.reordered = true
|
|
563
|
+
else if (!sameSet) {
|
|
564
|
+
// same cardinality, partial overlap -> a block was substituted
|
|
565
|
+
const overlap = lastSeq.filter((h) => firstIdx.has(h)).length
|
|
566
|
+
if (overlap > 0 && overlap < lastSeq.length) out.modified = true
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
return out
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
// ---------------------------------------------------------------------------
|
|
573
|
+
// GPT-5.6 cache root + namespace keys
|
|
574
|
+
//
|
|
575
|
+
// The GPT prompt_cache_key should represent the CACHE ROOT of the session tree
|
|
576
|
+
// (the topmost ancestor that a forked/child session shares a prompt prefix
|
|
577
|
+
// with), not the raw session id. In this OpenCode version (1.18.27) the
|
|
578
|
+
// canonical lineage field is Session.parentID (DB session.parent_id); it is
|
|
579
|
+
// populated when a session is created via Session.create({parentID}) -- the
|
|
580
|
+
// task/subagent tool does this (task.ts:159) -- but session.fork (message-copy)
|
|
581
|
+
// does NOT set it (verified empirically: forked children have empty parent_id).
|
|
582
|
+
// So root resolution climbs the parentID chain when present, and otherwise falls
|
|
583
|
+
// back to the session itself as root; the fallback is surfaced in telemetry.
|
|
584
|
+
// ---------------------------------------------------------------------------
|
|
585
|
+
|
|
586
|
+
// provider-safe length cap for prompt_cache_key values (session ids are ~24ch)
|
|
587
|
+
export const GPT_KEY_MAX_LENGTH = 256
|
|
588
|
+
|
|
589
|
+
// Pure root resolver over a synchronous parent lookup (used in tests; the
|
|
590
|
+
// plugin feeds it an async-backed chain built from client.session.get). Walks
|
|
591
|
+
// from `start` up parentID links until none/unknown/cycle/depth-cap.
|
|
592
|
+
// parentOf(id) => parent session id | null | undefined.
|
|
593
|
+
export function resolveCacheRootSync(start, parentOf, { maxHops = 16 } = {}) {
|
|
594
|
+
const seen = new Set()
|
|
595
|
+
let cur = start
|
|
596
|
+
let hops = 0
|
|
597
|
+
let source = "self"
|
|
598
|
+
while (typeof cur === "string" && cur.length > 0 && hops < maxHops) {
|
|
599
|
+
if (seen.has(cur)) {
|
|
600
|
+
source = "cycle"
|
|
601
|
+
break
|
|
602
|
+
}
|
|
603
|
+
seen.add(cur)
|
|
604
|
+
let parent
|
|
605
|
+
try {
|
|
606
|
+
parent = parentOf(cur)
|
|
607
|
+
} catch {
|
|
608
|
+
source = "unknown"
|
|
609
|
+
break
|
|
610
|
+
}
|
|
611
|
+
if (parent == null || typeof parent !== "string" || parent.length === 0 || parent === cur) break
|
|
612
|
+
if (seen.has(parent)) {
|
|
613
|
+
source = "cycle"
|
|
614
|
+
cur = parent
|
|
615
|
+
hops++
|
|
616
|
+
break
|
|
617
|
+
}
|
|
618
|
+
cur = parent
|
|
619
|
+
hops++
|
|
620
|
+
source = "parent"
|
|
621
|
+
}
|
|
622
|
+
if (hops >= maxHops && source === "parent") source = "depth"
|
|
623
|
+
return { root: typeof cur === "string" ? cur : start, hops, source }
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
// GPT prompt_cache_key for a given namespace. Compaction requests for the same
|
|
627
|
+
// cache root get a deterministic, stable, distinct namespace so a compaction
|
|
628
|
+
// cache write never interferes with the live-session cache. Returns null when
|
|
629
|
+
// the derived key would exceed provider constraints.
|
|
630
|
+
export function gptCacheKeyFor(cacheRoot, { compaction = false } = {}) {
|
|
631
|
+
if (typeof cacheRoot !== "string" || cacheRoot.length === 0) return null
|
|
632
|
+
const key = compaction ? `${cacheRoot}:compact` : cacheRoot
|
|
633
|
+
return key.length <= GPT_KEY_MAX_LENGTH ? key : null
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
// ---------------------------------------------------------------------------
|
|
637
|
+
// GPT-5.6 reasoning-effort diagnostics
|
|
638
|
+
//
|
|
639
|
+
// This runtime exposes the effective GPT reasoning effort on the merged options
|
|
640
|
+
// record seen by chat.params as `reasoningEffort` (flat camelCase; the runtime
|
|
641
|
+
// defaults gpt-5.x non-pro models to "medium" in ProviderTransform.options()).
|
|
642
|
+
// We only OBSERVE it across requests for the same session/cache root; we never
|
|
643
|
+
// change it. Unknown/absent is reported as unknown and never fabricates a value
|
|
644
|
+
// or a false change.
|
|
645
|
+
// ---------------------------------------------------------------------------
|
|
646
|
+
|
|
647
|
+
// Extract the reasoning-effort value from the chat.params options record.
|
|
648
|
+
// Returns { known, value } where value is a string when known, else null.
|
|
649
|
+
export function reasoningEffortFromOptions(options) {
|
|
650
|
+
const o = options && typeof options === "object" ? options : {}
|
|
651
|
+
const direct = o.reasoningEffort
|
|
652
|
+
if (typeof direct === "string" && direct.length > 0) return { known: true, value: direct }
|
|
653
|
+
// openrouter-style nesting: reasoning.effort
|
|
654
|
+
const nested = o.reasoning && typeof o.reasoning === "object" ? o.reasoning.effort : undefined
|
|
655
|
+
if (typeof nested === "string" && nested.length > 0) return { known: true, value: nested }
|
|
656
|
+
return { known: false, value: null }
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
// Step the reasoning-effort observer for one request.
|
|
660
|
+
// state: { known, value } (previous observation) or null (first observation).
|
|
661
|
+
// current: output of reasoningEffortFromOptions.
|
|
662
|
+
// Returns { event: "none"|"baseline"|"change", state, previous, current }.
|
|
663
|
+
// Rules:
|
|
664
|
+
// - first observation establishes the baseline (no change event)
|
|
665
|
+
// - unknown current => never a change; baseline stays unknown until a value
|
|
666
|
+
// - known value differing from a known baseline => "change"
|
|
667
|
+
export function observeReasoningEffort(state, current) {
|
|
668
|
+
if (!state || (state.known !== true && state.value === undefined)) {
|
|
669
|
+
return { event: "baseline", state: { known: current.known, value: current.value }, previous: null, current }
|
|
670
|
+
}
|
|
671
|
+
if (!current.known) {
|
|
672
|
+
return { event: "none", state, previous: state, current }
|
|
673
|
+
}
|
|
674
|
+
if (state.known && state.value === current.value) {
|
|
675
|
+
return { event: "none", state, previous: state, current }
|
|
676
|
+
}
|
|
677
|
+
return { event: "change", state: { known: true, value: current.value }, previous: state, current }
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
// ---------------------------------------------------------------------------
|
|
681
|
+
// Boundary reason classification (structured, local diagnostics only)
|
|
682
|
+
//
|
|
683
|
+
// These tokens describe WHICH structural dimension changed. They are causal
|
|
684
|
+
// diagnostics for the *observed* prefix shape -- never a claim that the
|
|
685
|
+
// provider cache key changed or that a cache miss occurred. The authoritative
|
|
686
|
+
// signal remains provider-reported usage.
|
|
687
|
+
// ---------------------------------------------------------------------------
|
|
688
|
+
|
|
689
|
+
// Map granular changed shape fields to structured reason tokens.
|
|
690
|
+
export function prefixChangeReasons(changedFields) {
|
|
691
|
+
const out = []
|
|
692
|
+
for (const f of changedFields || []) {
|
|
693
|
+
if (f === "stableSystemPrefixHash") out.push("system_stable_prefix_changed")
|
|
694
|
+
else if (f === "volatileSystemSuffixHash") out.push("system_volatile_suffix_changed")
|
|
695
|
+
else if (f === "semanticToolsHash") out.push("tools_semantic_changed")
|
|
696
|
+
else if (f === "wireToolsHash") out.push("tools_wire_changed")
|
|
697
|
+
else if (f === "fullSystemHash") {
|
|
698
|
+
// full changed but neither stable nor volatile reported granularly
|
|
699
|
+
if (!changedFields.includes("stableSystemPrefixHash") && !changedFields.includes("volatileSystemSuffixHash")) {
|
|
700
|
+
out.push("system_stable_prefix_changed")
|
|
701
|
+
}
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
if (out.length === 0) out.push("unknown")
|
|
705
|
+
return out
|
|
706
|
+
}
|
|
707
|
+
|
|
708
|
+
// Map reasoning-integrity flags to reason tokens (empty when none).
|
|
709
|
+
export function reasoningIssueReasons(issues) {
|
|
710
|
+
const out = []
|
|
711
|
+
if (!issues) return out
|
|
712
|
+
if (issues.withinDuplicates > 0) out.push("reasoning_duplicate_detected")
|
|
713
|
+
if (issues.crossDuplicates > 0) out.push("reasoning_duplicate_detected")
|
|
714
|
+
if (issues.reordered) out.push("reasoning_reordered")
|
|
715
|
+
if (issues.modified) out.push("reasoning_modified")
|
|
716
|
+
return out
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
// ---------------------------------------------------------------------------
|
|
720
|
+
// Compaction digest dedup guard
|
|
721
|
+
// ---------------------------------------------------------------------------
|
|
722
|
+
|
|
723
|
+
// Returns true when the digest template should be appended for this compaction
|
|
724
|
+
// invocation. Prevents duplicate insertion if the hook fires more than once for
|
|
725
|
+
// the same compaction operation.
|
|
726
|
+
export function digestDecision({ compactTemplate, pendingInsert }) {
|
|
727
|
+
return compactTemplate === true && pendingInsert === true
|
|
728
|
+
}
|