opencode-cache-engine 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +1032 -0
- package/examples/cache-engine.json +25 -0
- package/opencode-cache-engine-0.1.0.tgz +0 -0
- package/package.json +33 -0
- package/src/cache-engine-core.mjs +728 -0
- package/src/cache-engine.ts +753 -0
- package/test/cache-engine.test.mjs +765 -0
|
@@ -0,0 +1,753 @@
|
|
|
1
|
+
import type { Plugin } from "@opencode-ai/plugin"
|
|
2
|
+
import type { Event, Message, Part } from "@opencode-ai/sdk"
|
|
3
|
+
import {
|
|
4
|
+
DEFAULT_CONFIG_PATH,
|
|
5
|
+
DIGEST_TEMPLATE,
|
|
6
|
+
POLICY_GLM53,
|
|
7
|
+
POLICY_GPT56,
|
|
8
|
+
POLICY_NEUTRAL,
|
|
9
|
+
createRecorder,
|
|
10
|
+
detectPolicy,
|
|
11
|
+
detectReasoningIssues,
|
|
12
|
+
digestDecision,
|
|
13
|
+
ensureMetricsDir,
|
|
14
|
+
glmHitRatio,
|
|
15
|
+
gptCacheKeyFor,
|
|
16
|
+
gptCacheOptionsDelta,
|
|
17
|
+
hitRatePct,
|
|
18
|
+
loadConfig,
|
|
19
|
+
nextProcessedCursor,
|
|
20
|
+
observeReasoningEffort,
|
|
21
|
+
policyEnabled,
|
|
22
|
+
prefixChangeReasons,
|
|
23
|
+
reasoningEffortFromOptions,
|
|
24
|
+
reasoningIssueReasons,
|
|
25
|
+
relocateVolatileEnvBlock,
|
|
26
|
+
resolveCacheRootSync,
|
|
27
|
+
scanPage,
|
|
28
|
+
shapeDiff,
|
|
29
|
+
shapeFieldDiffs,
|
|
30
|
+
shouldAggregate,
|
|
31
|
+
systemShapeHashes,
|
|
32
|
+
toolFingerprint,
|
|
33
|
+
toolWireFingerprint,
|
|
34
|
+
} from "./cache-engine-core.mjs"
|
|
35
|
+
|
|
36
|
+
// ---------------------------------------------------------------------------
|
|
37
|
+
// cache-engine
|
|
38
|
+
//
|
|
39
|
+
// Provider-aware prompt-cache observability + conservative cache-shape
|
|
40
|
+
// preservation for ONE OpenCode TUI across three model families:
|
|
41
|
+
//
|
|
42
|
+
// DeepSeek V4 Flash -> pure passive. >99.66% hit rate is preserved by never
|
|
43
|
+
// mutating system/options/requests. Observability only.
|
|
44
|
+
// GPT-5.6 Luna -> ACTIVE cache-control: a stable session-derived
|
|
45
|
+
// prompt_cache_key + prompt_cache_options (implicit,
|
|
46
|
+
// ttl 30m) injected via the chat.params hook, which is
|
|
47
|
+
// the exact point where the runtime's own options are
|
|
48
|
+
// assembled (request.ts). Never sent to older models.
|
|
49
|
+
// GLM-5.3 Flash -> input-shape strategy: relocate the volatile env block
|
|
50
|
+
// (per-day date) to the tail of the system prompt so a
|
|
51
|
+
// date change only invalidates the suffix, keep tool +
|
|
52
|
+
// history stable, and INSTRUMENT preserved-thinking
|
|
53
|
+
// integrity (duplicate/reorder/modified reasoning). No
|
|
54
|
+
// invented cache key (Z.ai exposes none).
|
|
55
|
+
//
|
|
56
|
+
// The engine remains conservative: it observes, hashes, compares, records,
|
|
57
|
+
// appends a compaction continuation template, and (for GPT-5.6 only) injects
|
|
58
|
+
// documented cache options. It never rewrites message history, reorders tools,
|
|
59
|
+
// or alters user content. DeepSeek and neutral models are byte-untouched.
|
|
60
|
+
//
|
|
61
|
+
// IMPORTANT (terminology): local hashes describe the *observed* prefix shape.
|
|
62
|
+
// A changed hash means request bytes changed; it is NOT proof the provider's
|
|
63
|
+
// cache key changed or that a cache miss occurred. Provider-reported cache
|
|
64
|
+
// token counts are authoritative; hashes are diagnostics only.
|
|
65
|
+
// ---------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
const TOOL_FETCH_TTL_MS = 1500
|
|
68
|
+
const PAGE_SIZE = 100
|
|
69
|
+
const REASONING_SEEN_CAP = 5000
|
|
70
|
+
const ROOT_HOPS_MAX = 16
|
|
71
|
+
const ROOT_CACHE_TTL_MS = 30_000
|
|
72
|
+
|
|
73
|
+
// The runtime plugin client accepts these options even though the v1 SDK type
|
|
74
|
+
// only declares `path.id`/`query`; the empirical call shape is sessionID-based.
|
|
75
|
+
type MessagesOpts = { sessionID: string; limit?: number; before?: string }
|
|
76
|
+
type MessagePage = { info: Message; parts: Part[] }
|
|
77
|
+
type MessagesResult = { data?: MessagePage[]; response?: Response }
|
|
78
|
+
|
|
79
|
+
type ToolDef = { id: string; description: string; parameters: unknown }
|
|
80
|
+
|
|
81
|
+
type Shape = {
|
|
82
|
+
fullSystemHash: string | null
|
|
83
|
+
stableSystemPrefixHash: string | null
|
|
84
|
+
volatileSystemSuffixHash: string | null
|
|
85
|
+
semanticToolsHash: string | null
|
|
86
|
+
wireToolsHash: string | null
|
|
87
|
+
toolCount: number | null
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
type ModelInfo = { family: string; providerID: string; modelID: string }
|
|
91
|
+
|
|
92
|
+
type ToolCache = { semanticToolsHash: string | null; wireToolsHash: string | null }
|
|
93
|
+
|
|
94
|
+
type CacheRoot = { root: string; hops: number; source: string }
|
|
95
|
+
|
|
96
|
+
type EffortObs = { known: boolean; value: string | null }
|
|
97
|
+
|
|
98
|
+
type SessionState = {
|
|
99
|
+
shape: Shape | null
|
|
100
|
+
baselineSystem: string | null
|
|
101
|
+
modelInfo: ModelInfo | null
|
|
102
|
+
tools: ToolCache | null
|
|
103
|
+
toolCount: number | null
|
|
104
|
+
lastToolFetchAt: number | null
|
|
105
|
+
lastProcessedMessageID: string | null
|
|
106
|
+
read: number
|
|
107
|
+
write: number
|
|
108
|
+
input: number
|
|
109
|
+
usageSamples: number
|
|
110
|
+
pendingInsert: boolean
|
|
111
|
+
gptInjected: boolean
|
|
112
|
+
cacheRoot: CacheRoot | null
|
|
113
|
+
cacheRootAt: number | null
|
|
114
|
+
reasoningSeen: Map<string, number>
|
|
115
|
+
reasoningLastSeq: string[] | null
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const emptyShape = (): Shape => ({
|
|
119
|
+
fullSystemHash: null,
|
|
120
|
+
stableSystemPrefixHash: null,
|
|
121
|
+
volatileSystemSuffixHash: null,
|
|
122
|
+
semanticToolsHash: null,
|
|
123
|
+
wireToolsHash: null,
|
|
124
|
+
toolCount: null,
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
type ChatParamsModel = { providerID: string; id?: string; api?: { id?: string; npm?: string }; name?: string }
|
|
128
|
+
|
|
129
|
+
export const CacheEngine: Plugin = async ({ client, directory }) => {
|
|
130
|
+
const cfg = loadConfig({ configPath: DEFAULT_CONFIG_PATH, env: process.env })
|
|
131
|
+
if (!cfg.enabled) return {}
|
|
132
|
+
ensureMetricsDir(cfg.metricsFile)
|
|
133
|
+
const rec = createRecorder(cfg.metricsFile)
|
|
134
|
+
|
|
135
|
+
const sessions = new Map<string, SessionState>()
|
|
136
|
+
const get = (sid: string): SessionState => {
|
|
137
|
+
let s = sessions.get(sid)
|
|
138
|
+
if (!s) {
|
|
139
|
+
s = {
|
|
140
|
+
shape: null,
|
|
141
|
+
baselineSystem: null,
|
|
142
|
+
modelInfo: null,
|
|
143
|
+
tools: null,
|
|
144
|
+
toolCount: null,
|
|
145
|
+
lastToolFetchAt: null,
|
|
146
|
+
lastProcessedMessageID: null,
|
|
147
|
+
read: 0,
|
|
148
|
+
write: 0,
|
|
149
|
+
input: 0,
|
|
150
|
+
usageSamples: 0,
|
|
151
|
+
pendingInsert: true,
|
|
152
|
+
gptInjected: false,
|
|
153
|
+
cacheRoot: null,
|
|
154
|
+
cacheRootAt: null,
|
|
155
|
+
reasoningSeen: new Map(),
|
|
156
|
+
reasoningLastSeq: null,
|
|
157
|
+
}
|
|
158
|
+
sessions.set(sid, s)
|
|
159
|
+
}
|
|
160
|
+
return s
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// session.get client call: the v1 plugin client only accepts { path: { id } }.
|
|
164
|
+
// NOTE: methods are prototype methods using `this._client`, so we must invoke
|
|
165
|
+
// them as members (not detach them) or bind the receiver explicitly.
|
|
166
|
+
const sessionGet = (opts: { path: { id: string } }): Promise<{ data?: { parentID?: string | null } }> =>
|
|
167
|
+
(client.session.get as unknown as (o: { path: { id: string } }) => Promise<{ data?: { parentID?: string | null } }>).call(
|
|
168
|
+
client.session,
|
|
169
|
+
opts,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
// Resolve the cache root for a session by climbing the canonical parentID
|
|
173
|
+
// chain via client.session.get. Memoized per session with a short TTL so the
|
|
174
|
+
// per-request path stays cheap. On any lookup failure we fall back to the
|
|
175
|
+
// session itself as root and surface the reason in telemetry.
|
|
176
|
+
const resolveCacheRoot = async (sid: string): Promise<CacheRoot> => {
|
|
177
|
+
const s = get(sid)
|
|
178
|
+
const now = Date.now()
|
|
179
|
+
if (s.cacheRoot && s.cacheRootAt != null && now - s.cacheRootAt < ROOT_CACHE_TTL_MS) return s.cacheRoot
|
|
180
|
+
try {
|
|
181
|
+
const parentOf = async (id: string): Promise<string | null> => {
|
|
182
|
+
const res = await sessionGet({ path: { id } })
|
|
183
|
+
return res?.data?.parentID && res.data.parentID !== id ? res.data.parentID : null
|
|
184
|
+
}
|
|
185
|
+
// build a synchronous chain resolver fed by async lookups, hop by hop
|
|
186
|
+
const edges = new Map<string, string | null>()
|
|
187
|
+
let cur = sid
|
|
188
|
+
for (let i = 0; i < ROOT_HOPS_MAX; i++) {
|
|
189
|
+
if (edges.has(cur)) break
|
|
190
|
+
const parent = await parentOf(cur)
|
|
191
|
+
edges.set(cur, parent)
|
|
192
|
+
if (parent == null) break
|
|
193
|
+
cur = parent
|
|
194
|
+
}
|
|
195
|
+
const lookup = (id: string): string | null | undefined => {
|
|
196
|
+
// only consult edges we already resolved; unknown -> undefined (stop)
|
|
197
|
+
return edges.has(id) ? edges.get(id) : undefined
|
|
198
|
+
}
|
|
199
|
+
const resolved = resolveCacheRootSync(sid, lookup, { maxHops: ROOT_HOPS_MAX })
|
|
200
|
+
s.cacheRoot = resolved
|
|
201
|
+
s.cacheRootAt = now
|
|
202
|
+
if (resolved.source !== "self") {
|
|
203
|
+
log("info", "cache root resolved", { sid, root: resolved.root, source: resolved.source, hops: resolved.hops })
|
|
204
|
+
}
|
|
205
|
+
return resolved
|
|
206
|
+
} catch (e) {
|
|
207
|
+
rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
|
|
208
|
+
const fallback: CacheRoot = { root: sid, hops: 0, source: "fallback" }
|
|
209
|
+
s.cacheRoot = fallback
|
|
210
|
+
s.cacheRootAt = now
|
|
211
|
+
return fallback
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// Reasoning-effort diagnostics are per cache root: the value is a property of
|
|
216
|
+
// the request configuration, but forks share the cache root's key namespace.
|
|
217
|
+
const effortByRoot = new Map<string, EffortObs>()
|
|
218
|
+
const observeEffort = (root: string, current: EffortObs) => {
|
|
219
|
+
const prev = effortByRoot.get(root) ?? null
|
|
220
|
+
const step = observeReasoningEffort(prev, current)
|
|
221
|
+
effortByRoot.set(root, step.state)
|
|
222
|
+
return step
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// Best-effort structured log; must never break a request.
|
|
226
|
+
const log = (level: "debug" | "info" | "warn", message: string, extra: Record<string, unknown>): void => {
|
|
227
|
+
void client?.app
|
|
228
|
+
?.log({ body: { service: "cache-engine", level, message, extra } })
|
|
229
|
+
.catch(() => {})
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// Latch the first NON-neutral model observed for a session. Title/summary
|
|
233
|
+
// requests may use the small model; we never let that overwrite a real
|
|
234
|
+
// gpt56/glm53/deepseek classification once established.
|
|
235
|
+
const rememberModel = (sid: string, model: ChatParamsModel | undefined): ModelInfo | null => {
|
|
236
|
+
if (!model) return null
|
|
237
|
+
const family = detectPolicy(model)
|
|
238
|
+
const s = get(sid)
|
|
239
|
+
const info: ModelInfo = {
|
|
240
|
+
family,
|
|
241
|
+
providerID: String(model.providerID ?? ""),
|
|
242
|
+
modelID: String(model.api?.id ?? model.id ?? ""),
|
|
243
|
+
}
|
|
244
|
+
if (s.modelInfo == null || (s.modelInfo.family === POLICY_NEUTRAL && family !== POLICY_NEUTRAL)) {
|
|
245
|
+
s.modelInfo = info
|
|
246
|
+
// A title/summary request (neutral small model) may have established the
|
|
247
|
+
// system baseline first. Its "powered by the model named ..." env line
|
|
248
|
+
// differs from the real model's, so re-baseline on upgrade.
|
|
249
|
+
if (s.baselineSystem !== null && s.modelInfo.family !== POLICY_NEUTRAL) {
|
|
250
|
+
s.baselineSystem = null
|
|
251
|
+
s.shape = null
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
return s.modelInfo
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
// Fetch tool definitions (registry order, closest deterministic pre-wire
|
|
258
|
+
// representation available to a plugin). Computes BOTH fingerprints.
|
|
259
|
+
const refreshTools = async (sid: string, s: SessionState, model: { providerID: string; id?: string; api?: { id?: string } } | undefined): Promise<void> => {
|
|
260
|
+
const providerID = model?.providerID
|
|
261
|
+
const apiID = model?.api?.id ?? model?.id
|
|
262
|
+
if (!providerID || !apiID) {
|
|
263
|
+
s.tools = { semanticToolsHash: null, wireToolsHash: null }
|
|
264
|
+
return
|
|
265
|
+
}
|
|
266
|
+
const now = Date.now()
|
|
267
|
+
if (s.lastToolFetchAt != null && now - s.lastToolFetchAt < TOOL_FETCH_TTL_MS) return
|
|
268
|
+
try {
|
|
269
|
+
const res = await (client.tool.list as unknown as (opts: {
|
|
270
|
+
query: { directory: string; provider: string; model: string }
|
|
271
|
+
}) => Promise<{ data?: ToolDef[] }>)({
|
|
272
|
+
query: { directory, provider: providerID, model: apiID },
|
|
273
|
+
})
|
|
274
|
+
s.lastToolFetchAt = now
|
|
275
|
+
s.tools = {
|
|
276
|
+
semanticToolsHash: toolFingerprint(res?.data),
|
|
277
|
+
wireToolsHash: toolWireFingerprint(res?.data),
|
|
278
|
+
}
|
|
279
|
+
s.toolCount = Array.isArray(res?.data) ? res.data.length : null
|
|
280
|
+
} catch {
|
|
281
|
+
// Unknown tool shape: leave fingerprints null (no retry TTL) so a
|
|
282
|
+
// transient failure is never misreported as a tool change.
|
|
283
|
+
s.tools = { semanticToolsHash: null, wireToolsHash: null }
|
|
284
|
+
s.toolCount = null
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
// Aggregate cache tokens + reasoning-integrity signals from all messages
|
|
289
|
+
// newer than the last-processed boundary, paginating until the boundary (or
|
|
290
|
+
// the tail) is reached.
|
|
291
|
+
const collectUsage = async (sid: string): Promise<void> => {
|
|
292
|
+
try {
|
|
293
|
+
const listMessages = (opts: MessagesOpts): Promise<MessagesResult> =>
|
|
294
|
+
(client.session.messages as unknown as (o: MessagesOpts) => Promise<MessagesResult>).call(client.session, opts)
|
|
295
|
+
const s = get(sid)
|
|
296
|
+
const startCursor = s.lastProcessedMessageID
|
|
297
|
+
let before: string | undefined
|
|
298
|
+
let read = 0
|
|
299
|
+
let write = 0
|
|
300
|
+
let input = 0
|
|
301
|
+
let count = 0
|
|
302
|
+
let reachedStart = false
|
|
303
|
+
let firstPage: MessagePage[] | null = null
|
|
304
|
+
let lastPageSize = 0
|
|
305
|
+
const reasoningNewestFirst: { id: string; hashes: string[] }[] = []
|
|
306
|
+
let guard = 0
|
|
307
|
+
|
|
308
|
+
while (guard++ < 200) {
|
|
309
|
+
const res = await listMessages({ sessionID: sid, limit: PAGE_SIZE, before })
|
|
310
|
+
const page = res?.data ?? []
|
|
311
|
+
lastPageSize = page.length
|
|
312
|
+
if (firstPage === null && page.length > 0) firstPage = page
|
|
313
|
+
const scan = scanPage(page, startCursor)
|
|
314
|
+
read += scan.read
|
|
315
|
+
write += scan.write
|
|
316
|
+
input += scan.input
|
|
317
|
+
count += scan.count
|
|
318
|
+
reachedStart = scan.reachedStart
|
|
319
|
+
for (const r of scan.reasoning) reasoningNewestFirst.push(r)
|
|
320
|
+
if (reachedStart) break
|
|
321
|
+
|
|
322
|
+
const next = res?.response?.headers?.get("x-next-cursor")
|
|
323
|
+
if (!next || page.length === 0 || next === before) break
|
|
324
|
+
before = next
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
s.lastProcessedMessageID = nextProcessedCursor(firstPage, startCursor)
|
|
328
|
+
|
|
329
|
+
const family = s.modelInfo?.family
|
|
330
|
+
const glmIntegrity =
|
|
331
|
+
family === POLICY_GLM53 &&
|
|
332
|
+
policyEnabled(cfg, POLICY_GLM53) &&
|
|
333
|
+
cfg.policies?.[POLICY_GLM53]?.preserveThinkingIntegrity === true
|
|
334
|
+
|
|
335
|
+
if (glmIntegrity && reasoningNewestFirst.length > 0) {
|
|
336
|
+
// messages arrive newest-first; process oldest->newest so `seen` grows
|
|
337
|
+
// naturally and `lastSeq` reflects the previous assistant reasoning.
|
|
338
|
+
const chronological = [...reasoningNewestFirst].reverse()
|
|
339
|
+
for (const { id, hashes } of chronological) {
|
|
340
|
+
// within-message duplicate reasoning (duplicated reasoning blocks)
|
|
341
|
+
const issues = detectReasoningIssues(hashes, s.reasoningLastSeq, s.reasoningSeen)
|
|
342
|
+
// `seen` holds hashes from EARLIER messages; refresh it AFTER the check.
|
|
343
|
+
if (issues.crossDuplicates > 0 || issues.withinDuplicates > 0 || issues.reordered || issues.modified) {
|
|
344
|
+
const rReasons = reasoningIssueReasons(issues)
|
|
345
|
+
rec.record({
|
|
346
|
+
kind: "reasoning-integrity",
|
|
347
|
+
sid,
|
|
348
|
+
ts: Date.now(),
|
|
349
|
+
provider: s.modelInfo?.providerID,
|
|
350
|
+
model: s.modelInfo?.modelID,
|
|
351
|
+
policy: s.modelInfo?.family,
|
|
352
|
+
reason: rReasons[0] ?? "reasoning_changed",
|
|
353
|
+
reasons: rReasons,
|
|
354
|
+
msgId: id,
|
|
355
|
+
withinDuplicates: issues.withinDuplicates,
|
|
356
|
+
crossDuplicates: issues.crossDuplicates,
|
|
357
|
+
reordered: issues.reordered,
|
|
358
|
+
modified: issues.modified,
|
|
359
|
+
})
|
|
360
|
+
}
|
|
361
|
+
for (const h of new Set(hashes)) {
|
|
362
|
+
s.reasoningSeen.set(h, (s.reasoningSeen.get(h) ?? 0) + 1)
|
|
363
|
+
}
|
|
364
|
+
if (s.reasoningSeen.size > REASONING_SEEN_CAP) {
|
|
365
|
+
// bounded bookkeeping: drop the oldest half of the map
|
|
366
|
+
const drop = [...s.reasoningSeen.keys()].slice(0, Math.floor(s.reasoningSeen.size / 2))
|
|
367
|
+
for (const k of drop) s.reasoningSeen.delete(k)
|
|
368
|
+
}
|
|
369
|
+
if (hashes.length > 0) s.reasoningLastSeq = hashes
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
if (shouldAggregate(count, read, write)) {
|
|
374
|
+
s.read += read
|
|
375
|
+
s.write += write
|
|
376
|
+
s.input += input
|
|
377
|
+
s.usageSamples += count
|
|
378
|
+
const recFields: Record<string, unknown> = {
|
|
379
|
+
kind: "usage",
|
|
380
|
+
sid,
|
|
381
|
+
ts: Date.now(),
|
|
382
|
+
read,
|
|
383
|
+
write,
|
|
384
|
+
input,
|
|
385
|
+
messages: count,
|
|
386
|
+
sampleHitRate: hitRatePct(read, write),
|
|
387
|
+
cumulative: { read: s.read, write: s.write },
|
|
388
|
+
cumulativeHitRate: hitRatePct(s.read, s.write),
|
|
389
|
+
cursor: s.lastProcessedMessageID,
|
|
390
|
+
}
|
|
391
|
+
if (s.modelInfo) {
|
|
392
|
+
recFields.provider = s.modelInfo.providerID
|
|
393
|
+
recFields.model = s.modelInfo.modelID
|
|
394
|
+
recFields.policy = s.modelInfo.family
|
|
395
|
+
}
|
|
396
|
+
if (family === POLICY_GLM53) {
|
|
397
|
+
recFields.promptTokens = read + write + input
|
|
398
|
+
recFields.glmHitRate = glmHitRatio(read, write, input)
|
|
399
|
+
}
|
|
400
|
+
if (family === POLICY_GPT56 && s.gptInjected) {
|
|
401
|
+
recFields.keyStrategy = "session"
|
|
402
|
+
recFields.mode = cfg.policies?.[POLICY_GPT56]?.mode
|
|
403
|
+
recFields.ttl = cfg.policies?.[POLICY_GPT56]?.ttl
|
|
404
|
+
}
|
|
405
|
+
rec.record(recFields)
|
|
406
|
+
} else {
|
|
407
|
+
// No cache data observed for new messages (or none at all). We
|
|
408
|
+
// deliberately do NOT emit a zero-valued usage record here.
|
|
409
|
+
log("debug", "idle: no new assistant usage", {
|
|
410
|
+
sid,
|
|
411
|
+
messagesScanned: lastPageSize,
|
|
412
|
+
newAssistantMessages: count,
|
|
413
|
+
reachedBoundary: reachedStart,
|
|
414
|
+
})
|
|
415
|
+
}
|
|
416
|
+
} catch (e) {
|
|
417
|
+
rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
const prefixFields = (shape: Shape, toolCount: number | null) => ({
|
|
422
|
+
fullSystemHash: shape.fullSystemHash,
|
|
423
|
+
stableSystemPrefixHash: shape.stableSystemPrefixHash,
|
|
424
|
+
volatileSystemSuffixHash: shape.volatileSystemSuffixHash,
|
|
425
|
+
semanticToolsHash: shape.semanticToolsHash,
|
|
426
|
+
wireToolsHash: shape.wireToolsHash,
|
|
427
|
+
toolCount,
|
|
428
|
+
// legacy aliases for backward compatibility with earlier dashboards
|
|
429
|
+
systemHash: shape.fullSystemHash,
|
|
430
|
+
toolsHash: shape.semanticToolsHash,
|
|
431
|
+
})
|
|
432
|
+
|
|
433
|
+
return {
|
|
434
|
+
// Per-request model context: latch family/model for telemetry and, for
|
|
435
|
+
// GPT-5.6, inject the cache-root-derived prompt-cache key + implicit cache
|
|
436
|
+
// options at the exact point the runtime assembles the outgoing options.
|
|
437
|
+
//
|
|
438
|
+
// Cache root affinity: a forked/child session inherits the topmost
|
|
439
|
+
// ancestor's key (via Session.parentID chain) rather than the raw session
|
|
440
|
+
// id, so a fork that shares the parent's prompt prefix reuses the parent's
|
|
441
|
+
// GPT cache. Compaction requests (agent === "compaction") for the same root
|
|
442
|
+
// use a deterministic separate namespace (<root>:compact) so a compaction
|
|
443
|
+
// cache write never interferes with the useful live-session cache.
|
|
444
|
+
"chat.params": async (input, output) => {
|
|
445
|
+
try {
|
|
446
|
+
const info = rememberModel(input.sessionID, input.model as unknown as ChatParamsModel)
|
|
447
|
+
const family = info?.family
|
|
448
|
+
if (!(family === POLICY_GPT56 && policyEnabled(cfg, POLICY_GPT56))) {
|
|
449
|
+
// DeepSeek / GLM / neutral: nothing to inject. GLM has no cache-key API;
|
|
450
|
+
// DeepSeek caching is fully passive; we never mutate requests for them.
|
|
451
|
+
return
|
|
452
|
+
}
|
|
453
|
+
const gpol = cfg.policies?.[POLICY_GPT56]
|
|
454
|
+
const applyRoot = gpol?.cacheRootKey !== false
|
|
455
|
+
const compaction = input.agent === "compaction"
|
|
456
|
+
const isolated = compaction && gpol?.compactionCacheIsolation === true
|
|
457
|
+
const enableKey = gpol?.promptCacheKey !== false
|
|
458
|
+
|
|
459
|
+
// cache root resolution (only when used for key/effort baseline)
|
|
460
|
+
const rootRes = applyRoot || gpol?.reasoningEffortDiagnostics === true ? await resolveCacheRoot(input.sessionID) : null
|
|
461
|
+
const keyBase = applyRoot ? rootRes!.root : input.sessionID
|
|
462
|
+
const desiredKey = enableKey ? gptCacheKeyFor(keyBase, { compaction: isolated }) : null
|
|
463
|
+
|
|
464
|
+
// ---- reasoning-effort diagnostics (observe only, never change) ----
|
|
465
|
+
// The effort setting is a request-config property; we track it per
|
|
466
|
+
// cache root so forks compare against the same lineage baseline.
|
|
467
|
+
if (gpol?.reasoningEffortDiagnostics === true) {
|
|
468
|
+
const cur = reasoningEffortFromOptions(output.options)
|
|
469
|
+
const effKey = applyRoot ? rootRes!.root : input.sessionID
|
|
470
|
+
const step = observeEffort(effKey, cur)
|
|
471
|
+
if (step.event === "change") {
|
|
472
|
+
rec.record({
|
|
473
|
+
kind: "boundary",
|
|
474
|
+
sid: input.sessionID,
|
|
475
|
+
ts: Date.now(),
|
|
476
|
+
reason: "gpt_reasoning_effort_changed",
|
|
477
|
+
provider: info.providerID,
|
|
478
|
+
model: info.modelID,
|
|
479
|
+
policy: family,
|
|
480
|
+
cacheRoot: effKey,
|
|
481
|
+
cacheRootSource: applyRoot ? rootRes!.source : "self",
|
|
482
|
+
before: step.previous?.known ? step.previous.value : null,
|
|
483
|
+
after: step.current?.known ? step.current.value : null,
|
|
484
|
+
})
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
if (!enableKey || !desiredKey) return
|
|
489
|
+
|
|
490
|
+
// ---- cache-root affinity + compaction isolation key injection --------
|
|
491
|
+
const already = output.options.promptCacheKey
|
|
492
|
+
if (applyRoot && (already === undefined || already !== desiredKey)) {
|
|
493
|
+
output.options.promptCacheKey = desiredKey
|
|
494
|
+
}
|
|
495
|
+
const optsDelta = gptCacheOptionsDelta(output.options, {
|
|
496
|
+
key: desiredKey,
|
|
497
|
+
mode: gpol?.mode,
|
|
498
|
+
ttl: gpol?.ttl,
|
|
499
|
+
})
|
|
500
|
+
if (Object.keys(optsDelta).length > 0) Object.assign(output.options, optsDelta)
|
|
501
|
+
|
|
502
|
+
const s = get(input.sessionID)
|
|
503
|
+
const cacheCtxExtra = applyRoot
|
|
504
|
+
? {
|
|
505
|
+
cacheRoot: rootRes!.root,
|
|
506
|
+
cacheRootSource: rootRes!.source,
|
|
507
|
+
cacheRootHops: rootRes!.hops,
|
|
508
|
+
}
|
|
509
|
+
: {
|
|
510
|
+
cacheRoot: input.sessionID,
|
|
511
|
+
cacheRootSource: "session" as const,
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
if (!s.gptInjected) {
|
|
515
|
+
s.gptInjected = true
|
|
516
|
+
|
|
517
|
+
rec.record({
|
|
518
|
+
kind: "cache-options",
|
|
519
|
+
sid: input.sessionID,
|
|
520
|
+
ts: Date.now(),
|
|
521
|
+
policy: POLICY_GPT56,
|
|
522
|
+
provider: info.providerID,
|
|
523
|
+
model: info.modelID,
|
|
524
|
+
keyStrategy: applyRoot ? "cache-root" : "session",
|
|
525
|
+
...cacheCtxExtra,
|
|
526
|
+
compaction,
|
|
527
|
+
namespace: isolated ? "compact" : "live",
|
|
528
|
+
key: desiredKey,
|
|
529
|
+
options: { ...(output.options.promptCacheOptions ?? {}) },
|
|
530
|
+
})
|
|
531
|
+
|
|
532
|
+
if (applyRoot && rootRes!.source === "parent" && rootRes!.hops > 0) {
|
|
533
|
+
rec.record({
|
|
534
|
+
kind: "boundary",
|
|
535
|
+
sid: input.sessionID,
|
|
536
|
+
ts: Date.now(),
|
|
537
|
+
reason: "gpt_session_fork",
|
|
538
|
+
provider: info.providerID,
|
|
539
|
+
model: info.modelID,
|
|
540
|
+
policy: family,
|
|
541
|
+
...cacheCtxExtra,
|
|
542
|
+
hops: rootRes!.hops,
|
|
543
|
+
})
|
|
544
|
+
}
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
if (compaction) {
|
|
548
|
+
rec.record({
|
|
549
|
+
kind: "boundary",
|
|
550
|
+
sid: input.sessionID,
|
|
551
|
+
ts: Date.now(),
|
|
552
|
+
reason: "gpt_compaction",
|
|
553
|
+
provider: info.providerID,
|
|
554
|
+
model: info.modelID,
|
|
555
|
+
policy: family,
|
|
556
|
+
...cacheCtxExtra,
|
|
557
|
+
namespace: isolated ? "compact" : "live",
|
|
558
|
+
key: desiredKey,
|
|
559
|
+
})
|
|
560
|
+
}
|
|
561
|
+
} catch (e) {
|
|
562
|
+
rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
|
|
563
|
+
}
|
|
564
|
+
},
|
|
565
|
+
|
|
566
|
+
event: async ({ event }: { event: Event }) => {
|
|
567
|
+
try {
|
|
568
|
+
if (event.type === "session.idle") {
|
|
569
|
+
void collectUsage(event.properties.sessionID)
|
|
570
|
+
return
|
|
571
|
+
}
|
|
572
|
+
if (event.type === "session.compacted") {
|
|
573
|
+
const sid = event.properties.sessionID
|
|
574
|
+
const s = get(sid)
|
|
575
|
+
s.pendingInsert = true
|
|
576
|
+
rec.record({
|
|
577
|
+
kind: "compaction",
|
|
578
|
+
sid,
|
|
579
|
+
ts: Date.now(),
|
|
580
|
+
reason: "compaction",
|
|
581
|
+
usageSamples: s.usageSamples,
|
|
582
|
+
cumulative: { read: s.read, write: s.write },
|
|
583
|
+
...(s.cacheRoot ? { cacheRoot: s.cacheRoot.root, cacheRootSource: s.cacheRoot.source } : {}),
|
|
584
|
+
...(s.modelInfo
|
|
585
|
+
? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
|
|
586
|
+
: {}),
|
|
587
|
+
})
|
|
588
|
+
return
|
|
589
|
+
}
|
|
590
|
+
// Diagnostic only. `session.usage.updated` is emitted by the runtime
|
|
591
|
+
// (session-cumulative totals) but is not part of the typed Event union;
|
|
592
|
+
// it is never used as the authoritative aggregation path.
|
|
593
|
+
if ((event as { type?: string }).type === "session.usage.updated") {
|
|
594
|
+
const e = event as unknown as {
|
|
595
|
+
properties?: { sessionID?: string }
|
|
596
|
+
data?: { sessionID?: string; cost?: number; tokens?: { cache?: { read: number; write: number } } }
|
|
597
|
+
}
|
|
598
|
+
const t = e?.data?.tokens
|
|
599
|
+
if (t?.cache) {
|
|
600
|
+
const sid = e?.properties?.sessionID ?? e?.data?.sessionID
|
|
601
|
+
const s = sid ? get(sid) : null
|
|
602
|
+
rec.record({
|
|
603
|
+
kind: "usage-event",
|
|
604
|
+
sid,
|
|
605
|
+
ts: Date.now(),
|
|
606
|
+
read: t.cache.read,
|
|
607
|
+
write: t.cache.write,
|
|
608
|
+
cost: e?.data?.cost,
|
|
609
|
+
...(s?.modelInfo
|
|
610
|
+
? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
|
|
611
|
+
: {}),
|
|
612
|
+
})
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
} catch (e) {
|
|
616
|
+
rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
|
|
617
|
+
}
|
|
618
|
+
},
|
|
619
|
+
|
|
620
|
+
"experimental.chat.system.transform": async (input, output) => {
|
|
621
|
+
try {
|
|
622
|
+
const sid = input.sessionID
|
|
623
|
+
if (!sid) return
|
|
624
|
+
const model = input.model as unknown as ChatParamsModel
|
|
625
|
+
rememberModel(sid, model)
|
|
626
|
+
const s = get(sid)
|
|
627
|
+
const family = s.modelInfo?.family
|
|
628
|
+
|
|
629
|
+
// ---- GLM-5.3 input-shape stabilization -----------------------------
|
|
630
|
+
// Relocate the identifiable volatile env block (per-day date) to the
|
|
631
|
+
// tail of the single system string, content-preserving, ONLY when the
|
|
632
|
+
// model is GLM-5.3 and the block markers are present exactly. Never
|
|
633
|
+
// touches other content/order; never applied to other families.
|
|
634
|
+
//
|
|
635
|
+
// In-place mutation note: request.ts keeps using its own local `system`
|
|
636
|
+
// array after the hook (the trigger's returned output is ignored), so
|
|
637
|
+
// reassigning `output.system = [...]` would be lost. We rewrite the
|
|
638
|
+
// single element in place instead.
|
|
639
|
+
let systemText = output.system.join("\n")
|
|
640
|
+
if (
|
|
641
|
+
family === POLICY_GLM53 &&
|
|
642
|
+
policyEnabled(cfg, POLICY_GLM53) &&
|
|
643
|
+
cfg.policies?.[POLICY_GLM53]?.stabilizeSystem === true &&
|
|
644
|
+
output.system.length === 1
|
|
645
|
+
) {
|
|
646
|
+
const rel = relocateVolatileEnvBlock(output.system[0])
|
|
647
|
+
if (rel.changed) {
|
|
648
|
+
output.system[0] = rel.text
|
|
649
|
+
systemText = rel.text
|
|
650
|
+
log("debug", "glm system env block relocated to suffix", { sid })
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
const cur: Shape = { ...emptyShape() }
|
|
655
|
+
const curHashes = systemShapeHashes(s.baselineSystem ?? systemText, systemText)
|
|
656
|
+
cur.fullSystemHash = curHashes.fullSystemHash
|
|
657
|
+
cur.stableSystemPrefixHash = curHashes.stableSystemPrefixHash
|
|
658
|
+
cur.volatileSystemSuffixHash = curHashes.volatileSystemSuffixHash
|
|
659
|
+
if (s.baselineSystem === null) s.baselineSystem = systemText
|
|
660
|
+
await refreshTools(sid, s, model as { providerID: string; id?: string; api?: { id?: string } } | undefined)
|
|
661
|
+
cur.semanticToolsHash = s.tools?.semanticToolsHash ?? null
|
|
662
|
+
cur.wireToolsHash = s.tools?.wireToolsHash ?? null
|
|
663
|
+
cur.toolCount = s.toolCount
|
|
664
|
+
const toolCount = s.toolCount
|
|
665
|
+
|
|
666
|
+
const cacheCtx = (root: CacheRoot | null) =>
|
|
667
|
+
root ? { cacheRoot: root.root, cacheRootSource: root.source, cacheRootHops: root.hops } : {}
|
|
668
|
+
|
|
669
|
+
const prev = s.shape
|
|
670
|
+
if (prev === null) {
|
|
671
|
+
// First observation for this session: establish the baseline. The
|
|
672
|
+
// absence of a previous hash is not a change.
|
|
673
|
+
s.shape = cur
|
|
674
|
+
rec.record({
|
|
675
|
+
kind: "prefix-observation",
|
|
676
|
+
sid,
|
|
677
|
+
ts: Date.now(),
|
|
678
|
+
...(s.modelInfo
|
|
679
|
+
? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
|
|
680
|
+
: {}),
|
|
681
|
+
...cacheCtx(s.cacheRoot),
|
|
682
|
+
...prefixFields(cur, toolCount),
|
|
683
|
+
note: "initial",
|
|
684
|
+
})
|
|
685
|
+
return
|
|
686
|
+
}
|
|
687
|
+
|
|
688
|
+
const changed = shapeDiff(prev, cur)
|
|
689
|
+
if (changed.length > 0) {
|
|
690
|
+
const granular = shapeFieldDiffs(prev, cur, [
|
|
691
|
+
"fullSystemHash",
|
|
692
|
+
"stableSystemPrefixHash",
|
|
693
|
+
"volatileSystemSuffixHash",
|
|
694
|
+
"semanticToolsHash",
|
|
695
|
+
"wireToolsHash",
|
|
696
|
+
])
|
|
697
|
+
const reasons = prefixChangeReasons(granular)
|
|
698
|
+
const before: Record<string, unknown> = {}
|
|
699
|
+
const after: Record<string, unknown> = {}
|
|
700
|
+
for (const f of granular) {
|
|
701
|
+
before[f] = (prev as Record<string, unknown>)[f]
|
|
702
|
+
after[f] = (cur as Record<string, unknown>)[f]
|
|
703
|
+
}
|
|
704
|
+
// richer tool diagnostics: does this change carry a tool-count shift,
|
|
705
|
+
// a semantic tool change, and/or a wire (ordering/serialization) change?
|
|
706
|
+
const toolsChanged = granular.includes("semanticToolsHash") || granular.includes("wireToolsHash")
|
|
707
|
+
const semanticChanged = granular.includes("semanticToolsHash")
|
|
708
|
+
const wireChanged = granular.includes("wireToolsHash")
|
|
709
|
+
const prevCount = prev.toolCount ?? null
|
|
710
|
+
rec.record({
|
|
711
|
+
kind: "prefix-change",
|
|
712
|
+
sid,
|
|
713
|
+
ts: Date.now(),
|
|
714
|
+
...(s.modelInfo
|
|
715
|
+
? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
|
|
716
|
+
: {}),
|
|
717
|
+
...cacheCtx(s.cacheRoot),
|
|
718
|
+
dimensions: changed,
|
|
719
|
+
changedFields: granular,
|
|
720
|
+
reasons,
|
|
721
|
+
before,
|
|
722
|
+
after,
|
|
723
|
+
...(toolsChanged
|
|
724
|
+
? { toolCount, prevToolCount: prevCount, semanticToolsChanged: semanticChanged, wireToolsChanged: wireChanged }
|
|
725
|
+
: {}),
|
|
726
|
+
})
|
|
727
|
+
if (cfg.logPrefixChanges) {
|
|
728
|
+
log("warn", "observed prefix shape change", {
|
|
729
|
+
sid,
|
|
730
|
+
dimensions: changed,
|
|
731
|
+
changedFields: granular,
|
|
732
|
+
reasons,
|
|
733
|
+
})
|
|
734
|
+
}
|
|
735
|
+
}
|
|
736
|
+
s.shape = cur
|
|
737
|
+
} catch (e) {
|
|
738
|
+
rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
|
|
739
|
+
}
|
|
740
|
+
},
|
|
741
|
+
|
|
742
|
+
"experimental.session.compacting": async (input, output) => {
|
|
743
|
+
try {
|
|
744
|
+
const s = get(input.sessionID)
|
|
745
|
+
if (!digestDecision({ compactTemplate: cfg.compactTemplate, pendingInsert: s.pendingInsert })) return
|
|
746
|
+
output.context.push(DIGEST_TEMPLATE)
|
|
747
|
+
s.pendingInsert = false
|
|
748
|
+
} catch {
|
|
749
|
+
/* best-effort */
|
|
750
|
+
}
|
|
751
|
+
},
|
|
752
|
+
}
|
|
753
|
+
}
|