opencode-cache-engine 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,753 @@
1
+ import type { Plugin } from "@opencode-ai/plugin"
2
+ import type { Event, Message, Part } from "@opencode-ai/sdk"
3
+ import {
4
+ DEFAULT_CONFIG_PATH,
5
+ DIGEST_TEMPLATE,
6
+ POLICY_GLM53,
7
+ POLICY_GPT56,
8
+ POLICY_NEUTRAL,
9
+ createRecorder,
10
+ detectPolicy,
11
+ detectReasoningIssues,
12
+ digestDecision,
13
+ ensureMetricsDir,
14
+ glmHitRatio,
15
+ gptCacheKeyFor,
16
+ gptCacheOptionsDelta,
17
+ hitRatePct,
18
+ loadConfig,
19
+ nextProcessedCursor,
20
+ observeReasoningEffort,
21
+ policyEnabled,
22
+ prefixChangeReasons,
23
+ reasoningEffortFromOptions,
24
+ reasoningIssueReasons,
25
+ relocateVolatileEnvBlock,
26
+ resolveCacheRootSync,
27
+ scanPage,
28
+ shapeDiff,
29
+ shapeFieldDiffs,
30
+ shouldAggregate,
31
+ systemShapeHashes,
32
+ toolFingerprint,
33
+ toolWireFingerprint,
34
+ } from "./cache-engine-core.mjs"
35
+
36
+ // ---------------------------------------------------------------------------
37
+ // cache-engine
38
+ //
39
+ // Provider-aware prompt-cache observability + conservative cache-shape
40
+ // preservation for ONE OpenCode TUI across three model families:
41
+ //
42
+ // DeepSeek V4 Flash -> pure passive. >99.66% hit rate is preserved by never
43
+ // mutating system/options/requests. Observability only.
44
+ // GPT-5.6 Luna -> ACTIVE cache-control: a stable session-derived
45
+ // prompt_cache_key + prompt_cache_options (implicit,
46
+ // ttl 30m) injected via the chat.params hook, which is
47
+ // the exact point where the runtime's own options are
48
+ // assembled (request.ts). Never sent to older models.
49
+ // GLM-5.3 Flash -> input-shape strategy: relocate the volatile env block
50
+ // (per-day date) to the tail of the system prompt so a
51
+ // date change only invalidates the suffix, keep tool +
52
+ // history stable, and INSTRUMENT preserved-thinking
53
+ // integrity (duplicate/reorder/modified reasoning). No
54
+ // invented cache key (Z.ai exposes none).
55
+ //
56
+ // The engine remains conservative: it observes, hashes, compares, records,
57
+ // appends a compaction continuation template, and (for GPT-5.6 only) injects
58
+ // documented cache options. It never rewrites message history, reorders tools,
59
+ // or alters user content. DeepSeek and neutral models are byte-untouched.
60
+ //
61
+ // IMPORTANT (terminology): local hashes describe the *observed* prefix shape.
62
+ // A changed hash means request bytes changed; it is NOT proof the provider's
63
+ // cache key changed or that a cache miss occurred. Provider-reported cache
64
+ // token counts are authoritative; hashes are diagnostics only.
65
+ // ---------------------------------------------------------------------------
66
+
67
+ const TOOL_FETCH_TTL_MS = 1500
68
+ const PAGE_SIZE = 100
69
+ const REASONING_SEEN_CAP = 5000
70
+ const ROOT_HOPS_MAX = 16
71
+ const ROOT_CACHE_TTL_MS = 30_000
72
+
73
+ // The runtime plugin client accepts these options even though the v1 SDK type
74
+ // only declares `path.id`/`query`; the empirical call shape is sessionID-based.
75
+ type MessagesOpts = { sessionID: string; limit?: number; before?: string }
76
+ type MessagePage = { info: Message; parts: Part[] }
77
+ type MessagesResult = { data?: MessagePage[]; response?: Response }
78
+
79
+ type ToolDef = { id: string; description: string; parameters: unknown }
80
+
81
+ type Shape = {
82
+ fullSystemHash: string | null
83
+ stableSystemPrefixHash: string | null
84
+ volatileSystemSuffixHash: string | null
85
+ semanticToolsHash: string | null
86
+ wireToolsHash: string | null
87
+ toolCount: number | null
88
+ }
89
+
90
+ type ModelInfo = { family: string; providerID: string; modelID: string }
91
+
92
+ type ToolCache = { semanticToolsHash: string | null; wireToolsHash: string | null }
93
+
94
+ type CacheRoot = { root: string; hops: number; source: string }
95
+
96
+ type EffortObs = { known: boolean; value: string | null }
97
+
98
+ type SessionState = {
99
+ shape: Shape | null
100
+ baselineSystem: string | null
101
+ modelInfo: ModelInfo | null
102
+ tools: ToolCache | null
103
+ toolCount: number | null
104
+ lastToolFetchAt: number | null
105
+ lastProcessedMessageID: string | null
106
+ read: number
107
+ write: number
108
+ input: number
109
+ usageSamples: number
110
+ pendingInsert: boolean
111
+ gptInjected: boolean
112
+ cacheRoot: CacheRoot | null
113
+ cacheRootAt: number | null
114
+ reasoningSeen: Map<string, number>
115
+ reasoningLastSeq: string[] | null
116
+ }
117
+
118
+ const emptyShape = (): Shape => ({
119
+ fullSystemHash: null,
120
+ stableSystemPrefixHash: null,
121
+ volatileSystemSuffixHash: null,
122
+ semanticToolsHash: null,
123
+ wireToolsHash: null,
124
+ toolCount: null,
125
+ })
126
+
127
+ type ChatParamsModel = { providerID: string; id?: string; api?: { id?: string; npm?: string }; name?: string }
128
+
129
+ export const CacheEngine: Plugin = async ({ client, directory }) => {
130
+ const cfg = loadConfig({ configPath: DEFAULT_CONFIG_PATH, env: process.env })
131
+ if (!cfg.enabled) return {}
132
+ ensureMetricsDir(cfg.metricsFile)
133
+ const rec = createRecorder(cfg.metricsFile)
134
+
135
+ const sessions = new Map<string, SessionState>()
136
+ const get = (sid: string): SessionState => {
137
+ let s = sessions.get(sid)
138
+ if (!s) {
139
+ s = {
140
+ shape: null,
141
+ baselineSystem: null,
142
+ modelInfo: null,
143
+ tools: null,
144
+ toolCount: null,
145
+ lastToolFetchAt: null,
146
+ lastProcessedMessageID: null,
147
+ read: 0,
148
+ write: 0,
149
+ input: 0,
150
+ usageSamples: 0,
151
+ pendingInsert: true,
152
+ gptInjected: false,
153
+ cacheRoot: null,
154
+ cacheRootAt: null,
155
+ reasoningSeen: new Map(),
156
+ reasoningLastSeq: null,
157
+ }
158
+ sessions.set(sid, s)
159
+ }
160
+ return s
161
+ }
162
+
163
+ // session.get client call: the v1 plugin client only accepts { path: { id } }.
164
+ // NOTE: methods are prototype methods using `this._client`, so we must invoke
165
+ // them as members (not detach them) or bind the receiver explicitly.
166
+ const sessionGet = (opts: { path: { id: string } }): Promise<{ data?: { parentID?: string | null } }> =>
167
+ (client.session.get as unknown as (o: { path: { id: string } }) => Promise<{ data?: { parentID?: string | null } }>).call(
168
+ client.session,
169
+ opts,
170
+ )
171
+
172
+ // Resolve the cache root for a session by climbing the canonical parentID
173
+ // chain via client.session.get. Memoized per session with a short TTL so the
174
+ // per-request path stays cheap. On any lookup failure we fall back to the
175
+ // session itself as root and surface the reason in telemetry.
176
+ const resolveCacheRoot = async (sid: string): Promise<CacheRoot> => {
177
+ const s = get(sid)
178
+ const now = Date.now()
179
+ if (s.cacheRoot && s.cacheRootAt != null && now - s.cacheRootAt < ROOT_CACHE_TTL_MS) return s.cacheRoot
180
+ try {
181
+ const parentOf = async (id: string): Promise<string | null> => {
182
+ const res = await sessionGet({ path: { id } })
183
+ return res?.data?.parentID && res.data.parentID !== id ? res.data.parentID : null
184
+ }
185
+ // build a synchronous chain resolver fed by async lookups, hop by hop
186
+ const edges = new Map<string, string | null>()
187
+ let cur = sid
188
+ for (let i = 0; i < ROOT_HOPS_MAX; i++) {
189
+ if (edges.has(cur)) break
190
+ const parent = await parentOf(cur)
191
+ edges.set(cur, parent)
192
+ if (parent == null) break
193
+ cur = parent
194
+ }
195
+ const lookup = (id: string): string | null | undefined => {
196
+ // only consult edges we already resolved; unknown -> undefined (stop)
197
+ return edges.has(id) ? edges.get(id) : undefined
198
+ }
199
+ const resolved = resolveCacheRootSync(sid, lookup, { maxHops: ROOT_HOPS_MAX })
200
+ s.cacheRoot = resolved
201
+ s.cacheRootAt = now
202
+ if (resolved.source !== "self") {
203
+ log("info", "cache root resolved", { sid, root: resolved.root, source: resolved.source, hops: resolved.hops })
204
+ }
205
+ return resolved
206
+ } catch (e) {
207
+ rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
208
+ const fallback: CacheRoot = { root: sid, hops: 0, source: "fallback" }
209
+ s.cacheRoot = fallback
210
+ s.cacheRootAt = now
211
+ return fallback
212
+ }
213
+ }
214
+
215
+ // Reasoning-effort diagnostics are per cache root: the value is a property of
216
+ // the request configuration, but forks share the cache root's key namespace.
217
+ const effortByRoot = new Map<string, EffortObs>()
218
+ const observeEffort = (root: string, current: EffortObs) => {
219
+ const prev = effortByRoot.get(root) ?? null
220
+ const step = observeReasoningEffort(prev, current)
221
+ effortByRoot.set(root, step.state)
222
+ return step
223
+ }
224
+
225
+ // Best-effort structured log; must never break a request.
226
+ const log = (level: "debug" | "info" | "warn", message: string, extra: Record<string, unknown>): void => {
227
+ void client?.app
228
+ ?.log({ body: { service: "cache-engine", level, message, extra } })
229
+ .catch(() => {})
230
+ }
231
+
232
+ // Latch the first NON-neutral model observed for a session. Title/summary
233
+ // requests may use the small model; we never let that overwrite a real
234
+ // gpt56/glm53/deepseek classification once established.
235
+ const rememberModel = (sid: string, model: ChatParamsModel | undefined): ModelInfo | null => {
236
+ if (!model) return null
237
+ const family = detectPolicy(model)
238
+ const s = get(sid)
239
+ const info: ModelInfo = {
240
+ family,
241
+ providerID: String(model.providerID ?? ""),
242
+ modelID: String(model.api?.id ?? model.id ?? ""),
243
+ }
244
+ if (s.modelInfo == null || (s.modelInfo.family === POLICY_NEUTRAL && family !== POLICY_NEUTRAL)) {
245
+ s.modelInfo = info
246
+ // A title/summary request (neutral small model) may have established the
247
+ // system baseline first. Its "powered by the model named ..." env line
248
+ // differs from the real model's, so re-baseline on upgrade.
249
+ if (s.baselineSystem !== null && s.modelInfo.family !== POLICY_NEUTRAL) {
250
+ s.baselineSystem = null
251
+ s.shape = null
252
+ }
253
+ }
254
+ return s.modelInfo
255
+ }
256
+
257
+ // Fetch tool definitions (registry order, closest deterministic pre-wire
258
+ // representation available to a plugin). Computes BOTH fingerprints.
259
+ const refreshTools = async (sid: string, s: SessionState, model: { providerID: string; id?: string; api?: { id?: string } } | undefined): Promise<void> => {
260
+ const providerID = model?.providerID
261
+ const apiID = model?.api?.id ?? model?.id
262
+ if (!providerID || !apiID) {
263
+ s.tools = { semanticToolsHash: null, wireToolsHash: null }
264
+ return
265
+ }
266
+ const now = Date.now()
267
+ if (s.lastToolFetchAt != null && now - s.lastToolFetchAt < TOOL_FETCH_TTL_MS) return
268
+ try {
269
+ const res = await (client.tool.list as unknown as (opts: {
270
+ query: { directory: string; provider: string; model: string }
271
+ }) => Promise<{ data?: ToolDef[] }>)({
272
+ query: { directory, provider: providerID, model: apiID },
273
+ })
274
+ s.lastToolFetchAt = now
275
+ s.tools = {
276
+ semanticToolsHash: toolFingerprint(res?.data),
277
+ wireToolsHash: toolWireFingerprint(res?.data),
278
+ }
279
+ s.toolCount = Array.isArray(res?.data) ? res.data.length : null
280
+ } catch {
281
+ // Unknown tool shape: leave fingerprints null (no retry TTL) so a
282
+ // transient failure is never misreported as a tool change.
283
+ s.tools = { semanticToolsHash: null, wireToolsHash: null }
284
+ s.toolCount = null
285
+ }
286
+ }
287
+
288
+ // Aggregate cache tokens + reasoning-integrity signals from all messages
289
+ // newer than the last-processed boundary, paginating until the boundary (or
290
+ // the tail) is reached.
291
+ const collectUsage = async (sid: string): Promise<void> => {
292
+ try {
293
+ const listMessages = (opts: MessagesOpts): Promise<MessagesResult> =>
294
+ (client.session.messages as unknown as (o: MessagesOpts) => Promise<MessagesResult>).call(client.session, opts)
295
+ const s = get(sid)
296
+ const startCursor = s.lastProcessedMessageID
297
+ let before: string | undefined
298
+ let read = 0
299
+ let write = 0
300
+ let input = 0
301
+ let count = 0
302
+ let reachedStart = false
303
+ let firstPage: MessagePage[] | null = null
304
+ let lastPageSize = 0
305
+ const reasoningNewestFirst: { id: string; hashes: string[] }[] = []
306
+ let guard = 0
307
+
308
+ while (guard++ < 200) {
309
+ const res = await listMessages({ sessionID: sid, limit: PAGE_SIZE, before })
310
+ const page = res?.data ?? []
311
+ lastPageSize = page.length
312
+ if (firstPage === null && page.length > 0) firstPage = page
313
+ const scan = scanPage(page, startCursor)
314
+ read += scan.read
315
+ write += scan.write
316
+ input += scan.input
317
+ count += scan.count
318
+ reachedStart = scan.reachedStart
319
+ for (const r of scan.reasoning) reasoningNewestFirst.push(r)
320
+ if (reachedStart) break
321
+
322
+ const next = res?.response?.headers?.get("x-next-cursor")
323
+ if (!next || page.length === 0 || next === before) break
324
+ before = next
325
+ }
326
+
327
+ s.lastProcessedMessageID = nextProcessedCursor(firstPage, startCursor)
328
+
329
+ const family = s.modelInfo?.family
330
+ const glmIntegrity =
331
+ family === POLICY_GLM53 &&
332
+ policyEnabled(cfg, POLICY_GLM53) &&
333
+ cfg.policies?.[POLICY_GLM53]?.preserveThinkingIntegrity === true
334
+
335
+ if (glmIntegrity && reasoningNewestFirst.length > 0) {
336
+ // messages arrive newest-first; process oldest->newest so `seen` grows
337
+ // naturally and `lastSeq` reflects the previous assistant reasoning.
338
+ const chronological = [...reasoningNewestFirst].reverse()
339
+ for (const { id, hashes } of chronological) {
340
+ // within-message duplicate reasoning (duplicated reasoning blocks)
341
+ const issues = detectReasoningIssues(hashes, s.reasoningLastSeq, s.reasoningSeen)
342
+ // `seen` holds hashes from EARLIER messages; refresh it AFTER the check.
343
+ if (issues.crossDuplicates > 0 || issues.withinDuplicates > 0 || issues.reordered || issues.modified) {
344
+ const rReasons = reasoningIssueReasons(issues)
345
+ rec.record({
346
+ kind: "reasoning-integrity",
347
+ sid,
348
+ ts: Date.now(),
349
+ provider: s.modelInfo?.providerID,
350
+ model: s.modelInfo?.modelID,
351
+ policy: s.modelInfo?.family,
352
+ reason: rReasons[0] ?? "reasoning_changed",
353
+ reasons: rReasons,
354
+ msgId: id,
355
+ withinDuplicates: issues.withinDuplicates,
356
+ crossDuplicates: issues.crossDuplicates,
357
+ reordered: issues.reordered,
358
+ modified: issues.modified,
359
+ })
360
+ }
361
+ for (const h of new Set(hashes)) {
362
+ s.reasoningSeen.set(h, (s.reasoningSeen.get(h) ?? 0) + 1)
363
+ }
364
+ if (s.reasoningSeen.size > REASONING_SEEN_CAP) {
365
+ // bounded bookkeeping: drop the oldest half of the map
366
+ const drop = [...s.reasoningSeen.keys()].slice(0, Math.floor(s.reasoningSeen.size / 2))
367
+ for (const k of drop) s.reasoningSeen.delete(k)
368
+ }
369
+ if (hashes.length > 0) s.reasoningLastSeq = hashes
370
+ }
371
+ }
372
+
373
+ if (shouldAggregate(count, read, write)) {
374
+ s.read += read
375
+ s.write += write
376
+ s.input += input
377
+ s.usageSamples += count
378
+ const recFields: Record<string, unknown> = {
379
+ kind: "usage",
380
+ sid,
381
+ ts: Date.now(),
382
+ read,
383
+ write,
384
+ input,
385
+ messages: count,
386
+ sampleHitRate: hitRatePct(read, write),
387
+ cumulative: { read: s.read, write: s.write },
388
+ cumulativeHitRate: hitRatePct(s.read, s.write),
389
+ cursor: s.lastProcessedMessageID,
390
+ }
391
+ if (s.modelInfo) {
392
+ recFields.provider = s.modelInfo.providerID
393
+ recFields.model = s.modelInfo.modelID
394
+ recFields.policy = s.modelInfo.family
395
+ }
396
+ if (family === POLICY_GLM53) {
397
+ recFields.promptTokens = read + write + input
398
+ recFields.glmHitRate = glmHitRatio(read, write, input)
399
+ }
400
+ if (family === POLICY_GPT56 && s.gptInjected) {
401
+ recFields.keyStrategy = "session"
402
+ recFields.mode = cfg.policies?.[POLICY_GPT56]?.mode
403
+ recFields.ttl = cfg.policies?.[POLICY_GPT56]?.ttl
404
+ }
405
+ rec.record(recFields)
406
+ } else {
407
+ // No cache data observed for new messages (or none at all). We
408
+ // deliberately do NOT emit a zero-valued usage record here.
409
+ log("debug", "idle: no new assistant usage", {
410
+ sid,
411
+ messagesScanned: lastPageSize,
412
+ newAssistantMessages: count,
413
+ reachedBoundary: reachedStart,
414
+ })
415
+ }
416
+ } catch (e) {
417
+ rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
418
+ }
419
+ }
420
+
421
+ const prefixFields = (shape: Shape, toolCount: number | null) => ({
422
+ fullSystemHash: shape.fullSystemHash,
423
+ stableSystemPrefixHash: shape.stableSystemPrefixHash,
424
+ volatileSystemSuffixHash: shape.volatileSystemSuffixHash,
425
+ semanticToolsHash: shape.semanticToolsHash,
426
+ wireToolsHash: shape.wireToolsHash,
427
+ toolCount,
428
+ // legacy aliases for backward compatibility with earlier dashboards
429
+ systemHash: shape.fullSystemHash,
430
+ toolsHash: shape.semanticToolsHash,
431
+ })
432
+
433
+ return {
434
+ // Per-request model context: latch family/model for telemetry and, for
435
+ // GPT-5.6, inject the cache-root-derived prompt-cache key + implicit cache
436
+ // options at the exact point the runtime assembles the outgoing options.
437
+ //
438
+ // Cache root affinity: a forked/child session inherits the topmost
439
+ // ancestor's key (via Session.parentID chain) rather than the raw session
440
+ // id, so a fork that shares the parent's prompt prefix reuses the parent's
441
+ // GPT cache. Compaction requests (agent === "compaction") for the same root
442
+ // use a deterministic separate namespace (<root>:compact) so a compaction
443
+ // cache write never interferes with the useful live-session cache.
444
+ "chat.params": async (input, output) => {
445
+ try {
446
+ const info = rememberModel(input.sessionID, input.model as unknown as ChatParamsModel)
447
+ const family = info?.family
448
+ if (!(family === POLICY_GPT56 && policyEnabled(cfg, POLICY_GPT56))) {
449
+ // DeepSeek / GLM / neutral: nothing to inject. GLM has no cache-key API;
450
+ // DeepSeek caching is fully passive; we never mutate requests for them.
451
+ return
452
+ }
453
+ const gpol = cfg.policies?.[POLICY_GPT56]
454
+ const applyRoot = gpol?.cacheRootKey !== false
455
+ const compaction = input.agent === "compaction"
456
+ const isolated = compaction && gpol?.compactionCacheIsolation === true
457
+ const enableKey = gpol?.promptCacheKey !== false
458
+
459
+ // cache root resolution (only when used for key/effort baseline)
460
+ const rootRes = applyRoot || gpol?.reasoningEffortDiagnostics === true ? await resolveCacheRoot(input.sessionID) : null
461
+ const keyBase = applyRoot ? rootRes!.root : input.sessionID
462
+ const desiredKey = enableKey ? gptCacheKeyFor(keyBase, { compaction: isolated }) : null
463
+
464
+ // ---- reasoning-effort diagnostics (observe only, never change) ----
465
+ // The effort setting is a request-config property; we track it per
466
+ // cache root so forks compare against the same lineage baseline.
467
+ if (gpol?.reasoningEffortDiagnostics === true) {
468
+ const cur = reasoningEffortFromOptions(output.options)
469
+ const effKey = applyRoot ? rootRes!.root : input.sessionID
470
+ const step = observeEffort(effKey, cur)
471
+ if (step.event === "change") {
472
+ rec.record({
473
+ kind: "boundary",
474
+ sid: input.sessionID,
475
+ ts: Date.now(),
476
+ reason: "gpt_reasoning_effort_changed",
477
+ provider: info.providerID,
478
+ model: info.modelID,
479
+ policy: family,
480
+ cacheRoot: effKey,
481
+ cacheRootSource: applyRoot ? rootRes!.source : "self",
482
+ before: step.previous?.known ? step.previous.value : null,
483
+ after: step.current?.known ? step.current.value : null,
484
+ })
485
+ }
486
+ }
487
+
488
+ if (!enableKey || !desiredKey) return
489
+
490
+ // ---- cache-root affinity + compaction isolation key injection --------
491
+ const already = output.options.promptCacheKey
492
+ if (applyRoot && (already === undefined || already !== desiredKey)) {
493
+ output.options.promptCacheKey = desiredKey
494
+ }
495
+ const optsDelta = gptCacheOptionsDelta(output.options, {
496
+ key: desiredKey,
497
+ mode: gpol?.mode,
498
+ ttl: gpol?.ttl,
499
+ })
500
+ if (Object.keys(optsDelta).length > 0) Object.assign(output.options, optsDelta)
501
+
502
+ const s = get(input.sessionID)
503
+ const cacheCtxExtra = applyRoot
504
+ ? {
505
+ cacheRoot: rootRes!.root,
506
+ cacheRootSource: rootRes!.source,
507
+ cacheRootHops: rootRes!.hops,
508
+ }
509
+ : {
510
+ cacheRoot: input.sessionID,
511
+ cacheRootSource: "session" as const,
512
+ }
513
+
514
+ if (!s.gptInjected) {
515
+ s.gptInjected = true
516
+
517
+ rec.record({
518
+ kind: "cache-options",
519
+ sid: input.sessionID,
520
+ ts: Date.now(),
521
+ policy: POLICY_GPT56,
522
+ provider: info.providerID,
523
+ model: info.modelID,
524
+ keyStrategy: applyRoot ? "cache-root" : "session",
525
+ ...cacheCtxExtra,
526
+ compaction,
527
+ namespace: isolated ? "compact" : "live",
528
+ key: desiredKey,
529
+ options: { ...(output.options.promptCacheOptions ?? {}) },
530
+ })
531
+
532
+ if (applyRoot && rootRes!.source === "parent" && rootRes!.hops > 0) {
533
+ rec.record({
534
+ kind: "boundary",
535
+ sid: input.sessionID,
536
+ ts: Date.now(),
537
+ reason: "gpt_session_fork",
538
+ provider: info.providerID,
539
+ model: info.modelID,
540
+ policy: family,
541
+ ...cacheCtxExtra,
542
+ hops: rootRes!.hops,
543
+ })
544
+ }
545
+ }
546
+
547
+ if (compaction) {
548
+ rec.record({
549
+ kind: "boundary",
550
+ sid: input.sessionID,
551
+ ts: Date.now(),
552
+ reason: "gpt_compaction",
553
+ provider: info.providerID,
554
+ model: info.modelID,
555
+ policy: family,
556
+ ...cacheCtxExtra,
557
+ namespace: isolated ? "compact" : "live",
558
+ key: desiredKey,
559
+ })
560
+ }
561
+ } catch (e) {
562
+ rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
563
+ }
564
+ },
565
+
566
+ event: async ({ event }: { event: Event }) => {
567
+ try {
568
+ if (event.type === "session.idle") {
569
+ void collectUsage(event.properties.sessionID)
570
+ return
571
+ }
572
+ if (event.type === "session.compacted") {
573
+ const sid = event.properties.sessionID
574
+ const s = get(sid)
575
+ s.pendingInsert = true
576
+ rec.record({
577
+ kind: "compaction",
578
+ sid,
579
+ ts: Date.now(),
580
+ reason: "compaction",
581
+ usageSamples: s.usageSamples,
582
+ cumulative: { read: s.read, write: s.write },
583
+ ...(s.cacheRoot ? { cacheRoot: s.cacheRoot.root, cacheRootSource: s.cacheRoot.source } : {}),
584
+ ...(s.modelInfo
585
+ ? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
586
+ : {}),
587
+ })
588
+ return
589
+ }
590
+ // Diagnostic only. `session.usage.updated` is emitted by the runtime
591
+ // (session-cumulative totals) but is not part of the typed Event union;
592
+ // it is never used as the authoritative aggregation path.
593
+ if ((event as { type?: string }).type === "session.usage.updated") {
594
+ const e = event as unknown as {
595
+ properties?: { sessionID?: string }
596
+ data?: { sessionID?: string; cost?: number; tokens?: { cache?: { read: number; write: number } } }
597
+ }
598
+ const t = e?.data?.tokens
599
+ if (t?.cache) {
600
+ const sid = e?.properties?.sessionID ?? e?.data?.sessionID
601
+ const s = sid ? get(sid) : null
602
+ rec.record({
603
+ kind: "usage-event",
604
+ sid,
605
+ ts: Date.now(),
606
+ read: t.cache.read,
607
+ write: t.cache.write,
608
+ cost: e?.data?.cost,
609
+ ...(s?.modelInfo
610
+ ? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
611
+ : {}),
612
+ })
613
+ }
614
+ }
615
+ } catch (e) {
616
+ rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
617
+ }
618
+ },
619
+
620
+ "experimental.chat.system.transform": async (input, output) => {
621
+ try {
622
+ const sid = input.sessionID
623
+ if (!sid) return
624
+ const model = input.model as unknown as ChatParamsModel
625
+ rememberModel(sid, model)
626
+ const s = get(sid)
627
+ const family = s.modelInfo?.family
628
+
629
+ // ---- GLM-5.3 input-shape stabilization -----------------------------
630
+ // Relocate the identifiable volatile env block (per-day date) to the
631
+ // tail of the single system string, content-preserving, ONLY when the
632
+ // model is GLM-5.3 and the block markers are present exactly. Never
633
+ // touches other content/order; never applied to other families.
634
+ //
635
+ // In-place mutation note: request.ts keeps using its own local `system`
636
+ // array after the hook (the trigger's returned output is ignored), so
637
+ // reassigning `output.system = [...]` would be lost. We rewrite the
638
+ // single element in place instead.
639
+ let systemText = output.system.join("\n")
640
+ if (
641
+ family === POLICY_GLM53 &&
642
+ policyEnabled(cfg, POLICY_GLM53) &&
643
+ cfg.policies?.[POLICY_GLM53]?.stabilizeSystem === true &&
644
+ output.system.length === 1
645
+ ) {
646
+ const rel = relocateVolatileEnvBlock(output.system[0])
647
+ if (rel.changed) {
648
+ output.system[0] = rel.text
649
+ systemText = rel.text
650
+ log("debug", "glm system env block relocated to suffix", { sid })
651
+ }
652
+ }
653
+
654
+ const cur: Shape = { ...emptyShape() }
655
+ const curHashes = systemShapeHashes(s.baselineSystem ?? systemText, systemText)
656
+ cur.fullSystemHash = curHashes.fullSystemHash
657
+ cur.stableSystemPrefixHash = curHashes.stableSystemPrefixHash
658
+ cur.volatileSystemSuffixHash = curHashes.volatileSystemSuffixHash
659
+ if (s.baselineSystem === null) s.baselineSystem = systemText
660
+ await refreshTools(sid, s, model as { providerID: string; id?: string; api?: { id?: string } } | undefined)
661
+ cur.semanticToolsHash = s.tools?.semanticToolsHash ?? null
662
+ cur.wireToolsHash = s.tools?.wireToolsHash ?? null
663
+ cur.toolCount = s.toolCount
664
+ const toolCount = s.toolCount
665
+
666
+ const cacheCtx = (root: CacheRoot | null) =>
667
+ root ? { cacheRoot: root.root, cacheRootSource: root.source, cacheRootHops: root.hops } : {}
668
+
669
+ const prev = s.shape
670
+ if (prev === null) {
671
+ // First observation for this session: establish the baseline. The
672
+ // absence of a previous hash is not a change.
673
+ s.shape = cur
674
+ rec.record({
675
+ kind: "prefix-observation",
676
+ sid,
677
+ ts: Date.now(),
678
+ ...(s.modelInfo
679
+ ? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
680
+ : {}),
681
+ ...cacheCtx(s.cacheRoot),
682
+ ...prefixFields(cur, toolCount),
683
+ note: "initial",
684
+ })
685
+ return
686
+ }
687
+
688
+ const changed = shapeDiff(prev, cur)
689
+ if (changed.length > 0) {
690
+ const granular = shapeFieldDiffs(prev, cur, [
691
+ "fullSystemHash",
692
+ "stableSystemPrefixHash",
693
+ "volatileSystemSuffixHash",
694
+ "semanticToolsHash",
695
+ "wireToolsHash",
696
+ ])
697
+ const reasons = prefixChangeReasons(granular)
698
+ const before: Record<string, unknown> = {}
699
+ const after: Record<string, unknown> = {}
700
+ for (const f of granular) {
701
+ before[f] = (prev as Record<string, unknown>)[f]
702
+ after[f] = (cur as Record<string, unknown>)[f]
703
+ }
704
+ // richer tool diagnostics: does this change carry a tool-count shift,
705
+ // a semantic tool change, and/or a wire (ordering/serialization) change?
706
+ const toolsChanged = granular.includes("semanticToolsHash") || granular.includes("wireToolsHash")
707
+ const semanticChanged = granular.includes("semanticToolsHash")
708
+ const wireChanged = granular.includes("wireToolsHash")
709
+ const prevCount = prev.toolCount ?? null
710
+ rec.record({
711
+ kind: "prefix-change",
712
+ sid,
713
+ ts: Date.now(),
714
+ ...(s.modelInfo
715
+ ? { provider: s.modelInfo.providerID, model: s.modelInfo.modelID, policy: s.modelInfo.family }
716
+ : {}),
717
+ ...cacheCtx(s.cacheRoot),
718
+ dimensions: changed,
719
+ changedFields: granular,
720
+ reasons,
721
+ before,
722
+ after,
723
+ ...(toolsChanged
724
+ ? { toolCount, prevToolCount: prevCount, semanticToolsChanged: semanticChanged, wireToolsChanged: wireChanged }
725
+ : {}),
726
+ })
727
+ if (cfg.logPrefixChanges) {
728
+ log("warn", "observed prefix shape change", {
729
+ sid,
730
+ dimensions: changed,
731
+ changedFields: granular,
732
+ reasons,
733
+ })
734
+ }
735
+ }
736
+ s.shape = cur
737
+ } catch (e) {
738
+ rec.record({ kind: "telemetry-error", ts: Date.now(), error: String(e) })
739
+ }
740
+ },
741
+
742
+ "experimental.session.compacting": async (input, output) => {
743
+ try {
744
+ const s = get(input.sessionID)
745
+ if (!digestDecision({ compactTemplate: cfg.compactTemplate, pendingInsert: s.pendingInsert })) return
746
+ output.context.push(DIGEST_TEMPLATE)
747
+ s.pendingInsert = false
748
+ } catch {
749
+ /* best-effort */
750
+ }
751
+ },
752
+ }
753
+ }