@bitkyc08/opencodex 2.55.0 → 2.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/gui/dist/assets/{index-VuoiWj9J.js → index-D4zuyIxQ.js} +1 -1
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +2 -1
  4. package/src/adapters/base.ts +21 -0
  5. package/src/adapters/cursor/transport-retry.ts +46 -1
  6. package/src/adapters/cursor.ts +4 -0
  7. package/src/adapters/kiro/adapter.ts +42 -1
  8. package/src/adapters/kiro-retry.ts +23 -4
  9. package/src/adapters/openai-chat/errors.ts +116 -0
  10. package/src/adapters/openai-chat/messages.ts +346 -0
  11. package/src/adapters/openai-chat/passthrough.ts +146 -0
  12. package/src/adapters/openai-chat/response-events.ts +117 -0
  13. package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
  14. package/src/adapters/openai-chat/tool-schema.ts +477 -0
  15. package/src/adapters/openai-chat/wire.ts +50 -0
  16. package/src/adapters/openai-chat.ts +33 -1445
  17. package/src/adapters/openai-responses/canonical-forward.ts +202 -0
  18. package/src/adapters/openai-responses/image-gen.ts +406 -0
  19. package/src/adapters/openai-responses/internal.ts +3 -0
  20. package/src/adapters/openai-responses/passthrough.ts +611 -0
  21. package/src/adapters/openai-responses/prompt-cache.ts +83 -0
  22. package/src/adapters/openai-responses/reasoning.ts +220 -0
  23. package/src/adapters/openai-responses/request-strips.ts +185 -0
  24. package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
  25. package/src/adapters/openai-responses/tool-schema.ts +293 -0
  26. package/src/adapters/openai-responses/web-search.ts +156 -0
  27. package/src/adapters/openai-responses.ts +4 -2625
  28. package/src/bridge/errors.ts +34 -0
  29. package/src/bridge/internal.ts +174 -0
  30. package/src/bridge/response-json.ts +624 -0
  31. package/src/bridge/sse.ts +1444 -0
  32. package/src/bridge.ts +5 -2204
  33. package/src/chat/inbound.ts +12 -1
  34. package/src/codex/account-lifecycle.ts +3 -0
  35. package/src/codex/account-store.ts +71 -9
  36. package/src/codex/auth-api/account-list.ts +507 -0
  37. package/src/codex/auth-api/http.ts +32 -0
  38. package/src/codex/auth-api/login-flow.ts +554 -0
  39. package/src/codex/auth-api/login-state.ts +64 -0
  40. package/src/codex/auth-api/main-account-probe.ts +331 -0
  41. package/src/codex/auth-api/pool-mode-gate.ts +274 -0
  42. package/src/codex/auth-api/pool-quota-probe.ts +512 -0
  43. package/src/codex/auth-api/reset-credit-service.ts +422 -0
  44. package/src/codex/auth-api/routes.ts +425 -0
  45. package/src/codex/auth-api/runtime-config.ts +48 -0
  46. package/src/codex/auth-api.ts +27 -3118
  47. package/src/codex/auth-context.ts +95 -28
  48. package/src/codex/catalog/auto-review.ts +507 -0
  49. package/src/codex/catalog/build-entries.ts +981 -0
  50. package/src/codex/catalog/combo-member.ts +375 -0
  51. package/src/codex/catalog/derive-entry.ts +229 -0
  52. package/src/codex/catalog/effort.ts +0 -1
  53. package/src/codex/catalog/gated-native-warn.ts +63 -0
  54. package/src/codex/catalog/gather-capture.ts +533 -0
  55. package/src/codex/catalog/model-hints.ts +691 -0
  56. package/src/codex/catalog/model-visibility.ts +304 -0
  57. package/src/codex/catalog/provider-fetch.ts +52 -2942
  58. package/src/codex/catalog/provider-models.ts +685 -0
  59. package/src/codex/catalog/restore.ts +132 -0
  60. package/src/codex/catalog/retained-sync.ts +706 -0
  61. package/src/codex/catalog/routed-gather.ts +858 -0
  62. package/src/codex/catalog/subagent-roster.ts +176 -0
  63. package/src/codex/catalog/sync.ts +52 -2698
  64. package/src/codex/inject/config-toml.ts +563 -0
  65. package/src/codex/inject/remove.ts +192 -0
  66. package/src/codex/inject/restore.ts +540 -0
  67. package/src/codex/inject/routing-classify.ts +109 -0
  68. package/src/codex/inject/routing-target.ts +125 -0
  69. package/src/codex/inject.ts +81 -1436
  70. package/src/codex/lineage.ts +458 -0
  71. package/src/codex/pool-refresh-backoff.ts +152 -0
  72. package/src/codex/routing/active-account.ts +194 -0
  73. package/src/codex/routing/cooldown-math.ts +275 -0
  74. package/src/codex/routing/health-store.ts +402 -0
  75. package/src/codex/routing/probe-lease.ts +358 -0
  76. package/src/codex/routing/selection.ts +703 -0
  77. package/src/codex/routing/thread-affinity.ts +538 -0
  78. package/src/codex/routing.ts +353 -2234
  79. package/src/codex/shim-fingerprint.ts +223 -0
  80. package/src/codex/shim-inspect.ts +175 -0
  81. package/src/codex/shim-probe.ts +367 -0
  82. package/src/codex/shim-restore-lock.ts +169 -0
  83. package/src/codex/shim-state-file.ts +151 -0
  84. package/src/codex/shim-templates.ts +265 -0
  85. package/src/codex/shim.ts +48 -1268
  86. package/src/config/diagnostics.ts +705 -0
  87. package/src/config/feature-flags.ts +55 -0
  88. package/src/config/live-reconcile.ts +403 -0
  89. package/src/config/load-degrade.ts +880 -0
  90. package/src/config/mutation-lock.ts +244 -0
  91. package/src/config/openai-tier-backup.ts +268 -0
  92. package/src/config/persist-unlocked.ts +92 -0
  93. package/src/config/proxy-env.ts +188 -0
  94. package/src/config/salvage.ts +244 -0
  95. package/src/config/schema/config-schema.ts +640 -0
  96. package/src/config/schema/leaf-validators.ts +855 -0
  97. package/src/config/warn-memo.ts +28 -0
  98. package/src/config.ts +234 -4481
  99. package/src/generated/compatibility-version.json +539 -39
  100. package/src/lib/request-execution-budget.ts +69 -20
  101. package/src/lib/spend-reservation-ledger.ts +940 -0
  102. package/src/lib/upstream-retry.ts +55 -11
  103. package/src/lib/workflow-budget.ts +553 -30
  104. package/src/providers/quota/account-cache.ts +441 -0
  105. package/src/providers/quota/antigravity.ts +295 -0
  106. package/src/providers/quota/report-cache.ts +320 -0
  107. package/src/providers/quota/vendor-probes-key.ts +1243 -0
  108. package/src/providers/quota/vendor-probes-oauth.ts +590 -0
  109. package/src/providers/quota.ts +324 -3079
  110. package/src/providers/registry/entries-core.ts +1221 -0
  111. package/src/providers/registry/entries-extended.ts +1204 -0
  112. package/src/providers/registry/model-seeds.ts +908 -0
  113. package/src/providers/registry/types.ts +352 -0
  114. package/src/providers/registry.ts +24 -3536
  115. package/src/responses/continuation-ownership.ts +29 -0
  116. package/src/responses/state/replay-fingerprint.ts +80 -0
  117. package/src/responses/state/snapshot-codec.ts +104 -0
  118. package/src/responses/state/spill-failure.ts +118 -0
  119. package/src/responses/state/spill-queue.ts +665 -0
  120. package/src/responses/state/temp-recovery.ts +257 -0
  121. package/src/responses/state.ts +82 -1143
  122. package/src/routing/identity-domains.ts +449 -0
  123. package/src/routing/probe-lease.ts +511 -0
  124. package/src/server/index/bounded-request.ts +88 -0
  125. package/src/server/index/live-sideband.ts +565 -0
  126. package/src/server/index/serve-options.ts +1766 -0
  127. package/src/server/index/startup-warnings.ts +213 -0
  128. package/src/server/index/websocket-handler.ts +335 -0
  129. package/src/server/index.ts +40 -2547
  130. package/src/server/management/route-registry.ts +26 -23
  131. package/src/server/management/shared.ts +8 -5
  132. package/src/server/management/workflow-budget-routes.ts +133 -0
  133. package/src/server/management-api.ts +12 -0
  134. package/src/server/request-log-conversation.ts +9 -7
  135. package/src/server/request-log.ts +245 -1
  136. package/src/server/responses/account-change-state.ts +233 -0
  137. package/src/server/responses/adapter-continuation.ts +514 -0
  138. package/src/server/responses/adapter-delivery.ts +214 -0
  139. package/src/server/responses/adapter-dispatch.ts +971 -0
  140. package/src/server/responses/compact.ts +59 -4
  141. package/src/server/responses/completion-policy.ts +33 -0
  142. package/src/server/responses/core-auth.ts +527 -0
  143. package/src/server/responses/core-codex-account.ts +859 -0
  144. package/src/server/responses/core-combo-failure.ts +210 -0
  145. package/src/server/responses/core-combo.ts +707 -0
  146. package/src/server/responses/core-errors.ts +152 -0
  147. package/src/server/responses/core-lifetime.ts +95 -0
  148. package/src/server/responses/core-normalize.ts +350 -0
  149. package/src/server/responses/core-opaque-recovery.ts +380 -0
  150. package/src/server/responses/core-options.ts +159 -0
  151. package/src/server/responses/core-replay.ts +225 -0
  152. package/src/server/responses/core.ts +192 -8893
  153. package/src/server/responses/passthrough-delivery.ts +856 -0
  154. package/src/server/responses/passthrough-dispatch.ts +1476 -0
  155. package/src/server/responses/passthrough-execution.ts +54 -0
  156. package/src/server/responses/request-prepare.ts +970 -0
  157. package/src/server/responses/request-send-budget.ts +164 -0
  158. package/src/server/responses/request-sidecar-auth.ts +149 -0
  159. package/src/server/responses/request-transport.ts +744 -0
  160. package/src/server/responses/response-effects.ts +157 -0
  161. package/src/server/responses/run-turn-execution.ts +448 -0
  162. package/src/server/responses/sidecar-execution.ts +469 -0
  163. package/src/server/responses-image-gen-repair.ts +1 -1
  164. package/src/server/workflow-refusal.ts +84 -0
  165. package/src/types/config.ts +30 -0
  166. package/src/usage/log.ts +146 -0
  167. package/src/usage/summary.ts +171 -21
package/src/usage/log.ts CHANGED
@@ -80,6 +80,46 @@ export type AttemptRecoveryKind =
80
80
  /** Request-time upstream credential class, never a credential or account identifier. */
81
81
  export type UsageCredentialSource = "grok-oauth" | "xai-api-key";
82
82
 
83
+ /**
84
+ * Where a row's cache-token detail came from.
85
+ *
86
+ * Strict-client normalization emits zero-default token-detail objects on every bridged wire
87
+ * (`responsesUsage` in src/bridge.ts), so a `cached_tokens: 0` read back off that wire is a
88
+ * wire-compatibility artifact and not a measured cache miss. The three values stay distinct all
89
+ * the way to the summary because folding `synthesized` or `unknown` into `observed` is what
90
+ * lets a pool that discarded every warm prefix still report a plausible cache hit rate (#4546).
91
+ */
92
+ export type CacheTelemetryProvenance = "observed" | "synthesized" | "unknown";
93
+
94
+ const KNOWN_CACHE_PROVENANCE = new Set<CacheTelemetryProvenance>([
95
+ "observed", "synthesized", "unknown",
96
+ ]);
97
+
98
+ export function isKnownCacheTelemetryProvenance(value: unknown): value is CacheTelemetryProvenance {
99
+ return typeof value === "string" && KNOWN_CACHE_PROVENANCE.has(value as CacheTelemetryProvenance);
100
+ }
101
+
102
+ /**
103
+ * Classify one usage record's cache detail.
104
+ *
105
+ * `wireParsed` means the counts were read back off a response wire rather than reported raw by
106
+ * the adapter. An all-zero cache detail from that source cannot be told apart from the zero
107
+ * defaults the normalizer writes, so it is `synthesized`; the same shape reported raw is a real
108
+ * zero and stays `observed`. A record with no cache fields at all is `unknown`, which is not a
109
+ * zero either.
110
+ */
111
+ export function classifyCacheTelemetryProvenance(
112
+ usage: OcxUsage | undefined,
113
+ options: { wireParsed?: boolean } = {},
114
+ ): CacheTelemetryProvenance {
115
+ if (!usage) return "unknown";
116
+ const present = [usage.cachedInputTokens, usage.cacheReadInputTokens, usage.cacheCreationInputTokens]
117
+ .filter((value): value is number => typeof value === "number" && Number.isFinite(value));
118
+ if (present.length === 0) return "unknown";
119
+ if (present.some(value => value > 0)) return "observed";
120
+ return options.wireParsed === true ? "synthesized" : "observed";
121
+ }
122
+
83
123
  export interface PersistedUsageAttempt {
84
124
  ordinal: number;
85
125
  provider: string;
@@ -113,6 +153,11 @@ export interface PersistedUsageAttempt {
113
153
  usage?: OcxUsage;
114
154
  totalTokens?: number;
115
155
  errorCode?: string;
156
+ /**
157
+ * Provenance of this attempt's cache detail. Absent on rows written before the distinction
158
+ * existed, where `classifyCacheTelemetryProvenance` reconstructs the pre-existing reading.
159
+ */
160
+ cacheProvenance?: CacheTelemetryProvenance;
116
161
  /** Installation-local exact Compatibility Lab route-subject digest for this attempt. */
117
162
  labRouteSubjectId?: string;
118
163
  /** Target-specific reasoning intent and exact adapter-normalized wire parameter. */
@@ -132,9 +177,84 @@ export interface PersistedUsageAttempt {
132
177
  codexWsStage?: CodexWsStageRecord;
133
178
  }
134
179
 
180
+ /**
181
+ * What one logical request spent upstream, and why (#4546, devlog 040 slice D).
182
+ *
183
+ * `sendCount` counts physical sends per ATTEMPT, which answers the wrong question: a user turn
184
+ * that failed over twice and fanned out to three combo targets is one turn, and the number an
185
+ * operator needs is the total that reached upstream carrying the full prompt. These fields are
186
+ * that total, decomposed by how much of it is explained.
187
+ */
188
+ export interface PersistedRequestSpend {
189
+ /** Physical upstream sends summed across every attempt of this logical request, combo children included. */
190
+ sends: number;
191
+ /** Sends whose attempt reached a terminal status, so the spend has a known outcome. */
192
+ settled: number;
193
+ /**
194
+ * Sends charged with no terminal outcome behind them: an attempt abandoned mid-flight, or a
195
+ * budget charge no attempt row ever accounted for. Never folded into `settled` — an unexplained
196
+ * send is the exact quantity this record exists to make visible.
197
+ */
198
+ unresolved: number;
199
+ /** Model sends the request execution budget charged. Absent when no budget was attached. */
200
+ reserved?: number;
201
+ /** Budget profile that produced `reserved`, so a count can be read against the policy it obeyed. */
202
+ policyVersion?: string;
203
+ /**
204
+ * Why the pool binding moved during this request. A move discards the warmed prompt-cache
205
+ * prefix, so the reason belongs next to the send count rather than a page away from it.
206
+ */
207
+ moveReasons?: CodexAffinityReason[];
208
+ }
209
+
210
+ const MAX_PERSISTED_MOVE_REASONS = 8;
211
+ const LOGICAL_REQUEST_ID_RE = /^[A-Za-z0-9_.:-]{1,64}$/;
212
+
213
+ export function isLogicalRequestId(value: unknown): value is string {
214
+ return typeof value === "string" && LOGICAL_REQUEST_ID_RE.test(value);
215
+ }
216
+
217
+ /**
218
+ * A spend record is trusted only when every count is a non-negative integer and the decomposition
219
+ * holds. A hand-edited row that reports more settled spend than it sent would understate exactly
220
+ * the quantity the record exists to expose, so the whole record is dropped instead.
221
+ */
222
+ export function normalizeRequestSpend(value: unknown): PersistedRequestSpend | undefined {
223
+ if (!value || typeof value !== "object" || Array.isArray(value)) return undefined;
224
+ const spend = value as Record<string, unknown>;
225
+ const count = (raw: unknown): number | null =>
226
+ typeof raw === "number" && Number.isInteger(raw) && raw >= 0 ? raw : null;
227
+ const sends = count(spend.sends);
228
+ const settled = count(spend.settled);
229
+ const unresolved = count(spend.unresolved);
230
+ if (sends === null || settled === null || unresolved === null) return undefined;
231
+ if (settled > sends) return undefined;
232
+ const reserved = "reserved" in spend ? count(spend.reserved) : undefined;
233
+ if (reserved === null) return undefined;
234
+ const moveReasons = Array.isArray(spend.moveReasons)
235
+ ? [...new Set(spend.moveReasons.filter(isKnownAffinityReason))].slice(0, MAX_PERSISTED_MOVE_REASONS)
236
+ : [];
237
+ return {
238
+ sends,
239
+ settled,
240
+ unresolved,
241
+ ...(reserved !== undefined ? { reserved } : {}),
242
+ ...(typeof spend.policyVersion === "string" && spend.policyVersion
243
+ ? { policyVersion: capMetadataString(spend.policyVersion) }
244
+ : {}),
245
+ ...(moveReasons.length > 0 ? { moveReasons } : {}),
246
+ };
247
+ }
248
+
135
249
  export interface PersistedUsageEntry {
136
250
  requestedAlias?: string;
137
251
  requestId: string;
252
+ /**
253
+ * Identity of the ONE logical request this row belongs to (#4546), minted by
254
+ * `createRequestExecutionBudget` at ingress. `requestId` identifies a log row; a retry layer,
255
+ * a repair leg and a combo child are all the same logical request, and only this field says so.
256
+ */
257
+ logicalRequestId?: string;
138
258
  timestamp: number;
139
259
  provider: string;
140
260
  model: string;
@@ -177,6 +297,10 @@ export interface PersistedUsageEntry {
177
297
  usage?: OcxUsage;
178
298
  totalTokens?: number;
179
299
  attempts?: PersistedUsageAttempt[];
300
+ /** Aggregated upstream spend for this logical request; additive, older rows omit it. */
301
+ spend?: PersistedRequestSpend;
302
+ /** Provenance of this row's cache detail; absent rows are reconstructed, never assumed observed. */
303
+ cacheProvenance?: CacheTelemetryProvenance;
180
304
  // Failure diagnostics (devlog/_plan/260716_claudecode_hardening/030): persisted for
181
305
  // status>=400 or non-completed terminals so incidents survive the in-memory ring buffer.
182
306
  errorCode?: string;
@@ -195,6 +319,11 @@ export interface PersistedUsageEntry {
195
319
  */
196
320
  affinity?: CodexAffinityMove;
197
321
  affinityReason?: CodexAffinityReason;
322
+ /**
323
+ * Set when this request dropped account-bound continuation after a Codex pool
324
+ * account change. Never an account identifier.
325
+ */
326
+ conversationStateScrub?: "account-change";
198
327
  /**
199
328
  * Bounded route-decision trace (RI-01): why this provider/model/account was
200
329
  * selected. Additive field; old rows without it parse unchanged. Never
@@ -269,6 +398,9 @@ const KNOWN_AFFINITY_REASONS = new Set<NonNullable<PersistedUsageEntry["affinity
269
398
  "unusable", "paused", "plan_excluded", "cooldown", "quota_avoided", "generation",
270
399
  "expired", "model_lane",
271
400
  ]);
401
+ const KNOWN_CONVERSATION_STATE_SCRUBS = new Set<NonNullable<PersistedUsageEntry["conversationStateScrub"]>>([
402
+ "account-change",
403
+ ]);
272
404
 
273
405
  export function isKnownAffinityMove(value: unknown): value is NonNullable<PersistedUsageEntry["affinity"]> {
274
406
  return typeof value === "string" && KNOWN_AFFINITY_MOVES.has(value as NonNullable<PersistedUsageEntry["affinity"]>);
@@ -519,6 +651,9 @@ function normalizeUsageAttempt(raw: unknown): PersistedUsageAttempt | null {
519
651
  ? { totalTokens: attempt.totalTokens }
520
652
  : {}),
521
653
  ...(typeof attempt.errorCode === "string" ? { errorCode: attempt.errorCode } : {}),
654
+ ...(isKnownCacheTelemetryProvenance(attempt.cacheProvenance)
655
+ ? { cacheProvenance: attempt.cacheProvenance }
656
+ : {}),
522
657
  ...(isLabRouteSubjectId(attempt.labRouteSubjectId)
523
658
  ? { labRouteSubjectId: attempt.labRouteSubjectId }
524
659
  : {}),
@@ -625,11 +760,17 @@ function normalizeUsageEntry(entry: PersistedUsageEntry): PersistedUsageEntry {
625
760
  const affinityReason = affinity !== undefined && isKnownAffinityReason(entry.affinityReason)
626
761
  ? entry.affinityReason
627
762
  : undefined;
763
+ const conversationStateScrub = typeof entry.conversationStateScrub === "string"
764
+ && KNOWN_CONVERSATION_STATE_SCRUBS.has(entry.conversationStateScrub)
765
+ ? entry.conversationStateScrub
766
+ : undefined;
628
767
  const routeDecision = entry.routeDecision
629
768
  ? normalizeRouteDecisionTrace(entry.routeDecision)
630
769
  : undefined;
770
+ const spend = normalizeRequestSpend(entry.spend);
631
771
  return {
632
772
  requestId: entry.requestId,
773
+ ...(isLogicalRequestId(entry.logicalRequestId) ? { logicalRequestId: entry.logicalRequestId } : {}),
633
774
  timestamp: entry.timestamp,
634
775
  provider: entry.provider,
635
776
  model: entry.model,
@@ -693,10 +834,15 @@ function normalizeUsageEntry(entry: PersistedUsageEntry): PersistedUsageEntry {
693
834
  ...(entry.usage ? { usage: normalizeUsageValue(entry.usage) } : {}),
694
835
  ...(typeof entry.totalTokens === "number" ? { totalTokens: entry.totalTokens } : {}),
695
836
  ...(Array.isArray(entry.attempts) ? { attempts } : {}),
837
+ ...(spend ? { spend } : {}),
838
+ ...(isKnownCacheTelemetryProvenance(entry.cacheProvenance)
839
+ ? { cacheProvenance: entry.cacheProvenance }
840
+ : {}),
696
841
  ...(transportPhase ? { transportPhase } : {}),
697
842
  ...(terminalSource ? { terminalSource } : {}),
698
843
  ...(affinity ? { affinity } : {}),
699
844
  ...(affinityReason ? { affinityReason } : {}),
845
+ ...(conversationStateScrub ? { conversationStateScrub } : {}),
700
846
  ...(entry.errorCode ? { errorCode: entry.errorCode } : {}),
701
847
  ...(entry.terminalStatus ? { terminalStatus: entry.terminalStatus } : {}),
702
848
  ...(entry.closeReason ? { closeReason: entry.closeReason } : {}),
@@ -3,7 +3,13 @@ import { canonicalAntigravityUsageModel } from "../providers/antigravity-models"
3
3
  import { usageDisplayTotalTokens } from "./totals";
4
4
  import type { UsageTimeWindow } from "./time-range";
5
5
  import { isUnresolvedRequestedModel, usageModelPriceOptions } from "./model-identity";
6
- import { isCodexUsageAccountLogLabel, type PersistedUsageEntry, type UsageStatus } from "./log";
6
+ import {
7
+ classifyCacheTelemetryProvenance,
8
+ isCodexUsageAccountLogLabel,
9
+ type CacheTelemetryProvenance,
10
+ type PersistedUsageEntry,
11
+ type UsageStatus,
12
+ } from "./log";
7
13
  import { type AttemptCostEstimate, type CostEstimate, estimateAttemptCost, estimateRequestCost, serviceTierContext, type ServiceTierContext } from "./cost";
8
14
 
9
15
  /**
@@ -45,6 +51,28 @@ export interface UsageSummaryTotals {
45
51
  unpricedRequests: number;
46
52
  /** Requests whose usage itself is missing/unsupported, so no cost can be computed. */
47
53
  unmeteredRequests: number;
54
+ /**
55
+ * Physical upstream sends aggregated per logical request (#4546, devlog 040 slice D): attempts
56
+ * and combo children summed on the row, then summed over rows. `attemptCount` answers how many
57
+ * attempts were recorded, which is a smaller number — retry layers re-send inside one attempt.
58
+ *
59
+ * These are optional because the management read-failure fallback emits a zeroed summary of its
60
+ * own; absence means "not computed", never zero.
61
+ */
62
+ sends?: number;
63
+ /** Sends whose attempt reached a terminal status. */
64
+ settledSends?: number;
65
+ /** Sends charged with no terminal outcome behind them. Never folded into `settledSends`. */
66
+ unresolvedSends?: number;
67
+ /** Rows that carried a spend record, i.e. logical requests with send accounting. */
68
+ spendRequests?: number;
69
+ /** Input tokens whose row carried OBSERVED cache detail; the only honest hit-rate denominator. */
70
+ cacheObservedInputTokens?: number;
71
+ cacheObservedRequests?: number;
72
+ /** Rows whose cache detail is a wire-compatibility zero: present, and proof of nothing. */
73
+ cacheSynthesizedRequests?: number;
74
+ /** Rows with no cache detail at all. Not a miss, and not a zero. */
75
+ cacheUnknownRequests?: number;
48
76
  }
49
77
 
50
78
  export interface UsageDay {
@@ -71,6 +99,8 @@ export interface UsageDayModel {
71
99
  cacheReadInputTokens?: number;
72
100
  cacheCreationInputTokens?: number;
73
101
  cacheHitRate?: number | null;
102
+ /** Denominator behind `cacheHitRate`: input tokens whose cache detail was observed. */
103
+ cacheObservedInputTokens?: number;
74
104
  estimatedCostUsd?: number;
75
105
  }
76
106
 
@@ -92,6 +122,8 @@ export interface UsageModel {
92
122
  cacheReadInputTokens?: number;
93
123
  cacheCreationInputTokens?: number;
94
124
  cacheHitRate?: number | null;
125
+ /** Denominator behind `cacheHitRate`; below `inputTokens` whenever some rows never measured cache. */
126
+ cacheObservedInputTokens?: number;
95
127
  priceCoverageRatio?: number;
96
128
  pricedRequests?: number;
97
129
  unpricedRequests?: number;
@@ -113,6 +145,8 @@ export interface UsageProvider {
113
145
  cacheReadInputTokens?: number;
114
146
  cacheCreationInputTokens?: number;
115
147
  cacheHitRate?: number | null;
148
+ /** Denominator behind `cacheHitRate`; below `inputTokens` whenever some rows never measured cache. */
149
+ cacheObservedInputTokens?: number;
116
150
  priceCoverageRatio?: number;
117
151
  pricedRequests?: number;
118
152
  unpricedRequests?: number;
@@ -203,13 +237,34 @@ export function cacheTokensFromUsage(usage?: PersistedUsageEntry["usage"]): {
203
237
  return { read, creation, hasCacheTelemetry };
204
238
  }
205
239
 
240
+ /**
241
+ * Cache tokens plus the provenance that says whether they may be averaged.
242
+ *
243
+ * A persisted `cacheProvenance` wins; a row written before the field existed is reconstructed
244
+ * from its own shape, which reproduces the previous reading exactly (telemetry present is
245
+ * observed, absent is unknown) so historical rows do not change meaning. Only `observed` reaches
246
+ * a hit-rate denominator: a synthesized zero was emitted for wire compatibility and an unknown
247
+ * was never measured, and averaging either as a zero is how a cold pool reports a warm cache.
248
+ */
249
+ export function cacheObservationFromUsage(
250
+ usage: PersistedUsageEntry["usage"],
251
+ provenance: CacheTelemetryProvenance | undefined,
252
+ ): { read: number | undefined; creation: number | undefined; provenance: CacheTelemetryProvenance } {
253
+ const { read, creation, hasCacheTelemetry } = cacheTokensFromUsage(usage);
254
+ // A row cannot have observed what it does not carry, so a stored label never opens the
255
+ // denominator for a record with no cache fields in it.
256
+ if (!usage || !hasCacheTelemetry) return { read, creation, provenance: "unknown" };
257
+ return { read, creation, provenance: provenance ?? classifyCacheTelemetryProvenance(usage) };
258
+ }
259
+
206
260
  export function calculateCacheHitRate(
207
261
  cacheObserved: boolean,
208
- inputTokens: number,
262
+ /** Observed input tokens only. Passing the row's whole input total averages unknowns as zeros. */
263
+ observedInputTokens: number,
209
264
  cacheReadTokens: number,
210
265
  ): number | null {
211
- if (!cacheObserved || inputTokens <= 0) return null;
212
- return Math.max(0, Math.min(1, cacheReadTokens / inputTokens));
266
+ if (!cacheObserved || observedInputTokens <= 0) return null;
267
+ return Math.max(0, Math.min(1, cacheReadTokens / observedInputTokens));
213
268
  }
214
269
 
215
270
  export function computeEntryCost(entry: PersistedUsageEntry): EntryCostInfo {
@@ -340,6 +395,14 @@ function blankTotals(): UsageSummaryTotals {
340
395
  pricedRequests: 0,
341
396
  unpricedRequests: 0,
342
397
  unmeteredRequests: 0,
398
+ sends: 0,
399
+ settledSends: 0,
400
+ unresolvedSends: 0,
401
+ spendRequests: 0,
402
+ cacheObservedInputTokens: 0,
403
+ cacheObservedRequests: 0,
404
+ cacheSynthesizedRequests: 0,
405
+ cacheUnknownRequests: 0,
343
406
  };
344
407
  }
345
408
 
@@ -357,6 +420,8 @@ interface UsageAttribution {
357
420
  usageStatus: UsageStatus;
358
421
  usage?: PersistedUsageEntry["usage"];
359
422
  totalTokens?: number;
423
+ /** Attempt provenance when the row has one, else the entry's; never assumed observed. */
424
+ cacheProvenance?: CacheTelemetryProvenance;
360
425
  }
361
426
 
362
427
 
@@ -400,18 +465,26 @@ function usageAttributions(entry: PersistedUsageEntry): UsageAttribution[] {
400
465
  usageStatus: entry.usageStatus,
401
466
  ...(entry.usage ? { usage: entry.usage } : {}),
402
467
  ...(entry.totalTokens !== undefined ? { totalTokens: entry.totalTokens } : {}),
468
+ ...(entry.cacheProvenance ? { cacheProvenance: entry.cacheProvenance } : {}),
403
469
  }];
404
470
  }
405
- return entry.attempts.map(attempt => ({
406
- requestId: entry.requestId,
407
- provider: attempt.provider,
408
- ...usageModelIdentity(attempt.provider, attempt.model),
409
- ...(isUnresolvedRequestedModel(entry, attempt) ? { hasUnresolvedRequestedModel: true as const } : {}),
410
- ...(attempt.accountLogLabel ? { accountLogLabel: attempt.accountLogLabel } : {}),
411
- usageStatus: attempt.usageStatus,
412
- ...(attempt.usage ? { usage: attempt.usage } : {}),
413
- ...(attempt.totalTokens !== undefined ? { totalTokens: attempt.totalTokens } : {}),
414
- }));
471
+ return entry.attempts.map(attempt => {
472
+ // An attempt's own provenance wins; the row's is the fallback for a child written before
473
+ // attempt-level provenance existed. A child carrying no cache fields still resolves to
474
+ // unknown in cacheObservationFromUsage, so it cannot inherit a sibling's observation.
475
+ const cacheProvenance = attempt.cacheProvenance ?? entry.cacheProvenance;
476
+ return {
477
+ requestId: entry.requestId,
478
+ provider: attempt.provider,
479
+ ...usageModelIdentity(attempt.provider, attempt.model),
480
+ ...(isUnresolvedRequestedModel(entry, attempt) ? { hasUnresolvedRequestedModel: true as const } : {}),
481
+ ...(attempt.accountLogLabel ? { accountLogLabel: attempt.accountLogLabel } : {}),
482
+ usageStatus: attempt.usageStatus,
483
+ ...(attempt.usage ? { usage: attempt.usage } : {}),
484
+ ...(attempt.totalTokens !== undefined ? { totalTokens: attempt.totalTokens } : {}),
485
+ ...(cacheProvenance ? { cacheProvenance } : {}),
486
+ };
487
+ });
415
488
  }
416
489
 
417
490
  function projectedComboUsage(
@@ -497,6 +570,43 @@ function addTokens(
497
570
  totals.totalTokens += usageDisplayTotalTokens(entry.usage, entry.totalTokens) ?? 0;
498
571
  }
499
572
 
573
+ /**
574
+ * Fold one row's send accounting into the window totals.
575
+ *
576
+ * The row already aggregated its attempts and combo children, so this is a sum over logical
577
+ * requests. Rows written before the spend record existed contribute nothing rather than a zero:
578
+ * a request whose sends were never counted is not a request that sent nothing.
579
+ */
580
+ function addSpendTotals(
581
+ totals: UsageSummaryTotals,
582
+ entry: Pick<PersistedUsageEntry, "spend">,
583
+ ): void {
584
+ const spend = entry.spend;
585
+ if (!spend) return;
586
+ totals.sends = (totals.sends ?? 0) + spend.sends;
587
+ totals.settledSends = (totals.settledSends ?? 0) + spend.settled;
588
+ totals.unresolvedSends = (totals.unresolvedSends ?? 0) + spend.unresolved;
589
+ totals.spendRequests = (totals.spendRequests ?? 0) + 1;
590
+ }
591
+
592
+ /** Keep the three cache provenances countable, and let only observed input tokens be averaged. */
593
+ function addCacheProvenanceTotals(
594
+ totals: UsageSummaryTotals,
595
+ entry: Pick<PersistedUsageEntry, "usage" | "cacheProvenance">,
596
+ ): void {
597
+ const { provenance } = cacheObservationFromUsage(entry.usage, entry.cacheProvenance);
598
+ if (provenance === "observed") {
599
+ totals.cacheObservedRequests = (totals.cacheObservedRequests ?? 0) + 1;
600
+ totals.cacheObservedInputTokens = (totals.cacheObservedInputTokens ?? 0) + (entry.usage?.inputTokens ?? 0);
601
+ return;
602
+ }
603
+ if (provenance === "synthesized") {
604
+ totals.cacheSynthesizedRequests = (totals.cacheSynthesizedRequests ?? 0) + 1;
605
+ return;
606
+ }
607
+ totals.cacheUnknownRequests = (totals.cacheUnknownRequests ?? 0) + 1;
608
+ }
609
+
500
610
  function finalizeCoverage(totals: UsageSummaryTotals): void {
501
611
  totals.coverageRatio = totals.requests === 0 ? 0 : totals.measuredRequests / totals.requests;
502
612
  }
@@ -558,6 +668,8 @@ interface UsageModelAccumulator {
558
668
  cacheReadInputTokens: number;
559
669
  cacheCreationInputTokens: number;
560
670
  cacheObserved: boolean;
671
+ /** Input tokens from attributions with OBSERVED cache detail; the hit-rate denominator. */
672
+ cacheObservedInputTokens: number;
561
673
  estimatedCostUsd?: number;
562
674
  requestCounts: UsageRequestCounts;
563
675
  requestFacts?: Map<number, number>;
@@ -700,6 +812,20 @@ function mergeTotals(target: UsageSummaryTotals, source: UsageSummaryTotals): vo
700
812
  target.pricedRequests += source.pricedRequests;
701
813
  target.unpricedRequests += source.unpricedRequests;
702
814
  target.unmeteredRequests += source.unmeteredRequests;
815
+ target.sends = mergeOptionalTotal(target.sends, source.sends);
816
+ target.settledSends = mergeOptionalTotal(target.settledSends, source.settledSends);
817
+ target.unresolvedSends = mergeOptionalTotal(target.unresolvedSends, source.unresolvedSends);
818
+ target.spendRequests = mergeOptionalTotal(target.spendRequests, source.spendRequests);
819
+ target.cacheObservedInputTokens = mergeOptionalTotal(target.cacheObservedInputTokens, source.cacheObservedInputTokens);
820
+ target.cacheObservedRequests = mergeOptionalTotal(target.cacheObservedRequests, source.cacheObservedRequests);
821
+ target.cacheSynthesizedRequests = mergeOptionalTotal(target.cacheSynthesizedRequests, source.cacheSynthesizedRequests);
822
+ target.cacheUnknownRequests = mergeOptionalTotal(target.cacheUnknownRequests, source.cacheUnknownRequests);
823
+ }
824
+
825
+ /** Sum an optional total. Absent on one side means "not computed there", so it contributes nothing. */
826
+ function mergeOptionalTotal(target: number | undefined, source: number | undefined): number | undefined {
827
+ if (target === undefined && source === undefined) return undefined;
828
+ return (target ?? 0) + (source ?? 0);
703
829
  }
704
830
 
705
831
  function blankModelAccumulator(
@@ -722,6 +848,7 @@ function blankModelAccumulator(
722
848
  cacheReadInputTokens: 0,
723
849
  cacheCreationInputTokens: 0,
724
850
  cacheObserved: false,
851
+ cacheObservedInputTokens: 0,
725
852
  requestCounts: blankRequestCounts(),
726
853
  ...(mode === "exact" ? { requestFacts: new Map() } : {}),
727
854
  };
@@ -749,6 +876,7 @@ function mergeModelAccumulator(target: UsageModelAccumulator, source: UsageModel
749
876
  target.cacheReadInputTokens += source.cacheReadInputTokens;
750
877
  target.cacheCreationInputTokens += source.cacheCreationInputTokens;
751
878
  target.cacheObserved ||= source.cacheObserved;
879
+ target.cacheObservedInputTokens += source.cacheObservedInputTokens;
752
880
  if (source.estimatedCostUsd !== undefined) {
753
881
  target.estimatedCostUsd = (target.estimatedCostUsd ?? 0) + source.estimatedCostUsd;
754
882
  }
@@ -853,9 +981,17 @@ function projectedEntryForFilter(
853
981
  return filterMatchesAttribution(filter, attempt.provider, identity.model);
854
982
  });
855
983
  if (attempts.length === 0) return null;
856
- const { usage: _parentUsage, totalTokens: _parentTotalTokens, ...withoutParentUsage } = entry;
984
+ const { usage: _parentUsage, totalTokens: _parentTotalTokens, spend: parentSpend, ...withoutParentUsage } = entry;
857
985
  return {
858
- entry: { ...withoutParentUsage, attempts, ...projectedComboUsage(attempts) },
986
+ entry: {
987
+ ...withoutParentUsage,
988
+ // The spend record counts the whole logical request. A projection that dropped a combo
989
+ // child no longer describes it, so the record is dropped with the child rather than
990
+ // reporting a full-request send count against a partial row.
991
+ ...(parentSpend && attempts.length === entry.attempts.length ? { spend: parentSpend } : {}),
992
+ attempts,
993
+ ...projectedComboUsage(attempts),
994
+ },
859
995
  comboOverlap: entry.attempts.length > 1,
860
996
  };
861
997
  }
@@ -915,7 +1051,8 @@ function buildDayModels(
915
1051
  outputTokens: model.outputTokens,
916
1052
  cacheReadInputTokens: model.cacheReadInputTokens,
917
1053
  cacheCreationInputTokens: model.cacheCreationInputTokens,
918
- cacheHitRate: calculateCacheHitRate(model.cacheObserved, model.inputTokens, model.cacheReadInputTokens),
1054
+ cacheHitRate: calculateCacheHitRate(model.cacheObserved, model.cacheObservedInputTokens, model.cacheReadInputTokens),
1055
+ cacheObservedInputTokens: model.cacheObservedInputTokens,
919
1056
  ...(model.estimatedCostUsd !== undefined ? { estimatedCostUsd: model.estimatedCostUsd } : {}),
920
1057
  }));
921
1058
  }
@@ -947,7 +1084,8 @@ function buildUsageModels(
947
1084
  cachedInputTokens: model.cacheReadInputTokens,
948
1085
  cacheReadInputTokens: model.cacheReadInputTokens,
949
1086
  cacheCreationInputTokens: model.cacheCreationInputTokens,
950
- cacheHitRate: calculateCacheHitRate(model.cacheObserved, model.inputTokens, model.cacheReadInputTokens),
1087
+ cacheHitRate: calculateCacheHitRate(model.cacheObserved, model.cacheObservedInputTokens, model.cacheReadInputTokens),
1088
+ cacheObservedInputTokens: model.cacheObservedInputTokens,
951
1089
  priceCoverageRatio: requests > 0 ? counts.pricedRequests / requests : 0,
952
1090
  pricedRequests: counts.pricedRequests,
953
1091
  unpricedRequests: counts.unpricedRequests,
@@ -985,7 +1123,8 @@ function buildUsageProviders(
985
1123
  cachedInputTokens: provider.cacheReadInputTokens,
986
1124
  cacheReadInputTokens: provider.cacheReadInputTokens,
987
1125
  cacheCreationInputTokens: provider.cacheCreationInputTokens,
988
- cacheHitRate: calculateCacheHitRate(provider.cacheObserved, provider.inputTokens, provider.cacheReadInputTokens),
1126
+ cacheHitRate: calculateCacheHitRate(provider.cacheObserved, provider.cacheObservedInputTokens, provider.cacheReadInputTokens),
1127
+ cacheObservedInputTokens: provider.cacheObservedInputTokens,
989
1128
  priceCoverageRatio: requests > 0 ? counts.pricedRequests / requests : 0,
990
1129
  pricedRequests: counts.pricedRequests,
991
1130
  unpricedRequests: counts.unpricedRequests,
@@ -1159,8 +1298,17 @@ class StreamingUsageSummaryAccumulator implements UsageSummaryAccumulator {
1159
1298
  if (attribution.usage) {
1160
1299
  breakdown.inputTokens += attribution.usage.inputTokens;
1161
1300
  breakdown.outputTokens += attribution.usage.outputTokens;
1162
- const { read, creation, hasCacheTelemetry } = cacheTokensFromUsage(attribution.usage);
1163
- breakdown.cacheObserved ||= hasCacheTelemetry;
1301
+ const { read, creation, provenance } = cacheObservationFromUsage(
1302
+ attribution.usage,
1303
+ attribution.cacheProvenance,
1304
+ );
1305
+ // Only an observation opens the denominator. A synthesized zero and an unreported detail
1306
+ // both contribute their tokens to inputTokens and nothing to the cache average, which is
1307
+ // the difference between "no cache reads measured" and "no cache reads happened".
1308
+ if (provenance === "observed") {
1309
+ breakdown.cacheObserved = true;
1310
+ breakdown.cacheObservedInputTokens += attribution.usage.inputTokens;
1311
+ }
1164
1312
  if (typeof read === "number") breakdown.cacheReadInputTokens += read;
1165
1313
  if (typeof creation === "number") breakdown.cacheCreationInputTokens += creation;
1166
1314
  breakdown.summaryTotalTokens += usageDisplayTotalTokens(attribution.usage, attribution.totalTokens) ?? 0;
@@ -1313,6 +1461,8 @@ class StreamingUsageSummaryAccumulator implements UsageSummaryAccumulator {
1313
1461
  bumpStatus(partition.totals, entry.usageStatus);
1314
1462
  partition.totals.attemptCount += entry.attempts?.length ?? 1;
1315
1463
  addTokens(partition.totals, entry);
1464
+ addSpendTotals(partition.totals, entry);
1465
+ addCacheProvenanceTotals(partition.totals, entry);
1316
1466
  addEstimatedCost(partition.totals, entry, costInfo);
1317
1467
 
1318
1468
  const requestKey = this.mode === "exact" ? this.requestKey(entry.requestId) : null;