@bitkyc08/opencodex 2.55.0 → 2.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/gui/dist/assets/{index-VuoiWj9J.js → index-D4zuyIxQ.js} +1 -1
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +2 -1
  4. package/src/adapters/base.ts +21 -0
  5. package/src/adapters/cursor/transport-retry.ts +46 -1
  6. package/src/adapters/cursor.ts +4 -0
  7. package/src/adapters/kiro/adapter.ts +42 -1
  8. package/src/adapters/kiro-retry.ts +23 -4
  9. package/src/adapters/openai-chat/errors.ts +116 -0
  10. package/src/adapters/openai-chat/messages.ts +346 -0
  11. package/src/adapters/openai-chat/passthrough.ts +146 -0
  12. package/src/adapters/openai-chat/response-events.ts +117 -0
  13. package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
  14. package/src/adapters/openai-chat/tool-schema.ts +477 -0
  15. package/src/adapters/openai-chat/wire.ts +50 -0
  16. package/src/adapters/openai-chat.ts +33 -1445
  17. package/src/adapters/openai-responses/canonical-forward.ts +202 -0
  18. package/src/adapters/openai-responses/image-gen.ts +406 -0
  19. package/src/adapters/openai-responses/internal.ts +3 -0
  20. package/src/adapters/openai-responses/passthrough.ts +611 -0
  21. package/src/adapters/openai-responses/prompt-cache.ts +83 -0
  22. package/src/adapters/openai-responses/reasoning.ts +220 -0
  23. package/src/adapters/openai-responses/request-strips.ts +185 -0
  24. package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
  25. package/src/adapters/openai-responses/tool-schema.ts +293 -0
  26. package/src/adapters/openai-responses/web-search.ts +156 -0
  27. package/src/adapters/openai-responses.ts +4 -2625
  28. package/src/bridge/errors.ts +34 -0
  29. package/src/bridge/internal.ts +174 -0
  30. package/src/bridge/response-json.ts +624 -0
  31. package/src/bridge/sse.ts +1444 -0
  32. package/src/bridge.ts +5 -2204
  33. package/src/chat/inbound.ts +12 -1
  34. package/src/codex/account-lifecycle.ts +3 -0
  35. package/src/codex/account-store.ts +71 -9
  36. package/src/codex/auth-api/account-list.ts +507 -0
  37. package/src/codex/auth-api/http.ts +32 -0
  38. package/src/codex/auth-api/login-flow.ts +554 -0
  39. package/src/codex/auth-api/login-state.ts +64 -0
  40. package/src/codex/auth-api/main-account-probe.ts +331 -0
  41. package/src/codex/auth-api/pool-mode-gate.ts +274 -0
  42. package/src/codex/auth-api/pool-quota-probe.ts +512 -0
  43. package/src/codex/auth-api/reset-credit-service.ts +422 -0
  44. package/src/codex/auth-api/routes.ts +425 -0
  45. package/src/codex/auth-api/runtime-config.ts +48 -0
  46. package/src/codex/auth-api.ts +27 -3118
  47. package/src/codex/auth-context.ts +95 -28
  48. package/src/codex/catalog/auto-review.ts +507 -0
  49. package/src/codex/catalog/build-entries.ts +981 -0
  50. package/src/codex/catalog/combo-member.ts +375 -0
  51. package/src/codex/catalog/derive-entry.ts +229 -0
  52. package/src/codex/catalog/effort.ts +0 -1
  53. package/src/codex/catalog/gated-native-warn.ts +63 -0
  54. package/src/codex/catalog/gather-capture.ts +533 -0
  55. package/src/codex/catalog/model-hints.ts +691 -0
  56. package/src/codex/catalog/model-visibility.ts +304 -0
  57. package/src/codex/catalog/provider-fetch.ts +52 -2942
  58. package/src/codex/catalog/provider-models.ts +685 -0
  59. package/src/codex/catalog/restore.ts +132 -0
  60. package/src/codex/catalog/retained-sync.ts +706 -0
  61. package/src/codex/catalog/routed-gather.ts +858 -0
  62. package/src/codex/catalog/subagent-roster.ts +176 -0
  63. package/src/codex/catalog/sync.ts +52 -2698
  64. package/src/codex/inject/config-toml.ts +563 -0
  65. package/src/codex/inject/remove.ts +192 -0
  66. package/src/codex/inject/restore.ts +540 -0
  67. package/src/codex/inject/routing-classify.ts +109 -0
  68. package/src/codex/inject/routing-target.ts +125 -0
  69. package/src/codex/inject.ts +81 -1436
  70. package/src/codex/lineage.ts +458 -0
  71. package/src/codex/pool-refresh-backoff.ts +152 -0
  72. package/src/codex/routing/active-account.ts +194 -0
  73. package/src/codex/routing/cooldown-math.ts +275 -0
  74. package/src/codex/routing/health-store.ts +402 -0
  75. package/src/codex/routing/probe-lease.ts +358 -0
  76. package/src/codex/routing/selection.ts +703 -0
  77. package/src/codex/routing/thread-affinity.ts +538 -0
  78. package/src/codex/routing.ts +353 -2234
  79. package/src/codex/shim-fingerprint.ts +223 -0
  80. package/src/codex/shim-inspect.ts +175 -0
  81. package/src/codex/shim-probe.ts +367 -0
  82. package/src/codex/shim-restore-lock.ts +169 -0
  83. package/src/codex/shim-state-file.ts +151 -0
  84. package/src/codex/shim-templates.ts +265 -0
  85. package/src/codex/shim.ts +48 -1268
  86. package/src/config/diagnostics.ts +705 -0
  87. package/src/config/feature-flags.ts +55 -0
  88. package/src/config/live-reconcile.ts +403 -0
  89. package/src/config/load-degrade.ts +880 -0
  90. package/src/config/mutation-lock.ts +244 -0
  91. package/src/config/openai-tier-backup.ts +268 -0
  92. package/src/config/persist-unlocked.ts +92 -0
  93. package/src/config/proxy-env.ts +188 -0
  94. package/src/config/salvage.ts +244 -0
  95. package/src/config/schema/config-schema.ts +640 -0
  96. package/src/config/schema/leaf-validators.ts +855 -0
  97. package/src/config/warn-memo.ts +28 -0
  98. package/src/config.ts +234 -4481
  99. package/src/generated/compatibility-version.json +539 -39
  100. package/src/lib/request-execution-budget.ts +69 -20
  101. package/src/lib/spend-reservation-ledger.ts +940 -0
  102. package/src/lib/upstream-retry.ts +55 -11
  103. package/src/lib/workflow-budget.ts +553 -30
  104. package/src/providers/quota/account-cache.ts +441 -0
  105. package/src/providers/quota/antigravity.ts +295 -0
  106. package/src/providers/quota/report-cache.ts +320 -0
  107. package/src/providers/quota/vendor-probes-key.ts +1243 -0
  108. package/src/providers/quota/vendor-probes-oauth.ts +590 -0
  109. package/src/providers/quota.ts +324 -3079
  110. package/src/providers/registry/entries-core.ts +1221 -0
  111. package/src/providers/registry/entries-extended.ts +1204 -0
  112. package/src/providers/registry/model-seeds.ts +908 -0
  113. package/src/providers/registry/types.ts +352 -0
  114. package/src/providers/registry.ts +24 -3536
  115. package/src/responses/continuation-ownership.ts +29 -0
  116. package/src/responses/state/replay-fingerprint.ts +80 -0
  117. package/src/responses/state/snapshot-codec.ts +104 -0
  118. package/src/responses/state/spill-failure.ts +118 -0
  119. package/src/responses/state/spill-queue.ts +665 -0
  120. package/src/responses/state/temp-recovery.ts +257 -0
  121. package/src/responses/state.ts +82 -1143
  122. package/src/routing/identity-domains.ts +449 -0
  123. package/src/routing/probe-lease.ts +511 -0
  124. package/src/server/index/bounded-request.ts +88 -0
  125. package/src/server/index/live-sideband.ts +565 -0
  126. package/src/server/index/serve-options.ts +1766 -0
  127. package/src/server/index/startup-warnings.ts +213 -0
  128. package/src/server/index/websocket-handler.ts +335 -0
  129. package/src/server/index.ts +40 -2547
  130. package/src/server/management/route-registry.ts +26 -23
  131. package/src/server/management/shared.ts +8 -5
  132. package/src/server/management/workflow-budget-routes.ts +133 -0
  133. package/src/server/management-api.ts +12 -0
  134. package/src/server/request-log-conversation.ts +9 -7
  135. package/src/server/request-log.ts +245 -1
  136. package/src/server/responses/account-change-state.ts +233 -0
  137. package/src/server/responses/adapter-continuation.ts +514 -0
  138. package/src/server/responses/adapter-delivery.ts +214 -0
  139. package/src/server/responses/adapter-dispatch.ts +971 -0
  140. package/src/server/responses/compact.ts +59 -4
  141. package/src/server/responses/completion-policy.ts +33 -0
  142. package/src/server/responses/core-auth.ts +527 -0
  143. package/src/server/responses/core-codex-account.ts +859 -0
  144. package/src/server/responses/core-combo-failure.ts +210 -0
  145. package/src/server/responses/core-combo.ts +707 -0
  146. package/src/server/responses/core-errors.ts +152 -0
  147. package/src/server/responses/core-lifetime.ts +95 -0
  148. package/src/server/responses/core-normalize.ts +350 -0
  149. package/src/server/responses/core-opaque-recovery.ts +380 -0
  150. package/src/server/responses/core-options.ts +159 -0
  151. package/src/server/responses/core-replay.ts +225 -0
  152. package/src/server/responses/core.ts +192 -8893
  153. package/src/server/responses/passthrough-delivery.ts +856 -0
  154. package/src/server/responses/passthrough-dispatch.ts +1476 -0
  155. package/src/server/responses/passthrough-execution.ts +54 -0
  156. package/src/server/responses/request-prepare.ts +970 -0
  157. package/src/server/responses/request-send-budget.ts +164 -0
  158. package/src/server/responses/request-sidecar-auth.ts +149 -0
  159. package/src/server/responses/request-transport.ts +744 -0
  160. package/src/server/responses/response-effects.ts +157 -0
  161. package/src/server/responses/run-turn-execution.ts +448 -0
  162. package/src/server/responses/sidecar-execution.ts +469 -0
  163. package/src/server/responses-image-gen-repair.ts +1 -1
  164. package/src/server/workflow-refusal.ts +84 -0
  165. package/src/types/config.ts +30 -0
  166. package/src/usage/log.ts +146 -0
  167. package/src/usage/summary.ts +171 -21
@@ -1,30 +1,60 @@
1
- import { chmodSync, existsSync, lstatSync, mkdirSync, opendirSync, readFileSync, rmSync, statSync, unlinkSync } from "node:fs";
2
- import { uptime } from "node:os";
1
+ import { chmodSync, existsSync, mkdirSync, readFileSync, rmSync, statSync, unlinkSync } from "node:fs";
3
2
  import { dirname, join } from "node:path";
4
3
  import { atomicWriteFileAsync, getConfigDir, resolveWriteTarget } from "../config";
5
4
  import { enforceAppOwnedMemoryBudget, type RetainedStoreSnapshot } from "../lib/app-owned-memory";
6
5
  import { windowsSecretAclApplies } from "../lib/windows-secret-acl";
7
6
  import type { OcxProviderContinuationState } from "../types";
8
7
  import {
9
- cleanupSupersededResponseSpillPublication,
10
- createResponseSpillPublicationControl,
11
8
  deleteResponseSpill,
12
- MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
13
9
  noteStubSwapForTest,
14
10
  readResponseSpill,
15
11
  recoverOrphanedResponseSpills,
16
12
  responseSpillDirectory,
17
13
  responseSpillPayloadCap,
18
- markResponseSpillPublicationSuperseded,
19
- prospectiveResponseSpillBytes,
20
- type ResponseSpillPublicationControl,
21
14
  type ResponseSpillRef,
22
15
  writeResponseSpillDurably,
23
- writeResponseSpillDurablyAsync,
24
16
  } from "./spill-store";
17
+ import { clientCarriedPrefixLength, providerIssuedIdentity } from "./state/replay-fingerprint";
18
+ export type { ResponseStateTempRecoveryResult, ResponseStateTempRecoveryOptions } from "./state/temp-recovery";
19
+ export { recoverStaleResponseStateTemps, reclaimAbandonedResponseStateTemps, inspectAbandonedResponseStateTemps, sweepAbandonedResponseStateTemps } from "./state/temp-recovery";
20
+ import { recoverStaleResponseStateTemps } from "./state/temp-recovery";
21
+ export type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseSpillWriteFailureOrigin } from "./state/spill-failure";
22
+ import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseSpillWriteFailureOrigin } from "./state/spill-failure";
23
+ export { responseAdmissionCountersForTests } from "./state/spill-failure";
24
+ import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
25
+ import { loadSnapshotEntry } from "./state/snapshot-codec";
26
+ export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
27
+ import {
28
+ bindSpillQueueStore,
29
+ cancelPendingResponseSpill,
30
+ drainResponseSpillPublications,
31
+ queuePendingResponseSpill,
32
+ replaceWithPendingResponseSpill,
33
+ resetSpillQueueForTests,
34
+ spillQueueAccounting,
35
+ spillQueueHoldsResidentCandidate,
36
+ spillQueuePendingBytes,
37
+ spillQueueResidentCandidates,
38
+ spillQueueSupersededSpillFor,
39
+ } from "./state/spill-queue";
25
40
 
26
41
  const MAX_STORED_RESPONSES = 1_000;
27
- const RESPONSE_TTL_MS = 60 * 60 * 1_000;
42
+ /**
43
+ * Retention for locally replayed continuation state.
44
+ *
45
+ * A Codex client chained by `previous_response_id` sends ONLY the new turn and expects this
46
+ * process to hold everything before it, so this constant is the practical memory span of every
47
+ * conversation that does not go to the canonical ChatGPT backend. At the original one hour, a
48
+ * session resumed after lunch expanded to nothing and the delta — one user line — was all the
49
+ * provider ever saw, which reads to the operator as the model losing the conversation.
50
+ *
51
+ * A day is safe to hold because retention is no longer what bounds this store: the resident cap
52
+ * (MAX_STORED_RESPONSE_BYTES), the spill ceiling (MAX_SPILLED_RESPONSE_BYTES) and the entry count
53
+ * all evict oldest-first, and every turn re-stores the whole chain under a fresh id, so the live
54
+ * conversation is the last thing any of those three caps would drop. Raising the TTL therefore
55
+ * moves eviction from the clock to those budgets rather than growing the ceiling.
56
+ */
57
+ export const RESPONSE_TTL_MS = 24 * 60 * 60 * 1_000;
28
58
  const SNAPSHOT_DEBOUNCE_MS = 2_000;
29
59
  /** Snapshot size below which the debounce stays at its base value. */
30
60
  const SNAPSHOT_DEBOUNCE_SCALE_FROM_BYTES = 1 * 1024 * 1024;
@@ -67,28 +97,10 @@ const SNAPSHOT_TOTAL_MAX_BYTES = 24 * 1024 * 1024;
67
97
  * bound, so anything we wrote ourselves always loads; guards against externally
68
98
  * planted or pre-cap unbounded files being parsed whole). */
69
99
  const SNAPSHOT_FILE_MAX_BYTES = 32 * 1024 * 1024;
70
- const STALE_TEMP_GRACE_MS = 15 * 60 * 1_000;
71
- const STALE_TEMP_MAX_ENTRIES = 4_096;
72
- const STALE_TEMP_MAX_CLEANUPS = 512;
73
- /** Absorbs `os.uptime()` granularity only. It is deliberately NOT the safety margin:
74
- * the unconditional 15-minute grace above is (see the boot floor in the scan loop). */
75
- const BOOT_FLOOR_SKEW_MS = 60 * 1_000;
76
- /** Per-tick budget for the periodic reclaim. Smaller than the startup budget because the
77
- * periodic pass runs synchronously on the serving process's event loop every 60 s. */
78
- const PERIODIC_TEMP_MAX_ENTRIES = 512;
79
- const PERIODIC_TEMP_MAX_CLEANUPS = 64;
80
- /** Wall-clock ceiling for one periodic scan. An entry cap bounds syscalls, not time: on a
81
- * network-mounted config dir each `lstat` can cost 10-20 ms, which would stall in-flight
82
- * streams. Reclaim is idempotent, so a truncated tick simply resumes on the next one. */
83
- const PERIODIC_TEMP_SCAN_DEADLINE_MS = 25;
84
- const RESPONSE_STATE_TEMP_NAME = /^responses-state\.json\.ocx\.(\d+)\.(\d+)\.tmp$/;
85
100
  const MAX_SNAPSHOT_REWRITE_ATTEMPTS = 4;
86
- const RESPONSE_SPILL_SHUTDOWN_BUDGET_MS = 5_000;
87
- const RESPONSE_SPILL_SHUTDOWN_FALLBACK_RESERVE_MS = 4_000;
88
- const RESPONSE_SPILL_ASYNC_ACL_ATTEMPT_BUDGET_MS = 30_000;
89
101
  const RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES = MAX_STORED_RESPONSES + 1;
90
102
 
91
- interface ResidentResponseState {
103
+ export interface ResidentResponseState {
92
104
  kind: "resident";
93
105
  createdAt: number;
94
106
  clientThreadId?: string;
@@ -99,7 +111,7 @@ interface ResidentResponseState {
99
111
  sizeBytes: number;
100
112
  }
101
113
 
102
- interface SpilledResponseState {
114
+ export interface SpilledResponseState {
103
115
  kind: "spill";
104
116
  createdAt: number;
105
117
  clientThreadId?: string;
@@ -110,14 +122,14 @@ interface SpilledResponseState {
110
122
  sizeBytes: number;
111
123
  }
112
124
 
113
- interface SpillFailedResponseState {
125
+ export interface SpillFailedResponseState {
114
126
  kind: "spill-failed";
115
127
  createdAt: number;
116
128
  sizeBytes: number;
117
129
  }
118
130
 
119
- type StoredResponseState = ResidentResponseState | SpilledResponseState | SpillFailedResponseState;
120
- type ResidentInput = Omit<ResidentResponseState, "kind" | "sizeBytes">;
131
+ export type StoredResponseState = ResidentResponseState | SpilledResponseState | SpillFailedResponseState;
132
+ export type ResidentInput = Omit<ResidentResponseState, "kind" | "sizeBytes">;
121
133
 
122
134
  export type PreviousResponseReplayFailure = {
123
135
  code: "previous_response_not_found";
@@ -169,124 +181,8 @@ async function snapshotOnDiskMatches(path: string, payload: string, payloadBytes
169
181
  return false;
170
182
  }
171
183
  }
172
- const spillCounters = {
173
- writes: 0, writeFailures: 0, readFailures: 0,
174
- aclRetryReturnedTimeouts: 0, aclTimeoutMemoRefusals: 0,
175
- };
176
-
177
- export type ResponseSpillWriteFailureCode =
178
- | "EACLRETRYEXHAUSTED"
179
- | "ETIMEDOUT"
180
- | "EACCES"
181
- | "ENOSPC"
182
- | "EFBIG"
183
- | "EIO"
184
- | "ECAPACITY"
185
- | "ELOOP"
186
- | "EUNKNOWN";
187
-
188
- export type ResponseSpillWriteStatus = "initial" | "healthy" | "degraded";
189
-
190
- export type ResponseSpillWriteFailureOrigin =
191
- | "retry_returned_timeout"
192
- | "timeout_memo_refusal";
193
-
194
- interface ResponseSpillWriteHealth {
195
- consecutiveFailures: number;
196
- lastFailureCode: ResponseSpillWriteFailureCode | null;
197
- lastFailureOrigin: ResponseSpillWriteFailureOrigin | null;
198
- lastFailureAt: number | null;
199
- lastSuccessAt: number | null;
200
- }
201
-
202
- const spillWriteHealth: ResponseSpillWriteHealth = {
203
- consecutiveFailures: 0,
204
- lastFailureCode: null,
205
- lastFailureOrigin: null,
206
- lastFailureAt: null,
207
- lastSuccessAt: null,
208
- };
209
-
210
- /**
211
- * Collapse filesystem/runtime errors into a fixed privacy-safe diagnostic union.
212
- * Messages and paths are deliberately ignored: this projection is returned by the
213
- * authenticated memory endpoint, and a nested `cause` can contain a username or
214
- * workspace path even when the public wrapper does not.
215
- */
216
- function classifySpillWriteFailure(error: unknown): ResponseSpillWriteFailureCode {
217
- let cursor = error;
218
- for (let depth = 0; depth < 4 && cursor && typeof cursor === "object"; depth += 1) {
219
- const record = cursor as { code?: unknown; cause?: unknown };
220
- const code = typeof record.code === "string" ? record.code.toUpperCase() : "";
221
- switch (code) {
222
- case "EACLRETRYEXHAUSTED": return "EACLRETRYEXHAUSTED";
223
- case "ETIMEDOUT": return "ETIMEDOUT";
224
- case "EACCES":
225
- case "EPERM": return "EACCES";
226
- case "ENOSPC":
227
- case "EDQUOT": return "ENOSPC";
228
- case "EFBIG": return "EFBIG";
229
- case "EIO": return "EIO";
230
- case "ECAPACITY": return "ECAPACITY";
231
- case "ELOOP": return "ELOOP";
232
- }
233
- cursor = record.cause;
234
- }
235
- return "EUNKNOWN";
236
- }
237
-
238
- /** The spill writer preserves ACL errors in cause; only a fixed memo marker is diagnostic. */
239
- function spillAclMemoRefusalOrigin(error: unknown): "timeout_memo_refusal" | null {
240
- let cursor = error;
241
- for (let depth = 0; depth < 4 && cursor && typeof cursor === "object"; depth += 1) {
242
- const record = cursor as { code?: unknown; aclFailureOrigin?: unknown; cause?: unknown };
243
- if ((record.code === "ETIMEDOUT" || record.code === "EACLRETRYEXHAUSTED")
244
- && record.aclFailureOrigin === "timeout_memo_refusal") {
245
- return "timeout_memo_refusal";
246
- }
247
- cursor = record.cause;
248
- }
249
- return null;
250
- }
251
-
252
- function noteSpillWriteSuccess(): void {
253
- spillCounters.writes += 1;
254
- spillWriteHealth.consecutiveFailures = 0;
255
- spillWriteHealth.lastSuccessAt = now();
256
- }
257
-
258
- function noteSpillWriteFailure(
259
- error: unknown,
260
- override?: ResponseSpillWriteFailureCode,
261
- retryOrigin: ResponseSpillWriteFailureOrigin | null = null,
262
- ): void {
263
- const code = override ?? classifySpillWriteFailure(error);
264
- const origin = code === "ETIMEDOUT" || code === "EACLRETRYEXHAUSTED"
265
- ? spillAclMemoRefusalOrigin(error) ?? retryOrigin
266
- : null;
267
- spillCounters.writeFailures += 1;
268
- spillWriteHealth.consecutiveFailures += 1;
269
- spillWriteHealth.lastFailureCode = code;
270
- spillWriteHealth.lastFailureOrigin = origin;
271
- spillWriteHealth.lastFailureAt = now();
272
- // Count terminal publications, not ACL calls or a transient first attempt.
273
- if (origin === "retry_returned_timeout") spillCounters.aclRetryReturnedTimeouts += 1;
274
- else if (origin === "timeout_memo_refusal") spillCounters.aclTimeoutMemoRefusals += 1;
275
- }
276
- /**
277
- * Admission-boundary observability (test-visible). directSpills: oversized
278
- * candidates routed straight to durable spill without a resident stay or
279
- * unrelated demotion. oversizedDrops: candidates above the single-spill
280
- * payload ceiling, tombstoned instead of retained. snapshotOversizedRefusals:
281
- * snapshot files refused before parse.
282
- */
283
- const admissionCounters = { directSpills: 0, oversizedDrops: 0, snapshotOversizedRefusals: 0 };
284
184
  let replayScopeMismatchDrops = 0;
285
185
 
286
- /** Test-only: admission-boundary counters (proves the new paths fire). */
287
- export function responseAdmissionCountersForTests(): Readonly<typeof admissionCounters> {
288
- return admissionCounters;
289
- }
290
186
  // Superseded spill generations awaiting a durable snapshot before unlink
291
187
  // (review C1-1: unlinking at swap time races a crash against the debounced
292
188
  // snapshot — the reloaded OLD stub would point at a deleted file).
@@ -299,99 +195,6 @@ const pendingSpillUnlinks: ResponseSpillRef[] = [];
299
195
  // structured 400 — bounded-loss, never silent corruption or unbounded disk.
300
196
  const PENDING_SPILL_UNLINKS_MAX = 128;
301
197
 
302
- /**
303
- * Windows keeps the candidate replayable while required ACL hardening runs off the event loop.
304
- * Pending bytes are pinned, not evictable; cap them below the process-owned 512 MiB ceiling so an
305
- * icacls outage cannot turn the serialized queue into an unbounded resident backlog.
306
- */
307
- const MAX_PENDING_RESPONSE_SPILL_BYTES = MAX_RESPONSE_SPILL_PAYLOAD_BYTES;
308
-
309
- interface PendingResponseSpill {
310
- id: string;
311
- candidate: ResidentResponseState | null;
312
- supersededSpill?: ResponseSpillRef;
313
- directAdmission: boolean;
314
- running: boolean;
315
- cancelled: boolean;
316
- released: boolean;
317
- sizeBytes: number;
318
- /** Peak on-disk bytes reserved for this publication; released exactly once on settle. */
319
- reservedBytes: number;
320
- publicationControl: ResponseSpillPublicationControl;
321
- }
322
-
323
- const pendingResponseSpills = new Set<PendingResponseSpill>();
324
- const pendingResponseSpillById = new Map<string, PendingResponseSpill>();
325
- let pendingResponseSpillBytes = 0;
326
- /**
327
- * On-disk bytes a queued publication is about to occupy but has not yet installed into
328
- * `states`.
329
- *
330
- * `spilledResponseBytes()` walks installed spills and deferred unlinks — files that
331
- * already exist. It cannot see one that `writeResponseSpillDurablyAsync` is in the
332
- * middle of creating, and on Windows that middle can last as long as `icacls` takes.
333
- * Without a reservation the cap holds only when writes are fast, which is not a cap.
334
- *
335
- * The reserved figure is the PEAK footprint, not the payload: publication can fall back
336
- * from hard-linking to an exclusive copy, and during that fallback the destination copy
337
- * and the temp file exist simultaneously. Reserving one envelope would leave the overshoot
338
- * intact at half its magnitude.
339
- *
340
- * Ownership is single: a job holds its reservation from queue until
341
- * `releasePendingResponseSpill`, which every exit from the publication path reaches
342
- * through the `finally` in `runPendingResponseSpill` and through cancellation of a
343
- * not-yet-running job. A leaked reservation is monotonic — it would ratchet the usable
344
- * cap toward zero — so the release must stay on the settlement path rather than in a
345
- * parallel bookkeeping pass.
346
- */
347
- let reservedResponseSpillBytes = 0;
348
- /**
349
- * Paths a failed cleanup left on the volume, with the bytes each one occupies.
350
- *
351
- * A failed unlink leaves a real file behind, so the cap has to keep seeing it. But a
352
- * never-decremented total would be phantom debt: a Windows lock that clears a moment
353
- * later, or the async writer's own retry, can remove the file while the charge stays
354
- * forever — and with 256 MiB payloads two conservative charges consume the whole default
355
- * cap, after which nothing can spill for the life of the process.
356
- *
357
- * So the debt is per PATH, priced at what that path actually holds, and settled the
358
- * moment the path is gone. `reconcileUnreclaimableSpillPaths` re-checks on every read of
359
- * the accounted total, which is the same tick that would otherwise refuse an admission.
360
- */
361
- const unreclaimableSpillPaths = new Map<string, number>();
362
-
363
- function chargeUnreclaimableSpillPath(path: string | null | undefined, bytes: number): void {
364
- if (!path || bytes <= 0) return;
365
- unreclaimableSpillPaths.set(path, bytes);
366
- }
367
-
368
- /** Drop charges for paths that have since disappeared; returns the surviving total. */
369
- function reconcileUnreclaimableSpillPaths(): number {
370
- let total = 0;
371
- for (const [path, bytes] of [...unreclaimableSpillPaths]) {
372
- if (existsSync(path)) total += bytes;
373
- else unreclaimableSpillPaths.delete(path);
374
- }
375
- return total;
376
- }
377
-
378
- /**
379
- * Peak on-disk footprint of publishing this candidate: temp plus destination copy.
380
- *
381
- * Measured from the production serializer rather than from `candidate.sizeBytes`. The
382
- * resident measurement omits the `version` field the published envelope carries, so
383
- * pricing an admission by it undercounts and lets a request sitting exactly at the cap
384
- * still exceed it. Falls back to the resident figure only when serialization fails, which
385
- * is the same condition that will fail the publication itself.
386
- */
387
- function publicationFootprintBytes(id: string, candidate: ResidentResponseState): number {
388
- const exact = prospectiveResponseSpillBytes(id, spillPayloadForResident(candidate));
389
- return (exact ?? candidate.sizeBytes) * 2;
390
- }
391
- let responseSpillPublicationTail: Promise<void> = Promise.resolve();
392
- let responseSpillShutdownBudgetOverride: { totalMs: number; fallbackReserveMs: number } | null = null;
393
- let responseSpillShutdownTerminalizationPassLimitOverride: number | null = null;
394
- let responseSpillAsyncAclAttemptBudgetOverride: number | null = null;
395
198
 
396
199
  function deferSupersededSpill(ref: ResponseSpillRef | undefined): void {
397
200
  if (!ref) return;
@@ -401,470 +204,6 @@ function deferSupersededSpill(ref: ResponseSpillRef | undefined): void {
401
204
  }
402
205
  }
403
206
 
404
- function releasePendingResponseSpill(job: PendingResponseSpill): void {
405
- if (job.released) return;
406
- job.released = true;
407
- pendingResponseSpillBytes = Math.max(0, pendingResponseSpillBytes - job.sizeBytes);
408
- reservedResponseSpillBytes = Math.max(0, reservedResponseSpillBytes - job.reservedBytes);
409
- pendingResponseSpills.delete(job);
410
- if (pendingResponseSpillById.get(job.id) === job) pendingResponseSpillById.delete(job.id);
411
- job.candidate = null;
412
- }
413
-
414
- function cancelPendingResponseSpill(id: string): ResponseSpillRef | undefined {
415
- const job = pendingResponseSpillById.get(id);
416
- if (!job) return undefined;
417
- pendingResponseSpillById.delete(id);
418
- job.cancelled = true;
419
- markResponseSpillPublicationSuperseded(job.publicationControl);
420
- const superseded = job.supersededSpill;
421
- // Ownership TRANSFERS to the caller. Leaving the ref on the cancelled job would let the
422
- // accounting walk count the same physical file twice — once here and once on the
423
- // replacement — and an overcount evicts live continuations to make room for bytes that
424
- // are not there.
425
- delete job.supersededSpill;
426
- // A queued job has not captured the candidate in an async frame yet, so release it now.
427
- // A running job retains its accounting until settlement and will discard its stale file.
428
- if (!job.running) releasePendingResponseSpill(job);
429
- return superseded;
430
- }
431
-
432
- function isAclTimeout(error: unknown): boolean {
433
- return !!error && typeof error === "object" && "code" in error
434
- && String((error as { code?: unknown }).code) === "ETIMEDOUT";
435
- }
436
-
437
- function spillPayloadForResident(candidate: ResidentResponseState): Parameters<typeof writeResponseSpillDurably>[1] {
438
- return {
439
- createdAt: candidate.createdAt,
440
- ...(candidate.clientThreadId ? { clientThreadId: candidate.clientThreadId } : {}),
441
- items: candidate.items,
442
- ...(candidate.providerOutputStart !== undefined ? { providerOutputStart: candidate.providerOutputStart } : {}),
443
- ...(candidate.providers ? { providers: candidate.providers } : {}),
444
- };
445
- }
446
-
447
- async function runPendingResponseSpill(job: PendingResponseSpill): Promise<void> {
448
- if (job.cancelled || !job.candidate) return;
449
- job.running = true;
450
- const candidate = job.candidate;
451
- let ref: ResponseSpillRef | null = null;
452
- let exhaustedAclRetry = false;
453
- let aclRetryFailureOrigin: ResponseSpillWriteFailureOrigin | null = null;
454
- try {
455
- const state = spillPayloadForResident(candidate);
456
- try {
457
- ref = await writeResponseSpillDurablyAsync(job.id, state, {
458
- aclBudgetMs: responseSpillAsyncAclAttemptBudgetMs(),
459
- publicationControl: job.publicationControl,
460
- });
461
- } catch (error) {
462
- if (!isAclTimeout(error)) throw error;
463
- // The ACL helper permits exactly one caller-owned recovery budget. The resident generation
464
- // remains replayable during both attempts, so a transient timeout never becomes a tombstone.
465
- try {
466
- ref = await writeResponseSpillDurablyAsync(job.id, state, {
467
- aclBudgetMs: responseSpillAsyncAclAttemptBudgetMs(),
468
- retryTimedOutOnce: true,
469
- publicationControl: job.publicationControl,
470
- });
471
- } catch (retryError) {
472
- exhaustedAclRetry = isAclTimeout(retryError);
473
- // A returned timeout can also mean an exhausted budget before the next OS command.
474
- aclRetryFailureOrigin = spillAclMemoRefusalOrigin(retryError)
475
- ?? (exhaustedAclRetry ? "retry_returned_timeout" : null);
476
- throw retryError;
477
- }
478
- }
479
- if (ref.payloadBytes > responseSpillPayloadCap()) {
480
- deleteResponseSpill(ref);
481
- ref = null;
482
- if (job.directAdmission) admissionCounters.oversizedDrops += 1;
483
- throw Object.assign(new Error("Response spill payload exceeds replay ceiling"), { code: "EFBIG" });
484
- }
485
- if (states.get(job.id) !== candidate || job.cancelled) {
486
- deleteResponseSpill(ref);
487
- ref = null;
488
- return;
489
- }
490
- if (swapResidentForSpill(job.id, candidate, ref)) {
491
- ref = null;
492
- noteSpillWriteSuccess();
493
- if (job.directAdmission) admissionCounters.directSpills += 1;
494
- deferSupersededSpill(job.supersededSpill);
495
- }
496
- } catch (error) {
497
- if (ref) deleteResponseSpill(ref);
498
- if (states.get(job.id) === candidate && !job.cancelled) {
499
- noteSpillWriteFailure(error, exhaustedAclRetry ? "EACLRETRYEXHAUSTED" : undefined, aclRetryFailureOrigin);
500
- replaceWithSpillFailure(job.id, candidate);
501
- deferSupersededSpill(job.supersededSpill);
502
- }
503
- } finally {
504
- const cancelled = job.cancelled;
505
- releasePendingResponseSpill(job);
506
- recomputeOldestResident();
507
- if (!cancelled) {
508
- schedulePersist();
509
- pruneResponses();
510
- enforceAppOwnedMemoryBudget();
511
- }
512
- }
513
- }
514
-
515
- function queuePendingResponseSpill(
516
- id: string,
517
- candidate: ResidentResponseState,
518
- options: { supersededSpill?: ResponseSpillRef; directAdmission?: boolean } = {},
519
- ): void {
520
- const inheritedSpill = cancelPendingResponseSpill(id) ?? options.supersededSpill;
521
- if (pendingResponseSpillBytes + candidate.sizeBytes > MAX_PENDING_RESPONSE_SPILL_BYTES) {
522
- noteSpillWriteFailure(null, "ECAPACITY");
523
- replaceWithSpillFailure(id, candidate);
524
- deferSupersededSpill(inheritedSpill);
525
- return;
526
- }
527
- // Enforce the disk cap BEFORE the temp or destination file is created. Deleting the
528
- // overflow afterwards is not equivalent: on Windows the file can outlive the decision
529
- // by as long as ACL hardening takes, which is the window the measured 6.8 GiB
530
- // accumulated in. Reclaim first, and only refuse if the peak footprint still does not
531
- // fit — an eviction pass can free a live continuation's worth of room.
532
- const footprint = publicationFootprintBytes(id, candidate);
533
- // The superseded generation this job is about to own is already off `states` and not
534
- // yet on the job, so it is invisible to the walk. Price it here or admission decides
535
- // against a total that is short by a whole envelope.
536
- const inheritedBytes = inheritedSpill?.payloadBytes ?? 0;
537
- if (accountedResponseSpillBytes() + footprint + inheritedBytes > spillByteCap()) {
538
- enforceSpilledResponseBudget();
539
- if (accountedResponseSpillBytes() + footprint + inheritedBytes > spillByteCap()) {
540
- noteSpillWriteFailure(null, "ECAPACITY");
541
- replaceWithSpillFailure(id, candidate);
542
- deferSupersededSpill(inheritedSpill);
543
- return;
544
- }
545
- }
546
- const job: PendingResponseSpill = {
547
- id,
548
- candidate,
549
- ...(inheritedSpill ? { supersededSpill: inheritedSpill } : {}),
550
- directAdmission: options.directAdmission === true,
551
- running: false,
552
- cancelled: false,
553
- released: false,
554
- sizeBytes: candidate.sizeBytes,
555
- reservedBytes: footprint,
556
- publicationControl: createResponseSpillPublicationControl(),
557
- };
558
- pendingResponseSpills.add(job);
559
- pendingResponseSpillById.set(id, job);
560
- pendingResponseSpillBytes += job.sizeBytes;
561
- reservedResponseSpillBytes += job.reservedBytes;
562
- recomputeOldestResident();
563
- responseSpillPublicationTail = responseSpillPublicationTail
564
- .then(() => runPendingResponseSpill(job), () => runPendingResponseSpill(job));
565
- }
566
-
567
- function replaceWithPendingResponseSpill(
568
- id: string,
569
- candidate: ResidentResponseState,
570
- expected: StoredResponseState | undefined,
571
- options: { directAdmission?: boolean } = {},
572
- ): boolean {
573
- const inheritedSpill = pendingResponseSpillById.get(id)?.supersededSpill
574
- ?? (expected?.kind === "spill" ? expected.spill : undefined);
575
- if (!replaceMapEntry(id, candidate, expected)) return false;
576
- queuePendingResponseSpill(id, candidate, {
577
- ...(inheritedSpill ? { supersededSpill: inheritedSpill } : {}),
578
- directAdmission: options.directAdmission === true,
579
- });
580
- return true;
581
- }
582
-
583
- /** Test-only: settle every serialized Windows spill publication. */
584
- export async function flushPendingResponseSpillsForTests(): Promise<void> {
585
- await drainResponseSpillPublications();
586
- }
587
-
588
- /** Test-only: observe ordinary queue settlement without invoking shutdown fallback. */
589
- export async function awaitResponseSpillPublicationTailForTests(): Promise<void> {
590
- await responseSpillPublicationTail;
591
- }
592
-
593
- /** Test-only: observe the bounded queue without exposing payloads. */
594
- export function pendingResponseSpillMetricsForTests(): { count: number; bytes: number } {
595
- return { count: pendingResponseSpills.size, bytes: pendingResponseSpillBytes };
596
- }
597
-
598
- /** Test-only: shorten the shutdown drain/fallback budget (null restores production values). */
599
- export function setResponseSpillShutdownBudgetForTests(
600
- budget: { totalMs: number; fallbackReserveMs: number } | null,
601
- ): void {
602
- responseSpillShutdownBudgetOverride = budget;
603
- }
604
-
605
- /** Test-only: shorten the ordinary async whole-attempt ACL budget. */
606
- export function setResponseSpillAsyncAclAttemptBudgetForTests(budgetMs: number | null): void {
607
- responseSpillAsyncAclAttemptBudgetOverride = budgetMs;
608
- }
609
-
610
- function responseSpillAsyncAclAttemptBudgetMs(): number {
611
- return responseSpillAsyncAclAttemptBudgetOverride ?? RESPONSE_SPILL_ASYNC_ACL_ATTEMPT_BUDGET_MS;
612
- }
613
-
614
- /** Test-only: lower the hard terminalization pass guard (null restores production). */
615
- export function setResponseSpillShutdownTerminalizationPassLimitForTests(limit: number | null): void {
616
- responseSpillShutdownTerminalizationPassLimitOverride = limit;
617
- }
618
-
619
- function responseSpillShutdownTerminalizationPassLimit(): number {
620
- return responseSpillShutdownTerminalizationPassLimitOverride
621
- ?? RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES;
622
- }
623
-
624
- function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: number } {
625
- return responseSpillShutdownBudgetOverride ?? {
626
- totalMs: RESPONSE_SPILL_SHUTDOWN_BUDGET_MS,
627
- fallbackReserveMs: RESPONSE_SPILL_SHUTDOWN_FALLBACK_RESERVE_MS,
628
- };
629
- }
630
-
631
- function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
632
- const remaining = deadline - Date.now();
633
- if (remaining <= 0) return Promise.resolve(false);
634
- return new Promise(resolve => {
635
- let finished = false;
636
- const finish = (settled: boolean): void => {
637
- if (finished) return;
638
- finished = true;
639
- clearTimeout(timer);
640
- resolve(settled);
641
- };
642
- const timer = setTimeout(() => finish(false), remaining);
643
- observed.then(() => finish(true), () => finish(true));
644
- });
645
- }
646
-
647
- function installShutdownFallbackSpill(
648
- job: PendingResponseSpill,
649
- candidate: ResidentResponseState,
650
- aclBudgetMs: number,
651
- ): void {
652
- let ref: ResponseSpillRef | null = null;
653
- // Supersession released this job's reservation, but the synchronous write below is the
654
- // largest publication of the shutdown path and has its own link-then-copy fallback
655
- // holding a temp and a destination at once. Re-reserve for its duration so the cap is
656
- // not blind exactly where the drain does its heaviest work, and settle in `finally` so
657
- // every return, throw and mismatch releases it.
658
- const footprint = publicationFootprintBytes(job.id, candidate);
659
- reservedResponseSpillBytes += footprint;
660
- try {
661
- // Supersession released this job, so its superseded generation is no longer visible
662
- // to the accounting walk — but the file is still on the volume until
663
- // `deferSupersededSpill` or a delete takes it. Price it here or the fallback decides
664
- // against a total short by that whole envelope, which is exactly the gap that lets
665
- // `debt + footprint <= cap < old + debt + footprint` publish over budget.
666
- const supersededBytes = job.supersededSpill?.payloadBytes ?? 0;
667
- // The drain must not publish over the cap either. Reclaim first; if the footprint
668
- // still does not fit — which is what unreclaimable cleanup debt looks like — the
669
- // honest close-out is a tombstone, not another file on a volume that is already
670
- // over budget. `replaceWithSpillFailure` is the same fail-closed ending the budget
671
- // exhaustion path uses, so replay reports `spill_failed` and the client resends.
672
- if (accountedResponseSpillBytes() + supersededBytes > spillByteCap()) {
673
- enforceSpilledResponseBudget();
674
- if (accountedResponseSpillBytes() + supersededBytes > spillByteCap()) {
675
- if (states.get(job.id) === candidate) {
676
- noteSpillWriteFailure(null, "ECAPACITY");
677
- replaceWithSpillFailure(job.id, candidate);
678
- deferSupersededSpill(job.supersededSpill);
679
- }
680
- throw Object.assign(new Error("Response spill shutdown fallback exceeds the durable disk cap"), { code: "ENOSPC" });
681
- }
682
- }
683
- ref = writeResponseSpillDurably(job.id, spillPayloadForResident(candidate), { aclBudgetMs });
684
- if (ref.payloadBytes > responseSpillPayloadCap()) {
685
- deleteResponseSpill(ref);
686
- ref = null;
687
- if (job.directAdmission) admissionCounters.oversizedDrops += 1;
688
- throw Object.assign(new Error("Response spill payload exceeds replay ceiling"), { code: "EFBIG" });
689
- }
690
- if (states.get(job.id) !== candidate) {
691
- deleteResponseSpill(ref);
692
- ref = null;
693
- return;
694
- }
695
- if (swapResidentForSpill(job.id, candidate, ref)) {
696
- ref = null;
697
- noteSpillWriteSuccess();
698
- if (job.directAdmission) admissionCounters.directSpills += 1;
699
- deferSupersededSpill(job.supersededSpill);
700
- }
701
- } catch (error) {
702
- if (ref) deleteResponseSpill(ref);
703
- if (states.get(job.id) === candidate) {
704
- noteSpillWriteFailure(error);
705
- replaceWithSpillFailure(job.id, candidate);
706
- deferSupersededSpill(job.supersededSpill);
707
- }
708
- throw error;
709
- } finally {
710
- reservedResponseSpillBytes = Math.max(0, reservedResponseSpillBytes - footprint);
711
- }
712
- }
713
-
714
- function terminalizeShutdownFallbackCandidate(
715
- job: PendingResponseSpill,
716
- candidate: ResidentResponseState,
717
- failureCode: ResponseSpillWriteFailureCode = "ETIMEDOUT",
718
- ): void {
719
- if (states.get(job.id) !== candidate) return;
720
- noteSpillWriteFailure(null, failureCode);
721
- replaceWithSpillFailure(job.id, candidate);
722
- deferSupersededSpill(job.supersededSpill);
723
- }
724
-
725
- function pendingShutdownFallbackCandidates(): Array<{
726
- job: PendingResponseSpill;
727
- candidate: ResidentResponseState;
728
- }> {
729
- return [...pendingResponseSpills]
730
- .map(job => ({ job, candidate: job.candidate }))
731
- .filter((entry): entry is { job: PendingResponseSpill; candidate: ResidentResponseState } => !!entry.candidate);
732
- }
733
-
734
- function supersedeShutdownFallbackBatch(
735
- pending: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
736
- failures: Error[],
737
- ): void {
738
- for (const { job } of pending) {
739
- job.cancelled = true;
740
- markResponseSpillPublicationSuperseded(job.publicationControl);
741
- }
742
- for (const { job } of pending) {
743
- const cleanupFailure = cleanupSupersededResponseSpillPublication(job.publicationControl);
744
- if (cleanupFailure) {
745
- failures.push(cleanupFailure);
746
- // Cleanup failed, so an async temp or destination is STILL on the volume. Releasing
747
- // the reservation would un-account a file that exists, and the fallback write that
748
- // follows reserves only its own footprint — three envelopes on disk priced as two.
749
- //
750
- // Charge the surviving PATHS rather than a flat two envelopes: `clearOwnedPath`
751
- // nulls whichever it managed to remove, so one failure is one file, not two. The
752
- // charge is settled automatically once the path disappears, which a retried unlink
753
- // or a released Windows lock can still do.
754
- const perPath = Math.max(1, Math.floor(job.reservedBytes / 2));
755
- chargeUnreclaimableSpillPath(job.publicationControl.tempPath, perPath);
756
- chargeUnreclaimableSpillPath(job.publicationControl.destinationPath, perPath);
757
- }
758
- releasePendingResponseSpill(job);
759
- }
760
- }
761
-
762
- function stopAtShutdownTerminalizationPassLimit(
763
- pending: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
764
- failures: Error[],
765
- ): void {
766
- failures.push(Object.assign(new Error("Response spill shutdown terminalization pass limit exceeded"), { code: "ELOOP" }));
767
- supersedeShutdownFallbackBatch(pending, failures);
768
- for (const { job, candidate } of pending) {
769
- terminalizeShutdownFallbackCandidate(job, candidate, "ELOOP");
770
- }
771
- for (const [id, state] of [...states]) {
772
- if (state.kind !== "resident") continue;
773
- noteSpillWriteFailure(null, "ELOOP");
774
- replaceWithSpillFailure(id, state);
775
- }
776
- recomputeOldestResident();
777
- pruneResponses();
778
- enforceAppOwnedMemoryBudget();
779
- }
780
-
781
- function terminalizeExhaustedShutdownFallback(
782
- initial: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
783
- failures: Error[],
784
- ): void {
785
- let pending = initial;
786
- let passes = 0;
787
- const passLimit = responseSpillShutdownTerminalizationPassLimit();
788
- // Every pass replaces each captured resident with a tombstone. Pruning may expose
789
- // another finite batch, but resident count strictly decreases until none can requeue.
790
- while (pending.length > 0) {
791
- if (passes >= passLimit) {
792
- stopAtShutdownTerminalizationPassLimit(pending, failures);
793
- return;
794
- }
795
- passes += 1;
796
- supersedeShutdownFallbackBatch(pending, failures);
797
- for (const { job, candidate } of pending) {
798
- failures.push(Object.assign(new Error("Response spill shutdown fallback budget exhausted"), { code: "ETIMEDOUT" }));
799
- terminalizeShutdownFallbackCandidate(job, candidate);
800
- }
801
- recomputeOldestResident();
802
- pruneResponses();
803
- enforceAppOwnedMemoryBudget();
804
- pending = pendingShutdownFallbackCandidates();
805
- }
806
- }
807
-
808
- function fallbackPendingResponseSpills(reserveMs: number): Error[] {
809
- const deadline = Date.now() + reserveMs;
810
- const failures: Error[] = [];
811
- for (;;) {
812
- const pending = pendingShutdownFallbackCandidates();
813
- if (pending.length === 0) return failures;
814
- if (Date.now() >= deadline) {
815
- terminalizeExhaustedShutdownFallback(pending, failures);
816
- return failures;
817
- }
818
-
819
- supersedeShutdownFallbackBatch(pending, failures);
820
- let reserveExhausted = false;
821
- for (let index = 0; index < pending.length; index += 1) {
822
- const { job, candidate } = pending[index]!;
823
- if (states.get(job.id) !== candidate) continue;
824
- const remaining = deadline - Date.now();
825
- if (remaining <= 0) {
826
- reserveExhausted = true;
827
- for (const exhausted of pending.slice(index)) {
828
- failures.push(Object.assign(new Error("Response spill shutdown fallback budget exhausted"), { code: "ETIMEDOUT" }));
829
- terminalizeShutdownFallbackCandidate(exhausted.job, exhausted.candidate);
830
- }
831
- break;
832
- }
833
- try {
834
- installShutdownFallbackSpill(job, candidate, remaining);
835
- } catch (error) {
836
- failures.push(error instanceof Error ? error : new Error("Response spill shutdown fallback failed"));
837
- }
838
- }
839
- recomputeOldestResident();
840
- pruneResponses();
841
- enforceAppOwnedMemoryBudget();
842
- if (reserveExhausted || Date.now() >= deadline) {
843
- terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
844
- return failures;
845
- }
846
- }
847
- }
848
-
849
- async function drainResponseSpillPublications(): Promise<void> {
850
- const budget = responseSpillShutdownBudget();
851
- const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
852
- const drainDeadline = Date.now() + Math.max(0, budget.totalMs - fallbackReserveMs);
853
-
854
- for (;;) {
855
- if (pendingResponseSpills.size === 0) return;
856
- const observed = responseSpillPublicationTail;
857
- const settled = await awaitResponseSpillTailUntil(observed, drainDeadline);
858
- if (!settled) {
859
- const failures = fallbackPendingResponseSpills(fallbackReserveMs);
860
- if (failures.length > 0) {
861
- throw new AggregateError(failures, "Response spill shutdown fallback incomplete");
862
- }
863
- return;
864
- }
865
- if (observed === responseSpillPublicationTail) return;
866
- }
867
- }
868
207
 
869
208
  function byteCap(): number {
870
209
  return byteCapOverride ?? MAX_STORED_RESPONSE_BYTES;
@@ -920,12 +259,9 @@ function accountedResponseSpillBytes(): number {
920
259
  // counting only `states` plus `pendingSpillUnlinks` loses it for the whole publication
921
260
  // — during a copy fallback that is old generation + new temp + new destination, three
922
261
  // envelopes priced as two.
923
- let ownedBySpillJobs = 0;
924
- for (const job of pendingResponseSpills) {
925
- if (job.supersededSpill) ownedBySpillJobs += job.supersededSpill.payloadBytes;
926
- }
927
- return spilledResponseBytes() + reservedResponseSpillBytes + ownedBySpillJobs
928
- + reconcileUnreclaimableSpillPaths();
262
+ const accounting = spillQueueAccounting();
263
+ return spilledResponseBytes() + accounting.reservedBytes + accounting.jobOwnedBytes
264
+ + accounting.unreclaimableBytes;
929
265
  }
930
266
 
931
267
  /** Test-only: lower/restore the durable spill cap (null restores the default). */
@@ -969,7 +305,7 @@ function recomputeOldestResident(): void {
969
305
  oldestResidentAt = null;
970
306
  for (const [id, state] of states) {
971
307
  if (state.kind !== "resident") continue;
972
- if (pendingResponseSpillById.get(id)?.candidate === state) continue;
308
+ if (spillQueueHoldsResidentCandidate(id, state)) continue;
973
309
  if (oldestResidentAt !== null && state.createdAt >= oldestResidentAt) continue;
974
310
  oldestResidentId = id;
975
311
  oldestResidentAt = state.createdAt;
@@ -1134,8 +470,7 @@ function setResidentEntry(id: string, entry: ResidentInput): void {
1134
470
  pruneResponses();
1135
471
  return;
1136
472
  }
1137
- const pending = pendingResponseSpillById.get(id);
1138
- if (windowsSecretAclApplies() && (expected?.kind === "spill" || pending?.supersededSpill)) {
473
+ if (windowsSecretAclApplies() && (expected?.kind === "spill" || spillQueueSupersededSpillFor(id))) {
1139
474
  replaceWithPendingResponseSpill(id, candidate, expected);
1140
475
  pruneResponses();
1141
476
  return;
@@ -1219,6 +554,23 @@ function admitOversizedCandidate(
1219
554
  }
1220
555
  }
1221
556
 
557
+ bindSpillQueueStore({
558
+ swapResidentForSpill,
559
+ replaceWithSpillFailure,
560
+ deleteEntry,
561
+ deferSupersededSpill,
562
+ replaceMapEntry,
563
+ currentEntry: (id: string) => states.get(id),
564
+ residentEntries: () => [...states],
565
+ recomputeOldestResident,
566
+ schedulePersist,
567
+ pruneResponses,
568
+ accountedResponseSpillBytes,
569
+ spillByteCap,
570
+ enforceSpilledResponseBudget,
571
+ terminalizationMaxPasses: () => RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES,
572
+ });
573
+
1222
574
  // Replay provenance must stay proxy-private: a WeakMap distinguishes replayed history from the
1223
575
  // newly appended input suffix without adding an unknown field that native passthrough could send
1224
576
  // upstream. The parser uses this boundary to acknowledge historical compaction markers exactly
@@ -1241,284 +593,6 @@ function snapshotPath(): string {
1241
593
  return join(getConfigDir(), "responses-state.json");
1242
594
  }
1243
595
 
1244
- interface LegacySnapshotState {
1245
- createdAt?: unknown;
1246
- clientThreadId?: unknown;
1247
- items?: unknown;
1248
- providers?: OcxProviderContinuationState;
1249
- conversationId?: unknown;
1250
- cursorCheckpointUsable?: unknown;
1251
- }
1252
-
1253
- function isSpillRef(value: unknown): value is ResponseSpillRef {
1254
- if (!value || typeof value !== "object" || Array.isArray(value)) return false;
1255
- const ref = value as ResponseSpillRef;
1256
- return ref.version === 1
1257
- && typeof ref.fileName === "string"
1258
- && /^[0-9a-f]{64}$/.test(ref.digest)
1259
- && Number.isSafeInteger(ref.payloadBytes)
1260
- && ref.payloadBytes >= 0;
1261
- }
1262
-
1263
- function loadSnapshotEntry(id: string, value: unknown): void {
1264
- if (!value || typeof value !== "object" || Array.isArray(value)) return;
1265
- const rec = value as LegacySnapshotState & { kind?: unknown; spill?: unknown };
1266
- if (typeof rec.createdAt !== "number" || !Number.isFinite(rec.createdAt)) return;
1267
- const clientThreadId = typeof rec.clientThreadId === "string" && rec.clientThreadId.trim().length > 0
1268
- ? rec.clientThreadId.trim()
1269
- : undefined;
1270
- // A malformed boundary degrades to "never skip" rather than to a bad index: an untrusted
1271
- // snapshot must not be able to authorize dropping conversation history.
1272
- const anchorFor = (itemCount: number): number | undefined => {
1273
- const raw = (rec as { providerOutputStart?: unknown }).providerOutputStart;
1274
- return Number.isSafeInteger(raw) && (raw as number) >= 0 && (raw as number) <= itemCount
1275
- ? raw as number
1276
- : undefined;
1277
- };
1278
- if (rec.kind === "spill") {
1279
- if (!isSpillRef(rec.spill)) return;
1280
- const base: Omit<SpilledResponseState, "sizeBytes"> = {
1281
- kind: "spill",
1282
- createdAt: rec.createdAt,
1283
- ...(clientThreadId ? { clientThreadId } : {}),
1284
- // Item count is unknown until materialization, so accept any non-negative integer
1285
- // here; the spill payload validator re-checks it against the real array.
1286
- ...(anchorFor(Number.MAX_SAFE_INTEGER) !== undefined ? { providerOutputStart: anchorFor(Number.MAX_SAFE_INTEGER) } : {}),
1287
- ...(rec.providers ? { providers: rec.providers } : {}),
1288
- spill: rec.spill,
1289
- };
1290
- replaceMapEntry(id, { ...base, sizeBytes: stubSize(id, base) });
1291
- return;
1292
- }
1293
- if (rec.kind === "spill-failed") {
1294
- replaceMapEntry(id, tombstone(id, rec.createdAt));
1295
- return;
1296
- }
1297
- if (rec.kind !== undefined && rec.kind !== "resident") return;
1298
- if (!Array.isArray(rec.items)) return;
1299
- const providers = rec.providers ?? (typeof rec.conversationId === "string"
1300
- ? {
1301
- cursor: {
1302
- conversationId: rec.conversationId,
1303
- ...(typeof rec.cursorCheckpointUsable === "boolean"
1304
- ? { checkpointUsable: rec.cursorCheckpointUsable }
1305
- : {}),
1306
- },
1307
- }
1308
- : undefined);
1309
- const resident = measureResidentEntry(id, {
1310
- createdAt: rec.createdAt,
1311
- ...(clientThreadId ? { clientThreadId } : {}),
1312
- items: rec.items,
1313
- ...(anchorFor(rec.items.length) !== undefined ? { providerOutputStart: anchorFor(rec.items.length) } : {}),
1314
- ...(providers ? { providers } : {}),
1315
- });
1316
- if (!resident) {
1317
- replaceMapEntry(id, tombstone(id, rec.createdAt));
1318
- return;
1319
- }
1320
- // Same admission boundary as live writes: an oversized snapshot row goes
1321
- // straight to spill (or tombstone above the payload ceiling) instead of
1322
- // entering the resident map and demoting unrelated rows on the first prune.
1323
- if (resident.sizeBytes > byteCap()) {
1324
- admitOversizedCandidate(id, resident, undefined);
1325
- return;
1326
- }
1327
- replaceMapEntry(id, resident);
1328
- }
1329
-
1330
- export interface ResponseStateTempRecoveryResult {
1331
- matched: number;
1332
- removed: number;
1333
- failed: number;
1334
- bytesRemoved: number;
1335
- /** Entries that passed EVERY gate and would be reclaimed. In a dry run nothing is
1336
- * unlinked, so this is the only honest count to show an operator: `matched` is
1337
- * incremented before the file-type, age, boot-floor, and liveness gates. */
1338
- eligible: number;
1339
- /** Total size of the `eligible` entries. */
1340
- eligibleBytes: number;
1341
- /** The scan stopped on a budget (entry cap, cleanup cap, or deadline) rather than reaching
1342
- * the end of the directory, so the counts below describe a prefix of the backlog and not
1343
- * the backlog. `eligible > removed + failed` cannot express this: outside a dry run every
1344
- * eligible entry is unlinked or failed on the same iteration, so the two are always equal
1345
- * and a comparison between them is dead code. */
1346
- truncated: boolean;
1347
- }
1348
-
1349
- interface ResponseStateTempRecoveryIO {
1350
- now: () => number;
1351
- /** Approximate epoch ms of the current boot; see the boot floor in the scan loop. */
1352
- bootTime: () => number;
1353
- list: (dir: string) => Iterable<string>;
1354
- inspect: (path: string) => { isFile: boolean; mtimeMs: number; size: number };
1355
- isProcessAlive: (pid: number) => boolean;
1356
- unlink: (path: string) => void;
1357
- }
1358
-
1359
- export type ResponseStateTempRecoveryOptions = Partial<ResponseStateTempRecoveryIO> & {
1360
- maxEntries?: number;
1361
- maxCleanups?: number;
1362
- /** Wall-clock ceiling for the scan, or null/undefined for no deadline (startup path). */
1363
- deadlineMs?: number | null;
1364
- /** Report only: apply every gate, count what would be reclaimed, unlink nothing. */
1365
- dryRun?: boolean;
1366
- };
1367
-
1368
- function processIsAlive(pid: number): boolean {
1369
- if (pid === process.pid) return true;
1370
- try {
1371
- process.kill(pid, 0);
1372
- return true;
1373
- } catch (error) {
1374
- // EPERM means the process exists but cannot be signalled. Unknown platform errors
1375
- // are also protected; cleanup should prefer a false negative over touching a live writer.
1376
- return (error as NodeJS.ErrnoException).code !== "ESRCH";
1377
- }
1378
- }
1379
-
1380
- const responseStateTempRecoveryIO: ResponseStateTempRecoveryIO = {
1381
- now: Date.now,
1382
- bootTime: () => Date.now() - uptime() * 1_000,
1383
- list: function* list(dir) {
1384
- const handle = opendirSync(dir);
1385
- try {
1386
- for (let entry = handle.readSync(); entry; entry = handle.readSync()) yield entry.name;
1387
- } finally {
1388
- handle.closeSync();
1389
- }
1390
- },
1391
- inspect: path => {
1392
- const stat = lstatSync(path);
1393
- return { isFile: stat.isFile() && !stat.isSymbolicLink(), mtimeMs: stat.mtimeMs, size: stat.size };
1394
- },
1395
- isProcessAlive: processIsAlive,
1396
- unlink: unlinkSync,
1397
- };
1398
-
1399
- /**
1400
- * Recover only abandoned response-state atomic-write files. The exact basename,
1401
- * regular-file check, age gate, and PID liveness check protect unrelated/active files.
1402
- * Cleanup is capped and best-effort because continuation state is only a cache. Removal
1403
- * deliberately uses unlink only: path-based truncation could follow a replacement symlink.
1404
- */
1405
- export function recoverStaleResponseStateTemps(
1406
- dir = getConfigDir(),
1407
- options: ResponseStateTempRecoveryOptions = {},
1408
- ): ResponseStateTempRecoveryResult {
1409
- const {
1410
- maxEntries = STALE_TEMP_MAX_ENTRIES,
1411
- maxCleanups = STALE_TEMP_MAX_CLEANUPS,
1412
- deadlineMs = null,
1413
- dryRun = false,
1414
- ...overrides
1415
- } = options;
1416
- const io = { ...responseStateTempRecoveryIO, ...overrides };
1417
- const result: ResponseStateTempRecoveryResult = {
1418
- matched: 0,
1419
- removed: 0,
1420
- failed: 0,
1421
- bytesRemoved: 0,
1422
- eligible: 0,
1423
- eligibleBytes: 0,
1424
- truncated: false,
1425
- };
1426
- const startedAt = io.now();
1427
- // One probe per scan, not one per entry. A non-finite or future-dated boot is anomalous, and
1428
- // clamping it to "now" would be the WORST response: the floor would then retire the liveness
1429
- // probe for every file older than the skew, which is every file past the grace. Disable it
1430
- // instead -- an absent floor only costs a missed reclaim, never a wrong one.
1431
- const rawBoot = io.bootTime();
1432
- const bootMs = Number.isFinite(rawBoot) && rawBoot <= startedAt ? rawBoot : Number.NEGATIVE_INFINITY;
1433
- let names: Iterable<string>;
1434
- try { names = io.list(dir); } catch { return result; }
1435
- let iterator: Iterator<string>;
1436
- try { iterator = names[Symbol.iterator](); } catch { return result; }
1437
- let scanned = 0;
1438
- // Every early exit runs through this. The production `list` is a generator that closes its
1439
- // directory handle in a `finally`, and a `finally` does NOT run when the consumer simply
1440
- // stops calling `next()` -- only `return()` resumes the generator to completion. Breaking
1441
- // out of the loop directly therefore leaked one directory handle per truncated scan, and the
1442
- // periodic reclaim truncates on purpose (entry cap, cleanup cap, deadline), so on a slow
1443
- // filesystem that is a leak per tick, forever.
1444
- const stopScan = (): ResponseStateTempRecoveryResult => {
1445
- try { iterator.return?.(); } catch { /* closing is best-effort; never fail a reclaim on it */ }
1446
- return result;
1447
- };
1448
- for (;;) {
1449
- let next: IteratorResult<string>;
1450
- try { next = iterator.next(); } catch { return result; }
1451
- if (next.done) break;
1452
- const name = next.value;
1453
- scanned += 1;
1454
- // A dry run performs no cleanups, so bounding it by the cleanup budget would truncate
1455
- // the very report an operator uses to size the problem.
1456
- if (scanned > maxEntries) { result.truncated = true; return stopScan(); }
1457
- if (!dryRun && result.removed + result.failed >= maxCleanups) { result.truncated = true; return stopScan(); }
1458
- if (deadlineMs !== null && io.now() - startedAt > deadlineMs) { result.truncated = true; return stopScan(); }
1459
- const match = RESPONSE_STATE_TEMP_NAME.exec(name);
1460
- if (!match) continue;
1461
- result.matched += 1;
1462
- const pid = Number(match[1]);
1463
- const sequence = Number(match[2]);
1464
- if (!Number.isSafeInteger(pid) || pid <= 0 || !Number.isSafeInteger(sequence) || sequence <= 0) continue;
1465
- const path = join(dir, name);
1466
- let file: ReturnType<ResponseStateTempRecoveryIO["inspect"]>;
1467
- try { file = io.inspect(path); } catch { continue; }
1468
- if (!file.isFile || io.now() - file.mtimeMs < STALE_TEMP_GRACE_MS) continue;
1469
- // Boot floor. After a reboot the original writer's pid is routinely reused, which makes
1470
- // the liveness skip PERMANENT: the 15-minute grace above is a lower bound and never
1471
- // expires it, so the file is skipped on every future pass forever. A temp older than
1472
- // this boot cannot be owned by the pid we would probe, so the probe is vacuous and we
1473
- // retire it. This does NOT claim the file is provably dead: under a shared-volume
1474
- // container, suspend-excluding uptime, or a network config dir the computed boot can
1475
- // land after the real one. The unconditional 15-minute grace above remains the safety
1476
- // floor, and this process's own temps are never touched.
1477
- const predatesBoot = file.mtimeMs < bootMs - BOOT_FLOOR_SKEW_MS;
1478
- if (pid === process.pid) continue;
1479
- if (!predatesBoot && io.isProcessAlive(pid)) continue;
1480
-
1481
- result.eligible += 1;
1482
- result.eligibleBytes += file.size;
1483
- if (dryRun) continue;
1484
-
1485
- try {
1486
- io.unlink(path);
1487
- result.removed += 1;
1488
- result.bytesRemoved += file.size;
1489
- } catch (error) {
1490
- // Another proxy sharing this config dir may have won the race. A file that is already
1491
- // gone is reclaimed, not a failure -- reporting it as one would surface "in use or
1492
- // locked" to an operator for a file nobody holds.
1493
- if ((error as NodeJS.ErrnoException)?.code === "ENOENT") {
1494
- result.removed += 1;
1495
- continue;
1496
- }
1497
- // Locked files remain for a later startup. Do not truncate by path: a same-user
1498
- // replacement could turn that fallback into an arbitrary symlink-target write.
1499
- result.failed += 1;
1500
- }
1501
- }
1502
- return result;
1503
- }
1504
-
1505
- /**
1506
- * Literal config dir plus the snapshot's resolved dir. Atomic writes place their temp beside
1507
- * the RESOLVED target, so a symlinked snapshot (dotfiles-managed config dir) strands temps in
1508
- * the link's real directory where a scan of the literal dir would never see them. The two
1509
- * collapse to one when nothing is symlinked.
1510
- */
1511
- function responseStateSweepDirectories(): Set<string> {
1512
- const path = snapshotPath();
1513
- let resolvedDir = dirname(path);
1514
- try {
1515
- resolvedDir = dirname(resolveWriteTarget(path));
1516
- } catch {
1517
- /* unresolvable link: sweep the literal dir only */
1518
- }
1519
- return new Set([dirname(path), resolvedDir]);
1520
- }
1521
-
1522
596
  /**
1523
597
  * Best-effort disk snapshot so previous_response_id chains survive a proxy restart (the
1524
598
  * dominant expansion-miss cause: an in-memory-only store dies with the process, and the next
@@ -1565,7 +639,14 @@ function ensureLoaded(): void {
1565
639
  if ((raw.version === 1 || raw.version === 2) && Array.isArray(raw.states)) {
1566
640
  for (const entry of raw.states) {
1567
641
  if (!Array.isArray(entry) || entry.length !== 2 || typeof entry[0] !== "string") continue;
1568
- loadSnapshotEntry(entry[0], entry[1]);
642
+ loadSnapshotEntry(entry[0], entry[1], {
643
+ replaceMapEntry,
644
+ stubSize,
645
+ tombstone,
646
+ measureResidentEntry,
647
+ admitOversizedCandidate,
648
+ byteCap,
649
+ });
1569
650
  }
1570
651
  }
1571
652
  }
@@ -1748,89 +829,8 @@ function inputItems(input: unknown): unknown[] {
1748
829
  return [input];
1749
830
  }
1750
831
 
1751
- /** Hard cap for canonicalizing ANY item. Past it, the item is not comparable. */
1752
- const REPLAY_FINGERPRINT_MAX_BYTES = 8 * 1024;
1753
- /** Depth ceiling so a pathologically nested item cannot blow the canonicalizer. */
1754
- const REPLAY_FINGERPRINT_MAX_DEPTH = 64;
1755
-
1756
832
  let replayOverlapSkips = 0;
1757
833
 
1758
- /**
1759
- * Canonical, order-stable fingerprint for one input item, or null when the item cannot be
1760
- * compared safely.
1761
- *
1762
- * Byte-counted DURING the walk rather than serialize-then-measure: a tool result can be
1763
- * megabytes and this runs on the request path, so the point of the cap is to stop early,
1764
- * not to discover afterwards that we should have. Object keys are sorted so two
1765
- * semantically identical items cannot differ by key order alone.
1766
- *
1767
- * The cap applies to EVERY item. An `id`/`call_id` is additional occurrence evidence, never
1768
- * a substitute for content equality, so an over-cap identified tool item is non-comparable
1769
- * exactly like an over-cap message.
1770
- */
1771
- function replayItemFingerprint(item: unknown): string | null {
1772
- const out: string[] = [];
1773
- let bytes = 0;
1774
- const push = (text: string): boolean => {
1775
- bytes += Buffer.byteLength(text, "utf8");
1776
- if (bytes > REPLAY_FINGERPRINT_MAX_BYTES) return false;
1777
- out.push(text);
1778
- return true;
1779
- };
1780
- const walk = (value: unknown, depth: number): boolean => {
1781
- if (depth > REPLAY_FINGERPRINT_MAX_DEPTH) return false;
1782
- if (value === null || typeof value !== "object") return push(JSON.stringify(value) ?? "null");
1783
- if (Array.isArray(value)) {
1784
- if (!push("[")) return false;
1785
- for (const element of value) {
1786
- if (!walk(element, depth + 1)) return false;
1787
- if (!push(",")) return false;
1788
- }
1789
- return push("]");
1790
- }
1791
- if (!push("{")) return false;
1792
- for (const key of Object.keys(value as Record<string, unknown>).sort()) {
1793
- if (!push(JSON.stringify(key))) return false;
1794
- if (!walk((value as Record<string, unknown>)[key], depth + 1)) return false;
1795
- if (!push(",")) return false;
1796
- }
1797
- return push("}");
1798
- };
1799
- return walk(item, 0) ? out.join("") : null;
1800
- }
1801
-
1802
- /** Non-empty provider-issued `id`/`call_id` on an item, else null. */
1803
- function providerIssuedIdentity(item: unknown): string | null {
1804
- if (!item || typeof item !== "object" || Array.isArray(item)) return null;
1805
- const record = item as { id?: unknown; call_id?: unknown };
1806
- for (const candidate of [record.id, record.call_id]) {
1807
- if (typeof candidate === "string" && candidate.trim().length > 0) return candidate;
1808
- }
1809
- return null;
1810
- }
1811
-
1812
- /**
1813
- * Number of leading stored items the client already carries verbatim, or 0.
1814
- *
1815
- * Requires an exact ordered run: every stored item must match the client input item at the
1816
- * same index. Any not-comparable item aborts to 0 — skipping just that item could align two
1817
- * different occurrences and manufacture a false positive, and a false positive here deletes
1818
- * real conversation history.
1819
- *
1820
- * Known gap (FU-2): stored input can contain proxy-injected guidance the client never saw,
1821
- * and ids repaired after recording. Those sessions do not match here and expand as before.
1822
- */
1823
- function clientCarriedPrefixLength(stored: readonly unknown[], clientInput: readonly unknown[]): number {
1824
- if (stored.length === 0 || clientInput.length < stored.length) return 0;
1825
- for (let index = 0; index < stored.length; index += 1) {
1826
- const storedPrint = replayItemFingerprint(stored[index]);
1827
- if (storedPrint === null) return 0;
1828
- const clientPrint = replayItemFingerprint(clientInput[index]);
1829
- if (clientPrint === null || storedPrint !== clientPrint) return 0;
1830
- }
1831
- return stored.length;
1832
- }
1833
-
1834
834
  /** Test-only: replay prepends skipped because the client already carried the history. */
1835
835
  export function replayOverlapSkipsForTests(): number {
1836
836
  return replayOverlapSkips;
@@ -1903,9 +903,9 @@ function pruneResponses(at = now()): void {
1903
903
  // deleted only when even their bounded metadata cannot fit the override.
1904
904
  while (storedResponseBytes > byteCap() && states.size > 0) {
1905
905
  const oldestResident = [...states].find(([id, entry]) => entry.kind === "resident"
1906
- && pendingResponseSpillById.get(id)?.candidate !== entry);
906
+ && !spillQueueHoldsResidentCandidate(id, entry));
1907
907
  const hasPendingResident = !oldestResident && [...states].some(([id, entry]) => entry.kind === "resident"
1908
- && pendingResponseSpillById.get(id)?.candidate === entry);
908
+ && spillQueueHoldsResidentCandidate(id, entry));
1909
909
  if (hasPendingResident) break;
1910
910
  const oldestId = oldestResident?.[0] ?? states.keys().next().value as string | undefined;
1911
911
  if (!oldestId) break;
@@ -1950,70 +950,12 @@ export function sweepExpiredResponseStates(at = now()): number {
1950
950
  return removed;
1951
951
  }
1952
952
 
1953
- /**
1954
- * Periodic disk reclaim for abandoned atomic-write temps.
1955
- *
1956
- * `ensureLoaded` sweeps once per process, at load, BEFORE that process writes anything:
1957
- * every `schedulePersist` site is downstream of it. So a process that abandons a temp has
1958
- * already had its only look, the 15-minute grace hides the temp its predecessor's crash
1959
- * just produced, and `maxCleanups` caps a single pass below a large backlog. A restart
1960
- * loop therefore accumulates monotonically. Repeating the reclaim on a timer fixes all
1961
- * three: the grace expires into a later tick and the per-pass cap becomes a per-tick rate.
1962
- *
1963
- * Registered on the sweeper's LIVENESS tick, not the TTL tick: `sweepExpiredOnWrite` puts
1964
- * `sweepExpired` on hot write paths, and a directory scan does not belong there.
1965
- */
1966
- export function reclaimAbandonedResponseStateTemps(
1967
- options: ResponseStateTempRecoveryOptions = {},
1968
- ): ResponseStateTempRecoveryResult {
1969
- const total: ResponseStateTempRecoveryResult = {
1970
- matched: 0, removed: 0, failed: 0, bytesRemoved: 0, eligible: 0, eligibleBytes: 0, truncated: false,
1971
- };
1972
- // The try encloses responseStateSweepDirectories() deliberately: recoverStaleResponseStateTemps
1973
- // already swallows its own enumeration failures, so a catch around only that call would be
1974
- // unreachable. snapshotPath()/getConfigDir() are the paths that can genuinely throw.
1975
- try {
1976
- for (const dir of responseStateSweepDirectories()) {
1977
- const result = recoverStaleResponseStateTemps(dir, options);
1978
- total.matched += result.matched;
1979
- total.removed += result.removed;
1980
- total.failed += result.failed;
1981
- total.bytesRemoved += result.bytesRemoved;
1982
- total.eligible += result.eligible;
1983
- total.eligibleBytes += result.eligibleBytes;
1984
- // Truncation anywhere makes the whole total a prefix.
1985
- total.truncated ||= result.truncated;
1986
- }
1987
- } catch {
1988
- /* best-effort: disk reclaim must never destabilize the caller */
1989
- }
1990
- return total;
1991
- }
1992
-
1993
- /**
1994
- * Report-only counterpart for `ocx doctor`: applies every selection gate and unlinks
1995
- * nothing. It runs the SAME predicate as the reclaim, so the report and the subsequent
1996
- * removal cannot disagree about which files are reclaimable.
1997
- */
1998
- export function inspectAbandonedResponseStateTemps(): ResponseStateTempRecoveryResult {
1999
- return reclaimAbandonedResponseStateTemps({ dryRun: true });
2000
- }
2001
-
2002
- /** Sweeper adapter: narrows the reclaim to the `() => number` the liveness tick expects. */
2003
- export function sweepAbandonedResponseStateTemps(): number {
2004
- return reclaimAbandonedResponseStateTemps({
2005
- maxEntries: PERIODIC_TEMP_MAX_ENTRIES,
2006
- maxCleanups: PERIODIC_TEMP_MAX_CLEANUPS,
2007
- deadlineMs: PERIODIC_TEMP_SCAN_DEADLINE_MS,
2008
- }).removed;
2009
- }
2010
-
2011
953
  export function responseContinuationRetainedStoreSnapshot(): RetainedStoreSnapshot {
2012
954
  let currentPendingBytes = 0;
2013
- for (const job of pendingResponseSpills) {
2014
- if (job.candidate && states.get(job.id) === job.candidate) currentPendingBytes += job.sizeBytes;
955
+ for (const job of spillQueueResidentCandidates()) {
956
+ if (states.get(job.id) === job.candidate) currentPendingBytes += job.sizeBytes;
2015
957
  }
2016
- const detachedPendingBytes = Math.max(0, pendingResponseSpillBytes - currentPendingBytes);
958
+ const detachedPendingBytes = Math.max(0, spillQueuePendingBytes() - currentPendingBytes);
2017
959
  const bytes = storedResponseBytes + detachedPendingBytes;
2018
960
  const evictableBytes = Math.max(0, residentResponseBytes - currentPendingBytes);
2019
961
  return {
@@ -2326,7 +1268,7 @@ export function rememberResponseState(
2326
1268
  // `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
2327
1269
  // HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
2328
1270
  // The passthrough branch records with force so those chains can be expanded locally; the
2329
- // store stays in-memory with a 1h TTL, so this is a proxy-internal continuation cache, not
1271
+ // store stays in-memory under RESPONSE_TTL_MS, so this is a proxy-internal continuation cache, not
2330
1272
  // real server-side response storage.
2331
1273
  if (request.store === false && !opts?.force) return;
2332
1274
  if (typeof response.id !== "string" || !Array.isArray(response.output)) return;
@@ -2390,8 +1332,7 @@ export function clearResponseStateMemoryForTests(): void {
2390
1332
  persistTimer = null;
2391
1333
  }
2392
1334
  pendingPersistPath = null;
2393
- for (const id of [...pendingResponseSpillById.keys()]) cancelPendingResponseSpill(id);
2394
- pendingResponseSpillById.clear();
1335
+ resetSpillQueueForTests();
2395
1336
  states.clear();
2396
1337
  storedResponseBytes = 0;
2397
1338
  residentResponseBytes = 0;
@@ -2421,8 +1362,6 @@ export function clearResponseStateMemoryForTests(): void {
2421
1362
  export function clearResponseStateForTests(): void {
2422
1363
  for (const entry of states.values()) deleteOwnedSpills(entry);
2423
1364
  clearResponseStateMemoryForTests();
2424
- reservedResponseSpillBytes = 0;
2425
- unreclaimableSpillPaths.clear();
2426
1365
  try {
2427
1366
  unlinkSync(snapshotPath());
2428
1367
  } catch {