@bitkyc08/opencodex 2.55.0 → 2.56.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/{index-VuoiWj9J.js → index-D4zuyIxQ.js} +1 -1
- package/gui/dist/index.html +1 -1
- package/package.json +2 -1
- package/src/adapters/base.ts +21 -0
- package/src/adapters/cursor/transport-retry.ts +46 -1
- package/src/adapters/cursor.ts +4 -0
- package/src/adapters/kiro/adapter.ts +42 -1
- package/src/adapters/kiro-retry.ts +23 -4
- package/src/adapters/openai-chat/errors.ts +116 -0
- package/src/adapters/openai-chat/messages.ts +346 -0
- package/src/adapters/openai-chat/passthrough.ts +146 -0
- package/src/adapters/openai-chat/response-events.ts +117 -0
- package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
- package/src/adapters/openai-chat/tool-schema.ts +477 -0
- package/src/adapters/openai-chat/wire.ts +50 -0
- package/src/adapters/openai-chat.ts +33 -1445
- package/src/adapters/openai-responses/canonical-forward.ts +202 -0
- package/src/adapters/openai-responses/image-gen.ts +406 -0
- package/src/adapters/openai-responses/internal.ts +3 -0
- package/src/adapters/openai-responses/passthrough.ts +611 -0
- package/src/adapters/openai-responses/prompt-cache.ts +83 -0
- package/src/adapters/openai-responses/reasoning.ts +220 -0
- package/src/adapters/openai-responses/request-strips.ts +185 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
- package/src/adapters/openai-responses/tool-schema.ts +293 -0
- package/src/adapters/openai-responses/web-search.ts +156 -0
- package/src/adapters/openai-responses.ts +4 -2625
- package/src/bridge/errors.ts +34 -0
- package/src/bridge/internal.ts +174 -0
- package/src/bridge/response-json.ts +624 -0
- package/src/bridge/sse.ts +1444 -0
- package/src/bridge.ts +5 -2204
- package/src/chat/inbound.ts +12 -1
- package/src/codex/account-lifecycle.ts +3 -0
- package/src/codex/account-store.ts +71 -9
- package/src/codex/auth-api/account-list.ts +507 -0
- package/src/codex/auth-api/http.ts +32 -0
- package/src/codex/auth-api/login-flow.ts +554 -0
- package/src/codex/auth-api/login-state.ts +64 -0
- package/src/codex/auth-api/main-account-probe.ts +331 -0
- package/src/codex/auth-api/pool-mode-gate.ts +274 -0
- package/src/codex/auth-api/pool-quota-probe.ts +512 -0
- package/src/codex/auth-api/reset-credit-service.ts +422 -0
- package/src/codex/auth-api/routes.ts +425 -0
- package/src/codex/auth-api/runtime-config.ts +48 -0
- package/src/codex/auth-api.ts +27 -3118
- package/src/codex/auth-context.ts +95 -28
- package/src/codex/catalog/auto-review.ts +507 -0
- package/src/codex/catalog/build-entries.ts +981 -0
- package/src/codex/catalog/combo-member.ts +375 -0
- package/src/codex/catalog/derive-entry.ts +229 -0
- package/src/codex/catalog/effort.ts +0 -1
- package/src/codex/catalog/gated-native-warn.ts +63 -0
- package/src/codex/catalog/gather-capture.ts +533 -0
- package/src/codex/catalog/model-hints.ts +691 -0
- package/src/codex/catalog/model-visibility.ts +304 -0
- package/src/codex/catalog/provider-fetch.ts +52 -2942
- package/src/codex/catalog/provider-models.ts +685 -0
- package/src/codex/catalog/restore.ts +132 -0
- package/src/codex/catalog/retained-sync.ts +706 -0
- package/src/codex/catalog/routed-gather.ts +858 -0
- package/src/codex/catalog/subagent-roster.ts +176 -0
- package/src/codex/catalog/sync.ts +52 -2698
- package/src/codex/inject/config-toml.ts +563 -0
- package/src/codex/inject/remove.ts +192 -0
- package/src/codex/inject/restore.ts +540 -0
- package/src/codex/inject/routing-classify.ts +109 -0
- package/src/codex/inject/routing-target.ts +125 -0
- package/src/codex/inject.ts +81 -1436
- package/src/codex/lineage.ts +458 -0
- package/src/codex/pool-refresh-backoff.ts +152 -0
- package/src/codex/routing/active-account.ts +194 -0
- package/src/codex/routing/cooldown-math.ts +275 -0
- package/src/codex/routing/health-store.ts +402 -0
- package/src/codex/routing/probe-lease.ts +358 -0
- package/src/codex/routing/selection.ts +703 -0
- package/src/codex/routing/thread-affinity.ts +538 -0
- package/src/codex/routing.ts +353 -2234
- package/src/codex/shim-fingerprint.ts +223 -0
- package/src/codex/shim-inspect.ts +175 -0
- package/src/codex/shim-probe.ts +367 -0
- package/src/codex/shim-restore-lock.ts +169 -0
- package/src/codex/shim-state-file.ts +151 -0
- package/src/codex/shim-templates.ts +265 -0
- package/src/codex/shim.ts +48 -1268
- package/src/config/diagnostics.ts +705 -0
- package/src/config/feature-flags.ts +55 -0
- package/src/config/live-reconcile.ts +403 -0
- package/src/config/load-degrade.ts +880 -0
- package/src/config/mutation-lock.ts +244 -0
- package/src/config/openai-tier-backup.ts +268 -0
- package/src/config/persist-unlocked.ts +92 -0
- package/src/config/proxy-env.ts +188 -0
- package/src/config/salvage.ts +244 -0
- package/src/config/schema/config-schema.ts +640 -0
- package/src/config/schema/leaf-validators.ts +855 -0
- package/src/config/warn-memo.ts +28 -0
- package/src/config.ts +234 -4481
- package/src/generated/compatibility-version.json +539 -39
- package/src/lib/request-execution-budget.ts +69 -20
- package/src/lib/spend-reservation-ledger.ts +940 -0
- package/src/lib/upstream-retry.ts +55 -11
- package/src/lib/workflow-budget.ts +553 -30
- package/src/providers/quota/account-cache.ts +441 -0
- package/src/providers/quota/antigravity.ts +295 -0
- package/src/providers/quota/report-cache.ts +320 -0
- package/src/providers/quota/vendor-probes-key.ts +1243 -0
- package/src/providers/quota/vendor-probes-oauth.ts +590 -0
- package/src/providers/quota.ts +324 -3079
- package/src/providers/registry/entries-core.ts +1221 -0
- package/src/providers/registry/entries-extended.ts +1204 -0
- package/src/providers/registry/model-seeds.ts +908 -0
- package/src/providers/registry/types.ts +352 -0
- package/src/providers/registry.ts +24 -3536
- package/src/responses/continuation-ownership.ts +29 -0
- package/src/responses/state/replay-fingerprint.ts +80 -0
- package/src/responses/state/snapshot-codec.ts +104 -0
- package/src/responses/state/spill-failure.ts +118 -0
- package/src/responses/state/spill-queue.ts +665 -0
- package/src/responses/state/temp-recovery.ts +257 -0
- package/src/responses/state.ts +82 -1143
- package/src/routing/identity-domains.ts +449 -0
- package/src/routing/probe-lease.ts +511 -0
- package/src/server/index/bounded-request.ts +88 -0
- package/src/server/index/live-sideband.ts +565 -0
- package/src/server/index/serve-options.ts +1766 -0
- package/src/server/index/startup-warnings.ts +213 -0
- package/src/server/index/websocket-handler.ts +335 -0
- package/src/server/index.ts +40 -2547
- package/src/server/management/route-registry.ts +26 -23
- package/src/server/management/shared.ts +8 -5
- package/src/server/management/workflow-budget-routes.ts +133 -0
- package/src/server/management-api.ts +12 -0
- package/src/server/request-log-conversation.ts +9 -7
- package/src/server/request-log.ts +245 -1
- package/src/server/responses/account-change-state.ts +233 -0
- package/src/server/responses/adapter-continuation.ts +514 -0
- package/src/server/responses/adapter-delivery.ts +214 -0
- package/src/server/responses/adapter-dispatch.ts +971 -0
- package/src/server/responses/compact.ts +59 -4
- package/src/server/responses/completion-policy.ts +33 -0
- package/src/server/responses/core-auth.ts +527 -0
- package/src/server/responses/core-codex-account.ts +859 -0
- package/src/server/responses/core-combo-failure.ts +210 -0
- package/src/server/responses/core-combo.ts +707 -0
- package/src/server/responses/core-errors.ts +152 -0
- package/src/server/responses/core-lifetime.ts +95 -0
- package/src/server/responses/core-normalize.ts +350 -0
- package/src/server/responses/core-opaque-recovery.ts +380 -0
- package/src/server/responses/core-options.ts +159 -0
- package/src/server/responses/core-replay.ts +225 -0
- package/src/server/responses/core.ts +192 -8893
- package/src/server/responses/passthrough-delivery.ts +856 -0
- package/src/server/responses/passthrough-dispatch.ts +1476 -0
- package/src/server/responses/passthrough-execution.ts +54 -0
- package/src/server/responses/request-prepare.ts +970 -0
- package/src/server/responses/request-send-budget.ts +164 -0
- package/src/server/responses/request-sidecar-auth.ts +149 -0
- package/src/server/responses/request-transport.ts +744 -0
- package/src/server/responses/response-effects.ts +157 -0
- package/src/server/responses/run-turn-execution.ts +448 -0
- package/src/server/responses/sidecar-execution.ts +469 -0
- package/src/server/responses-image-gen-repair.ts +1 -1
- package/src/server/workflow-refusal.ts +84 -0
- package/src/types/config.ts +30 -0
- package/src/usage/log.ts +146 -0
- package/src/usage/summary.ts +171 -21
package/src/responses/state.ts
CHANGED
|
@@ -1,30 +1,60 @@
|
|
|
1
|
-
import { chmodSync, existsSync,
|
|
2
|
-
import { uptime } from "node:os";
|
|
1
|
+
import { chmodSync, existsSync, mkdirSync, readFileSync, rmSync, statSync, unlinkSync } from "node:fs";
|
|
3
2
|
import { dirname, join } from "node:path";
|
|
4
3
|
import { atomicWriteFileAsync, getConfigDir, resolveWriteTarget } from "../config";
|
|
5
4
|
import { enforceAppOwnedMemoryBudget, type RetainedStoreSnapshot } from "../lib/app-owned-memory";
|
|
6
5
|
import { windowsSecretAclApplies } from "../lib/windows-secret-acl";
|
|
7
6
|
import type { OcxProviderContinuationState } from "../types";
|
|
8
7
|
import {
|
|
9
|
-
cleanupSupersededResponseSpillPublication,
|
|
10
|
-
createResponseSpillPublicationControl,
|
|
11
8
|
deleteResponseSpill,
|
|
12
|
-
MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
|
|
13
9
|
noteStubSwapForTest,
|
|
14
10
|
readResponseSpill,
|
|
15
11
|
recoverOrphanedResponseSpills,
|
|
16
12
|
responseSpillDirectory,
|
|
17
13
|
responseSpillPayloadCap,
|
|
18
|
-
markResponseSpillPublicationSuperseded,
|
|
19
|
-
prospectiveResponseSpillBytes,
|
|
20
|
-
type ResponseSpillPublicationControl,
|
|
21
14
|
type ResponseSpillRef,
|
|
22
15
|
writeResponseSpillDurably,
|
|
23
|
-
writeResponseSpillDurablyAsync,
|
|
24
16
|
} from "./spill-store";
|
|
17
|
+
import { clientCarriedPrefixLength, providerIssuedIdentity } from "./state/replay-fingerprint";
|
|
18
|
+
export type { ResponseStateTempRecoveryResult, ResponseStateTempRecoveryOptions } from "./state/temp-recovery";
|
|
19
|
+
export { recoverStaleResponseStateTemps, reclaimAbandonedResponseStateTemps, inspectAbandonedResponseStateTemps, sweepAbandonedResponseStateTemps } from "./state/temp-recovery";
|
|
20
|
+
import { recoverStaleResponseStateTemps } from "./state/temp-recovery";
|
|
21
|
+
export type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseSpillWriteFailureOrigin } from "./state/spill-failure";
|
|
22
|
+
import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseSpillWriteFailureOrigin } from "./state/spill-failure";
|
|
23
|
+
export { responseAdmissionCountersForTests } from "./state/spill-failure";
|
|
24
|
+
import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
|
|
25
|
+
import { loadSnapshotEntry } from "./state/snapshot-codec";
|
|
26
|
+
export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
|
|
27
|
+
import {
|
|
28
|
+
bindSpillQueueStore,
|
|
29
|
+
cancelPendingResponseSpill,
|
|
30
|
+
drainResponseSpillPublications,
|
|
31
|
+
queuePendingResponseSpill,
|
|
32
|
+
replaceWithPendingResponseSpill,
|
|
33
|
+
resetSpillQueueForTests,
|
|
34
|
+
spillQueueAccounting,
|
|
35
|
+
spillQueueHoldsResidentCandidate,
|
|
36
|
+
spillQueuePendingBytes,
|
|
37
|
+
spillQueueResidentCandidates,
|
|
38
|
+
spillQueueSupersededSpillFor,
|
|
39
|
+
} from "./state/spill-queue";
|
|
25
40
|
|
|
26
41
|
const MAX_STORED_RESPONSES = 1_000;
|
|
27
|
-
|
|
42
|
+
/**
|
|
43
|
+
* Retention for locally replayed continuation state.
|
|
44
|
+
*
|
|
45
|
+
* A Codex client chained by `previous_response_id` sends ONLY the new turn and expects this
|
|
46
|
+
* process to hold everything before it, so this constant is the practical memory span of every
|
|
47
|
+
* conversation that does not go to the canonical ChatGPT backend. At the original one hour, a
|
|
48
|
+
* session resumed after lunch expanded to nothing and the delta — one user line — was all the
|
|
49
|
+
* provider ever saw, which reads to the operator as the model losing the conversation.
|
|
50
|
+
*
|
|
51
|
+
* A day is safe to hold because retention is no longer what bounds this store: the resident cap
|
|
52
|
+
* (MAX_STORED_RESPONSE_BYTES), the spill ceiling (MAX_SPILLED_RESPONSE_BYTES) and the entry count
|
|
53
|
+
* all evict oldest-first, and every turn re-stores the whole chain under a fresh id, so the live
|
|
54
|
+
* conversation is the last thing any of those three caps would drop. Raising the TTL therefore
|
|
55
|
+
* moves eviction from the clock to those budgets rather than growing the ceiling.
|
|
56
|
+
*/
|
|
57
|
+
export const RESPONSE_TTL_MS = 24 * 60 * 60 * 1_000;
|
|
28
58
|
const SNAPSHOT_DEBOUNCE_MS = 2_000;
|
|
29
59
|
/** Snapshot size below which the debounce stays at its base value. */
|
|
30
60
|
const SNAPSHOT_DEBOUNCE_SCALE_FROM_BYTES = 1 * 1024 * 1024;
|
|
@@ -67,28 +97,10 @@ const SNAPSHOT_TOTAL_MAX_BYTES = 24 * 1024 * 1024;
|
|
|
67
97
|
* bound, so anything we wrote ourselves always loads; guards against externally
|
|
68
98
|
* planted or pre-cap unbounded files being parsed whole). */
|
|
69
99
|
const SNAPSHOT_FILE_MAX_BYTES = 32 * 1024 * 1024;
|
|
70
|
-
const STALE_TEMP_GRACE_MS = 15 * 60 * 1_000;
|
|
71
|
-
const STALE_TEMP_MAX_ENTRIES = 4_096;
|
|
72
|
-
const STALE_TEMP_MAX_CLEANUPS = 512;
|
|
73
|
-
/** Absorbs `os.uptime()` granularity only. It is deliberately NOT the safety margin:
|
|
74
|
-
* the unconditional 15-minute grace above is (see the boot floor in the scan loop). */
|
|
75
|
-
const BOOT_FLOOR_SKEW_MS = 60 * 1_000;
|
|
76
|
-
/** Per-tick budget for the periodic reclaim. Smaller than the startup budget because the
|
|
77
|
-
* periodic pass runs synchronously on the serving process's event loop every 60 s. */
|
|
78
|
-
const PERIODIC_TEMP_MAX_ENTRIES = 512;
|
|
79
|
-
const PERIODIC_TEMP_MAX_CLEANUPS = 64;
|
|
80
|
-
/** Wall-clock ceiling for one periodic scan. An entry cap bounds syscalls, not time: on a
|
|
81
|
-
* network-mounted config dir each `lstat` can cost 10-20 ms, which would stall in-flight
|
|
82
|
-
* streams. Reclaim is idempotent, so a truncated tick simply resumes on the next one. */
|
|
83
|
-
const PERIODIC_TEMP_SCAN_DEADLINE_MS = 25;
|
|
84
|
-
const RESPONSE_STATE_TEMP_NAME = /^responses-state\.json\.ocx\.(\d+)\.(\d+)\.tmp$/;
|
|
85
100
|
const MAX_SNAPSHOT_REWRITE_ATTEMPTS = 4;
|
|
86
|
-
const RESPONSE_SPILL_SHUTDOWN_BUDGET_MS = 5_000;
|
|
87
|
-
const RESPONSE_SPILL_SHUTDOWN_FALLBACK_RESERVE_MS = 4_000;
|
|
88
|
-
const RESPONSE_SPILL_ASYNC_ACL_ATTEMPT_BUDGET_MS = 30_000;
|
|
89
101
|
const RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES = MAX_STORED_RESPONSES + 1;
|
|
90
102
|
|
|
91
|
-
interface ResidentResponseState {
|
|
103
|
+
export interface ResidentResponseState {
|
|
92
104
|
kind: "resident";
|
|
93
105
|
createdAt: number;
|
|
94
106
|
clientThreadId?: string;
|
|
@@ -99,7 +111,7 @@ interface ResidentResponseState {
|
|
|
99
111
|
sizeBytes: number;
|
|
100
112
|
}
|
|
101
113
|
|
|
102
|
-
interface SpilledResponseState {
|
|
114
|
+
export interface SpilledResponseState {
|
|
103
115
|
kind: "spill";
|
|
104
116
|
createdAt: number;
|
|
105
117
|
clientThreadId?: string;
|
|
@@ -110,14 +122,14 @@ interface SpilledResponseState {
|
|
|
110
122
|
sizeBytes: number;
|
|
111
123
|
}
|
|
112
124
|
|
|
113
|
-
interface SpillFailedResponseState {
|
|
125
|
+
export interface SpillFailedResponseState {
|
|
114
126
|
kind: "spill-failed";
|
|
115
127
|
createdAt: number;
|
|
116
128
|
sizeBytes: number;
|
|
117
129
|
}
|
|
118
130
|
|
|
119
|
-
type StoredResponseState = ResidentResponseState | SpilledResponseState | SpillFailedResponseState;
|
|
120
|
-
type ResidentInput = Omit<ResidentResponseState, "kind" | "sizeBytes">;
|
|
131
|
+
export type StoredResponseState = ResidentResponseState | SpilledResponseState | SpillFailedResponseState;
|
|
132
|
+
export type ResidentInput = Omit<ResidentResponseState, "kind" | "sizeBytes">;
|
|
121
133
|
|
|
122
134
|
export type PreviousResponseReplayFailure = {
|
|
123
135
|
code: "previous_response_not_found";
|
|
@@ -169,124 +181,8 @@ async function snapshotOnDiskMatches(path: string, payload: string, payloadBytes
|
|
|
169
181
|
return false;
|
|
170
182
|
}
|
|
171
183
|
}
|
|
172
|
-
const spillCounters = {
|
|
173
|
-
writes: 0, writeFailures: 0, readFailures: 0,
|
|
174
|
-
aclRetryReturnedTimeouts: 0, aclTimeoutMemoRefusals: 0,
|
|
175
|
-
};
|
|
176
|
-
|
|
177
|
-
export type ResponseSpillWriteFailureCode =
|
|
178
|
-
| "EACLRETRYEXHAUSTED"
|
|
179
|
-
| "ETIMEDOUT"
|
|
180
|
-
| "EACCES"
|
|
181
|
-
| "ENOSPC"
|
|
182
|
-
| "EFBIG"
|
|
183
|
-
| "EIO"
|
|
184
|
-
| "ECAPACITY"
|
|
185
|
-
| "ELOOP"
|
|
186
|
-
| "EUNKNOWN";
|
|
187
|
-
|
|
188
|
-
export type ResponseSpillWriteStatus = "initial" | "healthy" | "degraded";
|
|
189
|
-
|
|
190
|
-
export type ResponseSpillWriteFailureOrigin =
|
|
191
|
-
| "retry_returned_timeout"
|
|
192
|
-
| "timeout_memo_refusal";
|
|
193
|
-
|
|
194
|
-
interface ResponseSpillWriteHealth {
|
|
195
|
-
consecutiveFailures: number;
|
|
196
|
-
lastFailureCode: ResponseSpillWriteFailureCode | null;
|
|
197
|
-
lastFailureOrigin: ResponseSpillWriteFailureOrigin | null;
|
|
198
|
-
lastFailureAt: number | null;
|
|
199
|
-
lastSuccessAt: number | null;
|
|
200
|
-
}
|
|
201
|
-
|
|
202
|
-
const spillWriteHealth: ResponseSpillWriteHealth = {
|
|
203
|
-
consecutiveFailures: 0,
|
|
204
|
-
lastFailureCode: null,
|
|
205
|
-
lastFailureOrigin: null,
|
|
206
|
-
lastFailureAt: null,
|
|
207
|
-
lastSuccessAt: null,
|
|
208
|
-
};
|
|
209
|
-
|
|
210
|
-
/**
|
|
211
|
-
* Collapse filesystem/runtime errors into a fixed privacy-safe diagnostic union.
|
|
212
|
-
* Messages and paths are deliberately ignored: this projection is returned by the
|
|
213
|
-
* authenticated memory endpoint, and a nested `cause` can contain a username or
|
|
214
|
-
* workspace path even when the public wrapper does not.
|
|
215
|
-
*/
|
|
216
|
-
function classifySpillWriteFailure(error: unknown): ResponseSpillWriteFailureCode {
|
|
217
|
-
let cursor = error;
|
|
218
|
-
for (let depth = 0; depth < 4 && cursor && typeof cursor === "object"; depth += 1) {
|
|
219
|
-
const record = cursor as { code?: unknown; cause?: unknown };
|
|
220
|
-
const code = typeof record.code === "string" ? record.code.toUpperCase() : "";
|
|
221
|
-
switch (code) {
|
|
222
|
-
case "EACLRETRYEXHAUSTED": return "EACLRETRYEXHAUSTED";
|
|
223
|
-
case "ETIMEDOUT": return "ETIMEDOUT";
|
|
224
|
-
case "EACCES":
|
|
225
|
-
case "EPERM": return "EACCES";
|
|
226
|
-
case "ENOSPC":
|
|
227
|
-
case "EDQUOT": return "ENOSPC";
|
|
228
|
-
case "EFBIG": return "EFBIG";
|
|
229
|
-
case "EIO": return "EIO";
|
|
230
|
-
case "ECAPACITY": return "ECAPACITY";
|
|
231
|
-
case "ELOOP": return "ELOOP";
|
|
232
|
-
}
|
|
233
|
-
cursor = record.cause;
|
|
234
|
-
}
|
|
235
|
-
return "EUNKNOWN";
|
|
236
|
-
}
|
|
237
|
-
|
|
238
|
-
/** The spill writer preserves ACL errors in cause; only a fixed memo marker is diagnostic. */
|
|
239
|
-
function spillAclMemoRefusalOrigin(error: unknown): "timeout_memo_refusal" | null {
|
|
240
|
-
let cursor = error;
|
|
241
|
-
for (let depth = 0; depth < 4 && cursor && typeof cursor === "object"; depth += 1) {
|
|
242
|
-
const record = cursor as { code?: unknown; aclFailureOrigin?: unknown; cause?: unknown };
|
|
243
|
-
if ((record.code === "ETIMEDOUT" || record.code === "EACLRETRYEXHAUSTED")
|
|
244
|
-
&& record.aclFailureOrigin === "timeout_memo_refusal") {
|
|
245
|
-
return "timeout_memo_refusal";
|
|
246
|
-
}
|
|
247
|
-
cursor = record.cause;
|
|
248
|
-
}
|
|
249
|
-
return null;
|
|
250
|
-
}
|
|
251
|
-
|
|
252
|
-
function noteSpillWriteSuccess(): void {
|
|
253
|
-
spillCounters.writes += 1;
|
|
254
|
-
spillWriteHealth.consecutiveFailures = 0;
|
|
255
|
-
spillWriteHealth.lastSuccessAt = now();
|
|
256
|
-
}
|
|
257
|
-
|
|
258
|
-
function noteSpillWriteFailure(
|
|
259
|
-
error: unknown,
|
|
260
|
-
override?: ResponseSpillWriteFailureCode,
|
|
261
|
-
retryOrigin: ResponseSpillWriteFailureOrigin | null = null,
|
|
262
|
-
): void {
|
|
263
|
-
const code = override ?? classifySpillWriteFailure(error);
|
|
264
|
-
const origin = code === "ETIMEDOUT" || code === "EACLRETRYEXHAUSTED"
|
|
265
|
-
? spillAclMemoRefusalOrigin(error) ?? retryOrigin
|
|
266
|
-
: null;
|
|
267
|
-
spillCounters.writeFailures += 1;
|
|
268
|
-
spillWriteHealth.consecutiveFailures += 1;
|
|
269
|
-
spillWriteHealth.lastFailureCode = code;
|
|
270
|
-
spillWriteHealth.lastFailureOrigin = origin;
|
|
271
|
-
spillWriteHealth.lastFailureAt = now();
|
|
272
|
-
// Count terminal publications, not ACL calls or a transient first attempt.
|
|
273
|
-
if (origin === "retry_returned_timeout") spillCounters.aclRetryReturnedTimeouts += 1;
|
|
274
|
-
else if (origin === "timeout_memo_refusal") spillCounters.aclTimeoutMemoRefusals += 1;
|
|
275
|
-
}
|
|
276
|
-
/**
|
|
277
|
-
* Admission-boundary observability (test-visible). directSpills: oversized
|
|
278
|
-
* candidates routed straight to durable spill without a resident stay or
|
|
279
|
-
* unrelated demotion. oversizedDrops: candidates above the single-spill
|
|
280
|
-
* payload ceiling, tombstoned instead of retained. snapshotOversizedRefusals:
|
|
281
|
-
* snapshot files refused before parse.
|
|
282
|
-
*/
|
|
283
|
-
const admissionCounters = { directSpills: 0, oversizedDrops: 0, snapshotOversizedRefusals: 0 };
|
|
284
184
|
let replayScopeMismatchDrops = 0;
|
|
285
185
|
|
|
286
|
-
/** Test-only: admission-boundary counters (proves the new paths fire). */
|
|
287
|
-
export function responseAdmissionCountersForTests(): Readonly<typeof admissionCounters> {
|
|
288
|
-
return admissionCounters;
|
|
289
|
-
}
|
|
290
186
|
// Superseded spill generations awaiting a durable snapshot before unlink
|
|
291
187
|
// (review C1-1: unlinking at swap time races a crash against the debounced
|
|
292
188
|
// snapshot — the reloaded OLD stub would point at a deleted file).
|
|
@@ -299,99 +195,6 @@ const pendingSpillUnlinks: ResponseSpillRef[] = [];
|
|
|
299
195
|
// structured 400 — bounded-loss, never silent corruption or unbounded disk.
|
|
300
196
|
const PENDING_SPILL_UNLINKS_MAX = 128;
|
|
301
197
|
|
|
302
|
-
/**
|
|
303
|
-
* Windows keeps the candidate replayable while required ACL hardening runs off the event loop.
|
|
304
|
-
* Pending bytes are pinned, not evictable; cap them below the process-owned 512 MiB ceiling so an
|
|
305
|
-
* icacls outage cannot turn the serialized queue into an unbounded resident backlog.
|
|
306
|
-
*/
|
|
307
|
-
const MAX_PENDING_RESPONSE_SPILL_BYTES = MAX_RESPONSE_SPILL_PAYLOAD_BYTES;
|
|
308
|
-
|
|
309
|
-
interface PendingResponseSpill {
|
|
310
|
-
id: string;
|
|
311
|
-
candidate: ResidentResponseState | null;
|
|
312
|
-
supersededSpill?: ResponseSpillRef;
|
|
313
|
-
directAdmission: boolean;
|
|
314
|
-
running: boolean;
|
|
315
|
-
cancelled: boolean;
|
|
316
|
-
released: boolean;
|
|
317
|
-
sizeBytes: number;
|
|
318
|
-
/** Peak on-disk bytes reserved for this publication; released exactly once on settle. */
|
|
319
|
-
reservedBytes: number;
|
|
320
|
-
publicationControl: ResponseSpillPublicationControl;
|
|
321
|
-
}
|
|
322
|
-
|
|
323
|
-
const pendingResponseSpills = new Set<PendingResponseSpill>();
|
|
324
|
-
const pendingResponseSpillById = new Map<string, PendingResponseSpill>();
|
|
325
|
-
let pendingResponseSpillBytes = 0;
|
|
326
|
-
/**
|
|
327
|
-
* On-disk bytes a queued publication is about to occupy but has not yet installed into
|
|
328
|
-
* `states`.
|
|
329
|
-
*
|
|
330
|
-
* `spilledResponseBytes()` walks installed spills and deferred unlinks — files that
|
|
331
|
-
* already exist. It cannot see one that `writeResponseSpillDurablyAsync` is in the
|
|
332
|
-
* middle of creating, and on Windows that middle can last as long as `icacls` takes.
|
|
333
|
-
* Without a reservation the cap holds only when writes are fast, which is not a cap.
|
|
334
|
-
*
|
|
335
|
-
* The reserved figure is the PEAK footprint, not the payload: publication can fall back
|
|
336
|
-
* from hard-linking to an exclusive copy, and during that fallback the destination copy
|
|
337
|
-
* and the temp file exist simultaneously. Reserving one envelope would leave the overshoot
|
|
338
|
-
* intact at half its magnitude.
|
|
339
|
-
*
|
|
340
|
-
* Ownership is single: a job holds its reservation from queue until
|
|
341
|
-
* `releasePendingResponseSpill`, which every exit from the publication path reaches
|
|
342
|
-
* through the `finally` in `runPendingResponseSpill` and through cancellation of a
|
|
343
|
-
* not-yet-running job. A leaked reservation is monotonic — it would ratchet the usable
|
|
344
|
-
* cap toward zero — so the release must stay on the settlement path rather than in a
|
|
345
|
-
* parallel bookkeeping pass.
|
|
346
|
-
*/
|
|
347
|
-
let reservedResponseSpillBytes = 0;
|
|
348
|
-
/**
|
|
349
|
-
* Paths a failed cleanup left on the volume, with the bytes each one occupies.
|
|
350
|
-
*
|
|
351
|
-
* A failed unlink leaves a real file behind, so the cap has to keep seeing it. But a
|
|
352
|
-
* never-decremented total would be phantom debt: a Windows lock that clears a moment
|
|
353
|
-
* later, or the async writer's own retry, can remove the file while the charge stays
|
|
354
|
-
* forever — and with 256 MiB payloads two conservative charges consume the whole default
|
|
355
|
-
* cap, after which nothing can spill for the life of the process.
|
|
356
|
-
*
|
|
357
|
-
* So the debt is per PATH, priced at what that path actually holds, and settled the
|
|
358
|
-
* moment the path is gone. `reconcileUnreclaimableSpillPaths` re-checks on every read of
|
|
359
|
-
* the accounted total, which is the same tick that would otherwise refuse an admission.
|
|
360
|
-
*/
|
|
361
|
-
const unreclaimableSpillPaths = new Map<string, number>();
|
|
362
|
-
|
|
363
|
-
function chargeUnreclaimableSpillPath(path: string | null | undefined, bytes: number): void {
|
|
364
|
-
if (!path || bytes <= 0) return;
|
|
365
|
-
unreclaimableSpillPaths.set(path, bytes);
|
|
366
|
-
}
|
|
367
|
-
|
|
368
|
-
/** Drop charges for paths that have since disappeared; returns the surviving total. */
|
|
369
|
-
function reconcileUnreclaimableSpillPaths(): number {
|
|
370
|
-
let total = 0;
|
|
371
|
-
for (const [path, bytes] of [...unreclaimableSpillPaths]) {
|
|
372
|
-
if (existsSync(path)) total += bytes;
|
|
373
|
-
else unreclaimableSpillPaths.delete(path);
|
|
374
|
-
}
|
|
375
|
-
return total;
|
|
376
|
-
}
|
|
377
|
-
|
|
378
|
-
/**
|
|
379
|
-
* Peak on-disk footprint of publishing this candidate: temp plus destination copy.
|
|
380
|
-
*
|
|
381
|
-
* Measured from the production serializer rather than from `candidate.sizeBytes`. The
|
|
382
|
-
* resident measurement omits the `version` field the published envelope carries, so
|
|
383
|
-
* pricing an admission by it undercounts and lets a request sitting exactly at the cap
|
|
384
|
-
* still exceed it. Falls back to the resident figure only when serialization fails, which
|
|
385
|
-
* is the same condition that will fail the publication itself.
|
|
386
|
-
*/
|
|
387
|
-
function publicationFootprintBytes(id: string, candidate: ResidentResponseState): number {
|
|
388
|
-
const exact = prospectiveResponseSpillBytes(id, spillPayloadForResident(candidate));
|
|
389
|
-
return (exact ?? candidate.sizeBytes) * 2;
|
|
390
|
-
}
|
|
391
|
-
let responseSpillPublicationTail: Promise<void> = Promise.resolve();
|
|
392
|
-
let responseSpillShutdownBudgetOverride: { totalMs: number; fallbackReserveMs: number } | null = null;
|
|
393
|
-
let responseSpillShutdownTerminalizationPassLimitOverride: number | null = null;
|
|
394
|
-
let responseSpillAsyncAclAttemptBudgetOverride: number | null = null;
|
|
395
198
|
|
|
396
199
|
function deferSupersededSpill(ref: ResponseSpillRef | undefined): void {
|
|
397
200
|
if (!ref) return;
|
|
@@ -401,470 +204,6 @@ function deferSupersededSpill(ref: ResponseSpillRef | undefined): void {
|
|
|
401
204
|
}
|
|
402
205
|
}
|
|
403
206
|
|
|
404
|
-
function releasePendingResponseSpill(job: PendingResponseSpill): void {
|
|
405
|
-
if (job.released) return;
|
|
406
|
-
job.released = true;
|
|
407
|
-
pendingResponseSpillBytes = Math.max(0, pendingResponseSpillBytes - job.sizeBytes);
|
|
408
|
-
reservedResponseSpillBytes = Math.max(0, reservedResponseSpillBytes - job.reservedBytes);
|
|
409
|
-
pendingResponseSpills.delete(job);
|
|
410
|
-
if (pendingResponseSpillById.get(job.id) === job) pendingResponseSpillById.delete(job.id);
|
|
411
|
-
job.candidate = null;
|
|
412
|
-
}
|
|
413
|
-
|
|
414
|
-
function cancelPendingResponseSpill(id: string): ResponseSpillRef | undefined {
|
|
415
|
-
const job = pendingResponseSpillById.get(id);
|
|
416
|
-
if (!job) return undefined;
|
|
417
|
-
pendingResponseSpillById.delete(id);
|
|
418
|
-
job.cancelled = true;
|
|
419
|
-
markResponseSpillPublicationSuperseded(job.publicationControl);
|
|
420
|
-
const superseded = job.supersededSpill;
|
|
421
|
-
// Ownership TRANSFERS to the caller. Leaving the ref on the cancelled job would let the
|
|
422
|
-
// accounting walk count the same physical file twice — once here and once on the
|
|
423
|
-
// replacement — and an overcount evicts live continuations to make room for bytes that
|
|
424
|
-
// are not there.
|
|
425
|
-
delete job.supersededSpill;
|
|
426
|
-
// A queued job has not captured the candidate in an async frame yet, so release it now.
|
|
427
|
-
// A running job retains its accounting until settlement and will discard its stale file.
|
|
428
|
-
if (!job.running) releasePendingResponseSpill(job);
|
|
429
|
-
return superseded;
|
|
430
|
-
}
|
|
431
|
-
|
|
432
|
-
function isAclTimeout(error: unknown): boolean {
|
|
433
|
-
return !!error && typeof error === "object" && "code" in error
|
|
434
|
-
&& String((error as { code?: unknown }).code) === "ETIMEDOUT";
|
|
435
|
-
}
|
|
436
|
-
|
|
437
|
-
function spillPayloadForResident(candidate: ResidentResponseState): Parameters<typeof writeResponseSpillDurably>[1] {
|
|
438
|
-
return {
|
|
439
|
-
createdAt: candidate.createdAt,
|
|
440
|
-
...(candidate.clientThreadId ? { clientThreadId: candidate.clientThreadId } : {}),
|
|
441
|
-
items: candidate.items,
|
|
442
|
-
...(candidate.providerOutputStart !== undefined ? { providerOutputStart: candidate.providerOutputStart } : {}),
|
|
443
|
-
...(candidate.providers ? { providers: candidate.providers } : {}),
|
|
444
|
-
};
|
|
445
|
-
}
|
|
446
|
-
|
|
447
|
-
async function runPendingResponseSpill(job: PendingResponseSpill): Promise<void> {
|
|
448
|
-
if (job.cancelled || !job.candidate) return;
|
|
449
|
-
job.running = true;
|
|
450
|
-
const candidate = job.candidate;
|
|
451
|
-
let ref: ResponseSpillRef | null = null;
|
|
452
|
-
let exhaustedAclRetry = false;
|
|
453
|
-
let aclRetryFailureOrigin: ResponseSpillWriteFailureOrigin | null = null;
|
|
454
|
-
try {
|
|
455
|
-
const state = spillPayloadForResident(candidate);
|
|
456
|
-
try {
|
|
457
|
-
ref = await writeResponseSpillDurablyAsync(job.id, state, {
|
|
458
|
-
aclBudgetMs: responseSpillAsyncAclAttemptBudgetMs(),
|
|
459
|
-
publicationControl: job.publicationControl,
|
|
460
|
-
});
|
|
461
|
-
} catch (error) {
|
|
462
|
-
if (!isAclTimeout(error)) throw error;
|
|
463
|
-
// The ACL helper permits exactly one caller-owned recovery budget. The resident generation
|
|
464
|
-
// remains replayable during both attempts, so a transient timeout never becomes a tombstone.
|
|
465
|
-
try {
|
|
466
|
-
ref = await writeResponseSpillDurablyAsync(job.id, state, {
|
|
467
|
-
aclBudgetMs: responseSpillAsyncAclAttemptBudgetMs(),
|
|
468
|
-
retryTimedOutOnce: true,
|
|
469
|
-
publicationControl: job.publicationControl,
|
|
470
|
-
});
|
|
471
|
-
} catch (retryError) {
|
|
472
|
-
exhaustedAclRetry = isAclTimeout(retryError);
|
|
473
|
-
// A returned timeout can also mean an exhausted budget before the next OS command.
|
|
474
|
-
aclRetryFailureOrigin = spillAclMemoRefusalOrigin(retryError)
|
|
475
|
-
?? (exhaustedAclRetry ? "retry_returned_timeout" : null);
|
|
476
|
-
throw retryError;
|
|
477
|
-
}
|
|
478
|
-
}
|
|
479
|
-
if (ref.payloadBytes > responseSpillPayloadCap()) {
|
|
480
|
-
deleteResponseSpill(ref);
|
|
481
|
-
ref = null;
|
|
482
|
-
if (job.directAdmission) admissionCounters.oversizedDrops += 1;
|
|
483
|
-
throw Object.assign(new Error("Response spill payload exceeds replay ceiling"), { code: "EFBIG" });
|
|
484
|
-
}
|
|
485
|
-
if (states.get(job.id) !== candidate || job.cancelled) {
|
|
486
|
-
deleteResponseSpill(ref);
|
|
487
|
-
ref = null;
|
|
488
|
-
return;
|
|
489
|
-
}
|
|
490
|
-
if (swapResidentForSpill(job.id, candidate, ref)) {
|
|
491
|
-
ref = null;
|
|
492
|
-
noteSpillWriteSuccess();
|
|
493
|
-
if (job.directAdmission) admissionCounters.directSpills += 1;
|
|
494
|
-
deferSupersededSpill(job.supersededSpill);
|
|
495
|
-
}
|
|
496
|
-
} catch (error) {
|
|
497
|
-
if (ref) deleteResponseSpill(ref);
|
|
498
|
-
if (states.get(job.id) === candidate && !job.cancelled) {
|
|
499
|
-
noteSpillWriteFailure(error, exhaustedAclRetry ? "EACLRETRYEXHAUSTED" : undefined, aclRetryFailureOrigin);
|
|
500
|
-
replaceWithSpillFailure(job.id, candidate);
|
|
501
|
-
deferSupersededSpill(job.supersededSpill);
|
|
502
|
-
}
|
|
503
|
-
} finally {
|
|
504
|
-
const cancelled = job.cancelled;
|
|
505
|
-
releasePendingResponseSpill(job);
|
|
506
|
-
recomputeOldestResident();
|
|
507
|
-
if (!cancelled) {
|
|
508
|
-
schedulePersist();
|
|
509
|
-
pruneResponses();
|
|
510
|
-
enforceAppOwnedMemoryBudget();
|
|
511
|
-
}
|
|
512
|
-
}
|
|
513
|
-
}
|
|
514
|
-
|
|
515
|
-
function queuePendingResponseSpill(
|
|
516
|
-
id: string,
|
|
517
|
-
candidate: ResidentResponseState,
|
|
518
|
-
options: { supersededSpill?: ResponseSpillRef; directAdmission?: boolean } = {},
|
|
519
|
-
): void {
|
|
520
|
-
const inheritedSpill = cancelPendingResponseSpill(id) ?? options.supersededSpill;
|
|
521
|
-
if (pendingResponseSpillBytes + candidate.sizeBytes > MAX_PENDING_RESPONSE_SPILL_BYTES) {
|
|
522
|
-
noteSpillWriteFailure(null, "ECAPACITY");
|
|
523
|
-
replaceWithSpillFailure(id, candidate);
|
|
524
|
-
deferSupersededSpill(inheritedSpill);
|
|
525
|
-
return;
|
|
526
|
-
}
|
|
527
|
-
// Enforce the disk cap BEFORE the temp or destination file is created. Deleting the
|
|
528
|
-
// overflow afterwards is not equivalent: on Windows the file can outlive the decision
|
|
529
|
-
// by as long as ACL hardening takes, which is the window the measured 6.8 GiB
|
|
530
|
-
// accumulated in. Reclaim first, and only refuse if the peak footprint still does not
|
|
531
|
-
// fit — an eviction pass can free a live continuation's worth of room.
|
|
532
|
-
const footprint = publicationFootprintBytes(id, candidate);
|
|
533
|
-
// The superseded generation this job is about to own is already off `states` and not
|
|
534
|
-
// yet on the job, so it is invisible to the walk. Price it here or admission decides
|
|
535
|
-
// against a total that is short by a whole envelope.
|
|
536
|
-
const inheritedBytes = inheritedSpill?.payloadBytes ?? 0;
|
|
537
|
-
if (accountedResponseSpillBytes() + footprint + inheritedBytes > spillByteCap()) {
|
|
538
|
-
enforceSpilledResponseBudget();
|
|
539
|
-
if (accountedResponseSpillBytes() + footprint + inheritedBytes > spillByteCap()) {
|
|
540
|
-
noteSpillWriteFailure(null, "ECAPACITY");
|
|
541
|
-
replaceWithSpillFailure(id, candidate);
|
|
542
|
-
deferSupersededSpill(inheritedSpill);
|
|
543
|
-
return;
|
|
544
|
-
}
|
|
545
|
-
}
|
|
546
|
-
const job: PendingResponseSpill = {
|
|
547
|
-
id,
|
|
548
|
-
candidate,
|
|
549
|
-
...(inheritedSpill ? { supersededSpill: inheritedSpill } : {}),
|
|
550
|
-
directAdmission: options.directAdmission === true,
|
|
551
|
-
running: false,
|
|
552
|
-
cancelled: false,
|
|
553
|
-
released: false,
|
|
554
|
-
sizeBytes: candidate.sizeBytes,
|
|
555
|
-
reservedBytes: footprint,
|
|
556
|
-
publicationControl: createResponseSpillPublicationControl(),
|
|
557
|
-
};
|
|
558
|
-
pendingResponseSpills.add(job);
|
|
559
|
-
pendingResponseSpillById.set(id, job);
|
|
560
|
-
pendingResponseSpillBytes += job.sizeBytes;
|
|
561
|
-
reservedResponseSpillBytes += job.reservedBytes;
|
|
562
|
-
recomputeOldestResident();
|
|
563
|
-
responseSpillPublicationTail = responseSpillPublicationTail
|
|
564
|
-
.then(() => runPendingResponseSpill(job), () => runPendingResponseSpill(job));
|
|
565
|
-
}
|
|
566
|
-
|
|
567
|
-
function replaceWithPendingResponseSpill(
|
|
568
|
-
id: string,
|
|
569
|
-
candidate: ResidentResponseState,
|
|
570
|
-
expected: StoredResponseState | undefined,
|
|
571
|
-
options: { directAdmission?: boolean } = {},
|
|
572
|
-
): boolean {
|
|
573
|
-
const inheritedSpill = pendingResponseSpillById.get(id)?.supersededSpill
|
|
574
|
-
?? (expected?.kind === "spill" ? expected.spill : undefined);
|
|
575
|
-
if (!replaceMapEntry(id, candidate, expected)) return false;
|
|
576
|
-
queuePendingResponseSpill(id, candidate, {
|
|
577
|
-
...(inheritedSpill ? { supersededSpill: inheritedSpill } : {}),
|
|
578
|
-
directAdmission: options.directAdmission === true,
|
|
579
|
-
});
|
|
580
|
-
return true;
|
|
581
|
-
}
|
|
582
|
-
|
|
583
|
-
/** Test-only: settle every serialized Windows spill publication. */
|
|
584
|
-
export async function flushPendingResponseSpillsForTests(): Promise<void> {
|
|
585
|
-
await drainResponseSpillPublications();
|
|
586
|
-
}
|
|
587
|
-
|
|
588
|
-
/** Test-only: observe ordinary queue settlement without invoking shutdown fallback. */
|
|
589
|
-
export async function awaitResponseSpillPublicationTailForTests(): Promise<void> {
|
|
590
|
-
await responseSpillPublicationTail;
|
|
591
|
-
}
|
|
592
|
-
|
|
593
|
-
/** Test-only: observe the bounded queue without exposing payloads. */
|
|
594
|
-
export function pendingResponseSpillMetricsForTests(): { count: number; bytes: number } {
|
|
595
|
-
return { count: pendingResponseSpills.size, bytes: pendingResponseSpillBytes };
|
|
596
|
-
}
|
|
597
|
-
|
|
598
|
-
/** Test-only: shorten the shutdown drain/fallback budget (null restores production values). */
|
|
599
|
-
export function setResponseSpillShutdownBudgetForTests(
|
|
600
|
-
budget: { totalMs: number; fallbackReserveMs: number } | null,
|
|
601
|
-
): void {
|
|
602
|
-
responseSpillShutdownBudgetOverride = budget;
|
|
603
|
-
}
|
|
604
|
-
|
|
605
|
-
/** Test-only: shorten the ordinary async whole-attempt ACL budget. */
|
|
606
|
-
export function setResponseSpillAsyncAclAttemptBudgetForTests(budgetMs: number | null): void {
|
|
607
|
-
responseSpillAsyncAclAttemptBudgetOverride = budgetMs;
|
|
608
|
-
}
|
|
609
|
-
|
|
610
|
-
function responseSpillAsyncAclAttemptBudgetMs(): number {
|
|
611
|
-
return responseSpillAsyncAclAttemptBudgetOverride ?? RESPONSE_SPILL_ASYNC_ACL_ATTEMPT_BUDGET_MS;
|
|
612
|
-
}
|
|
613
|
-
|
|
614
|
-
/** Test-only: lower the hard terminalization pass guard (null restores production). */
|
|
615
|
-
export function setResponseSpillShutdownTerminalizationPassLimitForTests(limit: number | null): void {
|
|
616
|
-
responseSpillShutdownTerminalizationPassLimitOverride = limit;
|
|
617
|
-
}
|
|
618
|
-
|
|
619
|
-
function responseSpillShutdownTerminalizationPassLimit(): number {
|
|
620
|
-
return responseSpillShutdownTerminalizationPassLimitOverride
|
|
621
|
-
?? RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES;
|
|
622
|
-
}
|
|
623
|
-
|
|
624
|
-
function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: number } {
|
|
625
|
-
return responseSpillShutdownBudgetOverride ?? {
|
|
626
|
-
totalMs: RESPONSE_SPILL_SHUTDOWN_BUDGET_MS,
|
|
627
|
-
fallbackReserveMs: RESPONSE_SPILL_SHUTDOWN_FALLBACK_RESERVE_MS,
|
|
628
|
-
};
|
|
629
|
-
}
|
|
630
|
-
|
|
631
|
-
function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
|
|
632
|
-
const remaining = deadline - Date.now();
|
|
633
|
-
if (remaining <= 0) return Promise.resolve(false);
|
|
634
|
-
return new Promise(resolve => {
|
|
635
|
-
let finished = false;
|
|
636
|
-
const finish = (settled: boolean): void => {
|
|
637
|
-
if (finished) return;
|
|
638
|
-
finished = true;
|
|
639
|
-
clearTimeout(timer);
|
|
640
|
-
resolve(settled);
|
|
641
|
-
};
|
|
642
|
-
const timer = setTimeout(() => finish(false), remaining);
|
|
643
|
-
observed.then(() => finish(true), () => finish(true));
|
|
644
|
-
});
|
|
645
|
-
}
|
|
646
|
-
|
|
647
|
-
function installShutdownFallbackSpill(
|
|
648
|
-
job: PendingResponseSpill,
|
|
649
|
-
candidate: ResidentResponseState,
|
|
650
|
-
aclBudgetMs: number,
|
|
651
|
-
): void {
|
|
652
|
-
let ref: ResponseSpillRef | null = null;
|
|
653
|
-
// Supersession released this job's reservation, but the synchronous write below is the
|
|
654
|
-
// largest publication of the shutdown path and has its own link-then-copy fallback
|
|
655
|
-
// holding a temp and a destination at once. Re-reserve for its duration so the cap is
|
|
656
|
-
// not blind exactly where the drain does its heaviest work, and settle in `finally` so
|
|
657
|
-
// every return, throw and mismatch releases it.
|
|
658
|
-
const footprint = publicationFootprintBytes(job.id, candidate);
|
|
659
|
-
reservedResponseSpillBytes += footprint;
|
|
660
|
-
try {
|
|
661
|
-
// Supersession released this job, so its superseded generation is no longer visible
|
|
662
|
-
// to the accounting walk — but the file is still on the volume until
|
|
663
|
-
// `deferSupersededSpill` or a delete takes it. Price it here or the fallback decides
|
|
664
|
-
// against a total short by that whole envelope, which is exactly the gap that lets
|
|
665
|
-
// `debt + footprint <= cap < old + debt + footprint` publish over budget.
|
|
666
|
-
const supersededBytes = job.supersededSpill?.payloadBytes ?? 0;
|
|
667
|
-
// The drain must not publish over the cap either. Reclaim first; if the footprint
|
|
668
|
-
// still does not fit — which is what unreclaimable cleanup debt looks like — the
|
|
669
|
-
// honest close-out is a tombstone, not another file on a volume that is already
|
|
670
|
-
// over budget. `replaceWithSpillFailure` is the same fail-closed ending the budget
|
|
671
|
-
// exhaustion path uses, so replay reports `spill_failed` and the client resends.
|
|
672
|
-
if (accountedResponseSpillBytes() + supersededBytes > spillByteCap()) {
|
|
673
|
-
enforceSpilledResponseBudget();
|
|
674
|
-
if (accountedResponseSpillBytes() + supersededBytes > spillByteCap()) {
|
|
675
|
-
if (states.get(job.id) === candidate) {
|
|
676
|
-
noteSpillWriteFailure(null, "ECAPACITY");
|
|
677
|
-
replaceWithSpillFailure(job.id, candidate);
|
|
678
|
-
deferSupersededSpill(job.supersededSpill);
|
|
679
|
-
}
|
|
680
|
-
throw Object.assign(new Error("Response spill shutdown fallback exceeds the durable disk cap"), { code: "ENOSPC" });
|
|
681
|
-
}
|
|
682
|
-
}
|
|
683
|
-
ref = writeResponseSpillDurably(job.id, spillPayloadForResident(candidate), { aclBudgetMs });
|
|
684
|
-
if (ref.payloadBytes > responseSpillPayloadCap()) {
|
|
685
|
-
deleteResponseSpill(ref);
|
|
686
|
-
ref = null;
|
|
687
|
-
if (job.directAdmission) admissionCounters.oversizedDrops += 1;
|
|
688
|
-
throw Object.assign(new Error("Response spill payload exceeds replay ceiling"), { code: "EFBIG" });
|
|
689
|
-
}
|
|
690
|
-
if (states.get(job.id) !== candidate) {
|
|
691
|
-
deleteResponseSpill(ref);
|
|
692
|
-
ref = null;
|
|
693
|
-
return;
|
|
694
|
-
}
|
|
695
|
-
if (swapResidentForSpill(job.id, candidate, ref)) {
|
|
696
|
-
ref = null;
|
|
697
|
-
noteSpillWriteSuccess();
|
|
698
|
-
if (job.directAdmission) admissionCounters.directSpills += 1;
|
|
699
|
-
deferSupersededSpill(job.supersededSpill);
|
|
700
|
-
}
|
|
701
|
-
} catch (error) {
|
|
702
|
-
if (ref) deleteResponseSpill(ref);
|
|
703
|
-
if (states.get(job.id) === candidate) {
|
|
704
|
-
noteSpillWriteFailure(error);
|
|
705
|
-
replaceWithSpillFailure(job.id, candidate);
|
|
706
|
-
deferSupersededSpill(job.supersededSpill);
|
|
707
|
-
}
|
|
708
|
-
throw error;
|
|
709
|
-
} finally {
|
|
710
|
-
reservedResponseSpillBytes = Math.max(0, reservedResponseSpillBytes - footprint);
|
|
711
|
-
}
|
|
712
|
-
}
|
|
713
|
-
|
|
714
|
-
function terminalizeShutdownFallbackCandidate(
|
|
715
|
-
job: PendingResponseSpill,
|
|
716
|
-
candidate: ResidentResponseState,
|
|
717
|
-
failureCode: ResponseSpillWriteFailureCode = "ETIMEDOUT",
|
|
718
|
-
): void {
|
|
719
|
-
if (states.get(job.id) !== candidate) return;
|
|
720
|
-
noteSpillWriteFailure(null, failureCode);
|
|
721
|
-
replaceWithSpillFailure(job.id, candidate);
|
|
722
|
-
deferSupersededSpill(job.supersededSpill);
|
|
723
|
-
}
|
|
724
|
-
|
|
725
|
-
function pendingShutdownFallbackCandidates(): Array<{
|
|
726
|
-
job: PendingResponseSpill;
|
|
727
|
-
candidate: ResidentResponseState;
|
|
728
|
-
}> {
|
|
729
|
-
return [...pendingResponseSpills]
|
|
730
|
-
.map(job => ({ job, candidate: job.candidate }))
|
|
731
|
-
.filter((entry): entry is { job: PendingResponseSpill; candidate: ResidentResponseState } => !!entry.candidate);
|
|
732
|
-
}
|
|
733
|
-
|
|
734
|
-
function supersedeShutdownFallbackBatch(
|
|
735
|
-
pending: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
|
|
736
|
-
failures: Error[],
|
|
737
|
-
): void {
|
|
738
|
-
for (const { job } of pending) {
|
|
739
|
-
job.cancelled = true;
|
|
740
|
-
markResponseSpillPublicationSuperseded(job.publicationControl);
|
|
741
|
-
}
|
|
742
|
-
for (const { job } of pending) {
|
|
743
|
-
const cleanupFailure = cleanupSupersededResponseSpillPublication(job.publicationControl);
|
|
744
|
-
if (cleanupFailure) {
|
|
745
|
-
failures.push(cleanupFailure);
|
|
746
|
-
// Cleanup failed, so an async temp or destination is STILL on the volume. Releasing
|
|
747
|
-
// the reservation would un-account a file that exists, and the fallback write that
|
|
748
|
-
// follows reserves only its own footprint — three envelopes on disk priced as two.
|
|
749
|
-
//
|
|
750
|
-
// Charge the surviving PATHS rather than a flat two envelopes: `clearOwnedPath`
|
|
751
|
-
// nulls whichever it managed to remove, so one failure is one file, not two. The
|
|
752
|
-
// charge is settled automatically once the path disappears, which a retried unlink
|
|
753
|
-
// or a released Windows lock can still do.
|
|
754
|
-
const perPath = Math.max(1, Math.floor(job.reservedBytes / 2));
|
|
755
|
-
chargeUnreclaimableSpillPath(job.publicationControl.tempPath, perPath);
|
|
756
|
-
chargeUnreclaimableSpillPath(job.publicationControl.destinationPath, perPath);
|
|
757
|
-
}
|
|
758
|
-
releasePendingResponseSpill(job);
|
|
759
|
-
}
|
|
760
|
-
}
|
|
761
|
-
|
|
762
|
-
function stopAtShutdownTerminalizationPassLimit(
|
|
763
|
-
pending: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
|
|
764
|
-
failures: Error[],
|
|
765
|
-
): void {
|
|
766
|
-
failures.push(Object.assign(new Error("Response spill shutdown terminalization pass limit exceeded"), { code: "ELOOP" }));
|
|
767
|
-
supersedeShutdownFallbackBatch(pending, failures);
|
|
768
|
-
for (const { job, candidate } of pending) {
|
|
769
|
-
terminalizeShutdownFallbackCandidate(job, candidate, "ELOOP");
|
|
770
|
-
}
|
|
771
|
-
for (const [id, state] of [...states]) {
|
|
772
|
-
if (state.kind !== "resident") continue;
|
|
773
|
-
noteSpillWriteFailure(null, "ELOOP");
|
|
774
|
-
replaceWithSpillFailure(id, state);
|
|
775
|
-
}
|
|
776
|
-
recomputeOldestResident();
|
|
777
|
-
pruneResponses();
|
|
778
|
-
enforceAppOwnedMemoryBudget();
|
|
779
|
-
}
|
|
780
|
-
|
|
781
|
-
function terminalizeExhaustedShutdownFallback(
|
|
782
|
-
initial: Array<{ job: PendingResponseSpill; candidate: ResidentResponseState }>,
|
|
783
|
-
failures: Error[],
|
|
784
|
-
): void {
|
|
785
|
-
let pending = initial;
|
|
786
|
-
let passes = 0;
|
|
787
|
-
const passLimit = responseSpillShutdownTerminalizationPassLimit();
|
|
788
|
-
// Every pass replaces each captured resident with a tombstone. Pruning may expose
|
|
789
|
-
// another finite batch, but resident count strictly decreases until none can requeue.
|
|
790
|
-
while (pending.length > 0) {
|
|
791
|
-
if (passes >= passLimit) {
|
|
792
|
-
stopAtShutdownTerminalizationPassLimit(pending, failures);
|
|
793
|
-
return;
|
|
794
|
-
}
|
|
795
|
-
passes += 1;
|
|
796
|
-
supersedeShutdownFallbackBatch(pending, failures);
|
|
797
|
-
for (const { job, candidate } of pending) {
|
|
798
|
-
failures.push(Object.assign(new Error("Response spill shutdown fallback budget exhausted"), { code: "ETIMEDOUT" }));
|
|
799
|
-
terminalizeShutdownFallbackCandidate(job, candidate);
|
|
800
|
-
}
|
|
801
|
-
recomputeOldestResident();
|
|
802
|
-
pruneResponses();
|
|
803
|
-
enforceAppOwnedMemoryBudget();
|
|
804
|
-
pending = pendingShutdownFallbackCandidates();
|
|
805
|
-
}
|
|
806
|
-
}
|
|
807
|
-
|
|
808
|
-
function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
809
|
-
const deadline = Date.now() + reserveMs;
|
|
810
|
-
const failures: Error[] = [];
|
|
811
|
-
for (;;) {
|
|
812
|
-
const pending = pendingShutdownFallbackCandidates();
|
|
813
|
-
if (pending.length === 0) return failures;
|
|
814
|
-
if (Date.now() >= deadline) {
|
|
815
|
-
terminalizeExhaustedShutdownFallback(pending, failures);
|
|
816
|
-
return failures;
|
|
817
|
-
}
|
|
818
|
-
|
|
819
|
-
supersedeShutdownFallbackBatch(pending, failures);
|
|
820
|
-
let reserveExhausted = false;
|
|
821
|
-
for (let index = 0; index < pending.length; index += 1) {
|
|
822
|
-
const { job, candidate } = pending[index]!;
|
|
823
|
-
if (states.get(job.id) !== candidate) continue;
|
|
824
|
-
const remaining = deadline - Date.now();
|
|
825
|
-
if (remaining <= 0) {
|
|
826
|
-
reserveExhausted = true;
|
|
827
|
-
for (const exhausted of pending.slice(index)) {
|
|
828
|
-
failures.push(Object.assign(new Error("Response spill shutdown fallback budget exhausted"), { code: "ETIMEDOUT" }));
|
|
829
|
-
terminalizeShutdownFallbackCandidate(exhausted.job, exhausted.candidate);
|
|
830
|
-
}
|
|
831
|
-
break;
|
|
832
|
-
}
|
|
833
|
-
try {
|
|
834
|
-
installShutdownFallbackSpill(job, candidate, remaining);
|
|
835
|
-
} catch (error) {
|
|
836
|
-
failures.push(error instanceof Error ? error : new Error("Response spill shutdown fallback failed"));
|
|
837
|
-
}
|
|
838
|
-
}
|
|
839
|
-
recomputeOldestResident();
|
|
840
|
-
pruneResponses();
|
|
841
|
-
enforceAppOwnedMemoryBudget();
|
|
842
|
-
if (reserveExhausted || Date.now() >= deadline) {
|
|
843
|
-
terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
|
|
844
|
-
return failures;
|
|
845
|
-
}
|
|
846
|
-
}
|
|
847
|
-
}
|
|
848
|
-
|
|
849
|
-
async function drainResponseSpillPublications(): Promise<void> {
|
|
850
|
-
const budget = responseSpillShutdownBudget();
|
|
851
|
-
const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
|
|
852
|
-
const drainDeadline = Date.now() + Math.max(0, budget.totalMs - fallbackReserveMs);
|
|
853
|
-
|
|
854
|
-
for (;;) {
|
|
855
|
-
if (pendingResponseSpills.size === 0) return;
|
|
856
|
-
const observed = responseSpillPublicationTail;
|
|
857
|
-
const settled = await awaitResponseSpillTailUntil(observed, drainDeadline);
|
|
858
|
-
if (!settled) {
|
|
859
|
-
const failures = fallbackPendingResponseSpills(fallbackReserveMs);
|
|
860
|
-
if (failures.length > 0) {
|
|
861
|
-
throw new AggregateError(failures, "Response spill shutdown fallback incomplete");
|
|
862
|
-
}
|
|
863
|
-
return;
|
|
864
|
-
}
|
|
865
|
-
if (observed === responseSpillPublicationTail) return;
|
|
866
|
-
}
|
|
867
|
-
}
|
|
868
207
|
|
|
869
208
|
function byteCap(): number {
|
|
870
209
|
return byteCapOverride ?? MAX_STORED_RESPONSE_BYTES;
|
|
@@ -920,12 +259,9 @@ function accountedResponseSpillBytes(): number {
|
|
|
920
259
|
// counting only `states` plus `pendingSpillUnlinks` loses it for the whole publication
|
|
921
260
|
// — during a copy fallback that is old generation + new temp + new destination, three
|
|
922
261
|
// envelopes priced as two.
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
}
|
|
927
|
-
return spilledResponseBytes() + reservedResponseSpillBytes + ownedBySpillJobs
|
|
928
|
-
+ reconcileUnreclaimableSpillPaths();
|
|
262
|
+
const accounting = spillQueueAccounting();
|
|
263
|
+
return spilledResponseBytes() + accounting.reservedBytes + accounting.jobOwnedBytes
|
|
264
|
+
+ accounting.unreclaimableBytes;
|
|
929
265
|
}
|
|
930
266
|
|
|
931
267
|
/** Test-only: lower/restore the durable spill cap (null restores the default). */
|
|
@@ -969,7 +305,7 @@ function recomputeOldestResident(): void {
|
|
|
969
305
|
oldestResidentAt = null;
|
|
970
306
|
for (const [id, state] of states) {
|
|
971
307
|
if (state.kind !== "resident") continue;
|
|
972
|
-
if (
|
|
308
|
+
if (spillQueueHoldsResidentCandidate(id, state)) continue;
|
|
973
309
|
if (oldestResidentAt !== null && state.createdAt >= oldestResidentAt) continue;
|
|
974
310
|
oldestResidentId = id;
|
|
975
311
|
oldestResidentAt = state.createdAt;
|
|
@@ -1134,8 +470,7 @@ function setResidentEntry(id: string, entry: ResidentInput): void {
|
|
|
1134
470
|
pruneResponses();
|
|
1135
471
|
return;
|
|
1136
472
|
}
|
|
1137
|
-
|
|
1138
|
-
if (windowsSecretAclApplies() && (expected?.kind === "spill" || pending?.supersededSpill)) {
|
|
473
|
+
if (windowsSecretAclApplies() && (expected?.kind === "spill" || spillQueueSupersededSpillFor(id))) {
|
|
1139
474
|
replaceWithPendingResponseSpill(id, candidate, expected);
|
|
1140
475
|
pruneResponses();
|
|
1141
476
|
return;
|
|
@@ -1219,6 +554,23 @@ function admitOversizedCandidate(
|
|
|
1219
554
|
}
|
|
1220
555
|
}
|
|
1221
556
|
|
|
557
|
+
bindSpillQueueStore({
|
|
558
|
+
swapResidentForSpill,
|
|
559
|
+
replaceWithSpillFailure,
|
|
560
|
+
deleteEntry,
|
|
561
|
+
deferSupersededSpill,
|
|
562
|
+
replaceMapEntry,
|
|
563
|
+
currentEntry: (id: string) => states.get(id),
|
|
564
|
+
residentEntries: () => [...states],
|
|
565
|
+
recomputeOldestResident,
|
|
566
|
+
schedulePersist,
|
|
567
|
+
pruneResponses,
|
|
568
|
+
accountedResponseSpillBytes,
|
|
569
|
+
spillByteCap,
|
|
570
|
+
enforceSpilledResponseBudget,
|
|
571
|
+
terminalizationMaxPasses: () => RESPONSE_SPILL_SHUTDOWN_TERMINALIZATION_MAX_PASSES,
|
|
572
|
+
});
|
|
573
|
+
|
|
1222
574
|
// Replay provenance must stay proxy-private: a WeakMap distinguishes replayed history from the
|
|
1223
575
|
// newly appended input suffix without adding an unknown field that native passthrough could send
|
|
1224
576
|
// upstream. The parser uses this boundary to acknowledge historical compaction markers exactly
|
|
@@ -1241,284 +593,6 @@ function snapshotPath(): string {
|
|
|
1241
593
|
return join(getConfigDir(), "responses-state.json");
|
|
1242
594
|
}
|
|
1243
595
|
|
|
1244
|
-
interface LegacySnapshotState {
|
|
1245
|
-
createdAt?: unknown;
|
|
1246
|
-
clientThreadId?: unknown;
|
|
1247
|
-
items?: unknown;
|
|
1248
|
-
providers?: OcxProviderContinuationState;
|
|
1249
|
-
conversationId?: unknown;
|
|
1250
|
-
cursorCheckpointUsable?: unknown;
|
|
1251
|
-
}
|
|
1252
|
-
|
|
1253
|
-
function isSpillRef(value: unknown): value is ResponseSpillRef {
|
|
1254
|
-
if (!value || typeof value !== "object" || Array.isArray(value)) return false;
|
|
1255
|
-
const ref = value as ResponseSpillRef;
|
|
1256
|
-
return ref.version === 1
|
|
1257
|
-
&& typeof ref.fileName === "string"
|
|
1258
|
-
&& /^[0-9a-f]{64}$/.test(ref.digest)
|
|
1259
|
-
&& Number.isSafeInteger(ref.payloadBytes)
|
|
1260
|
-
&& ref.payloadBytes >= 0;
|
|
1261
|
-
}
|
|
1262
|
-
|
|
1263
|
-
function loadSnapshotEntry(id: string, value: unknown): void {
|
|
1264
|
-
if (!value || typeof value !== "object" || Array.isArray(value)) return;
|
|
1265
|
-
const rec = value as LegacySnapshotState & { kind?: unknown; spill?: unknown };
|
|
1266
|
-
if (typeof rec.createdAt !== "number" || !Number.isFinite(rec.createdAt)) return;
|
|
1267
|
-
const clientThreadId = typeof rec.clientThreadId === "string" && rec.clientThreadId.trim().length > 0
|
|
1268
|
-
? rec.clientThreadId.trim()
|
|
1269
|
-
: undefined;
|
|
1270
|
-
// A malformed boundary degrades to "never skip" rather than to a bad index: an untrusted
|
|
1271
|
-
// snapshot must not be able to authorize dropping conversation history.
|
|
1272
|
-
const anchorFor = (itemCount: number): number | undefined => {
|
|
1273
|
-
const raw = (rec as { providerOutputStart?: unknown }).providerOutputStart;
|
|
1274
|
-
return Number.isSafeInteger(raw) && (raw as number) >= 0 && (raw as number) <= itemCount
|
|
1275
|
-
? raw as number
|
|
1276
|
-
: undefined;
|
|
1277
|
-
};
|
|
1278
|
-
if (rec.kind === "spill") {
|
|
1279
|
-
if (!isSpillRef(rec.spill)) return;
|
|
1280
|
-
const base: Omit<SpilledResponseState, "sizeBytes"> = {
|
|
1281
|
-
kind: "spill",
|
|
1282
|
-
createdAt: rec.createdAt,
|
|
1283
|
-
...(clientThreadId ? { clientThreadId } : {}),
|
|
1284
|
-
// Item count is unknown until materialization, so accept any non-negative integer
|
|
1285
|
-
// here; the spill payload validator re-checks it against the real array.
|
|
1286
|
-
...(anchorFor(Number.MAX_SAFE_INTEGER) !== undefined ? { providerOutputStart: anchorFor(Number.MAX_SAFE_INTEGER) } : {}),
|
|
1287
|
-
...(rec.providers ? { providers: rec.providers } : {}),
|
|
1288
|
-
spill: rec.spill,
|
|
1289
|
-
};
|
|
1290
|
-
replaceMapEntry(id, { ...base, sizeBytes: stubSize(id, base) });
|
|
1291
|
-
return;
|
|
1292
|
-
}
|
|
1293
|
-
if (rec.kind === "spill-failed") {
|
|
1294
|
-
replaceMapEntry(id, tombstone(id, rec.createdAt));
|
|
1295
|
-
return;
|
|
1296
|
-
}
|
|
1297
|
-
if (rec.kind !== undefined && rec.kind !== "resident") return;
|
|
1298
|
-
if (!Array.isArray(rec.items)) return;
|
|
1299
|
-
const providers = rec.providers ?? (typeof rec.conversationId === "string"
|
|
1300
|
-
? {
|
|
1301
|
-
cursor: {
|
|
1302
|
-
conversationId: rec.conversationId,
|
|
1303
|
-
...(typeof rec.cursorCheckpointUsable === "boolean"
|
|
1304
|
-
? { checkpointUsable: rec.cursorCheckpointUsable }
|
|
1305
|
-
: {}),
|
|
1306
|
-
},
|
|
1307
|
-
}
|
|
1308
|
-
: undefined);
|
|
1309
|
-
const resident = measureResidentEntry(id, {
|
|
1310
|
-
createdAt: rec.createdAt,
|
|
1311
|
-
...(clientThreadId ? { clientThreadId } : {}),
|
|
1312
|
-
items: rec.items,
|
|
1313
|
-
...(anchorFor(rec.items.length) !== undefined ? { providerOutputStart: anchorFor(rec.items.length) } : {}),
|
|
1314
|
-
...(providers ? { providers } : {}),
|
|
1315
|
-
});
|
|
1316
|
-
if (!resident) {
|
|
1317
|
-
replaceMapEntry(id, tombstone(id, rec.createdAt));
|
|
1318
|
-
return;
|
|
1319
|
-
}
|
|
1320
|
-
// Same admission boundary as live writes: an oversized snapshot row goes
|
|
1321
|
-
// straight to spill (or tombstone above the payload ceiling) instead of
|
|
1322
|
-
// entering the resident map and demoting unrelated rows on the first prune.
|
|
1323
|
-
if (resident.sizeBytes > byteCap()) {
|
|
1324
|
-
admitOversizedCandidate(id, resident, undefined);
|
|
1325
|
-
return;
|
|
1326
|
-
}
|
|
1327
|
-
replaceMapEntry(id, resident);
|
|
1328
|
-
}
|
|
1329
|
-
|
|
1330
|
-
export interface ResponseStateTempRecoveryResult {
|
|
1331
|
-
matched: number;
|
|
1332
|
-
removed: number;
|
|
1333
|
-
failed: number;
|
|
1334
|
-
bytesRemoved: number;
|
|
1335
|
-
/** Entries that passed EVERY gate and would be reclaimed. In a dry run nothing is
|
|
1336
|
-
* unlinked, so this is the only honest count to show an operator: `matched` is
|
|
1337
|
-
* incremented before the file-type, age, boot-floor, and liveness gates. */
|
|
1338
|
-
eligible: number;
|
|
1339
|
-
/** Total size of the `eligible` entries. */
|
|
1340
|
-
eligibleBytes: number;
|
|
1341
|
-
/** The scan stopped on a budget (entry cap, cleanup cap, or deadline) rather than reaching
|
|
1342
|
-
* the end of the directory, so the counts below describe a prefix of the backlog and not
|
|
1343
|
-
* the backlog. `eligible > removed + failed` cannot express this: outside a dry run every
|
|
1344
|
-
* eligible entry is unlinked or failed on the same iteration, so the two are always equal
|
|
1345
|
-
* and a comparison between them is dead code. */
|
|
1346
|
-
truncated: boolean;
|
|
1347
|
-
}
|
|
1348
|
-
|
|
1349
|
-
interface ResponseStateTempRecoveryIO {
|
|
1350
|
-
now: () => number;
|
|
1351
|
-
/** Approximate epoch ms of the current boot; see the boot floor in the scan loop. */
|
|
1352
|
-
bootTime: () => number;
|
|
1353
|
-
list: (dir: string) => Iterable<string>;
|
|
1354
|
-
inspect: (path: string) => { isFile: boolean; mtimeMs: number; size: number };
|
|
1355
|
-
isProcessAlive: (pid: number) => boolean;
|
|
1356
|
-
unlink: (path: string) => void;
|
|
1357
|
-
}
|
|
1358
|
-
|
|
1359
|
-
export type ResponseStateTempRecoveryOptions = Partial<ResponseStateTempRecoveryIO> & {
|
|
1360
|
-
maxEntries?: number;
|
|
1361
|
-
maxCleanups?: number;
|
|
1362
|
-
/** Wall-clock ceiling for the scan, or null/undefined for no deadline (startup path). */
|
|
1363
|
-
deadlineMs?: number | null;
|
|
1364
|
-
/** Report only: apply every gate, count what would be reclaimed, unlink nothing. */
|
|
1365
|
-
dryRun?: boolean;
|
|
1366
|
-
};
|
|
1367
|
-
|
|
1368
|
-
function processIsAlive(pid: number): boolean {
|
|
1369
|
-
if (pid === process.pid) return true;
|
|
1370
|
-
try {
|
|
1371
|
-
process.kill(pid, 0);
|
|
1372
|
-
return true;
|
|
1373
|
-
} catch (error) {
|
|
1374
|
-
// EPERM means the process exists but cannot be signalled. Unknown platform errors
|
|
1375
|
-
// are also protected; cleanup should prefer a false negative over touching a live writer.
|
|
1376
|
-
return (error as NodeJS.ErrnoException).code !== "ESRCH";
|
|
1377
|
-
}
|
|
1378
|
-
}
|
|
1379
|
-
|
|
1380
|
-
const responseStateTempRecoveryIO: ResponseStateTempRecoveryIO = {
|
|
1381
|
-
now: Date.now,
|
|
1382
|
-
bootTime: () => Date.now() - uptime() * 1_000,
|
|
1383
|
-
list: function* list(dir) {
|
|
1384
|
-
const handle = opendirSync(dir);
|
|
1385
|
-
try {
|
|
1386
|
-
for (let entry = handle.readSync(); entry; entry = handle.readSync()) yield entry.name;
|
|
1387
|
-
} finally {
|
|
1388
|
-
handle.closeSync();
|
|
1389
|
-
}
|
|
1390
|
-
},
|
|
1391
|
-
inspect: path => {
|
|
1392
|
-
const stat = lstatSync(path);
|
|
1393
|
-
return { isFile: stat.isFile() && !stat.isSymbolicLink(), mtimeMs: stat.mtimeMs, size: stat.size };
|
|
1394
|
-
},
|
|
1395
|
-
isProcessAlive: processIsAlive,
|
|
1396
|
-
unlink: unlinkSync,
|
|
1397
|
-
};
|
|
1398
|
-
|
|
1399
|
-
/**
|
|
1400
|
-
* Recover only abandoned response-state atomic-write files. The exact basename,
|
|
1401
|
-
* regular-file check, age gate, and PID liveness check protect unrelated/active files.
|
|
1402
|
-
* Cleanup is capped and best-effort because continuation state is only a cache. Removal
|
|
1403
|
-
* deliberately uses unlink only: path-based truncation could follow a replacement symlink.
|
|
1404
|
-
*/
|
|
1405
|
-
export function recoverStaleResponseStateTemps(
|
|
1406
|
-
dir = getConfigDir(),
|
|
1407
|
-
options: ResponseStateTempRecoveryOptions = {},
|
|
1408
|
-
): ResponseStateTempRecoveryResult {
|
|
1409
|
-
const {
|
|
1410
|
-
maxEntries = STALE_TEMP_MAX_ENTRIES,
|
|
1411
|
-
maxCleanups = STALE_TEMP_MAX_CLEANUPS,
|
|
1412
|
-
deadlineMs = null,
|
|
1413
|
-
dryRun = false,
|
|
1414
|
-
...overrides
|
|
1415
|
-
} = options;
|
|
1416
|
-
const io = { ...responseStateTempRecoveryIO, ...overrides };
|
|
1417
|
-
const result: ResponseStateTempRecoveryResult = {
|
|
1418
|
-
matched: 0,
|
|
1419
|
-
removed: 0,
|
|
1420
|
-
failed: 0,
|
|
1421
|
-
bytesRemoved: 0,
|
|
1422
|
-
eligible: 0,
|
|
1423
|
-
eligibleBytes: 0,
|
|
1424
|
-
truncated: false,
|
|
1425
|
-
};
|
|
1426
|
-
const startedAt = io.now();
|
|
1427
|
-
// One probe per scan, not one per entry. A non-finite or future-dated boot is anomalous, and
|
|
1428
|
-
// clamping it to "now" would be the WORST response: the floor would then retire the liveness
|
|
1429
|
-
// probe for every file older than the skew, which is every file past the grace. Disable it
|
|
1430
|
-
// instead -- an absent floor only costs a missed reclaim, never a wrong one.
|
|
1431
|
-
const rawBoot = io.bootTime();
|
|
1432
|
-
const bootMs = Number.isFinite(rawBoot) && rawBoot <= startedAt ? rawBoot : Number.NEGATIVE_INFINITY;
|
|
1433
|
-
let names: Iterable<string>;
|
|
1434
|
-
try { names = io.list(dir); } catch { return result; }
|
|
1435
|
-
let iterator: Iterator<string>;
|
|
1436
|
-
try { iterator = names[Symbol.iterator](); } catch { return result; }
|
|
1437
|
-
let scanned = 0;
|
|
1438
|
-
// Every early exit runs through this. The production `list` is a generator that closes its
|
|
1439
|
-
// directory handle in a `finally`, and a `finally` does NOT run when the consumer simply
|
|
1440
|
-
// stops calling `next()` -- only `return()` resumes the generator to completion. Breaking
|
|
1441
|
-
// out of the loop directly therefore leaked one directory handle per truncated scan, and the
|
|
1442
|
-
// periodic reclaim truncates on purpose (entry cap, cleanup cap, deadline), so on a slow
|
|
1443
|
-
// filesystem that is a leak per tick, forever.
|
|
1444
|
-
const stopScan = (): ResponseStateTempRecoveryResult => {
|
|
1445
|
-
try { iterator.return?.(); } catch { /* closing is best-effort; never fail a reclaim on it */ }
|
|
1446
|
-
return result;
|
|
1447
|
-
};
|
|
1448
|
-
for (;;) {
|
|
1449
|
-
let next: IteratorResult<string>;
|
|
1450
|
-
try { next = iterator.next(); } catch { return result; }
|
|
1451
|
-
if (next.done) break;
|
|
1452
|
-
const name = next.value;
|
|
1453
|
-
scanned += 1;
|
|
1454
|
-
// A dry run performs no cleanups, so bounding it by the cleanup budget would truncate
|
|
1455
|
-
// the very report an operator uses to size the problem.
|
|
1456
|
-
if (scanned > maxEntries) { result.truncated = true; return stopScan(); }
|
|
1457
|
-
if (!dryRun && result.removed + result.failed >= maxCleanups) { result.truncated = true; return stopScan(); }
|
|
1458
|
-
if (deadlineMs !== null && io.now() - startedAt > deadlineMs) { result.truncated = true; return stopScan(); }
|
|
1459
|
-
const match = RESPONSE_STATE_TEMP_NAME.exec(name);
|
|
1460
|
-
if (!match) continue;
|
|
1461
|
-
result.matched += 1;
|
|
1462
|
-
const pid = Number(match[1]);
|
|
1463
|
-
const sequence = Number(match[2]);
|
|
1464
|
-
if (!Number.isSafeInteger(pid) || pid <= 0 || !Number.isSafeInteger(sequence) || sequence <= 0) continue;
|
|
1465
|
-
const path = join(dir, name);
|
|
1466
|
-
let file: ReturnType<ResponseStateTempRecoveryIO["inspect"]>;
|
|
1467
|
-
try { file = io.inspect(path); } catch { continue; }
|
|
1468
|
-
if (!file.isFile || io.now() - file.mtimeMs < STALE_TEMP_GRACE_MS) continue;
|
|
1469
|
-
// Boot floor. After a reboot the original writer's pid is routinely reused, which makes
|
|
1470
|
-
// the liveness skip PERMANENT: the 15-minute grace above is a lower bound and never
|
|
1471
|
-
// expires it, so the file is skipped on every future pass forever. A temp older than
|
|
1472
|
-
// this boot cannot be owned by the pid we would probe, so the probe is vacuous and we
|
|
1473
|
-
// retire it. This does NOT claim the file is provably dead: under a shared-volume
|
|
1474
|
-
// container, suspend-excluding uptime, or a network config dir the computed boot can
|
|
1475
|
-
// land after the real one. The unconditional 15-minute grace above remains the safety
|
|
1476
|
-
// floor, and this process's own temps are never touched.
|
|
1477
|
-
const predatesBoot = file.mtimeMs < bootMs - BOOT_FLOOR_SKEW_MS;
|
|
1478
|
-
if (pid === process.pid) continue;
|
|
1479
|
-
if (!predatesBoot && io.isProcessAlive(pid)) continue;
|
|
1480
|
-
|
|
1481
|
-
result.eligible += 1;
|
|
1482
|
-
result.eligibleBytes += file.size;
|
|
1483
|
-
if (dryRun) continue;
|
|
1484
|
-
|
|
1485
|
-
try {
|
|
1486
|
-
io.unlink(path);
|
|
1487
|
-
result.removed += 1;
|
|
1488
|
-
result.bytesRemoved += file.size;
|
|
1489
|
-
} catch (error) {
|
|
1490
|
-
// Another proxy sharing this config dir may have won the race. A file that is already
|
|
1491
|
-
// gone is reclaimed, not a failure -- reporting it as one would surface "in use or
|
|
1492
|
-
// locked" to an operator for a file nobody holds.
|
|
1493
|
-
if ((error as NodeJS.ErrnoException)?.code === "ENOENT") {
|
|
1494
|
-
result.removed += 1;
|
|
1495
|
-
continue;
|
|
1496
|
-
}
|
|
1497
|
-
// Locked files remain for a later startup. Do not truncate by path: a same-user
|
|
1498
|
-
// replacement could turn that fallback into an arbitrary symlink-target write.
|
|
1499
|
-
result.failed += 1;
|
|
1500
|
-
}
|
|
1501
|
-
}
|
|
1502
|
-
return result;
|
|
1503
|
-
}
|
|
1504
|
-
|
|
1505
|
-
/**
|
|
1506
|
-
* Literal config dir plus the snapshot's resolved dir. Atomic writes place their temp beside
|
|
1507
|
-
* the RESOLVED target, so a symlinked snapshot (dotfiles-managed config dir) strands temps in
|
|
1508
|
-
* the link's real directory where a scan of the literal dir would never see them. The two
|
|
1509
|
-
* collapse to one when nothing is symlinked.
|
|
1510
|
-
*/
|
|
1511
|
-
function responseStateSweepDirectories(): Set<string> {
|
|
1512
|
-
const path = snapshotPath();
|
|
1513
|
-
let resolvedDir = dirname(path);
|
|
1514
|
-
try {
|
|
1515
|
-
resolvedDir = dirname(resolveWriteTarget(path));
|
|
1516
|
-
} catch {
|
|
1517
|
-
/* unresolvable link: sweep the literal dir only */
|
|
1518
|
-
}
|
|
1519
|
-
return new Set([dirname(path), resolvedDir]);
|
|
1520
|
-
}
|
|
1521
|
-
|
|
1522
596
|
/**
|
|
1523
597
|
* Best-effort disk snapshot so previous_response_id chains survive a proxy restart (the
|
|
1524
598
|
* dominant expansion-miss cause: an in-memory-only store dies with the process, and the next
|
|
@@ -1565,7 +639,14 @@ function ensureLoaded(): void {
|
|
|
1565
639
|
if ((raw.version === 1 || raw.version === 2) && Array.isArray(raw.states)) {
|
|
1566
640
|
for (const entry of raw.states) {
|
|
1567
641
|
if (!Array.isArray(entry) || entry.length !== 2 || typeof entry[0] !== "string") continue;
|
|
1568
|
-
loadSnapshotEntry(entry[0], entry[1]
|
|
642
|
+
loadSnapshotEntry(entry[0], entry[1], {
|
|
643
|
+
replaceMapEntry,
|
|
644
|
+
stubSize,
|
|
645
|
+
tombstone,
|
|
646
|
+
measureResidentEntry,
|
|
647
|
+
admitOversizedCandidate,
|
|
648
|
+
byteCap,
|
|
649
|
+
});
|
|
1569
650
|
}
|
|
1570
651
|
}
|
|
1571
652
|
}
|
|
@@ -1748,89 +829,8 @@ function inputItems(input: unknown): unknown[] {
|
|
|
1748
829
|
return [input];
|
|
1749
830
|
}
|
|
1750
831
|
|
|
1751
|
-
/** Hard cap for canonicalizing ANY item. Past it, the item is not comparable. */
|
|
1752
|
-
const REPLAY_FINGERPRINT_MAX_BYTES = 8 * 1024;
|
|
1753
|
-
/** Depth ceiling so a pathologically nested item cannot blow the canonicalizer. */
|
|
1754
|
-
const REPLAY_FINGERPRINT_MAX_DEPTH = 64;
|
|
1755
|
-
|
|
1756
832
|
let replayOverlapSkips = 0;
|
|
1757
833
|
|
|
1758
|
-
/**
|
|
1759
|
-
* Canonical, order-stable fingerprint for one input item, or null when the item cannot be
|
|
1760
|
-
* compared safely.
|
|
1761
|
-
*
|
|
1762
|
-
* Byte-counted DURING the walk rather than serialize-then-measure: a tool result can be
|
|
1763
|
-
* megabytes and this runs on the request path, so the point of the cap is to stop early,
|
|
1764
|
-
* not to discover afterwards that we should have. Object keys are sorted so two
|
|
1765
|
-
* semantically identical items cannot differ by key order alone.
|
|
1766
|
-
*
|
|
1767
|
-
* The cap applies to EVERY item. An `id`/`call_id` is additional occurrence evidence, never
|
|
1768
|
-
* a substitute for content equality, so an over-cap identified tool item is non-comparable
|
|
1769
|
-
* exactly like an over-cap message.
|
|
1770
|
-
*/
|
|
1771
|
-
function replayItemFingerprint(item: unknown): string | null {
|
|
1772
|
-
const out: string[] = [];
|
|
1773
|
-
let bytes = 0;
|
|
1774
|
-
const push = (text: string): boolean => {
|
|
1775
|
-
bytes += Buffer.byteLength(text, "utf8");
|
|
1776
|
-
if (bytes > REPLAY_FINGERPRINT_MAX_BYTES) return false;
|
|
1777
|
-
out.push(text);
|
|
1778
|
-
return true;
|
|
1779
|
-
};
|
|
1780
|
-
const walk = (value: unknown, depth: number): boolean => {
|
|
1781
|
-
if (depth > REPLAY_FINGERPRINT_MAX_DEPTH) return false;
|
|
1782
|
-
if (value === null || typeof value !== "object") return push(JSON.stringify(value) ?? "null");
|
|
1783
|
-
if (Array.isArray(value)) {
|
|
1784
|
-
if (!push("[")) return false;
|
|
1785
|
-
for (const element of value) {
|
|
1786
|
-
if (!walk(element, depth + 1)) return false;
|
|
1787
|
-
if (!push(",")) return false;
|
|
1788
|
-
}
|
|
1789
|
-
return push("]");
|
|
1790
|
-
}
|
|
1791
|
-
if (!push("{")) return false;
|
|
1792
|
-
for (const key of Object.keys(value as Record<string, unknown>).sort()) {
|
|
1793
|
-
if (!push(JSON.stringify(key))) return false;
|
|
1794
|
-
if (!walk((value as Record<string, unknown>)[key], depth + 1)) return false;
|
|
1795
|
-
if (!push(",")) return false;
|
|
1796
|
-
}
|
|
1797
|
-
return push("}");
|
|
1798
|
-
};
|
|
1799
|
-
return walk(item, 0) ? out.join("") : null;
|
|
1800
|
-
}
|
|
1801
|
-
|
|
1802
|
-
/** Non-empty provider-issued `id`/`call_id` on an item, else null. */
|
|
1803
|
-
function providerIssuedIdentity(item: unknown): string | null {
|
|
1804
|
-
if (!item || typeof item !== "object" || Array.isArray(item)) return null;
|
|
1805
|
-
const record = item as { id?: unknown; call_id?: unknown };
|
|
1806
|
-
for (const candidate of [record.id, record.call_id]) {
|
|
1807
|
-
if (typeof candidate === "string" && candidate.trim().length > 0) return candidate;
|
|
1808
|
-
}
|
|
1809
|
-
return null;
|
|
1810
|
-
}
|
|
1811
|
-
|
|
1812
|
-
/**
|
|
1813
|
-
* Number of leading stored items the client already carries verbatim, or 0.
|
|
1814
|
-
*
|
|
1815
|
-
* Requires an exact ordered run: every stored item must match the client input item at the
|
|
1816
|
-
* same index. Any not-comparable item aborts to 0 — skipping just that item could align two
|
|
1817
|
-
* different occurrences and manufacture a false positive, and a false positive here deletes
|
|
1818
|
-
* real conversation history.
|
|
1819
|
-
*
|
|
1820
|
-
* Known gap (FU-2): stored input can contain proxy-injected guidance the client never saw,
|
|
1821
|
-
* and ids repaired after recording. Those sessions do not match here and expand as before.
|
|
1822
|
-
*/
|
|
1823
|
-
function clientCarriedPrefixLength(stored: readonly unknown[], clientInput: readonly unknown[]): number {
|
|
1824
|
-
if (stored.length === 0 || clientInput.length < stored.length) return 0;
|
|
1825
|
-
for (let index = 0; index < stored.length; index += 1) {
|
|
1826
|
-
const storedPrint = replayItemFingerprint(stored[index]);
|
|
1827
|
-
if (storedPrint === null) return 0;
|
|
1828
|
-
const clientPrint = replayItemFingerprint(clientInput[index]);
|
|
1829
|
-
if (clientPrint === null || storedPrint !== clientPrint) return 0;
|
|
1830
|
-
}
|
|
1831
|
-
return stored.length;
|
|
1832
|
-
}
|
|
1833
|
-
|
|
1834
834
|
/** Test-only: replay prepends skipped because the client already carried the history. */
|
|
1835
835
|
export function replayOverlapSkipsForTests(): number {
|
|
1836
836
|
return replayOverlapSkips;
|
|
@@ -1903,9 +903,9 @@ function pruneResponses(at = now()): void {
|
|
|
1903
903
|
// deleted only when even their bounded metadata cannot fit the override.
|
|
1904
904
|
while (storedResponseBytes > byteCap() && states.size > 0) {
|
|
1905
905
|
const oldestResident = [...states].find(([id, entry]) => entry.kind === "resident"
|
|
1906
|
-
&&
|
|
906
|
+
&& !spillQueueHoldsResidentCandidate(id, entry));
|
|
1907
907
|
const hasPendingResident = !oldestResident && [...states].some(([id, entry]) => entry.kind === "resident"
|
|
1908
|
-
&&
|
|
908
|
+
&& spillQueueHoldsResidentCandidate(id, entry));
|
|
1909
909
|
if (hasPendingResident) break;
|
|
1910
910
|
const oldestId = oldestResident?.[0] ?? states.keys().next().value as string | undefined;
|
|
1911
911
|
if (!oldestId) break;
|
|
@@ -1950,70 +950,12 @@ export function sweepExpiredResponseStates(at = now()): number {
|
|
|
1950
950
|
return removed;
|
|
1951
951
|
}
|
|
1952
952
|
|
|
1953
|
-
/**
|
|
1954
|
-
* Periodic disk reclaim for abandoned atomic-write temps.
|
|
1955
|
-
*
|
|
1956
|
-
* `ensureLoaded` sweeps once per process, at load, BEFORE that process writes anything:
|
|
1957
|
-
* every `schedulePersist` site is downstream of it. So a process that abandons a temp has
|
|
1958
|
-
* already had its only look, the 15-minute grace hides the temp its predecessor's crash
|
|
1959
|
-
* just produced, and `maxCleanups` caps a single pass below a large backlog. A restart
|
|
1960
|
-
* loop therefore accumulates monotonically. Repeating the reclaim on a timer fixes all
|
|
1961
|
-
* three: the grace expires into a later tick and the per-pass cap becomes a per-tick rate.
|
|
1962
|
-
*
|
|
1963
|
-
* Registered on the sweeper's LIVENESS tick, not the TTL tick: `sweepExpiredOnWrite` puts
|
|
1964
|
-
* `sweepExpired` on hot write paths, and a directory scan does not belong there.
|
|
1965
|
-
*/
|
|
1966
|
-
export function reclaimAbandonedResponseStateTemps(
|
|
1967
|
-
options: ResponseStateTempRecoveryOptions = {},
|
|
1968
|
-
): ResponseStateTempRecoveryResult {
|
|
1969
|
-
const total: ResponseStateTempRecoveryResult = {
|
|
1970
|
-
matched: 0, removed: 0, failed: 0, bytesRemoved: 0, eligible: 0, eligibleBytes: 0, truncated: false,
|
|
1971
|
-
};
|
|
1972
|
-
// The try encloses responseStateSweepDirectories() deliberately: recoverStaleResponseStateTemps
|
|
1973
|
-
// already swallows its own enumeration failures, so a catch around only that call would be
|
|
1974
|
-
// unreachable. snapshotPath()/getConfigDir() are the paths that can genuinely throw.
|
|
1975
|
-
try {
|
|
1976
|
-
for (const dir of responseStateSweepDirectories()) {
|
|
1977
|
-
const result = recoverStaleResponseStateTemps(dir, options);
|
|
1978
|
-
total.matched += result.matched;
|
|
1979
|
-
total.removed += result.removed;
|
|
1980
|
-
total.failed += result.failed;
|
|
1981
|
-
total.bytesRemoved += result.bytesRemoved;
|
|
1982
|
-
total.eligible += result.eligible;
|
|
1983
|
-
total.eligibleBytes += result.eligibleBytes;
|
|
1984
|
-
// Truncation anywhere makes the whole total a prefix.
|
|
1985
|
-
total.truncated ||= result.truncated;
|
|
1986
|
-
}
|
|
1987
|
-
} catch {
|
|
1988
|
-
/* best-effort: disk reclaim must never destabilize the caller */
|
|
1989
|
-
}
|
|
1990
|
-
return total;
|
|
1991
|
-
}
|
|
1992
|
-
|
|
1993
|
-
/**
|
|
1994
|
-
* Report-only counterpart for `ocx doctor`: applies every selection gate and unlinks
|
|
1995
|
-
* nothing. It runs the SAME predicate as the reclaim, so the report and the subsequent
|
|
1996
|
-
* removal cannot disagree about which files are reclaimable.
|
|
1997
|
-
*/
|
|
1998
|
-
export function inspectAbandonedResponseStateTemps(): ResponseStateTempRecoveryResult {
|
|
1999
|
-
return reclaimAbandonedResponseStateTemps({ dryRun: true });
|
|
2000
|
-
}
|
|
2001
|
-
|
|
2002
|
-
/** Sweeper adapter: narrows the reclaim to the `() => number` the liveness tick expects. */
|
|
2003
|
-
export function sweepAbandonedResponseStateTemps(): number {
|
|
2004
|
-
return reclaimAbandonedResponseStateTemps({
|
|
2005
|
-
maxEntries: PERIODIC_TEMP_MAX_ENTRIES,
|
|
2006
|
-
maxCleanups: PERIODIC_TEMP_MAX_CLEANUPS,
|
|
2007
|
-
deadlineMs: PERIODIC_TEMP_SCAN_DEADLINE_MS,
|
|
2008
|
-
}).removed;
|
|
2009
|
-
}
|
|
2010
|
-
|
|
2011
953
|
export function responseContinuationRetainedStoreSnapshot(): RetainedStoreSnapshot {
|
|
2012
954
|
let currentPendingBytes = 0;
|
|
2013
|
-
for (const job of
|
|
2014
|
-
if (
|
|
955
|
+
for (const job of spillQueueResidentCandidates()) {
|
|
956
|
+
if (states.get(job.id) === job.candidate) currentPendingBytes += job.sizeBytes;
|
|
2015
957
|
}
|
|
2016
|
-
const detachedPendingBytes = Math.max(0,
|
|
958
|
+
const detachedPendingBytes = Math.max(0, spillQueuePendingBytes() - currentPendingBytes);
|
|
2017
959
|
const bytes = storedResponseBytes + detachedPendingBytes;
|
|
2018
960
|
const evictableBytes = Math.max(0, residentResponseBytes - currentPendingBytes);
|
|
2019
961
|
return {
|
|
@@ -2326,7 +1268,7 @@ export function rememberResponseState(
|
|
|
2326
1268
|
// `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
|
|
2327
1269
|
// HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
|
|
2328
1270
|
// The passthrough branch records with force so those chains can be expanded locally; the
|
|
2329
|
-
// store stays in-memory
|
|
1271
|
+
// store stays in-memory under RESPONSE_TTL_MS, so this is a proxy-internal continuation cache, not
|
|
2330
1272
|
// real server-side response storage.
|
|
2331
1273
|
if (request.store === false && !opts?.force) return;
|
|
2332
1274
|
if (typeof response.id !== "string" || !Array.isArray(response.output)) return;
|
|
@@ -2390,8 +1332,7 @@ export function clearResponseStateMemoryForTests(): void {
|
|
|
2390
1332
|
persistTimer = null;
|
|
2391
1333
|
}
|
|
2392
1334
|
pendingPersistPath = null;
|
|
2393
|
-
|
|
2394
|
-
pendingResponseSpillById.clear();
|
|
1335
|
+
resetSpillQueueForTests();
|
|
2395
1336
|
states.clear();
|
|
2396
1337
|
storedResponseBytes = 0;
|
|
2397
1338
|
residentResponseBytes = 0;
|
|
@@ -2421,8 +1362,6 @@ export function clearResponseStateMemoryForTests(): void {
|
|
|
2421
1362
|
export function clearResponseStateForTests(): void {
|
|
2422
1363
|
for (const entry of states.values()) deleteOwnedSpills(entry);
|
|
2423
1364
|
clearResponseStateMemoryForTests();
|
|
2424
|
-
reservedResponseSpillBytes = 0;
|
|
2425
|
-
unreclaimableSpillPaths.clear();
|
|
2426
1365
|
try {
|
|
2427
1366
|
unlinkSync(snapshotPath());
|
|
2428
1367
|
} catch {
|