@tanstack/ai-sandbox 0.2.4 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
@@ -0,0 +1,187 @@
1
+ import { LockStore } from '@tanstack/ai/locks';
2
+ import { InternalLogger } from '@tanstack/ai/adapter-internals';
3
+ import { RunStore, StreamDurability } from '@tanstack/ai';
4
+ /** Quiescence window before a successor's first append. */
5
+ export declare const DEFAULT_FENCE_QUIET_MS = 5000;
6
+ /**
7
+ * Appends a fenced log makes between `driverEpoch` re-reads.
8
+ *
9
+ * Deliberately a COUNT, not an interval: `pipeToRunLog` appends one chunk per
10
+ * call, so this bounds a superseded driver to at most 31 further chunk batches
11
+ * (the bump can land immediately after a check) regardless of how fast the run
12
+ * streams. At 500 chunks/sec that worst case is ~62ms of writes; at 5
13
+ * chunks/sec it is ~6s of writes — either way 31 chunks, never ~1000.
14
+ *
15
+ * The cost of a smaller number is one extra `RunStore.get` per 32 chunks.
16
+ */
17
+ export declare const DEFAULT_EPOCH_RECHECK_APPENDS = 32;
18
+ /** Lock key for a run's driver. Per-run, so two runs never serialize. */
19
+ export declare function runDriverLockKey(runId: string): string;
20
+ /** The claim was never acquired, so the caller must not drive the run. */
21
+ export declare class RunClaimNotAcquiredError extends Error {
22
+ readonly runId: string;
23
+ readonly reason: 'terminal' | 'unknown' | 'superseded';
24
+ constructor(runId: string, reason: 'terminal' | 'unknown' | 'superseded');
25
+ }
26
+ /** The claim was held and has been superseded; stop writing immediately. */
27
+ export declare class RunClaimLostError extends Error {
28
+ readonly runId: string;
29
+ readonly heldEpoch: number;
30
+ readonly observedEpoch: number | 'lease-lost';
31
+ constructor(runId: string, heldEpoch: number, observedEpoch: number | 'lease-lost');
32
+ }
33
+ /** A held claim on one run. */
34
+ export interface RunClaim {
35
+ runId: string;
36
+ /** This driver's fencing token; strictly greater than any predecessor's. */
37
+ epoch: number;
38
+ /** Aborts when the lock can no longer guarantee ownership. */
39
+ signal: AbortSignal;
40
+ }
41
+ export interface WithRunClaimOptions {
42
+ runs: RunStore;
43
+ locks: LockStore;
44
+ runId: string;
45
+ /**
46
+ * Quiescence window for {@link awaitLogQuiescence}. Defaults to
47
+ * {@link DEFAULT_FENCE_QUIET_MS}.
48
+ *
49
+ * `withRunClaim` itself does not read this: it has no durability handle. It
50
+ * lives here so a caller assembling a drive passes ONE options object to
51
+ * `withRunClaim`, `awaitLogQuiescence`, and {@link fenceDurability} instead of
52
+ * three that can drift apart.
53
+ */
54
+ fenceQuietMs?: number;
55
+ /**
56
+ * Forwarded to {@link fenceDurability}. Defaults to
57
+ * {@link DEFAULT_EPOCH_RECHECK_APPENDS}. Same rationale as `fenceQuietMs`.
58
+ */
59
+ epochRecheckAppends?: number;
60
+ logger?: InternalLogger;
61
+ }
62
+ /**
63
+ * Claim exclusive driver rights on `runId` for the duration of `fn`.
64
+ *
65
+ * The ENTIRE body runs inside the lock, so a snapshot taken by `fn` and every
66
+ * append that follows it sit in one critical section.
67
+ *
68
+ * Rejects with {@link RunClaimNotAcquiredError} when the run is unknown or
69
+ * already terminal — a terminal run has nothing left to drive, and bumping its
70
+ * epoch would fence out nobody while confusing an operator reading the record.
71
+ *
72
+ * The epoch is bumped INSIDE the lock and only after those checks pass, so a
73
+ * refused claim leaves `driverEpoch` untouched.
74
+ */
75
+ export declare function withRunClaim<T>(options: WithRunClaimOptions, fn: (claim: RunClaim) => Promise<T>): Promise<T>;
76
+ /**
77
+ * Wait until the stored log stops growing, then answer how many entries it
78
+ * holds.
79
+ *
80
+ * Uses `snapshot()`, never `read()`: `read` tails and only resolves once the log
81
+ * is terminalized or the caller aborts, and a taken-over run's log is open by
82
+ * definition — the host that would have closed it is the host that died.
83
+ *
84
+ * Rejects rather than looping forever. A log that never quiesces means a
85
+ * predecessor is still actively writing, which is a condition to surface, not to
86
+ * append into.
87
+ *
88
+ * This only detects a CONCURRENT predecessor, which means it can only fire when
89
+ * the two drivers are in different processes. Within one process an
90
+ * `InMemoryLockStore` serializes the claims, so the predecessor has already
91
+ * stopped by the time the successor probes.
92
+ */
93
+ export declare function awaitLogQuiescence<TOffset extends string = string>(durability: StreamDurability<TOffset>, quietMs: number): Promise<number>;
94
+ /**
95
+ * Wrap a log so every `append` is fenced by `claim`.
96
+ *
97
+ * `append` is the ONLY fenced method, deliberately:
98
+ *
99
+ * - `close()` must never be fenced. It runs on every teardown path including
100
+ * the teardown caused by losing the claim, and a fenced `close` would leave
101
+ * the record wedged at `'running'` with every live tailer parked forever (a
102
+ * `read` only ends when the log closes).
103
+ * - `read` / `snapshot` / `resumeFrom` do not mutate, so a superseded host
104
+ * reading them is harmless.
105
+ *
106
+ * The lease check is synchronous and happens before any I/O, so a fenced append
107
+ * cannot half-land. The epoch re-check is throttled to `epochRecheckAppends`
108
+ * because it costs a store read and the append path is hot.
109
+ *
110
+ * ONE REFUSAL CLOSES THE FENCE FOR GOOD. The first `append` that is refused —
111
+ * for EITHER cause, lost lease or moved epoch — latches this wrapper shut, and
112
+ * every later `append` refuses immediately without consulting the throttle and
113
+ * without a store read. This is not a nicety:
114
+ *
115
+ * - Losing a claim is not transient. Epochs only move forward and a lease is
116
+ * never handed back, so a wrapper that has refused once can never legitimately
117
+ * append again. Re-deciding per append can only produce a WRONG answer.
118
+ * - The throttle makes that wrong answer reachable. A refusal consumes the
119
+ * re-read budget, so the very next append rides a fresh throttle window and is
120
+ * NOT re-checked. `pipeToRunLog`'s recovery path appends a `RUN_ERROR` right
121
+ * after the refusal it is recovering from, and that log belongs to the
122
+ * SUCCESSOR: a terminal `RUN_ERROR` from a dead host would fail the stream for
123
+ * every client attached to the live, healthy run.
124
+ * - It is also strictly cheaper: a latched boolean replaces a store read.
125
+ *
126
+ * The latch deliberately does NOT extend to `close()` — see above.
127
+ *
128
+ * PASSES THE OFFSET TYPE THROUGH, rather than collapsing it to `string`. The
129
+ * fence sits mid-chain between a caller's log and `pipeToRunLog`, so widening
130
+ * here would reintroduce the branded-offset wall one layer in: a
131
+ * `StreamDurability<DurableStreamOffset>` would go in and a
132
+ * `StreamDurability<string>` would come out, which is not assignable back to
133
+ * the caller's own type.
134
+ */
135
+ export declare function fenceDurability<TOffset extends string = string>(durability: StreamDurability<TOffset>, claim: RunClaim, options: {
136
+ runs: RunStore;
137
+ epochRecheckAppends?: number;
138
+ }): StreamDurability<TOffset>;
139
+ /**
140
+ * Wrap a run store so a TERMINAL record write is fenced by `claim`.
141
+ *
142
+ * The record is the run's other authoritative channel, and the same rule applies
143
+ * to it: a host that has lost its claim must not state that the run is over. It
144
+ * reaches this seam by the most ordinary route — `pipeToRunLog` catches the
145
+ * `RunClaimLostError` its refused append threw, folds it in, and calls
146
+ * `finish(ctx, 'failed', …)` — so fencing the log alone only moves where the harm
147
+ * surfaces. `'completed'` and `'aborted'` arrive the same way (an empty stream
148
+ * that never appended; a lease loss that aborts `claim.signal`, which
149
+ * `pipeToRunLog` reads as an abort before it appends anything), which is why the
150
+ * gate is {@link isTerminalRunStatus} and not "did an append refuse".
151
+ *
152
+ * SUPPRESSED, NOT ATTEMPTED-AND-SWALLOWED, and not thrown either. `update`
153
+ * resolves without writing. `pipeToRunLog` must not reject — `RunController.start`
154
+ * consumes its promise fire-and-forget — and a rejection here would additionally
155
+ * make `finish` report the run through the local rebuilt record as if the store
156
+ * had broken, which is a different and false fact.
157
+ *
158
+ * WHAT IS *NOT* FENCED, deliberately:
159
+ *
160
+ * - **`close()`** is not on this seam at all, and must stay off it: see
161
+ * {@link fenceDurability}. A wedged `'running'` record with tailers parked
162
+ * forever is worse than the write being prevented.
163
+ * - **Non-terminal writes pass through**, including `detachedSince` and
164
+ * `sandboxKey` written by a superseded host. They are stale, but staleness is
165
+ * not the harm being fixed: none of them can make a live run look finished, so
166
+ * none can mislead `isTerminalRunStatus`, `findActiveRun`, or the reaper. They
167
+ * are also self-healing — the successor owns those fields and overwrites them —
168
+ * whereas over-suppressing strands a record: `createOrResume` is how the row
169
+ * comes into existence at all, and refusing a non-terminal write on a
170
+ * mis-observed loss would leave a run with no record to recover from. Suppress
171
+ * the writes that assert an outcome; let bookkeeping through.
172
+ * - **Reads** (`get`, `listByThread`, `listReclaimable`, `findActiveRun`) do not
173
+ * mutate, so a superseded host reading them is harmless. `finish`'s terminal
174
+ * re-read therefore still works and answers with the SUCCESSOR's live record,
175
+ * which is the truthful thing to resolve with.
176
+ * - **Another run's record.** The fence knows about `claim.runId` only; a write
177
+ * aimed elsewhere is not this claim's to judge.
178
+ *
179
+ * The OPTIONAL methods (`listByThread`, `listReclaimable`) are forwarded only
180
+ * when the wrapped store actually has them: consumers feature-detect
181
+ * (`store.listReclaimable?.(…)`), so materializing one that delegates to a
182
+ * missing method would turn a graceful degrade into a `TypeError`.
183
+ * `findActiveRun` is required on the contract, so it forwards unconditionally.
184
+ */
185
+ export declare function fenceRunStore(runs: RunStore, claim: RunClaim, options?: {
186
+ logger?: InternalLogger;
187
+ }): RunStore;
@@ -0,0 +1,349 @@
1
+ import { isTerminalRunStatus } from "@tanstack/ai";
2
+ //#region src/claim.ts
3
+ /**
4
+ * The single-writer claim: what makes a takeover safe to attempt at all.
5
+ *
6
+ * WHY THIS MODULE EXISTS. `alignToStoredLog` decides where its appends start by
7
+ * reading `durability.snapshot()`, and `snapshot()` carries NO LOCK — core says
8
+ * so explicitly (`packages/ai/src/stream-durability.ts`: "a concurrent `append`
9
+ * may land immediately after the snapshot is taken"). If two hosts drive one
10
+ * run, both snapshot, both compute a "remainder", and both append it. The log
11
+ * then holds the same logical chunk twice under two different offsets, and the
12
+ * client CANNOT survive that: `ai-client`'s de-dup is keyed on the adapter's
13
+ * offset string, so a re-appended chunk looks new, and the stream processor
14
+ * applies text and tool-argument deltas unconditionally. The visible result is
15
+ * doubled message text and `{"a":1}{"a":1}` tool arguments.
16
+ *
17
+ * Takeover is by definition two hosts wanting one run, so nothing may read a
18
+ * journal for a run it has not claimed.
19
+ *
20
+ * THREE LAYERS, strongest first:
21
+ *
22
+ * 1. **The lease.** {@link withRunClaim} runs the whole drive inside
23
+ * `LockStore.withLock('run-driver:<runId>', …)`, so the snapshot and every
24
+ * append that follows are one critical section. A lease-backed lock aborts
25
+ * the callback signal the moment ownership is lost, and
26
+ * {@link fenceDurability} turns that into a thrown {@link RunClaimLostError}
27
+ * BEFORE the append reaches the log.
28
+ * 2. **The epoch.** Each successful claim bumps `RunRecord.driverEpoch`.
29
+ * {@link fenceDurability} re-reads it every
30
+ * {@link DEFAULT_EPOCH_RECHECK_APPENDS} appends and refuses to append once a
31
+ * higher epoch exists. This covers what a lease cannot: an
32
+ * `InMemoryLockStore`, whose signal is a fresh `AbortController().signal`
33
+ * that is never aborted, and any backend whose renewal is coarser than the
34
+ * run's append rate. Once EITHER fence has refused an append, the fence
35
+ * latches shut and every later append refuses without re-reading anything.
36
+ * 3. **Quiescence.** {@link awaitLogQuiescence} requires the stored log to stop
37
+ * growing before the successor appends anything, so a predecessor that is
38
+ * still writing is OBSERVED rather than raced.
39
+ *
40
+ * THE LOG IS NOT THE ONLY AUTHORITATIVE CHANNEL. A host that has lost its claim
41
+ * must not write authoritative facts about the run through ANY seam, and there
42
+ * are two: the event log and the run RECORD. Fencing only the log moves the harm
43
+ * rather than removing it — a superseded driver whose append was refused folds
44
+ * that refusal into a terminal `runs.update`, so the record reads `'failed'` for
45
+ * a run the successor is healthily streaming, and `isTerminalRunStatus` (which
46
+ * `findActiveRun`, the resume driver, and `reapDetachedRuns` all branch on) then
47
+ * answers `true` for a live run. {@link fenceRunStore} closes that seam; both
48
+ * fences share one per-claim latch so they can never disagree about whether the
49
+ * claim is still held.
50
+ *
51
+ * WHY THE EPOCH RE-CHECK COUNTS APPENDS, NOT MILLISECONDS. `pipeToRunLog`
52
+ * appends ONE chunk per call, so a time-based interval couples the fence's
53
+ * resolution to the run's chunk rate: at 500 chunks/sec a 2s interval lets a
54
+ * superseded driver write ~1000 chunks before it notices. A count gives a hard
55
+ * bound independent of rate — see {@link DEFAULT_EPOCH_RECHECK_APPENDS}.
56
+ *
57
+ * WHAT THIS IS NOT. It is not airtight fencing.
58
+ *
59
+ * - A predecessor paused (GC, VM suspend) for longer than the quiescence
60
+ * window, between its last fence check and its append landing at the backend,
61
+ * can still write one batch. Closing that requires a compare-and-set on the
62
+ * durability write; `StreamDurability.append` has no such parameter and this
63
+ * phase deliberately does not add one.
64
+ * - Layer 3 is only meaningful across PROCESSES. On a single-process
65
+ * `InMemoryLockStore` the two claims are serialized by the lock, not
66
+ * concurrent, so `awaitLogQuiescence` can never observe a predecessor still
67
+ * writing there — and consequently no unit test on that backend proves layer
68
+ * 3 does anything. What the tests do prove on that backend is layer 2.
69
+ *
70
+ * The mitigation for both is deployment-level: use a lease-backed distributed
71
+ * `LockStore`, and keep `fenceQuietMs` above the lease's renewal interval.
72
+ */
73
+ /** Quiescence window before a successor's first append. */
74
+ var DEFAULT_FENCE_QUIET_MS = 5e3;
75
+ /** Probes {@link awaitLogQuiescence} makes before giving up. */
76
+ var MAX_QUIESCENCE_PROBES = 6;
77
+ /** Lock key for a run's driver. Per-run, so two runs never serialize. */
78
+ function runDriverLockKey(runId) {
79
+ return `run-driver:${runId}`;
80
+ }
81
+ /** The claim was never acquired, so the caller must not drive the run. */
82
+ var RunClaimNotAcquiredError = class extends Error {
83
+ runId;
84
+ reason;
85
+ constructor(runId, reason) {
86
+ super(`run ${runId}: driver claim not acquired (${reason})`);
87
+ this.runId = runId;
88
+ this.reason = reason;
89
+ this.name = "RunClaimNotAcquiredError";
90
+ }
91
+ };
92
+ /** The claim was held and has been superseded; stop writing immediately. */
93
+ var RunClaimLostError = class extends Error {
94
+ runId;
95
+ heldEpoch;
96
+ observedEpoch;
97
+ constructor(runId, heldEpoch, observedEpoch) {
98
+ super(`run ${runId}: driver claim lost (held epoch ${heldEpoch}, observed ${observedEpoch})`);
99
+ this.runId = runId;
100
+ this.heldEpoch = heldEpoch;
101
+ this.observedEpoch = observedEpoch;
102
+ this.name = "RunClaimLostError";
103
+ }
104
+ };
105
+ /**
106
+ * Claim exclusive driver rights on `runId` for the duration of `fn`.
107
+ *
108
+ * The ENTIRE body runs inside the lock, so a snapshot taken by `fn` and every
109
+ * append that follows it sit in one critical section.
110
+ *
111
+ * Rejects with {@link RunClaimNotAcquiredError} when the run is unknown or
112
+ * already terminal — a terminal run has nothing left to drive, and bumping its
113
+ * epoch would fence out nobody while confusing an operator reading the record.
114
+ *
115
+ * The epoch is bumped INSIDE the lock and only after those checks pass, so a
116
+ * refused claim leaves `driverEpoch` untouched.
117
+ */
118
+ async function withRunClaim(options, fn) {
119
+ const { runs, locks, runId, logger } = options;
120
+ return locks.withLock(runDriverLockKey(runId), async (signal) => {
121
+ const record = await runs.get(runId);
122
+ if (record === null) throw new RunClaimNotAcquiredError(runId, "unknown");
123
+ if (isTerminalRunStatus(record.status)) throw new RunClaimNotAcquiredError(runId, "terminal");
124
+ const epoch = (record.driverEpoch ?? 0) + 1;
125
+ await runs.update(runId, { driverEpoch: epoch });
126
+ logger?.sandbox(`run ${runId}: driver claim acquired at epoch ${epoch}`, {
127
+ runId,
128
+ epoch
129
+ });
130
+ return fn({
131
+ runId,
132
+ epoch,
133
+ signal
134
+ });
135
+ });
136
+ }
137
+ /**
138
+ * Wait until the stored log stops growing, then answer how many entries it
139
+ * holds.
140
+ *
141
+ * Uses `snapshot()`, never `read()`: `read` tails and only resolves once the log
142
+ * is terminalized or the caller aborts, and a taken-over run's log is open by
143
+ * definition — the host that would have closed it is the host that died.
144
+ *
145
+ * Rejects rather than looping forever. A log that never quiesces means a
146
+ * predecessor is still actively writing, which is a condition to surface, not to
147
+ * append into.
148
+ *
149
+ * This only detects a CONCURRENT predecessor, which means it can only fire when
150
+ * the two drivers are in different processes. Within one process an
151
+ * `InMemoryLockStore` serializes the claims, so the predecessor has already
152
+ * stopped by the time the successor probes.
153
+ */
154
+ async function awaitLogQuiescence(durability, quietMs) {
155
+ let previous = (await durability.snapshot()).length;
156
+ for (let probe = 0; probe < MAX_QUIESCENCE_PROBES; probe += 1) {
157
+ await sleep(quietMs);
158
+ const current = (await durability.snapshot()).length;
159
+ if (current === previous) return current;
160
+ previous = current;
161
+ }
162
+ throw new Error(`journal takeover: the event log never quiesced after ${MAX_QUIESCENCE_PROBES} probes (${previous} entries and still growing); another host is still driving this run`);
163
+ }
164
+ function sleep(ms) {
165
+ if (ms <= 0) return Promise.resolve();
166
+ return new Promise((resolve) => setTimeout(resolve, ms));
167
+ }
168
+ var CLAIM_LATCHES = /* @__PURE__ */ new WeakMap();
169
+ function latchFor(claim) {
170
+ const existing = CLAIM_LATCHES.get(claim);
171
+ if (existing !== void 0) return existing;
172
+ const latch = { lost: void 0 };
173
+ CLAIM_LATCHES.set(claim, latch);
174
+ return latch;
175
+ }
176
+ /**
177
+ * The I/O-free half of the check: the latch and the lease. Synchronous on
178
+ * purpose — a fenced write must be refused BEFORE anything can half-land.
179
+ */
180
+ function claimLostSynchronously(claim, latch) {
181
+ if (latch.lost !== void 0) return latch.lost;
182
+ if (claim.signal.aborted) {
183
+ latch.lost = new RunClaimLostError(claim.runId, claim.epoch, "lease-lost");
184
+ return latch.lost;
185
+ }
186
+ }
187
+ /**
188
+ * The other half: re-read `driverEpoch` and refuse once a successor exists.
189
+ *
190
+ * A store failure is NOT treated as loss. The lease is the primary fence and it
191
+ * has not fired, so fencing ourselves out on a store blip would kill a healthy
192
+ * driver — and, for the record fence, would suppress a legitimate terminal write
193
+ * and strand the run at `'running'`, which is worse than the write it prevents.
194
+ */
195
+ async function claimLostByEpoch(claim, latch, runs) {
196
+ let observed;
197
+ try {
198
+ observed = (await runs.get(claim.runId))?.driverEpoch;
199
+ } catch {
200
+ return;
201
+ }
202
+ if (observed !== void 0 && observed > claim.epoch) {
203
+ latch.lost = new RunClaimLostError(claim.runId, claim.epoch, observed);
204
+ return latch.lost;
205
+ }
206
+ }
207
+ /**
208
+ * Wrap a log so every `append` is fenced by `claim`.
209
+ *
210
+ * `append` is the ONLY fenced method, deliberately:
211
+ *
212
+ * - `close()` must never be fenced. It runs on every teardown path including
213
+ * the teardown caused by losing the claim, and a fenced `close` would leave
214
+ * the record wedged at `'running'` with every live tailer parked forever (a
215
+ * `read` only ends when the log closes).
216
+ * - `read` / `snapshot` / `resumeFrom` do not mutate, so a superseded host
217
+ * reading them is harmless.
218
+ *
219
+ * The lease check is synchronous and happens before any I/O, so a fenced append
220
+ * cannot half-land. The epoch re-check is throttled to `epochRecheckAppends`
221
+ * because it costs a store read and the append path is hot.
222
+ *
223
+ * ONE REFUSAL CLOSES THE FENCE FOR GOOD. The first `append` that is refused —
224
+ * for EITHER cause, lost lease or moved epoch — latches this wrapper shut, and
225
+ * every later `append` refuses immediately without consulting the throttle and
226
+ * without a store read. This is not a nicety:
227
+ *
228
+ * - Losing a claim is not transient. Epochs only move forward and a lease is
229
+ * never handed back, so a wrapper that has refused once can never legitimately
230
+ * append again. Re-deciding per append can only produce a WRONG answer.
231
+ * - The throttle makes that wrong answer reachable. A refusal consumes the
232
+ * re-read budget, so the very next append rides a fresh throttle window and is
233
+ * NOT re-checked. `pipeToRunLog`'s recovery path appends a `RUN_ERROR` right
234
+ * after the refusal it is recovering from, and that log belongs to the
235
+ * SUCCESSOR: a terminal `RUN_ERROR` from a dead host would fail the stream for
236
+ * every client attached to the live, healthy run.
237
+ * - It is also strictly cheaper: a latched boolean replaces a store read.
238
+ *
239
+ * The latch deliberately does NOT extend to `close()` — see above.
240
+ *
241
+ * PASSES THE OFFSET TYPE THROUGH, rather than collapsing it to `string`. The
242
+ * fence sits mid-chain between a caller's log and `pipeToRunLog`, so widening
243
+ * here would reintroduce the branded-offset wall one layer in: a
244
+ * `StreamDurability<DurableStreamOffset>` would go in and a
245
+ * `StreamDurability<string>` would come out, which is not assignable back to
246
+ * the caller's own type.
247
+ */
248
+ function fenceDurability(durability, claim, options) {
249
+ const recheckAppends = Math.max(1, Math.trunc(options.epochRecheckAppends ?? 32));
250
+ let appendsSinceEpochRead = recheckAppends;
251
+ const latch = latchFor(claim);
252
+ async function assertHeld() {
253
+ const synchronous = claimLostSynchronously(claim, latch);
254
+ if (synchronous !== void 0) throw synchronous;
255
+ if (appendsSinceEpochRead < recheckAppends) {
256
+ appendsSinceEpochRead += 1;
257
+ return;
258
+ }
259
+ appendsSinceEpochRead = 1;
260
+ const byEpoch = await claimLostByEpoch(claim, latch, options.runs);
261
+ if (byEpoch !== void 0) throw byEpoch;
262
+ }
263
+ return {
264
+ resumeFrom: () => durability.resumeFrom(),
265
+ append: async (chunks) => {
266
+ await assertHeld();
267
+ return durability.append(chunks);
268
+ },
269
+ read: (offset, signal) => durability.read(offset, signal),
270
+ close: () => durability.close(),
271
+ snapshot: () => durability.snapshot()
272
+ };
273
+ }
274
+ /**
275
+ * Wrap a run store so a TERMINAL record write is fenced by `claim`.
276
+ *
277
+ * The record is the run's other authoritative channel, and the same rule applies
278
+ * to it: a host that has lost its claim must not state that the run is over. It
279
+ * reaches this seam by the most ordinary route — `pipeToRunLog` catches the
280
+ * `RunClaimLostError` its refused append threw, folds it in, and calls
281
+ * `finish(ctx, 'failed', …)` — so fencing the log alone only moves where the harm
282
+ * surfaces. `'completed'` and `'aborted'` arrive the same way (an empty stream
283
+ * that never appended; a lease loss that aborts `claim.signal`, which
284
+ * `pipeToRunLog` reads as an abort before it appends anything), which is why the
285
+ * gate is {@link isTerminalRunStatus} and not "did an append refuse".
286
+ *
287
+ * SUPPRESSED, NOT ATTEMPTED-AND-SWALLOWED, and not thrown either. `update`
288
+ * resolves without writing. `pipeToRunLog` must not reject — `RunController.start`
289
+ * consumes its promise fire-and-forget — and a rejection here would additionally
290
+ * make `finish` report the run through the local rebuilt record as if the store
291
+ * had broken, which is a different and false fact.
292
+ *
293
+ * WHAT IS *NOT* FENCED, deliberately:
294
+ *
295
+ * - **`close()`** is not on this seam at all, and must stay off it: see
296
+ * {@link fenceDurability}. A wedged `'running'` record with tailers parked
297
+ * forever is worse than the write being prevented.
298
+ * - **Non-terminal writes pass through**, including `detachedSince` and
299
+ * `sandboxKey` written by a superseded host. They are stale, but staleness is
300
+ * not the harm being fixed: none of them can make a live run look finished, so
301
+ * none can mislead `isTerminalRunStatus`, `findActiveRun`, or the reaper. They
302
+ * are also self-healing — the successor owns those fields and overwrites them —
303
+ * whereas over-suppressing strands a record: `createOrResume` is how the row
304
+ * comes into existence at all, and refusing a non-terminal write on a
305
+ * mis-observed loss would leave a run with no record to recover from. Suppress
306
+ * the writes that assert an outcome; let bookkeeping through.
307
+ * - **Reads** (`get`, `listByThread`, `listReclaimable`, `findActiveRun`) do not
308
+ * mutate, so a superseded host reading them is harmless. `finish`'s terminal
309
+ * re-read therefore still works and answers with the SUCCESSOR's live record,
310
+ * which is the truthful thing to resolve with.
311
+ * - **Another run's record.** The fence knows about `claim.runId` only; a write
312
+ * aimed elsewhere is not this claim's to judge.
313
+ *
314
+ * The OPTIONAL methods (`listByThread`, `listReclaimable`) are forwarded only
315
+ * when the wrapped store actually has them: consumers feature-detect
316
+ * (`store.listReclaimable?.(…)`), so materializing one that delegates to a
317
+ * missing method would turn a graceful degrade into a `TypeError`.
318
+ * `findActiveRun` is required on the contract, so it forwards unconditionally.
319
+ */
320
+ function fenceRunStore(runs, claim, options = {}) {
321
+ const latch = latchFor(claim);
322
+ const listByThread = runs.listByThread?.bind(runs);
323
+ const listReclaimable = runs.listReclaimable?.bind(runs);
324
+ return {
325
+ createOrResume: (input) => runs.createOrResume(input),
326
+ get: (runId) => runs.get(runId),
327
+ findActiveRun: (threadId) => runs.findActiveRun(threadId),
328
+ update: async (runId, patch) => {
329
+ const status = patch.status;
330
+ if (runId !== claim.runId || status === void 0 || !isTerminalRunStatus(status)) return runs.update(runId, patch);
331
+ const lost = claimLostSynchronously(claim, latch) ?? await claimLostByEpoch(claim, latch, runs);
332
+ if (lost === void 0) return runs.update(runId, patch);
333
+ try {
334
+ options.logger?.sandbox(`run ${runId}: suppressed a terminal '${status}' record write from a superseded driver`, {
335
+ runId,
336
+ status,
337
+ heldEpoch: claim.epoch,
338
+ error: lost
339
+ });
340
+ } catch {}
341
+ },
342
+ ...listByThread === void 0 ? {} : { listByThread },
343
+ ...listReclaimable === void 0 ? {} : { listReclaimable }
344
+ };
345
+ }
346
+ //#endregion
347
+ export { DEFAULT_FENCE_QUIET_MS, RunClaimLostError, RunClaimNotAcquiredError, awaitLogQuiescence, fenceDurability, fenceRunStore, runDriverLockKey, withRunClaim };
348
+
349
+ //# sourceMappingURL=claim.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"claim.js","names":[],"sources":["../../src/claim.ts"],"sourcesContent":["/**\n * The single-writer claim: what makes a takeover safe to attempt at all.\n *\n * WHY THIS MODULE EXISTS. `alignToStoredLog` decides where its appends start by\n * reading `durability.snapshot()`, and `snapshot()` carries NO LOCK — core says\n * so explicitly (`packages/ai/src/stream-durability.ts`: \"a concurrent `append`\n * may land immediately after the snapshot is taken\"). If two hosts drive one\n * run, both snapshot, both compute a \"remainder\", and both append it. The log\n * then holds the same logical chunk twice under two different offsets, and the\n * client CANNOT survive that: `ai-client`'s de-dup is keyed on the adapter's\n * offset string, so a re-appended chunk looks new, and the stream processor\n * applies text and tool-argument deltas unconditionally. The visible result is\n * doubled message text and `{\"a\":1}{\"a\":1}` tool arguments.\n *\n * Takeover is by definition two hosts wanting one run, so nothing may read a\n * journal for a run it has not claimed.\n *\n * THREE LAYERS, strongest first:\n *\n * 1. **The lease.** {@link withRunClaim} runs the whole drive inside\n * `LockStore.withLock('run-driver:<runId>', …)`, so the snapshot and every\n * append that follows are one critical section. A lease-backed lock aborts\n * the callback signal the moment ownership is lost, and\n * {@link fenceDurability} turns that into a thrown {@link RunClaimLostError}\n * BEFORE the append reaches the log.\n * 2. **The epoch.** Each successful claim bumps `RunRecord.driverEpoch`.\n * {@link fenceDurability} re-reads it every\n * {@link DEFAULT_EPOCH_RECHECK_APPENDS} appends and refuses to append once a\n * higher epoch exists. This covers what a lease cannot: an\n * `InMemoryLockStore`, whose signal is a fresh `AbortController().signal`\n * that is never aborted, and any backend whose renewal is coarser than the\n * run's append rate. Once EITHER fence has refused an append, the fence\n * latches shut and every later append refuses without re-reading anything.\n * 3. **Quiescence.** {@link awaitLogQuiescence} requires the stored log to stop\n * growing before the successor appends anything, so a predecessor that is\n * still writing is OBSERVED rather than raced.\n *\n * THE LOG IS NOT THE ONLY AUTHORITATIVE CHANNEL. A host that has lost its claim\n * must not write authoritative facts about the run through ANY seam, and there\n * are two: the event log and the run RECORD. Fencing only the log moves the harm\n * rather than removing it — a superseded driver whose append was refused folds\n * that refusal into a terminal `runs.update`, so the record reads `'failed'` for\n * a run the successor is healthily streaming, and `isTerminalRunStatus` (which\n * `findActiveRun`, the resume driver, and `reapDetachedRuns` all branch on) then\n * answers `true` for a live run. {@link fenceRunStore} closes that seam; both\n * fences share one per-claim latch so they can never disagree about whether the\n * claim is still held.\n *\n * WHY THE EPOCH RE-CHECK COUNTS APPENDS, NOT MILLISECONDS. `pipeToRunLog`\n * appends ONE chunk per call, so a time-based interval couples the fence's\n * resolution to the run's chunk rate: at 500 chunks/sec a 2s interval lets a\n * superseded driver write ~1000 chunks before it notices. A count gives a hard\n * bound independent of rate — see {@link DEFAULT_EPOCH_RECHECK_APPENDS}.\n *\n * WHAT THIS IS NOT. It is not airtight fencing.\n *\n * - A predecessor paused (GC, VM suspend) for longer than the quiescence\n * window, between its last fence check and its append landing at the backend,\n * can still write one batch. Closing that requires a compare-and-set on the\n * durability write; `StreamDurability.append` has no such parameter and this\n * phase deliberately does not add one.\n * - Layer 3 is only meaningful across PROCESSES. On a single-process\n * `InMemoryLockStore` the two claims are serialized by the lock, not\n * concurrent, so `awaitLogQuiescence` can never observe a predecessor still\n * writing there — and consequently no unit test on that backend proves layer\n * 3 does anything. What the tests do prove on that backend is layer 2.\n *\n * The mitigation for both is deployment-level: use a lease-backed distributed\n * `LockStore`, and keep `fenceQuietMs` above the lease's renewal interval.\n */\nimport { isTerminalRunStatus } from '@tanstack/ai'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'\n\n/** Quiescence window before a successor's first append. */\nexport const DEFAULT_FENCE_QUIET_MS = 5_000\n\n/**\n * Appends a fenced log makes between `driverEpoch` re-reads.\n *\n * Deliberately a COUNT, not an interval: `pipeToRunLog` appends one chunk per\n * call, so this bounds a superseded driver to at most 31 further chunk batches\n * (the bump can land immediately after a check) regardless of how fast the run\n * streams. At 500 chunks/sec that worst case is ~62ms of writes; at 5\n * chunks/sec it is ~6s of writes — either way 31 chunks, never ~1000.\n *\n * The cost of a smaller number is one extra `RunStore.get` per 32 chunks.\n */\nexport const DEFAULT_EPOCH_RECHECK_APPENDS = 32\n\n/** Probes {@link awaitLogQuiescence} makes before giving up. */\nconst MAX_QUIESCENCE_PROBES = 6\n\n/** Lock key for a run's driver. Per-run, so two runs never serialize. */\nexport function runDriverLockKey(runId: string): string {\n return `run-driver:${runId}`\n}\n\n/** The claim was never acquired, so the caller must not drive the run. */\nexport class RunClaimNotAcquiredError extends Error {\n constructor(\n readonly runId: string,\n readonly reason: 'terminal' | 'unknown' | 'superseded',\n ) {\n super(`run ${runId}: driver claim not acquired (${reason})`)\n this.name = 'RunClaimNotAcquiredError'\n }\n}\n\n/** The claim was held and has been superseded; stop writing immediately. */\nexport class RunClaimLostError extends Error {\n constructor(\n readonly runId: string,\n readonly heldEpoch: number,\n readonly observedEpoch: number | 'lease-lost',\n ) {\n super(\n `run ${runId}: driver claim lost (held epoch ${heldEpoch}, observed ${observedEpoch})`,\n )\n this.name = 'RunClaimLostError'\n }\n}\n\n/** A held claim on one run. */\nexport interface RunClaim {\n runId: string\n /** This driver's fencing token; strictly greater than any predecessor's. */\n epoch: number\n /** Aborts when the lock can no longer guarantee ownership. */\n signal: AbortSignal\n}\n\nexport interface WithRunClaimOptions {\n runs: RunStore\n locks: LockStore\n runId: string\n /**\n * Quiescence window for {@link awaitLogQuiescence}. Defaults to\n * {@link DEFAULT_FENCE_QUIET_MS}.\n *\n * `withRunClaim` itself does not read this: it has no durability handle. It\n * lives here so a caller assembling a drive passes ONE options object to\n * `withRunClaim`, `awaitLogQuiescence`, and {@link fenceDurability} instead of\n * three that can drift apart.\n */\n fenceQuietMs?: number\n /**\n * Forwarded to {@link fenceDurability}. Defaults to\n * {@link DEFAULT_EPOCH_RECHECK_APPENDS}. Same rationale as `fenceQuietMs`.\n */\n epochRecheckAppends?: number\n logger?: InternalLogger\n}\n\n/**\n * Claim exclusive driver rights on `runId` for the duration of `fn`.\n *\n * The ENTIRE body runs inside the lock, so a snapshot taken by `fn` and every\n * append that follows it sit in one critical section.\n *\n * Rejects with {@link RunClaimNotAcquiredError} when the run is unknown or\n * already terminal — a terminal run has nothing left to drive, and bumping its\n * epoch would fence out nobody while confusing an operator reading the record.\n *\n * The epoch is bumped INSIDE the lock and only after those checks pass, so a\n * refused claim leaves `driverEpoch` untouched.\n */\nexport async function withRunClaim<T>(\n options: WithRunClaimOptions,\n fn: (claim: RunClaim) => Promise<T>,\n): Promise<T> {\n const { runs, locks, runId, logger } = options\n return locks.withLock(runDriverLockKey(runId), async (signal) => {\n const record = await runs.get(runId)\n if (record === null) {\n throw new RunClaimNotAcquiredError(runId, 'unknown')\n }\n if (isTerminalRunStatus(record.status)) {\n throw new RunClaimNotAcquiredError(runId, 'terminal')\n }\n const epoch = (record.driverEpoch ?? 0) + 1\n await runs.update(runId, { driverEpoch: epoch })\n logger?.sandbox(`run ${runId}: driver claim acquired at epoch ${epoch}`, {\n runId,\n epoch,\n })\n return fn({ runId, epoch, signal })\n })\n}\n\n/**\n * Wait until the stored log stops growing, then answer how many entries it\n * holds.\n *\n * Uses `snapshot()`, never `read()`: `read` tails and only resolves once the log\n * is terminalized or the caller aborts, and a taken-over run's log is open by\n * definition — the host that would have closed it is the host that died.\n *\n * Rejects rather than looping forever. A log that never quiesces means a\n * predecessor is still actively writing, which is a condition to surface, not to\n * append into.\n *\n * This only detects a CONCURRENT predecessor, which means it can only fire when\n * the two drivers are in different processes. Within one process an\n * `InMemoryLockStore` serializes the claims, so the predecessor has already\n * stopped by the time the successor probes.\n */\nexport async function awaitLogQuiescence<TOffset extends string = string>(\n durability: StreamDurability<TOffset>,\n quietMs: number,\n): Promise<number> {\n let previous = (await durability.snapshot()).length\n for (let probe = 0; probe < MAX_QUIESCENCE_PROBES; probe += 1) {\n await sleep(quietMs)\n const current = (await durability.snapshot()).length\n if (current === previous) return current\n previous = current\n }\n throw new Error(\n `journal takeover: the event log never quiesced after ${MAX_QUIESCENCE_PROBES} probes (${previous} entries and still growing); another host is still driving this run`,\n )\n}\n\nfunction sleep(ms: number): Promise<void> {\n if (ms <= 0) return Promise.resolve()\n return new Promise<void>((resolve) => setTimeout(resolve, ms))\n}\n\n/**\n * The one-way \"this claim is gone\" flag, latched by the first refusal.\n *\n * Keyed by the claim rather than held in one wrapper's closure because a claim\n * has TWO fenced seams — its log ({@link fenceDurability}) and its record\n * ({@link fenceRunStore}) — and a latch per wrapper would let them disagree: a\n * lease that flaps back to `aborted === false`, or an epoch read that fails,\n * would re-open the fence that had not refused yet. Losing a claim is not\n * transient, so one observation must close both.\n *\n * A `WeakMap` and not a field on {@link RunClaim} so the claim stays the plain\n * data structure core's `RunDriverOptions.claim` types it as, and so the latch is\n * collected with the claim.\n */\ninterface ClaimLatch {\n /** `undefined` while the fence is open; otherwise the refusal to replay. */\n lost: RunClaimLostError | undefined\n}\n\nconst CLAIM_LATCHES = new WeakMap<RunClaim, ClaimLatch>()\n\nfunction latchFor(claim: RunClaim): ClaimLatch {\n const existing = CLAIM_LATCHES.get(claim)\n if (existing !== undefined) return existing\n const latch: ClaimLatch = { lost: undefined }\n CLAIM_LATCHES.set(claim, latch)\n return latch\n}\n\n/**\n * The I/O-free half of the check: the latch and the lease. Synchronous on\n * purpose — a fenced write must be refused BEFORE anything can half-land.\n */\nfunction claimLostSynchronously(\n claim: RunClaim,\n latch: ClaimLatch,\n): RunClaimLostError | undefined {\n if (latch.lost !== undefined) return latch.lost\n if (claim.signal.aborted) {\n latch.lost = new RunClaimLostError(claim.runId, claim.epoch, 'lease-lost')\n return latch.lost\n }\n return undefined\n}\n\n/**\n * The other half: re-read `driverEpoch` and refuse once a successor exists.\n *\n * A store failure is NOT treated as loss. The lease is the primary fence and it\n * has not fired, so fencing ourselves out on a store blip would kill a healthy\n * driver — and, for the record fence, would suppress a legitimate terminal write\n * and strand the run at `'running'`, which is worse than the write it prevents.\n */\nasync function claimLostByEpoch(\n claim: RunClaim,\n latch: ClaimLatch,\n runs: RunStore,\n): Promise<RunClaimLostError | undefined> {\n let observed: number | undefined\n try {\n observed = (await runs.get(claim.runId))?.driverEpoch\n } catch {\n return undefined\n }\n if (observed !== undefined && observed > claim.epoch) {\n latch.lost = new RunClaimLostError(claim.runId, claim.epoch, observed)\n return latch.lost\n }\n return undefined\n}\n\n/**\n * Wrap a log so every `append` is fenced by `claim`.\n *\n * `append` is the ONLY fenced method, deliberately:\n *\n * - `close()` must never be fenced. It runs on every teardown path including\n * the teardown caused by losing the claim, and a fenced `close` would leave\n * the record wedged at `'running'` with every live tailer parked forever (a\n * `read` only ends when the log closes).\n * - `read` / `snapshot` / `resumeFrom` do not mutate, so a superseded host\n * reading them is harmless.\n *\n * The lease check is synchronous and happens before any I/O, so a fenced append\n * cannot half-land. The epoch re-check is throttled to `epochRecheckAppends`\n * because it costs a store read and the append path is hot.\n *\n * ONE REFUSAL CLOSES THE FENCE FOR GOOD. The first `append` that is refused —\n * for EITHER cause, lost lease or moved epoch — latches this wrapper shut, and\n * every later `append` refuses immediately without consulting the throttle and\n * without a store read. This is not a nicety:\n *\n * - Losing a claim is not transient. Epochs only move forward and a lease is\n * never handed back, so a wrapper that has refused once can never legitimately\n * append again. Re-deciding per append can only produce a WRONG answer.\n * - The throttle makes that wrong answer reachable. A refusal consumes the\n * re-read budget, so the very next append rides a fresh throttle window and is\n * NOT re-checked. `pipeToRunLog`'s recovery path appends a `RUN_ERROR` right\n * after the refusal it is recovering from, and that log belongs to the\n * SUCCESSOR: a terminal `RUN_ERROR` from a dead host would fail the stream for\n * every client attached to the live, healthy run.\n * - It is also strictly cheaper: a latched boolean replaces a store read.\n *\n * The latch deliberately does NOT extend to `close()` — see above.\n *\n * PASSES THE OFFSET TYPE THROUGH, rather than collapsing it to `string`. The\n * fence sits mid-chain between a caller's log and `pipeToRunLog`, so widening\n * here would reintroduce the branded-offset wall one layer in: a\n * `StreamDurability<DurableStreamOffset>` would go in and a\n * `StreamDurability<string>` would come out, which is not assignable back to\n * the caller's own type.\n */\nexport function fenceDurability<TOffset extends string = string>(\n durability: StreamDurability<TOffset>,\n claim: RunClaim,\n options: { runs: RunStore; epochRecheckAppends?: number },\n): StreamDurability<TOffset> {\n const recheckAppends = Math.max(\n 1,\n Math.trunc(options.epochRecheckAppends ?? DEFAULT_EPOCH_RECHECK_APPENDS),\n )\n // Seeded at the threshold so the FIRST append always re-reads the epoch: a\n // successor may have claimed between this fence being built and its first\n // write.\n let appendsSinceEpochRead = recheckAppends\n // Latched by the FIRST refusal and never cleared, and SHARED with this claim's\n // record fence so the two seams cannot disagree.\n const latch = latchFor(claim)\n\n async function assertHeld(): Promise<void> {\n // Layer 1 plus the latch: no I/O, so nothing has been written yet, and once\n // refused no throttle and no store read can let a later append through.\n const synchronous = claimLostSynchronously(claim, latch)\n if (synchronous !== undefined) throw synchronous\n if (appendsSinceEpochRead < recheckAppends) {\n appendsSinceEpochRead += 1\n return\n }\n appendsSinceEpochRead = 1\n // Layer 2, throttled because it costs a store read.\n const byEpoch = await claimLostByEpoch(claim, latch, options.runs)\n if (byEpoch !== undefined) throw byEpoch\n }\n\n return {\n resumeFrom: () => durability.resumeFrom(),\n append: async (chunks: Array<StreamChunk>) => {\n await assertHeld()\n return durability.append(chunks)\n },\n read: (offset, signal) => durability.read(offset, signal),\n close: () => durability.close(),\n snapshot: () => durability.snapshot(),\n }\n}\n\n/**\n * Wrap a run store so a TERMINAL record write is fenced by `claim`.\n *\n * The record is the run's other authoritative channel, and the same rule applies\n * to it: a host that has lost its claim must not state that the run is over. It\n * reaches this seam by the most ordinary route — `pipeToRunLog` catches the\n * `RunClaimLostError` its refused append threw, folds it in, and calls\n * `finish(ctx, 'failed', …)` — so fencing the log alone only moves where the harm\n * surfaces. `'completed'` and `'aborted'` arrive the same way (an empty stream\n * that never appended; a lease loss that aborts `claim.signal`, which\n * `pipeToRunLog` reads as an abort before it appends anything), which is why the\n * gate is {@link isTerminalRunStatus} and not \"did an append refuse\".\n *\n * SUPPRESSED, NOT ATTEMPTED-AND-SWALLOWED, and not thrown either. `update`\n * resolves without writing. `pipeToRunLog` must not reject — `RunController.start`\n * consumes its promise fire-and-forget — and a rejection here would additionally\n * make `finish` report the run through the local rebuilt record as if the store\n * had broken, which is a different and false fact.\n *\n * WHAT IS *NOT* FENCED, deliberately:\n *\n * - **`close()`** is not on this seam at all, and must stay off it: see\n * {@link fenceDurability}. A wedged `'running'` record with tailers parked\n * forever is worse than the write being prevented.\n * - **Non-terminal writes pass through**, including `detachedSince` and\n * `sandboxKey` written by a superseded host. They are stale, but staleness is\n * not the harm being fixed: none of them can make a live run look finished, so\n * none can mislead `isTerminalRunStatus`, `findActiveRun`, or the reaper. They\n * are also self-healing — the successor owns those fields and overwrites them —\n * whereas over-suppressing strands a record: `createOrResume` is how the row\n * comes into existence at all, and refusing a non-terminal write on a\n * mis-observed loss would leave a run with no record to recover from. Suppress\n * the writes that assert an outcome; let bookkeeping through.\n * - **Reads** (`get`, `listByThread`, `listReclaimable`, `findActiveRun`) do not\n * mutate, so a superseded host reading them is harmless. `finish`'s terminal\n * re-read therefore still works and answers with the SUCCESSOR's live record,\n * which is the truthful thing to resolve with.\n * - **Another run's record.** The fence knows about `claim.runId` only; a write\n * aimed elsewhere is not this claim's to judge.\n *\n * The OPTIONAL methods (`listByThread`, `listReclaimable`) are forwarded only\n * when the wrapped store actually has them: consumers feature-detect\n * (`store.listReclaimable?.(…)`), so materializing one that delegates to a\n * missing method would turn a graceful degrade into a `TypeError`.\n * `findActiveRun` is required on the contract, so it forwards unconditionally.\n */\nexport function fenceRunStore(\n runs: RunStore,\n claim: RunClaim,\n options: { logger?: InternalLogger } = {},\n): RunStore {\n const latch = latchFor(claim)\n // Bound, not merely captured: the store may be a class instance\n // (`InMemoryRunStore`), whose methods need their receiver.\n const listByThread = runs.listByThread?.bind(runs)\n const listReclaimable = runs.listReclaimable?.bind(runs)\n\n return {\n createOrResume: (input) => runs.createOrResume(input),\n get: (runId) => runs.get(runId),\n findActiveRun: (threadId) => runs.findActiveRun(threadId),\n update: async (runId, patch) => {\n const status = patch.status\n if (\n runId !== claim.runId ||\n status === undefined ||\n !isTerminalRunStatus(status)\n ) {\n return runs.update(runId, patch)\n }\n const lost =\n claimLostSynchronously(claim, latch) ??\n // Unthrottled, unlike the append path: a terminal write happens once per\n // run, so one store read is not a hot cost — and it is the read that\n // catches a superseded driver whose stream ended without ever appending.\n (await claimLostByEpoch(claim, latch, runs))\n if (lost === undefined) return runs.update(runId, patch)\n // Absorbing this silently would make it invisible: a detached run has no\n // caller to report to. The logger is consumer-supplied, so a throwing sink\n // must not turn a suppression into a rejection.\n try {\n options.logger?.sandbox(\n `run ${runId}: suppressed a terminal '${status}' record write from a superseded driver`,\n { runId, status, heldEpoch: claim.epoch, error: lost },\n )\n } catch {\n // Intentionally empty: there is no second channel to report on.\n }\n return undefined\n },\n ...(listByThread === undefined ? {} : { listByThread }),\n ...(listReclaimable === undefined ? {} : { listReclaimable }),\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4EA,IAAa,yBAAyB;;AAgBtC,IAAM,wBAAwB;;AAG9B,SAAgB,iBAAiB,OAAuB;CACtD,OAAO,cAAc;AACvB;;AAGA,IAAa,2BAAb,cAA8C,MAAM;CAEvC;CACA;CAFX,YACE,OACA,QACA;EACA,MAAM,OAAO,MAAM,+BAA+B,OAAO,EAAE;EAHlD,KAAA,QAAA;EACA,KAAA,SAAA;EAGT,KAAK,OAAO;CACd;AACF;;AAGA,IAAa,oBAAb,cAAuC,MAAM;CAEhC;CACA;CACA;CAHX,YACE,OACA,WACA,eACA;EACA,MACE,OAAO,MAAM,kCAAkC,UAAU,aAAa,cAAc,EACtF;EANS,KAAA,QAAA;EACA,KAAA,YAAA;EACA,KAAA,gBAAA;EAKT,KAAK,OAAO;CACd;AACF;;;;;;;;;;;;;;AA8CA,eAAsB,aACpB,SACA,IACY;CACZ,MAAM,EAAE,MAAM,OAAO,OAAO,WAAW;CACvC,OAAO,MAAM,SAAS,iBAAiB,KAAK,GAAG,OAAO,WAAW;EAC/D,MAAM,SAAS,MAAM,KAAK,IAAI,KAAK;EACnC,IAAI,WAAW,MACb,MAAM,IAAI,yBAAyB,OAAO,SAAS;EAErD,IAAI,oBAAoB,OAAO,MAAM,GACnC,MAAM,IAAI,yBAAyB,OAAO,UAAU;EAEtD,MAAM,SAAS,OAAO,eAAe,KAAK;EAC1C,MAAM,KAAK,OAAO,OAAO,EAAE,aAAa,MAAM,CAAC;EAC/C,QAAQ,QAAQ,OAAO,MAAM,mCAAmC,SAAS;GACvE;GACA;EACF,CAAC;EACD,OAAO,GAAG;GAAE;GAAO;GAAO;EAAO,CAAC;CACpC,CAAC;AACH;;;;;;;;;;;;;;;;;;AAmBA,eAAsB,mBACpB,YACA,SACiB;CACjB,IAAI,YAAY,MAAM,WAAW,SAAS,EAAA,CAAG;CAC7C,KAAK,IAAI,QAAQ,GAAG,QAAQ,uBAAuB,SAAS,GAAG;EAC7D,MAAM,MAAM,OAAO;EACnB,MAAM,WAAW,MAAM,WAAW,SAAS,EAAA,CAAG;EAC9C,IAAI,YAAY,UAAU,OAAO;EACjC,WAAW;CACb;CACA,MAAM,IAAI,MACR,wDAAwD,sBAAsB,WAAW,SAAS,oEACpG;AACF;AAEA,SAAS,MAAM,IAA2B;CACxC,IAAI,MAAM,GAAG,OAAO,QAAQ,QAAQ;CACpC,OAAO,IAAI,SAAe,YAAY,WAAW,SAAS,EAAE,CAAC;AAC/D;AAqBA,IAAM,gCAAgB,IAAI,QAA8B;AAExD,SAAS,SAAS,OAA6B;CAC7C,MAAM,WAAW,cAAc,IAAI,KAAK;CACxC,IAAI,aAAa,KAAA,GAAW,OAAO;CACnC,MAAM,QAAoB,EAAE,MAAM,KAAA,EAAU;CAC5C,cAAc,IAAI,OAAO,KAAK;CAC9B,OAAO;AACT;;;;;AAMA,SAAS,uBACP,OACA,OAC+B;CAC/B,IAAI,MAAM,SAAS,KAAA,GAAW,OAAO,MAAM;CAC3C,IAAI,MAAM,OAAO,SAAS;EACxB,MAAM,OAAO,IAAI,kBAAkB,MAAM,OAAO,MAAM,OAAO,YAAY;EACzE,OAAO,MAAM;CACf;AAEF;;;;;;;;;AAUA,eAAe,iBACb,OACA,OACA,MACwC;CACxC,IAAI;CACJ,IAAI;EACF,YAAY,MAAM,KAAK,IAAI,MAAM,KAAK,EAAA,EAAI;CAC5C,QAAQ;EACN;CACF;CACA,IAAI,aAAa,KAAA,KAAa,WAAW,MAAM,OAAO;EACpD,MAAM,OAAO,IAAI,kBAAkB,MAAM,OAAO,MAAM,OAAO,QAAQ;EACrE,OAAO,MAAM;CACf;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2CA,SAAgB,gBACd,YACA,OACA,SAC2B;CAC3B,MAAM,iBAAiB,KAAK,IAC1B,GACA,KAAK,MAAM,QAAQ,uBAAA,EAAoD,CACzE;CAIA,IAAI,wBAAwB;CAG5B,MAAM,QAAQ,SAAS,KAAK;CAE5B,eAAe,aAA4B;EAGzC,MAAM,cAAc,uBAAuB,OAAO,KAAK;EACvD,IAAI,gBAAgB,KAAA,GAAW,MAAM;EACrC,IAAI,wBAAwB,gBAAgB;GAC1C,yBAAyB;GACzB;EACF;EACA,wBAAwB;EAExB,MAAM,UAAU,MAAM,iBAAiB,OAAO,OAAO,QAAQ,IAAI;EACjE,IAAI,YAAY,KAAA,GAAW,MAAM;CACnC;CAEA,OAAO;EACL,kBAAkB,WAAW,WAAW;EACxC,QAAQ,OAAO,WAA+B;GAC5C,MAAM,WAAW;GACjB,OAAO,WAAW,OAAO,MAAM;EACjC;EACA,OAAO,QAAQ,WAAW,WAAW,KAAK,QAAQ,MAAM;EACxD,aAAa,WAAW,MAAM;EAC9B,gBAAgB,WAAW,SAAS;CACtC;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAgDA,SAAgB,cACd,MACA,OACA,UAAuC,CAAC,GAC9B;CACV,MAAM,QAAQ,SAAS,KAAK;CAG5B,MAAM,eAAe,KAAK,cAAc,KAAK,IAAI;CACjD,MAAM,kBAAkB,KAAK,iBAAiB,KAAK,IAAI;CAEvD,OAAO;EACL,iBAAiB,UAAU,KAAK,eAAe,KAAK;EACpD,MAAM,UAAU,KAAK,IAAI,KAAK;EAC9B,gBAAgB,aAAa,KAAK,cAAc,QAAQ;EACxD,QAAQ,OAAO,OAAO,UAAU;GAC9B,MAAM,SAAS,MAAM;GACrB,IACE,UAAU,MAAM,SAChB,WAAW,KAAA,KACX,CAAC,oBAAoB,MAAM,GAE3B,OAAO,KAAK,OAAO,OAAO,KAAK;GAEjC,MAAM,OACJ,uBAAuB,OAAO,KAAK,KAIlC,MAAM,iBAAiB,OAAO,OAAO,IAAI;GAC5C,IAAI,SAAS,KAAA,GAAW,OAAO,KAAK,OAAO,OAAO,KAAK;GAIvD,IAAI;IACF,QAAQ,QAAQ,QACd,OAAO,MAAM,2BAA2B,OAAO,0CAC/C;KAAE;KAAO;KAAQ,WAAW,MAAM;KAAO,OAAO;IAAK,CACvD;GACF,QAAQ,CAER;EAEF;EACA,GAAI,iBAAiB,KAAA,IAAY,CAAC,IAAI,EAAE,aAAa;EACrD,GAAI,oBAAoB,KAAA,IAAY,CAAC,IAAI,EAAE,gBAAgB;CAC7D;AACF"}
@@ -20,6 +20,19 @@ export interface SandboxCapabilities {
20
20
  * file + shell redirection.
21
21
  */
22
22
  writableStdin: boolean;
23
+ /**
24
+ * A spawned process can be forcibly terminated via {@link SpawnHandle.kill}
25
+ * and aborted mid-flight via the {@link ProcessOptions.signal} passed to
26
+ * {@link SandboxProcess.spawn}. `true` for host/Docker; some edge providers
27
+ * (e.g. Cloudflare) implement `kill()` as a no-op and drop the abort signal
28
+ * entirely, so a long-running follower process (e.g. `tail -f`) started
29
+ * there can never be stopped by the caller — only polled and abandoned.
30
+ * Callers MUST branch on this before relying on `kill`/abort to reclaim a
31
+ * background process: a bring-your-own provider that omits it would
32
+ * otherwise be silently treated as killable, leaking an unstoppable process
33
+ * inside the sandbox.
34
+ */
35
+ killableProcesses: boolean;
23
36
  /** Capture/restore filesystem snapshots via {@link SandboxHandle.snapshot}. */
24
37
  snapshots: boolean;
25
38
  /** Declarative network egress allow/deny policy. */