@tanstack/ai-sandbox 0.2.4 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
@@ -0,0 +1,154 @@
1
+ /**
2
+ * Deterministic chunk identity — the prerequisite that makes journal replay
3
+ * exact.
4
+ *
5
+ * The design premise is that re-translating a journal prefix reproduces the
6
+ * chunks a previous host already delivered, so a successor can recognize and
7
+ * skip them (see `align.ts`, a later task). Two things in the default path
8
+ * break that premise:
9
+ *
10
+ * 1. `ChatAdapter.generateId()` — `packages/ai/src/activities/chat/adapter.ts:227`,
11
+ * read directly for this task — is:
12
+ *
13
+ * ```ts
14
+ * protected generateId(): string {
15
+ * return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
16
+ * }
17
+ * ```
18
+ *
19
+ * and every harness translator mints message ids through it (wired as
20
+ * `genId` in the Grok Build, Claude Code, and Codex text adapters). Both
21
+ * `Date.now()` and `Math.random()` are non-reproducible: replaying the same
22
+ * journal bytes through a second `generateId()` call produces a different
23
+ * id every time, so "same bytes ⇒ same chunks" is false on the journaled
24
+ * path today. {@link createRunScopedIdGen} replaces it with a run-scoped
25
+ * counter that has neither a clock nor randomness, so two generators built
26
+ * from the same `runId` always produce the same sequence.
27
+ * 2. Chunks also carry `timestamp: Date.now()`, which cannot be reproduced at
28
+ * all, deterministic id or not. {@link chunkFingerprint} therefore excludes
29
+ * exactly that field — nothing downstream keys on a chunk's timestamp, so
30
+ * leaving it wall-clock is safe, but every other field must participate in
31
+ * the comparison or a real divergence would go undetected.
32
+ */
33
+ import type { StreamChunk } from '@tanstack/ai'
34
+
35
+ /**
36
+ * A deterministic id generator scoped to one run.
37
+ *
38
+ * Passed as the harness translators' `genId`, so translating the same journal
39
+ * prefix twice mints the same message ids. The counter is per-generator, so a
40
+ * replay must create a fresh one and start from the journal's first byte —
41
+ * which is exactly what the alignment step (a later task) assumes.
42
+ *
43
+ * No clock, no `Math.random`, no crypto: `next` is the only state, and it is
44
+ * seeded fresh for every call to this factory.
45
+ */
46
+ export function createRunScopedIdGen(runId: string): () => string {
47
+ let next = 0
48
+ return () => {
49
+ const id = `${runId}-${next}`
50
+ next += 1
51
+ return id
52
+ }
53
+ }
54
+
55
+ /**
56
+ * Fields excluded from a fingerprint because they are wall-clock and therefore
57
+ * unreproducible. Kept as an explicit set so adding one is a deliberate,
58
+ * reviewable act rather than a silent loosening of the comparison.
59
+ */
60
+ const VOLATILE_FIELDS: ReadonlySet<string> = new Set(['timestamp'])
61
+
62
+ /**
63
+ * The conversation id every chunk carries. Excluded from
64
+ * {@link chunkFingerprintIgnoringThreadId} — and ONLY from that variant — so
65
+ * alignment can tell an id-only mismatch from a real content divergence.
66
+ */
67
+ const THREAD_ID_FIELD = 'threadId'
68
+
69
+ const VOLATILE_AND_THREAD_ID: ReadonlySet<string> = new Set([
70
+ ...VOLATILE_FIELDS,
71
+ THREAD_ID_FIELD,
72
+ ])
73
+
74
+ /**
75
+ * `dropped` applies at the TOP LEVEL only (nested calls pass `undefined`).
76
+ * A `threadId` nested inside, say, a tool call's arguments is real content and
77
+ * must keep participating in the comparison.
78
+ */
79
+ function stableStringify(
80
+ value: unknown,
81
+ dropped: ReadonlySet<string> | undefined,
82
+ ): string {
83
+ if (value === null) return 'null'
84
+ if (Array.isArray(value)) {
85
+ return `[${value.map((item) => stableStringify(item, undefined)).join(',')}]`
86
+ }
87
+ if (typeof value === 'object') {
88
+ const record: Record<string, unknown> = value as Record<string, unknown>
89
+ const keys = Object.keys(record)
90
+ .filter((key) => dropped === undefined || !dropped.has(key))
91
+ .sort()
92
+ const parts = keys.map((key) => {
93
+ const entry = record[key]
94
+ const encoded =
95
+ entry === undefined
96
+ ? '"__undefined__"'
97
+ : stableStringify(entry, undefined)
98
+ return `${JSON.stringify(key)}:${encoded}`
99
+ })
100
+ return `{${parts.join(',')}}`
101
+ }
102
+ const encoded = JSON.stringify(value)
103
+ return encoded === undefined ? 'null' : encoded
104
+ }
105
+
106
+ /**
107
+ * A stable, order-independent identity for a chunk, excluding wall-clock
108
+ * fields. Used to recognize the chunks a previous host already appended.
109
+ *
110
+ * - **Key-order independent**: object keys are sorted before stringifying, so
111
+ * a JSON round trip through the journal (which does not preserve key order)
112
+ * cannot spuriously diverge.
113
+ * - **Recurses into nested arrays and objects**: tool-call arguments are
114
+ * nested, and a shallow fingerprint would miss a changed argument.
115
+ * - **Excludes exactly `VOLATILE_FIELDS`** (`timestamp`) — everything else
116
+ * participates, including fields whose value is `undefined`.
117
+ * - **Distinguishes present-but-`undefined` from absent**: `undefined` is
118
+ * encoded as the sentinel string `"__undefined__"` rather than dropped, so
119
+ * `{a: undefined}` and `{}` do not collide. A translator emitting an
120
+ * explicit `undefined` is a different chunk shape and must fingerprint
121
+ * differently.
122
+ */
123
+ export function chunkFingerprint(chunk: StreamChunk): string {
124
+ return stableStringify(chunk, VOLATILE_FIELDS)
125
+ }
126
+
127
+ /**
128
+ * {@link chunkFingerprint} with the chunk's own `threadId` also excluded.
129
+ *
130
+ * NOT an alternative identity — never use it to decide that two chunks are the
131
+ * same. Its single purpose is DIAGNOSIS: when a replay diverges from the stored
132
+ * log, comparing both fingerprints answers "did the agent behave differently, or
133
+ * did only the conversation id move?". Two chunks that match here but not under
134
+ * {@link chunkFingerprint} differ in `threadId` and nothing else, which is a
135
+ * misconfigured attach route rather than a determinism regression (see
136
+ * `JournalReplayThreadIdMismatchError` in `align.ts`).
137
+ */
138
+ export function chunkFingerprintIgnoringThreadId(chunk: StreamChunk): string {
139
+ return stableStringify(chunk, VOLATILE_AND_THREAD_ID)
140
+ }
141
+
142
+ /**
143
+ * A chunk's own `threadId`, or `undefined` when it carries none.
144
+ *
145
+ * Reads the field structurally rather than narrowing on `chunk.type`: nearly
146
+ * every member of the `StreamChunk` union declares `threadId?: string`, and an
147
+ * exhaustive switch would have to be revisited for each new member while adding
148
+ * nothing — a chunk with no `threadId` is exactly the `undefined` case.
149
+ */
150
+ export function chunkThreadId(chunk: StreamChunk): string | undefined {
151
+ const record: Record<string, unknown> = chunk as Record<string, unknown>
152
+ const value = record[THREAD_ID_FIELD]
153
+ return typeof value === 'string' ? value : undefined
154
+ }
package/src/claim.ts ADDED
@@ -0,0 +1,479 @@
1
+ /**
2
+ * The single-writer claim: what makes a takeover safe to attempt at all.
3
+ *
4
+ * WHY THIS MODULE EXISTS. `alignToStoredLog` decides where its appends start by
5
+ * reading `durability.snapshot()`, and `snapshot()` carries NO LOCK — core says
6
+ * so explicitly (`packages/ai/src/stream-durability.ts`: "a concurrent `append`
7
+ * may land immediately after the snapshot is taken"). If two hosts drive one
8
+ * run, both snapshot, both compute a "remainder", and both append it. The log
9
+ * then holds the same logical chunk twice under two different offsets, and the
10
+ * client CANNOT survive that: `ai-client`'s de-dup is keyed on the adapter's
11
+ * offset string, so a re-appended chunk looks new, and the stream processor
12
+ * applies text and tool-argument deltas unconditionally. The visible result is
13
+ * doubled message text and `{"a":1}{"a":1}` tool arguments.
14
+ *
15
+ * Takeover is by definition two hosts wanting one run, so nothing may read a
16
+ * journal for a run it has not claimed.
17
+ *
18
+ * THREE LAYERS, strongest first:
19
+ *
20
+ * 1. **The lease.** {@link withRunClaim} runs the whole drive inside
21
+ * `LockStore.withLock('run-driver:<runId>', …)`, so the snapshot and every
22
+ * append that follows are one critical section. A lease-backed lock aborts
23
+ * the callback signal the moment ownership is lost, and
24
+ * {@link fenceDurability} turns that into a thrown {@link RunClaimLostError}
25
+ * BEFORE the append reaches the log.
26
+ * 2. **The epoch.** Each successful claim bumps `RunRecord.driverEpoch`.
27
+ * {@link fenceDurability} re-reads it every
28
+ * {@link DEFAULT_EPOCH_RECHECK_APPENDS} appends and refuses to append once a
29
+ * higher epoch exists. This covers what a lease cannot: an
30
+ * `InMemoryLockStore`, whose signal is a fresh `AbortController().signal`
31
+ * that is never aborted, and any backend whose renewal is coarser than the
32
+ * run's append rate. Once EITHER fence has refused an append, the fence
33
+ * latches shut and every later append refuses without re-reading anything.
34
+ * 3. **Quiescence.** {@link awaitLogQuiescence} requires the stored log to stop
35
+ * growing before the successor appends anything, so a predecessor that is
36
+ * still writing is OBSERVED rather than raced.
37
+ *
38
+ * THE LOG IS NOT THE ONLY AUTHORITATIVE CHANNEL. A host that has lost its claim
39
+ * must not write authoritative facts about the run through ANY seam, and there
40
+ * are two: the event log and the run RECORD. Fencing only the log moves the harm
41
+ * rather than removing it — a superseded driver whose append was refused folds
42
+ * that refusal into a terminal `runs.update`, so the record reads `'failed'` for
43
+ * a run the successor is healthily streaming, and `isTerminalRunStatus` (which
44
+ * `findActiveRun`, the resume driver, and `reapDetachedRuns` all branch on) then
45
+ * answers `true` for a live run. {@link fenceRunStore} closes that seam; both
46
+ * fences share one per-claim latch so they can never disagree about whether the
47
+ * claim is still held.
48
+ *
49
+ * WHY THE EPOCH RE-CHECK COUNTS APPENDS, NOT MILLISECONDS. `pipeToRunLog`
50
+ * appends ONE chunk per call, so a time-based interval couples the fence's
51
+ * resolution to the run's chunk rate: at 500 chunks/sec a 2s interval lets a
52
+ * superseded driver write ~1000 chunks before it notices. A count gives a hard
53
+ * bound independent of rate — see {@link DEFAULT_EPOCH_RECHECK_APPENDS}.
54
+ *
55
+ * WHAT THIS IS NOT. It is not airtight fencing.
56
+ *
57
+ * - A predecessor paused (GC, VM suspend) for longer than the quiescence
58
+ * window, between its last fence check and its append landing at the backend,
59
+ * can still write one batch. Closing that requires a compare-and-set on the
60
+ * durability write; `StreamDurability.append` has no such parameter and this
61
+ * phase deliberately does not add one.
62
+ * - Layer 3 is only meaningful across PROCESSES. On a single-process
63
+ * `InMemoryLockStore` the two claims are serialized by the lock, not
64
+ * concurrent, so `awaitLogQuiescence` can never observe a predecessor still
65
+ * writing there — and consequently no unit test on that backend proves layer
66
+ * 3 does anything. What the tests do prove on that backend is layer 2.
67
+ *
68
+ * The mitigation for both is deployment-level: use a lease-backed distributed
69
+ * `LockStore`, and keep `fenceQuietMs` above the lease's renewal interval.
70
+ */
71
+ import { isTerminalRunStatus } from '@tanstack/ai'
72
+ import type { LockStore } from '@tanstack/ai/locks'
73
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
74
+ import type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'
75
+
76
+ /** Quiescence window before a successor's first append. */
77
+ export const DEFAULT_FENCE_QUIET_MS = 5_000
78
+
79
+ /**
80
+ * Appends a fenced log makes between `driverEpoch` re-reads.
81
+ *
82
+ * Deliberately a COUNT, not an interval: `pipeToRunLog` appends one chunk per
83
+ * call, so this bounds a superseded driver to at most 31 further chunk batches
84
+ * (the bump can land immediately after a check) regardless of how fast the run
85
+ * streams. At 500 chunks/sec that worst case is ~62ms of writes; at 5
86
+ * chunks/sec it is ~6s of writes — either way 31 chunks, never ~1000.
87
+ *
88
+ * The cost of a smaller number is one extra `RunStore.get` per 32 chunks.
89
+ */
90
+ export const DEFAULT_EPOCH_RECHECK_APPENDS = 32
91
+
92
+ /** Probes {@link awaitLogQuiescence} makes before giving up. */
93
+ const MAX_QUIESCENCE_PROBES = 6
94
+
95
+ /** Lock key for a run's driver. Per-run, so two runs never serialize. */
96
+ export function runDriverLockKey(runId: string): string {
97
+ return `run-driver:${runId}`
98
+ }
99
+
100
+ /** The claim was never acquired, so the caller must not drive the run. */
101
+ export class RunClaimNotAcquiredError extends Error {
102
+ constructor(
103
+ readonly runId: string,
104
+ readonly reason: 'terminal' | 'unknown' | 'superseded',
105
+ ) {
106
+ super(`run ${runId}: driver claim not acquired (${reason})`)
107
+ this.name = 'RunClaimNotAcquiredError'
108
+ }
109
+ }
110
+
111
+ /** The claim was held and has been superseded; stop writing immediately. */
112
+ export class RunClaimLostError extends Error {
113
+ constructor(
114
+ readonly runId: string,
115
+ readonly heldEpoch: number,
116
+ readonly observedEpoch: number | 'lease-lost',
117
+ ) {
118
+ super(
119
+ `run ${runId}: driver claim lost (held epoch ${heldEpoch}, observed ${observedEpoch})`,
120
+ )
121
+ this.name = 'RunClaimLostError'
122
+ }
123
+ }
124
+
125
+ /** A held claim on one run. */
126
+ export interface RunClaim {
127
+ runId: string
128
+ /** This driver's fencing token; strictly greater than any predecessor's. */
129
+ epoch: number
130
+ /** Aborts when the lock can no longer guarantee ownership. */
131
+ signal: AbortSignal
132
+ }
133
+
134
+ export interface WithRunClaimOptions {
135
+ runs: RunStore
136
+ locks: LockStore
137
+ runId: string
138
+ /**
139
+ * Quiescence window for {@link awaitLogQuiescence}. Defaults to
140
+ * {@link DEFAULT_FENCE_QUIET_MS}.
141
+ *
142
+ * `withRunClaim` itself does not read this: it has no durability handle. It
143
+ * lives here so a caller assembling a drive passes ONE options object to
144
+ * `withRunClaim`, `awaitLogQuiescence`, and {@link fenceDurability} instead of
145
+ * three that can drift apart.
146
+ */
147
+ fenceQuietMs?: number
148
+ /**
149
+ * Forwarded to {@link fenceDurability}. Defaults to
150
+ * {@link DEFAULT_EPOCH_RECHECK_APPENDS}. Same rationale as `fenceQuietMs`.
151
+ */
152
+ epochRecheckAppends?: number
153
+ logger?: InternalLogger
154
+ }
155
+
156
+ /**
157
+ * Claim exclusive driver rights on `runId` for the duration of `fn`.
158
+ *
159
+ * The ENTIRE body runs inside the lock, so a snapshot taken by `fn` and every
160
+ * append that follows it sit in one critical section.
161
+ *
162
+ * Rejects with {@link RunClaimNotAcquiredError} when the run is unknown or
163
+ * already terminal — a terminal run has nothing left to drive, and bumping its
164
+ * epoch would fence out nobody while confusing an operator reading the record.
165
+ *
166
+ * The epoch is bumped INSIDE the lock and only after those checks pass, so a
167
+ * refused claim leaves `driverEpoch` untouched.
168
+ */
169
+ export async function withRunClaim<T>(
170
+ options: WithRunClaimOptions,
171
+ fn: (claim: RunClaim) => Promise<T>,
172
+ ): Promise<T> {
173
+ const { runs, locks, runId, logger } = options
174
+ return locks.withLock(runDriverLockKey(runId), async (signal) => {
175
+ const record = await runs.get(runId)
176
+ if (record === null) {
177
+ throw new RunClaimNotAcquiredError(runId, 'unknown')
178
+ }
179
+ if (isTerminalRunStatus(record.status)) {
180
+ throw new RunClaimNotAcquiredError(runId, 'terminal')
181
+ }
182
+ const epoch = (record.driverEpoch ?? 0) + 1
183
+ await runs.update(runId, { driverEpoch: epoch })
184
+ logger?.sandbox(`run ${runId}: driver claim acquired at epoch ${epoch}`, {
185
+ runId,
186
+ epoch,
187
+ })
188
+ return fn({ runId, epoch, signal })
189
+ })
190
+ }
191
+
192
+ /**
193
+ * Wait until the stored log stops growing, then answer how many entries it
194
+ * holds.
195
+ *
196
+ * Uses `snapshot()`, never `read()`: `read` tails and only resolves once the log
197
+ * is terminalized or the caller aborts, and a taken-over run's log is open by
198
+ * definition — the host that would have closed it is the host that died.
199
+ *
200
+ * Rejects rather than looping forever. A log that never quiesces means a
201
+ * predecessor is still actively writing, which is a condition to surface, not to
202
+ * append into.
203
+ *
204
+ * This only detects a CONCURRENT predecessor, which means it can only fire when
205
+ * the two drivers are in different processes. Within one process an
206
+ * `InMemoryLockStore` serializes the claims, so the predecessor has already
207
+ * stopped by the time the successor probes.
208
+ */
209
+ export async function awaitLogQuiescence<TOffset extends string = string>(
210
+ durability: StreamDurability<TOffset>,
211
+ quietMs: number,
212
+ ): Promise<number> {
213
+ let previous = (await durability.snapshot()).length
214
+ for (let probe = 0; probe < MAX_QUIESCENCE_PROBES; probe += 1) {
215
+ await sleep(quietMs)
216
+ const current = (await durability.snapshot()).length
217
+ if (current === previous) return current
218
+ previous = current
219
+ }
220
+ throw new Error(
221
+ `journal takeover: the event log never quiesced after ${MAX_QUIESCENCE_PROBES} probes (${previous} entries and still growing); another host is still driving this run`,
222
+ )
223
+ }
224
+
225
+ function sleep(ms: number): Promise<void> {
226
+ if (ms <= 0) return Promise.resolve()
227
+ return new Promise<void>((resolve) => setTimeout(resolve, ms))
228
+ }
229
+
230
+ /**
231
+ * The one-way "this claim is gone" flag, latched by the first refusal.
232
+ *
233
+ * Keyed by the claim rather than held in one wrapper's closure because a claim
234
+ * has TWO fenced seams — its log ({@link fenceDurability}) and its record
235
+ * ({@link fenceRunStore}) — and a latch per wrapper would let them disagree: a
236
+ * lease that flaps back to `aborted === false`, or an epoch read that fails,
237
+ * would re-open the fence that had not refused yet. Losing a claim is not
238
+ * transient, so one observation must close both.
239
+ *
240
+ * A `WeakMap` and not a field on {@link RunClaim} so the claim stays the plain
241
+ * data structure core's `RunDriverOptions.claim` types it as, and so the latch is
242
+ * collected with the claim.
243
+ */
244
+ interface ClaimLatch {
245
+ /** `undefined` while the fence is open; otherwise the refusal to replay. */
246
+ lost: RunClaimLostError | undefined
247
+ }
248
+
249
+ const CLAIM_LATCHES = new WeakMap<RunClaim, ClaimLatch>()
250
+
251
+ function latchFor(claim: RunClaim): ClaimLatch {
252
+ const existing = CLAIM_LATCHES.get(claim)
253
+ if (existing !== undefined) return existing
254
+ const latch: ClaimLatch = { lost: undefined }
255
+ CLAIM_LATCHES.set(claim, latch)
256
+ return latch
257
+ }
258
+
259
+ /**
260
+ * The I/O-free half of the check: the latch and the lease. Synchronous on
261
+ * purpose — a fenced write must be refused BEFORE anything can half-land.
262
+ */
263
+ function claimLostSynchronously(
264
+ claim: RunClaim,
265
+ latch: ClaimLatch,
266
+ ): RunClaimLostError | undefined {
267
+ if (latch.lost !== undefined) return latch.lost
268
+ if (claim.signal.aborted) {
269
+ latch.lost = new RunClaimLostError(claim.runId, claim.epoch, 'lease-lost')
270
+ return latch.lost
271
+ }
272
+ return undefined
273
+ }
274
+
275
+ /**
276
+ * The other half: re-read `driverEpoch` and refuse once a successor exists.
277
+ *
278
+ * A store failure is NOT treated as loss. The lease is the primary fence and it
279
+ * has not fired, so fencing ourselves out on a store blip would kill a healthy
280
+ * driver — and, for the record fence, would suppress a legitimate terminal write
281
+ * and strand the run at `'running'`, which is worse than the write it prevents.
282
+ */
283
+ async function claimLostByEpoch(
284
+ claim: RunClaim,
285
+ latch: ClaimLatch,
286
+ runs: RunStore,
287
+ ): Promise<RunClaimLostError | undefined> {
288
+ let observed: number | undefined
289
+ try {
290
+ observed = (await runs.get(claim.runId))?.driverEpoch
291
+ } catch {
292
+ return undefined
293
+ }
294
+ if (observed !== undefined && observed > claim.epoch) {
295
+ latch.lost = new RunClaimLostError(claim.runId, claim.epoch, observed)
296
+ return latch.lost
297
+ }
298
+ return undefined
299
+ }
300
+
301
+ /**
302
+ * Wrap a log so every `append` is fenced by `claim`.
303
+ *
304
+ * `append` is the ONLY fenced method, deliberately:
305
+ *
306
+ * - `close()` must never be fenced. It runs on every teardown path including
307
+ * the teardown caused by losing the claim, and a fenced `close` would leave
308
+ * the record wedged at `'running'` with every live tailer parked forever (a
309
+ * `read` only ends when the log closes).
310
+ * - `read` / `snapshot` / `resumeFrom` do not mutate, so a superseded host
311
+ * reading them is harmless.
312
+ *
313
+ * The lease check is synchronous and happens before any I/O, so a fenced append
314
+ * cannot half-land. The epoch re-check is throttled to `epochRecheckAppends`
315
+ * because it costs a store read and the append path is hot.
316
+ *
317
+ * ONE REFUSAL CLOSES THE FENCE FOR GOOD. The first `append` that is refused —
318
+ * for EITHER cause, lost lease or moved epoch — latches this wrapper shut, and
319
+ * every later `append` refuses immediately without consulting the throttle and
320
+ * without a store read. This is not a nicety:
321
+ *
322
+ * - Losing a claim is not transient. Epochs only move forward and a lease is
323
+ * never handed back, so a wrapper that has refused once can never legitimately
324
+ * append again. Re-deciding per append can only produce a WRONG answer.
325
+ * - The throttle makes that wrong answer reachable. A refusal consumes the
326
+ * re-read budget, so the very next append rides a fresh throttle window and is
327
+ * NOT re-checked. `pipeToRunLog`'s recovery path appends a `RUN_ERROR` right
328
+ * after the refusal it is recovering from, and that log belongs to the
329
+ * SUCCESSOR: a terminal `RUN_ERROR` from a dead host would fail the stream for
330
+ * every client attached to the live, healthy run.
331
+ * - It is also strictly cheaper: a latched boolean replaces a store read.
332
+ *
333
+ * The latch deliberately does NOT extend to `close()` — see above.
334
+ *
335
+ * PASSES THE OFFSET TYPE THROUGH, rather than collapsing it to `string`. The
336
+ * fence sits mid-chain between a caller's log and `pipeToRunLog`, so widening
337
+ * here would reintroduce the branded-offset wall one layer in: a
338
+ * `StreamDurability<DurableStreamOffset>` would go in and a
339
+ * `StreamDurability<string>` would come out, which is not assignable back to
340
+ * the caller's own type.
341
+ */
342
+ export function fenceDurability<TOffset extends string = string>(
343
+ durability: StreamDurability<TOffset>,
344
+ claim: RunClaim,
345
+ options: { runs: RunStore; epochRecheckAppends?: number },
346
+ ): StreamDurability<TOffset> {
347
+ const recheckAppends = Math.max(
348
+ 1,
349
+ Math.trunc(options.epochRecheckAppends ?? DEFAULT_EPOCH_RECHECK_APPENDS),
350
+ )
351
+ // Seeded at the threshold so the FIRST append always re-reads the epoch: a
352
+ // successor may have claimed between this fence being built and its first
353
+ // write.
354
+ let appendsSinceEpochRead = recheckAppends
355
+ // Latched by the FIRST refusal and never cleared, and SHARED with this claim's
356
+ // record fence so the two seams cannot disagree.
357
+ const latch = latchFor(claim)
358
+
359
+ async function assertHeld(): Promise<void> {
360
+ // Layer 1 plus the latch: no I/O, so nothing has been written yet, and once
361
+ // refused no throttle and no store read can let a later append through.
362
+ const synchronous = claimLostSynchronously(claim, latch)
363
+ if (synchronous !== undefined) throw synchronous
364
+ if (appendsSinceEpochRead < recheckAppends) {
365
+ appendsSinceEpochRead += 1
366
+ return
367
+ }
368
+ appendsSinceEpochRead = 1
369
+ // Layer 2, throttled because it costs a store read.
370
+ const byEpoch = await claimLostByEpoch(claim, latch, options.runs)
371
+ if (byEpoch !== undefined) throw byEpoch
372
+ }
373
+
374
+ return {
375
+ resumeFrom: () => durability.resumeFrom(),
376
+ append: async (chunks: Array<StreamChunk>) => {
377
+ await assertHeld()
378
+ return durability.append(chunks)
379
+ },
380
+ read: (offset, signal) => durability.read(offset, signal),
381
+ close: () => durability.close(),
382
+ snapshot: () => durability.snapshot(),
383
+ }
384
+ }
385
+
386
+ /**
387
+ * Wrap a run store so a TERMINAL record write is fenced by `claim`.
388
+ *
389
+ * The record is the run's other authoritative channel, and the same rule applies
390
+ * to it: a host that has lost its claim must not state that the run is over. It
391
+ * reaches this seam by the most ordinary route — `pipeToRunLog` catches the
392
+ * `RunClaimLostError` its refused append threw, folds it in, and calls
393
+ * `finish(ctx, 'failed', …)` — so fencing the log alone only moves where the harm
394
+ * surfaces. `'completed'` and `'aborted'` arrive the same way (an empty stream
395
+ * that never appended; a lease loss that aborts `claim.signal`, which
396
+ * `pipeToRunLog` reads as an abort before it appends anything), which is why the
397
+ * gate is {@link isTerminalRunStatus} and not "did an append refuse".
398
+ *
399
+ * SUPPRESSED, NOT ATTEMPTED-AND-SWALLOWED, and not thrown either. `update`
400
+ * resolves without writing. `pipeToRunLog` must not reject — `RunController.start`
401
+ * consumes its promise fire-and-forget — and a rejection here would additionally
402
+ * make `finish` report the run through the local rebuilt record as if the store
403
+ * had broken, which is a different and false fact.
404
+ *
405
+ * WHAT IS *NOT* FENCED, deliberately:
406
+ *
407
+ * - **`close()`** is not on this seam at all, and must stay off it: see
408
+ * {@link fenceDurability}. A wedged `'running'` record with tailers parked
409
+ * forever is worse than the write being prevented.
410
+ * - **Non-terminal writes pass through**, including `detachedSince` and
411
+ * `sandboxKey` written by a superseded host. They are stale, but staleness is
412
+ * not the harm being fixed: none of them can make a live run look finished, so
413
+ * none can mislead `isTerminalRunStatus`, `findActiveRun`, or the reaper. They
414
+ * are also self-healing — the successor owns those fields and overwrites them —
415
+ * whereas over-suppressing strands a record: `createOrResume` is how the row
416
+ * comes into existence at all, and refusing a non-terminal write on a
417
+ * mis-observed loss would leave a run with no record to recover from. Suppress
418
+ * the writes that assert an outcome; let bookkeeping through.
419
+ * - **Reads** (`get`, `listByThread`, `listReclaimable`, `findActiveRun`) do not
420
+ * mutate, so a superseded host reading them is harmless. `finish`'s terminal
421
+ * re-read therefore still works and answers with the SUCCESSOR's live record,
422
+ * which is the truthful thing to resolve with.
423
+ * - **Another run's record.** The fence knows about `claim.runId` only; a write
424
+ * aimed elsewhere is not this claim's to judge.
425
+ *
426
+ * The OPTIONAL methods (`listByThread`, `listReclaimable`) are forwarded only
427
+ * when the wrapped store actually has them: consumers feature-detect
428
+ * (`store.listReclaimable?.(…)`), so materializing one that delegates to a
429
+ * missing method would turn a graceful degrade into a `TypeError`.
430
+ * `findActiveRun` is required on the contract, so it forwards unconditionally.
431
+ */
432
+ export function fenceRunStore(
433
+ runs: RunStore,
434
+ claim: RunClaim,
435
+ options: { logger?: InternalLogger } = {},
436
+ ): RunStore {
437
+ const latch = latchFor(claim)
438
+ // Bound, not merely captured: the store may be a class instance
439
+ // (`InMemoryRunStore`), whose methods need their receiver.
440
+ const listByThread = runs.listByThread?.bind(runs)
441
+ const listReclaimable = runs.listReclaimable?.bind(runs)
442
+
443
+ return {
444
+ createOrResume: (input) => runs.createOrResume(input),
445
+ get: (runId) => runs.get(runId),
446
+ findActiveRun: (threadId) => runs.findActiveRun(threadId),
447
+ update: async (runId, patch) => {
448
+ const status = patch.status
449
+ if (
450
+ runId !== claim.runId ||
451
+ status === undefined ||
452
+ !isTerminalRunStatus(status)
453
+ ) {
454
+ return runs.update(runId, patch)
455
+ }
456
+ const lost =
457
+ claimLostSynchronously(claim, latch) ??
458
+ // Unthrottled, unlike the append path: a terminal write happens once per
459
+ // run, so one store read is not a hot cost — and it is the read that
460
+ // catches a superseded driver whose stream ended without ever appending.
461
+ (await claimLostByEpoch(claim, latch, runs))
462
+ if (lost === undefined) return runs.update(runId, patch)
463
+ // Absorbing this silently would make it invisible: a detached run has no
464
+ // caller to report to. The logger is consumer-supplied, so a throwing sink
465
+ // must not turn a suppression into a rejection.
466
+ try {
467
+ options.logger?.sandbox(
468
+ `run ${runId}: suppressed a terminal '${status}' record write from a superseded driver`,
469
+ { runId, status, heldEpoch: claim.epoch, error: lost },
470
+ )
471
+ } catch {
472
+ // Intentionally empty: there is no second channel to report on.
473
+ }
474
+ return undefined
475
+ },
476
+ ...(listByThread === undefined ? {} : { listByThread }),
477
+ ...(listReclaimable === undefined ? {} : { listReclaimable }),
478
+ }
479
+ }
package/src/contracts.ts CHANGED
@@ -36,6 +36,19 @@ export interface SandboxCapabilities {
36
36
  * file + shell redirection.
37
37
  */
38
38
  writableStdin: boolean
39
+ /**
40
+ * A spawned process can be forcibly terminated via {@link SpawnHandle.kill}
41
+ * and aborted mid-flight via the {@link ProcessOptions.signal} passed to
42
+ * {@link SandboxProcess.spawn}. `true` for host/Docker; some edge providers
43
+ * (e.g. Cloudflare) implement `kill()` as a no-op and drop the abort signal
44
+ * entirely, so a long-running follower process (e.g. `tail -f`) started
45
+ * there can never be stopped by the caller — only polled and abandoned.
46
+ * Callers MUST branch on this before relying on `kill`/abort to reclaim a
47
+ * background process: a bring-your-own provider that omits it would
48
+ * otherwise be silently treated as killable, leaking an unstoppable process
49
+ * inside the sandbox.
50
+ */
51
+ killableProcesses: boolean
39
52
  /** Capture/restore filesystem snapshots via {@link SandboxHandle.snapshot}. */
40
53
  snapshots: boolean
41
54
  /** Declarative network egress allow/deny policy. */