@tanstack/ai-sandbox 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
@@ -0,0 +1,1040 @@
1
+ /**
2
+ * Provider conformance for TAKEOVER: a second driver picking up a run whose
3
+ * first driver died, against a REAL sandbox.
4
+ *
5
+ * WHY THIS EXISTS SEPARATELY FROM THE UNIT TESTS. Every takeover unit test in
6
+ * this package drives fakes — a scripted `spawn`, a `test -f` that answers from
7
+ * a boolean, a log that is an array. Fakes model what we believe the shell and
8
+ * the filesystem do, and on this feature that belief has been wrong three times:
9
+ * `base64` delivers zero bytes on a live pipe, `tail -f` on a missing file exits
10
+ * instead of waiting, and a provider's `kill` does not always reap a grandchild.
11
+ * Each one passed every fake. So the four properties a takeover actually rests
12
+ * on are asserted here through a provider's real `spawn`/`exec` against a real
13
+ * journal file:
14
+ *
15
+ * 1. **The delivered sequence is the run's sequence, with no duplicated
16
+ * prefix.** Asserted as a TRANSCRIPT, never as "chunks arrived": a takeover
17
+ * that replays the whole journal and re-appends everything satisfies the weak
18
+ * assertion while showing the user the entire run twice. That is the exact
19
+ * failure `alignToStoredLog` exists to prevent, and the only assertion that
20
+ * can see it is one that compares the stored log to the expected sequence
21
+ * element for element.
22
+ * 2. **The attach preflight decides, or fails, but never hangs.** It probes with
23
+ * the provider's real `exec` (`test -f`), which is the layer where a fake's
24
+ * assumptions break, and its three verdicts (`unknown-run`, `terminal-run`,
25
+ * `journal-timeout`) plus the legitimate late-journal race are all timing
26
+ * against a real filesystem.
27
+ * 3. **The epoch fence and its latch hold under real concurrency.** Two drivers
28
+ * reading one real journal at once: the second wins, the first appends
29
+ * NOTHING — not even `pipeToRunLog`'s recovery `RUN_ERROR` — and cannot
30
+ * terminalize the record out from under the live successor.
31
+ * 4. **A terminal run's journal is deleted, and a later attach says so.** The
32
+ * deletion is a real `rm` of real files, and the follow-up attach must report
33
+ * `terminal-run` rather than tailing the file that `journalFollowCommand`
34
+ * would helpfully re-create.
35
+ *
36
+ * WHAT IS REAL HERE. The provider (its `spawn`, `exec`, and shell), the journal
37
+ * (a real NDJSON file the agent's stdout is redirected into), the agent (a real
38
+ * process writing real lines with a real pause in the middle), the reader
39
+ * (`readJournalNdjson`, including the follow/poll strategy split and the attach
40
+ * preflight), the alignment (`alignedIfAttaching` over the real
41
+ * `resolveSandboxDurability` output), the claim and BOTH fences
42
+ * (`sandboxRunDriver`), and the run record (`InMemoryRunStore`). The event log is
43
+ * in-process, exactly as the recommended `memoryStream` backend is.
44
+ *
45
+ * A provider that cannot satisfy the contract MUST declare `unsupported.reason`.
46
+ * As in the journal suite there is deliberately no silent-skip path: a
47
+ * conformance case that quietly returns prints as a pass, which is how an
48
+ * unimplemented capability ships green.
49
+ *
50
+ * FOUND BY THIS SUITE, FIXED IN THE PROVIDER, STILL NOT ASSERTED HERE. On
51
+ * local-process under Windows (git-bash `sh`), the follow read's `tail`
52
+ * grandchild used to SURVIVE `proc.kill()`: `LocalProcessHandle.killTree` ran
53
+ * `taskkill /PID <sh> /T /F` and returned as soon as `spawnSync` reported no
54
+ * `error`. Two things were wrong. It never checked taskkill's exit status — and
55
+ * that alone would not have caught it, because MSYS's fork emulation leaves the
56
+ * `tail.exe` pointing at an intermediate shell that has already exited, so
57
+ * `taskkill /T` (live parent links only) cannot reach it and still exits `0`.
58
+ * Measured by counting `tail.exe` before and after a run: this suite leaked 4 per
59
+ * run and the shipped journal suite 2, accumulating for the life of the machine.
60
+ * It was a provider defect, not a takeover defect — every case here still
61
+ * delivered the right transcript, because `untilAborted` (see
62
+ * `journal-reader.ts`) stops honoring the pipe once the signal fires rather than
63
+ * waiting for the kill, which is exactly why it never failed a test.
64
+ * `killTree` now resolves the tree through MSYS's own process table and verifies
65
+ * the survivors are gone (0 per run), covered in
66
+ * `ai-sandbox-local-process/tests/kill-tree.test.ts`.
67
+ * Deliberately still NOT asserted in this suite: a per-provider process census is
68
+ * not portable (Docker's `tail` dies with its container), and a conformance case
69
+ * that counted host processes would fail for reasons unrelated to takeover.
70
+ *
71
+ * EVERY WAIT IN THIS FILE IS BOUNDED. A hang stalls CI instead of failing it, so
72
+ * each journal read carries a timeout signal, each poll loop carries a deadline
73
+ * and a message naming what never happened, and each case carries an explicit
74
+ * per-test timeout.
75
+ *
76
+ * Vitest is an OPTIONAL peer dependency: this module is imported only from test
77
+ * files, which already run under Vitest.
78
+ */
79
+ import { describe, expect, it } from 'vitest'
80
+ import { EventType, InMemoryRunStore } from '@tanstack/ai'
81
+ import { InMemoryLockStore } from '@tanstack/ai/locks'
82
+ import {
83
+ journalCleanupCommand,
84
+ journalExistsCommand,
85
+ journalPaths,
86
+ journaledCommand,
87
+ } from '../journal'
88
+ import { readJournalNdjson, startJournaledAgent } from '../runner'
89
+ import {
90
+ JournalAttachUnavailableError,
91
+ awaitAttachableJournal,
92
+ } from '../attach-preflight'
93
+ import {
94
+ alignedIfAttaching,
95
+ journalOptionsFor,
96
+ resolveSandboxDurability,
97
+ } from '../durability'
98
+ import { sandboxRunDriver } from '../driver'
99
+ import { fenceDurability, withRunClaim } from '../claim'
100
+ import { chunkFingerprint, createRunScopedIdGen } from '../chunk-identity'
101
+ import type { SandboxRunDurability } from '../durability'
102
+ import type { JournalOptions } from '../runner'
103
+ import type { SandboxHandle } from '../contracts'
104
+ import type { LockStore } from '@tanstack/ai/locks'
105
+ import type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'
106
+
107
+ export interface TakeoverConformanceConfig {
108
+ /** Provider name, used in the describe title. */
109
+ name: string
110
+ /** Create a live sandbox plus its teardown. */
111
+ createHandle: () => Promise<{
112
+ handle: SandboxHandle
113
+ dispose: () => Promise<void>
114
+ }>
115
+ /**
116
+ * Declare that this provider cannot support takeover, with the reason.
117
+ * Registers a skipped case whose title carries the reason — a NAMED skip,
118
+ * visible in the reporter. Omit it and the suite runs.
119
+ */
120
+ unsupported?: { reason: string }
121
+ }
122
+
123
+ /**
124
+ * Journal directory for this suite, deliberately NOT
125
+ * {@link DEFAULT_JOURNAL_DIR}: on local-process the sandbox shell shares the
126
+ * host's real `/tmp`, so conformance runs must not write where an application's
127
+ * runs live.
128
+ */
129
+ const CONFORMANCE_JOURNAL_DIR = '/tmp/tanstack-takeover-conformance'
130
+
131
+ /** Poll interval handed to providers that cannot follow a growing file. */
132
+ const POLL_INTERVAL_MS = 50
133
+
134
+ /**
135
+ * Quiescence window for the successor's first append. Short because the
136
+ * predecessor in these cases has provably stopped (the suite sequenced it) —
137
+ * the gate still runs, it just does not need to wait 5s to observe nothing.
138
+ */
139
+ const FENCE_QUIET_MS = 25
140
+
141
+ /**
142
+ * Bound on a real journal read, so a reader that delivers nothing FAILS instead
143
+ * of parking CI.
144
+ *
145
+ * Never an assertion, and deliberately far above anything a healthy read needs
146
+ * (measured: 10–18s for the follow cases on both providers). Every use site
147
+ * pairs it with a `backstopped: false` witness, so a read the CLOCK ended fails
148
+ * naming this backstop rather than as a downstream transcript mismatch — which
149
+ * means this number can be raised freely and must never be the thing a case is
150
+ * tuned against.
151
+ */
152
+ const READ_BACKSTOP_MS = 90_000
153
+
154
+ /**
155
+ * Unique per case, and it must be: `journalPaths` derives the file name from the
156
+ * `runId` and the journal is append-only, so a reused id appends BEHIND the
157
+ * previous run's `{"__exit":N}` sentinel and the new run appears to emit nothing
158
+ * at all (see `journal.ts`). The counter covers two cases created inside the
159
+ * same millisecond; the random suffix covers two suites sharing one `/tmp`.
160
+ */
161
+ let caseCounter = 0
162
+ function uniqueRunId(label: string): string {
163
+ caseCounter += 1
164
+ const suffix = Math.random().toString(36).slice(2, 8)
165
+ return `tko-${label}-${Date.now()}-${caseCounter}-${suffix}`
166
+ }
167
+
168
+ /**
169
+ * An in-process event log with real accumulated state, plus the two facts the
170
+ * assertions need: what is stored (in append order) and how many times `close()`
171
+ * ran.
172
+ *
173
+ * `snapshot()` returns fresh objects, per the `StreamDurability` contract, so a
174
+ * caller cannot reach the stored log through the result.
175
+ */
176
+ interface ConformanceLog {
177
+ log: StreamDurability
178
+ /** Stored chunks, in append order. The transcript under test. */
179
+ stored: () => Array<StreamChunk>
180
+ /** `close()` calls — proof that `close` is NOT fenced. */
181
+ closes: () => number
182
+ }
183
+
184
+ function conformanceLog(): ConformanceLog {
185
+ const entries: Array<{ offset: string; chunk: StreamChunk }> = []
186
+ let closes = 0
187
+ return {
188
+ log: {
189
+ resumeFrom: () => null,
190
+ append: (chunks) =>
191
+ Promise.resolve(
192
+ chunks.map((chunk) => {
193
+ const offset = `conf:${entries.length}`
194
+ entries.push({ offset, chunk })
195
+ return offset
196
+ }),
197
+ ),
198
+ // Nothing in this suite tails the log — every assertion reads the stored
199
+ // transcript with `snapshot()`, which is also what `alignToStoredLog`
200
+ // uses, and a `read` would park until `close()` (see `align.ts`).
201
+ read: () => (async function* empty() {})(),
202
+ close: () => {
203
+ closes += 1
204
+ return Promise.resolve()
205
+ },
206
+ snapshot: () => Promise.resolve(entries.map((entry) => ({ ...entry }))),
207
+ },
208
+ stored: () => entries.map((entry) => entry.chunk),
209
+ closes: () => closes,
210
+ }
211
+ }
212
+
213
+ /**
214
+ * A lock that grants every request immediately and never reports a loss.
215
+ *
216
+ * `InMemoryLockStore` SERIALIZES claims within one process, so a second attach
217
+ * waits for the first to finish and the two drivers are never concurrent — which
218
+ * means the epoch fence can never be observed there. `claim.ts` says exactly
219
+ * that: in one process only layer 2, the `driverEpoch` fence, is provable. This
220
+ * models a lease-less lock so the two drives overlap and layer 2 does the work.
221
+ */
222
+ const permissiveLocks: LockStore = {
223
+ withLock: (_key, fn) => fn(new AbortController().signal),
224
+ }
225
+
226
+ /** The event a journal line translates into. `timestamp` is excluded from `chunkFingerprint`. */
227
+ function contentChunk(messageId: string, delta: string): StreamChunk {
228
+ return {
229
+ type: EventType.TEXT_MESSAGE_CONTENT,
230
+ messageId,
231
+ delta,
232
+ timestamp: Date.now(),
233
+ }
234
+ }
235
+
236
+ /**
237
+ * Narrow one parsed journal line into its chunk.
238
+ *
239
+ * Fields are validated and the chunk is REBUILT from them rather than asserted
240
+ * into shape: a cast would let a provider that mangles the bytes (a folded
241
+ * stderr diagnostic, a truncated line) reach `chunkFingerprint` as a
242
+ * structurally invalid chunk and fail somewhere unrelated.
243
+ */
244
+ function toChunk(
245
+ runId: string,
246
+ messageId: string,
247
+ value: unknown,
248
+ ): StreamChunk {
249
+ if (typeof value !== 'object' || value === null || !('delta' in value)) {
250
+ throw new Error(
251
+ `takeover conformance: run ${runId} journal line is not an agent event: ${JSON.stringify(value)}`,
252
+ )
253
+ }
254
+ const delta = value.delta
255
+ if (typeof delta !== 'string') {
256
+ throw new Error(
257
+ `takeover conformance: run ${runId} journal line has a non-string delta: ${JSON.stringify(value)}`,
258
+ )
259
+ }
260
+ return contentChunk(messageId, delta)
261
+ }
262
+
263
+ /**
264
+ * The translator. Deterministic by construction, which is what makes alignment
265
+ * possible at all: the message id comes from {@link createRunScopedIdGen}, so
266
+ * re-translating the same journal from byte 0 reproduces byte-identical chunks
267
+ * (modulo `timestamp`, the one field `chunkFingerprint` excludes).
268
+ */
269
+ async function* translate(
270
+ runId: string,
271
+ lines: AsyncIterable<unknown>,
272
+ ): AsyncIterable<StreamChunk> {
273
+ const messageId = createRunScopedIdGen(runId)()
274
+ for await (const line of lines) yield toChunk(runId, messageId, line)
275
+ }
276
+
277
+ /**
278
+ * A comparable transcript: each chunk reduced to its {@link chunkFingerprint}.
279
+ *
280
+ * The fingerprint, not the chunk object, and for the same reason alignment uses
281
+ * it — `timestamp` is wall-clock and unreproducible, so a raw `toEqual` on
282
+ * chunks would fail on the one field the feature deliberately ignores. Every
283
+ * other field participates, so a duplicated prefix, a dropped chunk, or a
284
+ * reordered one still fails.
285
+ */
286
+ function transcript(chunks: Array<StreamChunk>): Array<string> {
287
+ return chunks.map(chunkFingerprint)
288
+ }
289
+
290
+ /** The chunks a run over `deltas` must deliver, exactly once and in order. */
291
+ function expectedTranscript(
292
+ runId: string,
293
+ deltas: Array<string>,
294
+ ): Array<StreamChunk> {
295
+ const messageId = createRunScopedIdGen(runId)()
296
+ return deltas.map((delta) => contentChunk(messageId, delta))
297
+ }
298
+
299
+ /**
300
+ * A real agent: a shell command that prints one NDJSON line per delta, with an
301
+ * optional real pause partway through, then exits.
302
+ *
303
+ * `printf '%s\n' a b c` reuses the format for every operand on GNU coreutils and
304
+ * on busybox alike, so this needs no loop. The JSON contains only double quotes,
305
+ * so it is safe inside the POSIX single-quoted words this builds.
306
+ */
307
+ function agentCommand(deltas: Array<string>, pauseAfter: number): string {
308
+ const line = (delta: string): string => `'{"delta":"${delta}"}'`
309
+ const head = deltas.slice(0, pauseAfter)
310
+ const tail = deltas.slice(pauseAfter)
311
+ const parts = [`printf '%s\\n' ${head.map(line).join(' ')}`]
312
+ if (tail.length > 0) {
313
+ // A real sleep, so the takeover below happens while the agent is genuinely
314
+ // still writing rather than against a finished file.
315
+ parts.push('sleep 2', `printf '%s\\n' ${tail.map(line).join(' ')}`)
316
+ }
317
+ return parts.join('; ')
318
+ }
319
+
320
+ /** Resolve durability through the production resolver, fresh or attaching. */
321
+ function durabilityFor(
322
+ runs: RunStore,
323
+ log: StreamDurability,
324
+ attach: boolean,
325
+ ): SandboxRunDurability {
326
+ const resolved = resolveSandboxDurability({
327
+ runs,
328
+ durability: {
329
+ adapter: log,
330
+ journal: CONFORMANCE_JOURNAL_DIR,
331
+ attach,
332
+ pollIntervalMs: POLL_INTERVAL_MS,
333
+ },
334
+ })
335
+ if (resolved === undefined) {
336
+ throw new Error(
337
+ 'takeover conformance: resolveSandboxDurability returned undefined for a fully wired run',
338
+ )
339
+ }
340
+ return resolved
341
+ }
342
+
343
+ /**
344
+ * The reader's journal options for a resolved durability.
345
+ *
346
+ * `journalOptionsFor` answers `undefined` for a NON-durable run, which cannot
347
+ * happen here — every run in this suite is fully wired. Narrowing it with a
348
+ * thrown error rather than a non-null assertion keeps the impossible case loud
349
+ * if the resolver's contract ever changes.
350
+ */
351
+ function journalOptions(
352
+ durability: SandboxRunDurability,
353
+ runId: string,
354
+ ): JournalOptions {
355
+ const options = journalOptionsFor(durability, runId)
356
+ if (options === undefined) {
357
+ throw new Error(
358
+ `takeover conformance: journalOptionsFor answered undefined for durable run ${runId}`,
359
+ )
360
+ }
361
+ return options
362
+ }
363
+
364
+ /** A `'running'` record for `runId`, ready to be claimed. */
365
+ async function runningRun(
366
+ runId: string,
367
+ threadId: string,
368
+ ): Promise<InMemoryRunStore> {
369
+ const runs = new InMemoryRunStore()
370
+ await runs.createOrResume({ runId, threadId, startedAt: Date.now() })
371
+ return runs
372
+ }
373
+
374
+ /**
375
+ * Wrap a handle so the `process.exec` calls ONE operation makes can be counted.
376
+ *
377
+ * This is how the attach preflight's fail-fast cases are anchored, and the reason
378
+ * they are not anchored on elapsed time. `awaitAttachableJournal` runs exactly one
379
+ * `test -f` before it consults the run store, so a decision made from the record
380
+ * costs one `exec` and a decision made by waiting costs one per
381
+ * `probeIntervalMs`. The count separates those two behaviors exactly; elapsed time
382
+ * does not, because a single `exec` is a provider round-trip whose latency the
383
+ * suite does not control — a `docker exec` on a loaded daemon has been measured at
384
+ * 9.6s, which fails a `< 4_000ms` bound while the preflight under test did
385
+ * precisely the right thing. A timing bound that goes red on a busy machine
386
+ * teaches people to ignore the suite.
387
+ *
388
+ * The spread copies the handle's own methods, so everything except `exec` is the
389
+ * provider's; the wrapper delegates rather than reimplementing.
390
+ */
391
+ function countingExec(handle: SandboxHandle): {
392
+ handle: SandboxHandle
393
+ execs: () => number
394
+ } {
395
+ let execs = 0
396
+ return {
397
+ handle: {
398
+ ...handle,
399
+ process: {
400
+ ...handle.process,
401
+ exec: (command, options) => {
402
+ execs += 1
403
+ return handle.process.exec(command, options)
404
+ },
405
+ },
406
+ },
407
+ execs: () => execs,
408
+ }
409
+ }
410
+
411
+ /** Poll `check` until it answers true, or fail with a message naming what never happened. */
412
+ async function waitUntil(
413
+ check: () => Promise<boolean>,
414
+ options: { timeoutMs: number; message: string },
415
+ ): Promise<void> {
416
+ const deadline = Date.now() + options.timeoutMs
417
+ for (;;) {
418
+ if (await check()) return
419
+ if (Date.now() > deadline) {
420
+ throw new Error(
421
+ `takeover conformance: ${options.message} within ${options.timeoutMs}ms`,
422
+ )
423
+ }
424
+ await sleep(25)
425
+ }
426
+ }
427
+
428
+ function sleep(ms: number): Promise<void> {
429
+ return new Promise((resolve) => setTimeout(resolve, ms))
430
+ }
431
+
432
+ interface Gate {
433
+ promise: Promise<void>
434
+ open: () => void
435
+ }
436
+
437
+ /** A one-shot gate, for sequencing two concurrent drivers deterministically. */
438
+ function gate(): Gate {
439
+ let open = (): void => {}
440
+ const promise = new Promise<void>((resolve) => {
441
+ open = () => resolve()
442
+ })
443
+ return { promise, open }
444
+ }
445
+
446
+ /**
447
+ * Build the driver a host would build for one run.
448
+ *
449
+ * `drive` is the real journal path: read the run's journal from byte 0 (through
450
+ * the attach preflight when attaching), translate, and align against the stored
451
+ * log — `alignedIfAttaching`, so alignment runs on an attach and only on an
452
+ * attach.
453
+ *
454
+ * Returns the driver alongside `backstopped()`, the causal witness for
455
+ * {@link READ_BACKSTOP_MS}: every case that drives this must assert it is
456
+ * `false` before its transcript assertions, so a read the CLOCK ended fails
457
+ * naming the backstop instead of as a truncated-transcript diff.
458
+ */
459
+ function driverFor(input: {
460
+ handle: SandboxHandle
461
+ runs: RunStore
462
+ locks: LockStore
463
+ log: StreamDurability
464
+ runId: string
465
+ attach: boolean
466
+ /** Awaited before the FIRST translated chunk is yielded, never after. */
467
+ beforeFirstChunk?: () => Promise<void>
468
+ }): {
469
+ driver: ReturnType<typeof sandboxRunDriver>
470
+ /** True if any read this driver started was ended by the backstop clock. */
471
+ backstopped: () => boolean
472
+ } {
473
+ const durability = durabilityFor(input.runs, input.log, input.attach)
474
+ // One entry per `drive` invocation, so a re-drive cannot hide a backstopped
475
+ // read behind a healthy one.
476
+ const backstops: Array<AbortSignal> = []
477
+ const driver = sandboxRunDriver({
478
+ request: new Request(
479
+ `http://takeover.local/attach?runId=${encodeURIComponent(input.runId)}&offset=-1`,
480
+ ),
481
+ runs: input.runs,
482
+ locks: input.locks,
483
+ durability: () => input.log,
484
+ fenceQuietMs: FENCE_QUIET_MS,
485
+ drive: ({ runId, signal }) => {
486
+ // The read is bounded independently of `signal`: an `InMemoryLockStore`
487
+ // hands out a signal it never aborts, so a journal that stops growing
488
+ // would otherwise park this read forever and turn a broken takeover into a
489
+ // hung CI job instead of a failing assertion.
490
+ //
491
+ // Not the assertion — see {@link READ_BACKSTOP_MS}. `backstopped()` below
492
+ // is what proves the clock was not what ended the read.
493
+ const backstop = AbortSignal.timeout(READ_BACKSTOP_MS)
494
+ backstops.push(backstop)
495
+ const bounded = AbortSignal.any([signal, backstop])
496
+ const lines = readJournalNdjson(input.handle, {
497
+ signal: bounded,
498
+ journal: journalOptions(durability, runId),
499
+ })
500
+ const gated = input.beforeFirstChunk
501
+ const source =
502
+ gated === undefined
503
+ ? lines
504
+ : (async function* afterGate() {
505
+ let first = true
506
+ for await (const value of lines) {
507
+ if (first) {
508
+ first = false
509
+ await gated()
510
+ }
511
+ yield value
512
+ }
513
+ })()
514
+ return alignedIfAttaching(translate(runId, source), durability)
515
+ },
516
+ })
517
+ return { driver, backstopped: () => backstops.some((s) => s.aborted) }
518
+ }
519
+
520
+ /** Exactly what core's `startRunDriver` does: claim, then pipe the drive. */
521
+ function takeOver(
522
+ driver: ReturnType<typeof sandboxRunDriver>,
523
+ input: { runs: RunStore; runId: string; threadId: string },
524
+ ): Promise<unknown> {
525
+ const { runs, runId, threadId } = input
526
+ return driver.claim({ runs, locks: driver.locks, runId }, (claim) =>
527
+ driver.pipe(driver.drive({ runId, threadId, signal: claim.signal }), {
528
+ runId,
529
+ threadId,
530
+ signal: claim.signal,
531
+ }),
532
+ )
533
+ }
534
+
535
+ /** Best-effort removal of a case's journal files, through the shell (rule 3). */
536
+ async function cleanup(handle: SandboxHandle, runId: string): Promise<void> {
537
+ try {
538
+ await handle.process.exec(
539
+ journalCleanupCommand(journalPaths(runId, CONFORMANCE_JOURNAL_DIR)),
540
+ )
541
+ } catch {
542
+ // The sandbox may already be gone. Nothing under test depends on the files
543
+ // being absent afterwards — the cases that DO assert deletion assert it
544
+ // directly.
545
+ }
546
+ }
547
+
548
+ /**
549
+ * Assert `createHandle` satisfies the takeover conformance contract. Each `it`
550
+ * gets a fresh sandbox via `createHandle`/`dispose`, and a unique `runId`, so no
551
+ * case can observe another's journal.
552
+ */
553
+ export function runTakeoverConformance(
554
+ config: TakeoverConformanceConfig,
555
+ ): void {
556
+ describe(`takeover conformance — ${config.name}`, () => {
557
+ if (config.unsupported) {
558
+ it.skip(`unsupported: ${config.unsupported.reason}`, () => {
559
+ expect(true).toBe(true)
560
+ })
561
+ return
562
+ }
563
+
564
+ // ---------------------------------------------------------------------
565
+ // 1. A real takeover, end to end.
566
+ // ---------------------------------------------------------------------
567
+ it(
568
+ 'delivers the run sequence exactly once when a second driver takes over mid-stream',
569
+ { timeout: 180_000 },
570
+ async () => {
571
+ const { handle, dispose } = await config.createHandle()
572
+ const runId = uniqueRunId('e2e')
573
+ const threadId = `${runId}-t`
574
+ const deltas = ['1', '2', '3', '4', '5', '6']
575
+ const prefixLength = 3
576
+ const expected = expectedTranscript(runId, deltas)
577
+ const runs = await runningRun(runId, threadId)
578
+ const log = conformanceLog()
579
+ try {
580
+ const fresh = durabilityFor(runs, log.log, false)
581
+
582
+ // THE HOST THAT DIES. A real claim, a real fence, a real journal read
583
+ // of a real agent — and then it stops after `prefixLength` chunks
584
+ // without closing the log and without terminalizing the record, which
585
+ // is what a host vanishing looks like from the outside.
586
+ const deliveredByFirst: Array<StreamChunk> = []
587
+ // A backstop, so a reader that delivers nothing fails instead of
588
+ // parking CI. Not the assertion — `backstopped` below proves it was not
589
+ // what ended the loop.
590
+ const firstBackstop = AbortSignal.timeout(READ_BACKSTOP_MS)
591
+ await withRunClaim(
592
+ { runs, locks: new InMemoryLockStore(), runId },
593
+ async (claim) => {
594
+ const fenced = fenceDurability(log.log, claim, { runs })
595
+ await startJournaledAgent(
596
+ handle,
597
+ agentCommand(deltas, prefixLength),
598
+ { journal: journalOptions(fresh, runId) },
599
+ )
600
+ const lines = readJournalNdjson(handle, {
601
+ signal: firstBackstop,
602
+ journal: journalOptions(fresh, runId),
603
+ })
604
+ for await (const chunk of translate(runId, lines)) {
605
+ await fenced.append([chunk])
606
+ deliveredByFirst.push(chunk)
607
+ // Breaking ends the reader's `tail` before this host walks away;
608
+ // the AGENT keeps running, which is the whole premise.
609
+ if (deliveredByFirst.length === prefixLength) break
610
+ }
611
+ },
612
+ )
613
+ // The causal witness for the dying host's read: it must stop because
614
+ // the consumer broke at `prefixLength`, not because the clock ran out.
615
+ // A backstopped read here delivers a short prefix and the takeover the
616
+ // case exists to exercise would start from the wrong offset.
617
+ expect({ backstopped: firstBackstop.aborted }).toEqual({
618
+ backstopped: false,
619
+ })
620
+ expect(transcript(deliveredByFirst)).toEqual(
621
+ transcript(expected.slice(0, prefixLength)),
622
+ )
623
+
624
+ // THE SUCCESSOR. Same runId, same journal, a fresh claim.
625
+ const successor = driverFor({
626
+ handle,
627
+ runs,
628
+ locks: new InMemoryLockStore(),
629
+ log: log.log,
630
+ runId,
631
+ attach: true,
632
+ })
633
+ const record = await takeOver(successor.driver, {
634
+ runs,
635
+ runId,
636
+ threadId,
637
+ })
638
+
639
+ // THE CAUSAL WITNESS, first — see {@link READ_BACKSTOP_MS}. The
640
+ // transcript assertions below can only speak about chunks that
641
+ // arrived; this one says the successor's read ended because the
642
+ // journal ended, not because the clock did. Without it a backstopped
643
+ // read reports as a confusing short-transcript diff.
644
+ expect({ backstopped: successor.backstopped() }).toEqual({
645
+ backstopped: false,
646
+ })
647
+
648
+ // THE TRANSCRIPT, element for element. This is the assertion that can
649
+ // see the failure the feature exists to prevent: a takeover that
650
+ // replays the journal from byte 0 without aligning re-appends the
651
+ // prefix, so `stored` would be 9 entries beginning `1,2,3,1,2,3,…` —
652
+ // and the user would watch the first half of the run twice. "Chunks
653
+ // arrived" passes against that; this does not.
654
+ expect(transcript(log.stored())).toEqual(transcript(expected))
655
+ // Stated separately so a failure reads as what it is rather than as a
656
+ // 9-vs-6 array diff.
657
+ expect(log.stored()).toHaveLength(deltas.length)
658
+ expect(transcript(log.stored().slice(prefixLength))).toEqual(
659
+ transcript(expected.slice(prefixLength)),
660
+ )
661
+
662
+ const finalRecord = await runs.get(runId)
663
+ expect(finalRecord?.status).toBe('completed')
664
+ // The successor's claim, not the predecessor's: a hardcoded epoch
665
+ // would read 1 here and every takeover would be fenced out.
666
+ expect(finalRecord?.driverEpoch).toBe(2)
667
+ expect(record).not.toBeUndefined()
668
+ } finally {
669
+ await cleanup(handle, runId)
670
+ await dispose()
671
+ }
672
+ },
673
+ )
674
+
675
+ // ---------------------------------------------------------------------
676
+ // 2. The attach preflight, against a real filesystem.
677
+ // ---------------------------------------------------------------------
678
+ it(
679
+ 'fails an attach to an unknown runId with unknown-run, without waiting it out',
680
+ { timeout: 120_000 },
681
+ async () => {
682
+ const { handle, dispose } = await config.createHandle()
683
+ const runId = uniqueRunId('unknown')
684
+ try {
685
+ expect.hasAssertions()
686
+ const probes = countingExec(handle)
687
+ const error = await awaitAttachableJournal(probes.handle, {
688
+ paths: journalPaths(runId, CONFORMANCE_JOURNAL_DIR),
689
+ runId,
690
+ runs: new InMemoryRunStore(),
691
+ // Generous on purpose: were the store verdict skipped, this would
692
+ // poll for the full 8s and the probe count below would catch it.
693
+ waitMs: 8_000,
694
+ probeIntervalMs: POLL_INTERVAL_MS,
695
+ }).then(
696
+ () => null,
697
+ (reason: unknown) => reason,
698
+ )
699
+ expect(error).toBeInstanceOf(JournalAttachUnavailableError)
700
+ if (!(error instanceof JournalAttachUnavailableError)) return
701
+ expect(error.reason).toBe('unknown-run')
702
+ // Decided from the RECORD, not by waiting it out: one `test -f`, then
703
+ // the store. A preflight that polled to the deadline would run ~80
704
+ // probes here. See `countingExec` for why this is not a stopwatch.
705
+ expect(probes.execs()).toBe(1)
706
+ } finally {
707
+ await dispose()
708
+ }
709
+ },
710
+ )
711
+
712
+ it(
713
+ 'fails an attach to a terminal run whose journal is gone with terminal-run',
714
+ { timeout: 120_000 },
715
+ async () => {
716
+ const { handle, dispose } = await config.createHandle()
717
+ const runId = uniqueRunId('terminal')
718
+ const threadId = `${runId}-t`
719
+ try {
720
+ expect.hasAssertions()
721
+ const runs = await runningRun(runId, threadId)
722
+ await runs.update(runId, { status: 'completed', finishedAt: 2 })
723
+ const probes = countingExec(handle)
724
+ const error = await awaitAttachableJournal(probes.handle, {
725
+ paths: journalPaths(runId, CONFORMANCE_JOURNAL_DIR),
726
+ runId,
727
+ runs,
728
+ waitMs: 8_000,
729
+ probeIntervalMs: POLL_INTERVAL_MS,
730
+ }).then(
731
+ () => null,
732
+ (reason: unknown) => reason,
733
+ )
734
+ expect(error).toBeInstanceOf(JournalAttachUnavailableError)
735
+ if (!(error instanceof JournalAttachUnavailableError)) return
736
+ expect(error.reason).toBe('terminal-run')
737
+ // One `test -f`, then the record. Not a stopwatch — `countingExec`.
738
+ expect(probes.execs()).toBe(1)
739
+ } finally {
740
+ await dispose()
741
+ }
742
+ },
743
+ )
744
+
745
+ it(
746
+ 'waits for a live run whose journal appears late — the legitimate race',
747
+ { timeout: 120_000 },
748
+ async () => {
749
+ const { handle, dispose } = await config.createHandle()
750
+ const runId = uniqueRunId('race')
751
+ const threadId = `${runId}-t`
752
+ const paths = journalPaths(runId, CONFORMANCE_JOURNAL_DIR)
753
+ try {
754
+ const runs = await runningRun(runId, threadId)
755
+ // A real driver writing its real first line ~400ms after the attach
756
+ // starts probing. This is the NORMAL case — `journalFollowCommand`'s
757
+ // `: >> file` exists for it — so failing fast here would reintroduce
758
+ // the defect that fix cured.
759
+ const writer = sleep(400).then(() =>
760
+ handle.process.exec(
761
+ journaledCommand(`printf '{"delta":"1"}\\n'`, paths),
762
+ ),
763
+ )
764
+ try {
765
+ await awaitAttachableJournal(handle, {
766
+ paths,
767
+ runId,
768
+ runs,
769
+ // Comfortably longer than the write above; the per-test timeout is
770
+ // what turns a never-resolving wait into a failure.
771
+ waitMs: 20_000,
772
+ probeIntervalMs: POLL_INTERVAL_MS,
773
+ })
774
+ } finally {
775
+ await writer
776
+ }
777
+ // Resolving at all is the assertion; this pins the premise that it
778
+ // resolved because the file really is there now.
779
+ expect(
780
+ (await handle.process.exec(journalExistsCommand(paths))).exitCode,
781
+ ).toBe(0)
782
+ } finally {
783
+ await cleanup(handle, runId)
784
+ await dispose()
785
+ }
786
+ },
787
+ )
788
+
789
+ it(
790
+ 'bounds the wait for a live run whose journal never appears, with journal-timeout',
791
+ { timeout: 120_000 },
792
+ async () => {
793
+ const { handle, dispose } = await config.createHandle()
794
+ const runId = uniqueRunId('timeout')
795
+ const threadId = `${runId}-t`
796
+ try {
797
+ expect.hasAssertions()
798
+ const runs = await runningRun(runId, threadId)
799
+ const error = await awaitAttachableJournal(handle, {
800
+ paths: journalPaths(runId, CONFORMANCE_JOURNAL_DIR),
801
+ runId,
802
+ runs,
803
+ waitMs: 600,
804
+ probeIntervalMs: POLL_INTERVAL_MS,
805
+ }).then(
806
+ () => null,
807
+ (reason: unknown) => reason,
808
+ )
809
+ expect(error).toBeInstanceOf(JournalAttachUnavailableError)
810
+ if (!(error instanceof JournalAttachUnavailableError)) return
811
+ expect(error.reason).toBe('journal-timeout')
812
+ // The three assertions above ARE the proof the bound was applied: an
813
+ // unbounded wait never produces a `JournalAttachUnavailableError` at
814
+ // all, and `'600ms'` in the message is the configured bound reported
815
+ // back. No stopwatch assertion here on purpose — the case's own
816
+ // `{ timeout: 120_000 }` already converts an unbounded wait into a
817
+ // failure, and a wall-clock ceiling would red a CORRECT implementation
818
+ // on a machine where one `docker exec` was measured at 95s.
819
+ expect(error.message).toContain('600ms')
820
+ } finally {
821
+ await dispose()
822
+ }
823
+ },
824
+ )
825
+
826
+ // ---------------------------------------------------------------------
827
+ // 3. The epoch fence and the shared latch, under real concurrency.
828
+ // ---------------------------------------------------------------------
829
+ it(
830
+ 'lets the second of two concurrent drivers win, and the loser appends nothing at all',
831
+ { timeout: 180_000 },
832
+ async () => {
833
+ const { handle, dispose } = await config.createHandle()
834
+ const runId = uniqueRunId('fence')
835
+ const threadId = `${runId}-t`
836
+ const deltas = ['1', '2', '3']
837
+ const expected = expectedTranscript(runId, deltas)
838
+ try {
839
+ const runs = await runningRun(runId, threadId)
840
+ const log = conformanceLog()
841
+ // One agent, one journal, two drivers reading it concurrently.
842
+ await startJournaledAgent(
843
+ handle,
844
+ agentCommand(deltas, deltas.length),
845
+ {
846
+ journal: journalOptions(
847
+ durabilityFor(runs, log.log, false),
848
+ runId,
849
+ ),
850
+ },
851
+ )
852
+
853
+ // The LOSER: the original host, so it does not align (there is nothing
854
+ // stored when it starts). Gated before its first chunk reaches the
855
+ // log, which is where the fence has to catch it — after the successor
856
+ // has claimed and finished. The gate sits INSIDE the source stream, so
857
+ // the loser's alignment snapshot (were it attaching) and its first
858
+ // append both happen after the release, exactly as a host paused by a
859
+ // GC or a VM suspend would.
860
+ const released = gate()
861
+ const losingDriver = driverFor({
862
+ handle,
863
+ runs,
864
+ locks: permissiveLocks,
865
+ log: log.log,
866
+ runId,
867
+ attach: false,
868
+ beforeFirstChunk: () => released.promise,
869
+ })
870
+ const loser = takeOver(losingDriver.driver, {
871
+ runs,
872
+ runId,
873
+ threadId,
874
+ })
875
+ await waitUntil(
876
+ async () => ((await runs.get(runId))?.driverEpoch ?? 0) >= 1,
877
+ {
878
+ timeoutMs: 30_000,
879
+ message: `the first driver never claimed run ${runId}`,
880
+ },
881
+ )
882
+
883
+ // The WINNER: claims at a higher epoch and drives the run to the end.
884
+ const winner = driverFor({
885
+ handle,
886
+ runs,
887
+ locks: permissiveLocks,
888
+ log: log.log,
889
+ runId,
890
+ attach: true,
891
+ })
892
+ await takeOver(winner.driver, { runs, runId, threadId })
893
+ // The causal witness, before the transcript — see
894
+ // {@link READ_BACKSTOP_MS}. The winner drives the run to its sentinel,
895
+ // so a clock-ended read here must say so rather than surface as a
896
+ // missing chunk.
897
+ expect({ backstopped: winner.backstopped() }).toEqual({
898
+ backstopped: false,
899
+ })
900
+ expect(transcript(log.stored())).toEqual(transcript(expected))
901
+
902
+ // Now let the superseded host try to write.
903
+ released.open()
904
+ await loser
905
+ // The causal witness, and here it is load-bearing rather than merely
906
+ // diagnostic: the loser's gate is awaited from INSIDE its read, so a
907
+ // backstopped read would abandon the stream during the wait, the loser
908
+ // would never attempt an append at all, and every "nothing lands"
909
+ // assertion below would pass vacuously without the fence ever running.
910
+ expect({ backstopped: losingDriver.backstopped() }).toEqual({
911
+ backstopped: false,
912
+ })
913
+
914
+ // NOTHING lands — not the run's chunks a second time, and not
915
+ // `pipeToRunLog`'s recovery `RUN_ERROR` either. That log belongs to the
916
+ // WINNER: a terminal `RUN_ERROR` from a dead host would fail the stream
917
+ // for every client attached to the live, healthy run.
918
+ expect(transcript(log.stored())).toEqual(transcript(expected))
919
+ expect(
920
+ log.stored().some((chunk) => chunk.type === EventType.RUN_ERROR),
921
+ ).toBe(false)
922
+ // And nothing lands on the RECORD either: `isTerminalRunStatus` must
923
+ // not answer for the loser's view of a run the winner completed.
924
+ const record = await runs.get(runId)
925
+ expect(record?.status).toBe('completed')
926
+ expect(record?.error).toBeUndefined()
927
+ expect(record?.driverEpoch).toBe(2)
928
+ // `close()` is deliberately OUTSIDE both fences: it runs on the very
929
+ // teardown caused by losing the claim, and a fenced close would wedge
930
+ // the record at `'running'` with every live tailer parked forever. Two
931
+ // drivers, two closes.
932
+ expect(log.closes()).toBe(2)
933
+ } finally {
934
+ await cleanup(handle, runId)
935
+ await dispose()
936
+ }
937
+ },
938
+ )
939
+
940
+ // ---------------------------------------------------------------------
941
+ // 4. Journal cleanup on a terminal run, and the attach that follows it.
942
+ // ---------------------------------------------------------------------
943
+ it(
944
+ "deletes a terminal run's journal, and a later attach reports terminal-run instead of hanging",
945
+ { timeout: 180_000 },
946
+ async () => {
947
+ const { handle, dispose } = await config.createHandle()
948
+ const runId = uniqueRunId('cleanup')
949
+ const threadId = `${runId}-t`
950
+ const deltas = ['1', '2']
951
+ const paths = journalPaths(runId, CONFORMANCE_JOURNAL_DIR)
952
+ try {
953
+ const runs = await runningRun(runId, threadId)
954
+ const log = conformanceLog()
955
+ const fresh = durabilityFor(runs, log.log, false)
956
+ await startJournaledAgent(
957
+ handle,
958
+ agentCommand(deltas, deltas.length),
959
+ { journal: journalOptions(fresh, runId) },
960
+ )
961
+ const seen: Array<StreamChunk> = []
962
+ // A backstop, so a reader that delivers nothing fails instead of
963
+ // parking CI. Not the assertion — `backstopped` below proves it was not
964
+ // what ended the loop.
965
+ const backstop = AbortSignal.timeout(READ_BACKSTOP_MS)
966
+ for await (const chunk of translate(
967
+ runId,
968
+ readJournalNdjson(handle, {
969
+ signal: backstop,
970
+ journal: journalOptions(fresh, runId),
971
+ }),
972
+ )) {
973
+ seen.push(chunk)
974
+ }
975
+ // The causal witness, first: this loop has no `break`, so the ONLY
976
+ // honest reasons for it to end are the sentinel or the backstop. A
977
+ // clock-ended read must say so rather than report a short transcript.
978
+ expect({ backstopped: backstop.aborted }).toEqual({
979
+ backstopped: false,
980
+ })
981
+ // Reaching the sentinel is what makes the run terminal, and it is the
982
+ // precondition for the deletion below.
983
+ expect(transcript(seen)).toEqual(
984
+ transcript(expectedTranscript(runId, deltas)),
985
+ )
986
+
987
+ // Real files, really gone — asserted through the shell, never
988
+ // `handle.fs.exists`: on local-process the two resolve `/tmp`
989
+ // differently, so an `fs` probe would answer about a path the journal
990
+ // was never written to (`journal.ts` rule 3).
991
+ // Named rather than two bare `.not.toBe(0)` assertions on an exit
992
+ // code, so a regression reports WHICH file survived instead of
993
+ // `expected +0 not to be +0`.
994
+ const journalProbe = await handle.process.exec(
995
+ journalExistsCommand(paths),
996
+ )
997
+ const stderrProbe = await handle.process.exec(
998
+ journalExistsCommand({ ...paths, journal: paths.stderr }),
999
+ )
1000
+ expect({
1001
+ journalDeleted: journalProbe.exitCode !== 0,
1002
+ stderrSidecarDeleted: stderrProbe.exitCode !== 0,
1003
+ }).toEqual({ journalDeleted: true, stderrSidecarDeleted: true })
1004
+
1005
+ // The run is over, so the record says so — and the attach that follows
1006
+ // must answer from the record rather than tail the journal, which
1007
+ // `journalFollowCommand` would obligingly re-create as an empty file
1008
+ // that no sentinel can ever arrive in.
1009
+ await runs.update(runId, {
1010
+ status: 'completed',
1011
+ finishedAt: Date.now(),
1012
+ })
1013
+ const probes = countingExec(handle)
1014
+ const error = await awaitAttachableJournal(probes.handle, {
1015
+ paths,
1016
+ runId,
1017
+ runs,
1018
+ waitMs: 8_000,
1019
+ probeIntervalMs: POLL_INTERVAL_MS,
1020
+ }).then(
1021
+ () => null,
1022
+ (reason: unknown) => reason,
1023
+ )
1024
+ expect(error).toBeInstanceOf(JournalAttachUnavailableError)
1025
+ if (!(error instanceof JournalAttachUnavailableError)) return
1026
+ expect(error.reason).toBe('terminal-run')
1027
+ // The journal really is gone (asserted above), so the preflight takes
1028
+ // the record arm: one `test -f`, then the store, no wait. This is the
1029
+ // bound that failed as `expected 9652 to be less than 4000` under
1030
+ // parallel Docker load, where the 9.6s was one `docker exec`
1031
+ // round-trip and not the preflight — see `countingExec`.
1032
+ expect(probes.execs()).toBe(1)
1033
+ } finally {
1034
+ await cleanup(handle, runId)
1035
+ await dispose()
1036
+ }
1037
+ },
1038
+ )
1039
+ })
1040
+ }