@tanstack/ai-sandbox 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
package/src/reap.ts ADDED
@@ -0,0 +1,723 @@
1
+ /**
2
+ * The sweep `RunStore.listReclaimable` was always missing a consumer for: take a
3
+ * detached run whose viewer never came back, save its transcript, terminalize its
4
+ * record, and tear its sandbox down.
5
+ *
6
+ * THE ONE RULE THAT SHAPES EVERYTHING HERE: **never drive a run to find out
7
+ * whether it finished.**
8
+ *
9
+ * The obvious design — hand the run to `pipeToRunLog` under a short
10
+ * `runBudgetMs` and see whether it terminalizes — was measured and is broken.
11
+ * `pipeToRunLog` is total by construction: it ALWAYS writes a terminal status and
12
+ * ALWAYS calls `durability.close()`. Against a run that has not finished, all
13
+ * three producer shapes are destructive:
14
+ *
15
+ * | producer's reaction to the budget signal | stored status | `close()` |
16
+ * | ---------------------------------------- | ------------- | --------- |
17
+ * | ignores it and keeps producing | `aborted` | called |
18
+ * | returns on abort (the realistic `drive`) | `aborted` | called |
19
+ * | throws an AbortError | `failed` | called |
20
+ *
21
+ * The middle row USED to read `completed`, which was the fatal one: a signal-aware
22
+ * producer exits its loop NORMALLY, and `pipeToRunLog` only checked its signal
23
+ * per chunk, so a healthy mid-flight run was recorded as `'completed'` with a
24
+ * `finishedAt` — a false transcript. That gap is fixed (`run.ts` re-checks the
25
+ * signal after the loop), so the status is now honest on all three rows. The rule
26
+ * above is UNCHANGED, because the status was never the whole harm: every row
27
+ * writes a terminal record and closes a log that commit `5a1f821c9` deliberately
28
+ * leaves OPEN for takeover (ending every attached client's stream), and a terminal
29
+ * record drops out of `listReclaimable` forever, so TTL expiry can never reclaim
30
+ * that run's sandbox. A cost leak with no recovery path. There is therefore no
31
+ * "still running" outcome in {@link ReapRunOutcome}: it is unreachable by
32
+ * construction, not merely unlikely.
33
+ *
34
+ * So sentinel-reached is detected OUT OF BAND, through the in-sandbox journal
35
+ * ({@link probeRunExit}), and `pipeToRunLog` is entered only for a run already
36
+ * KNOWN to have finished, or for one whose TTL has expired (terminal either way).
37
+ * On the FINALIZATION path `runBudgetMs` therefore degrades from a load-bearing
38
+ * mechanism into a safety net whose expiry is a genuine anomaly — see
39
+ * `'budget-exceeded'`. On the EXPIRY path it stays load-bearing: nothing polls the
40
+ * cancel this module records, so the budget is what ends the drive of an expired
41
+ * run whose agent is still producing, and its expiry there is the designed path.
42
+ *
43
+ * WHY THE PROBE IS INJECTED (`ReapOptions.hasFinished`) rather than resolved
44
+ * here, exactly like `ReapOptions.reclaim`:
45
+ *
46
+ * - It cannot read `durability.snapshot()`. After a detach nothing appends to the
47
+ * delivery log — the host that would have appended is the host that left — so
48
+ * the log is frozen at the last delivered chunk while the JOURNAL keeps
49
+ * growing. The log can only ever say "no news".
50
+ * - It cannot resolve a `SandboxHandle` either. `SandboxInstanceStore` is
51
+ * `get`/`upsert`/`delete` with no `list` (see `reclaim.ts` for why that is
52
+ * deliberate), and only the application maps a `sandboxKey` to a live handle.
53
+ *
54
+ * NEVER REJECTS. This runs from a cron, an `alarm()`, or a `waitUntil` with
55
+ * nobody to catch it, so every per-run failure is logged and folded into
56
+ * {@link ReapResult} rather than escaping.
57
+ *
58
+ * NEVER CLEARS `detachedSince`. That field is what the reaper SELECTS on, and
59
+ * `packages/ai/src/stream-to-response.ts`'s `startRunDriver` clears it because a
60
+ * real viewer stopping the TTL clock is the opposite job. Its comment there names
61
+ * borrowing that path "the single most likely bug in this phase"; clearing the
62
+ * marker would reset the TTL on every sweep and a detached run would never
63
+ * expire.
64
+ */
65
+ import { isTerminalRunStatus, requestRunCancel } from '@tanstack/ai'
66
+ import {
67
+ DEFAULT_FENCE_QUIET_MS,
68
+ RunClaimLostError,
69
+ // Thrown, not merely caught: the expiry re-derivation under the lock refuses
70
+ // its own claim when the run's viewer has come back.
71
+ RunClaimNotAcquiredError,
72
+ awaitLogQuiescence,
73
+ fenceDurability,
74
+ fenceRunStore,
75
+ withRunClaim,
76
+ } from './claim'
77
+ import { pipeToRunLog } from './run'
78
+ import {
79
+ journalExitProbeCommand,
80
+ journalPaths,
81
+ parseJournalExit,
82
+ } from './journal'
83
+ import { decodeBase64Stream } from './journal-bytes'
84
+ import type { SandboxHandle } from './contracts'
85
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
86
+ import type { LockStore } from '@tanstack/ai/locks'
87
+ import type {
88
+ RunRecord,
89
+ RunStatus,
90
+ RunStore,
91
+ StreamChunk,
92
+ StreamDurability,
93
+ } from '@tanstack/ai'
94
+
95
+ /**
96
+ * Safety net for a single run's drive. Not the mechanism that decides whether a
97
+ * run finished — see the module doc for why that design was rejected — so this is
98
+ * generous rather than tight: it only has to stop a drive that has genuinely
99
+ * wedged on a run the journal already said was over.
100
+ *
101
+ * On the expiry path it is not merely a net: it is what stops a still-producing
102
+ * agent, since nothing polls the cancel recorded before that drive. A caller that
103
+ * expires live agents may want a tighter value there than a finalization replay
104
+ * needs.
105
+ */
106
+ export const DEFAULT_RUN_BUDGET_MS = 30_000
107
+
108
+ /**
109
+ * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU
110
+ * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands
111
+ * would be killed mid-run rather than finishing 25 and returning; the next tick
112
+ * takes the next batch.
113
+ */
114
+ export const DEFAULT_MAX_RUNS = 25
115
+
116
+ /** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */
117
+ export const DEFAULT_EXIT_PROBE_BYTES = 4096
118
+
119
+ /**
120
+ * What the out-of-band probe learned about a detached run's agent.
121
+ *
122
+ * THREE ARMS, not a boolean, because "could not tell" must not be
123
+ * indistinguishable from "still working": both leave the run alone, but only one
124
+ * of them is a condition an operator should see. A two-valued probe would also
125
+ * invite the caller to treat a provider `exec` failure as "finished" and drive a
126
+ * live run — the exact defect this module exists to prevent.
127
+ */
128
+ export type RunExitProbe =
129
+ /** The `{"__exit":N}` sentinel is in the journal. The agent is over. */
130
+ | { state: 'finished'; exitCode: number }
131
+ /** No sentinel. The agent is mid-flight (or never started). LEAVE IT ALONE. */
132
+ | { state: 'producing' }
133
+ /** The probe could not answer — no sandbox, `exec` rejected, frame undecodable. */
134
+ | { state: 'unknown'; error?: unknown }
135
+
136
+ /** What one sweep did to one run. */
137
+ export type ReapRunOutcome =
138
+ /**
139
+ * The probe saw `{"__exit":N}`, the run was driven to a terminal status, and
140
+ * its transcript is saved. The happy path.
141
+ */
142
+ | 'finalized'
143
+ /**
144
+ * Past `detachedRunTtlMs`. Cancelled first, then driven to terminal. The probe
145
+ * is skipped: the outcome is terminal whether the agent finished or not.
146
+ *
147
+ * Reported even when {@link ReapOptions.runBudgetMs} is what ended the drive —
148
+ * on this path that is the mechanism rather than an anomaly, so `'expired'` is
149
+ * the truthful outcome. The run's own `status` distinguishes the two shapes:
150
+ * an agent that had already finished replays to `'completed'`, while one still
151
+ * producing when the budget fired is `'aborted'`.
152
+ */
153
+ | 'expired'
154
+ /**
155
+ * Still producing. `pipeToRunLog` was NEVER entered — nothing appended, no
156
+ * terminal record written, `close()` not called, `detachedSince` untouched.
157
+ */
158
+ | 'producing'
159
+ /** The probe could not answer. Left exactly as untouched as `'producing'`. */
160
+ | 'unknown'
161
+ /**
162
+ * ANOMALY. The drive outran {@link ReapOptions.runBudgetMs} on a run the
163
+ * journal already said was finished. The record IS terminal and the log IS
164
+ * closed (`pipeToRunLog` guarantees both), so this is a diagnostic, not a leak
165
+ * — but a finished run that would not replay in 30s means the journal read, the
166
+ * translation, or the log is misbehaving.
167
+ *
168
+ * FINALIZATION ONLY. An expired run that outran its budget reports `'expired'`:
169
+ * there was no probe on that path and the agent may legitimately still have been
170
+ * producing, so the budget firing is the designed stop, not a misbehaving replay.
171
+ */
172
+ | 'budget-exceeded'
173
+ /**
174
+ * Another host holds the claim, or held it and superseded us mid-drive. Normal:
175
+ * a real viewer attaching mid-sweep is exactly this. Also covers a run that
176
+ * reached terminal in another host's hands between the listing and the claim.
177
+ */
178
+ | 'not-claimed'
179
+ /**
180
+ * The transcript IS saved and the record IS terminal — only
181
+ * {@link ReapOptions.reclaim} threw, so the sandbox is still up.
182
+ *
183
+ * A DISTINCT outcome rather than `'failed'`, because the two need opposite
184
+ * operator responses and `'failed'` cannot express this one: it carries no
185
+ * `status` and no `exitCode`, so "transcript saved, sandbox NOT reclaimed"
186
+ * read identically to "the sweep failed and the run was never finalized".
187
+ *
188
+ * NOT RETRYABLE BY THE SWEEP. The record is terminal by now, so the run has
189
+ * left `listReclaimable` for good; the sandbox leaks until something else
190
+ * tears it down. This entry, with its `error`, is the only notice of that.
191
+ *
192
+ * OVERWRITES `'budget-exceeded'` when both happened, because the leak is what
193
+ * needs acting on — {@link ReapRunEntry.terminalizedAnyway} is what preserves
194
+ * the budget half of that pair.
195
+ *
196
+ * `sandboxReclaimer` REJECTS on its `'destroy-failed'` arm precisely so this
197
+ * outcome is reachable through the shipped reclaimer and not only through a
198
+ * custom one; see `SandboxReclaimFailedError` in `reclaim.ts`.
199
+ */
200
+ | 'reclaim-failed'
201
+ /** Something threw. Logged, recorded here, and the sweep continued. */
202
+ | 'failed'
203
+
204
+ /** One run's line in the sweep summary. */
205
+ export interface ReapRunEntry {
206
+ runId: string
207
+ outcome: ReapRunOutcome
208
+ /** The run's status after the sweep, when the run was driven. */
209
+ status?: RunStatus
210
+ /** The agent's exit code, when the probe read one. */
211
+ exitCode?: number
212
+ /**
213
+ * THE BUDGET ANOMALY MARKER, and the only field whose mere PRESENCE carries a
214
+ * fact: it is set if and only if the drive outran
215
+ * {@link ReapOptions.runBudgetMs} on the finalization path — the condition
216
+ * `'budget-exceeded'` names. Its value is whether the record nonetheless
217
+ * reached a terminal status, practically always `true` since `pipeToRunLog` is
218
+ * total; it is reported rather than assumed so an operator does not have to
219
+ * infer it.
220
+ *
221
+ * SURVIVES A FAILED RECLAIM. `reclaim` runs after the outcome is classified
222
+ * and overwrites it with `'reclaim-failed'`, which is the more urgent fact (a
223
+ * leaked sandbox nothing will retry) and so wins the single `outcome` slot.
224
+ * This field is therefore what keeps the budget anomaly on the entry: an
225
+ * operator seeing `'reclaim-failed'` WITH `terminalizedAnyway` present is
226
+ * looking at a run that blew its budget and then leaked, and needs both halves.
227
+ */
228
+ terminalizedAnyway?: boolean
229
+ error?: unknown
230
+ }
231
+
232
+ export interface ReapResult {
233
+ /** Runs in this batch — i.e. after the {@link ReapOptions.maxRuns} cap. */
234
+ considered: number
235
+ /** Runs {@link ReapOptions.hasFinished} was actually called for. */
236
+ probed: number
237
+ outcomes: Record<ReapRunOutcome, number>
238
+ runs: Array<ReapRunEntry>
239
+ }
240
+
241
+ export interface ReapOptions<TOffset extends string = string> {
242
+ runs: RunStore
243
+ locks: LockStore
244
+ /**
245
+ * Per-run event log factory, same shape `RunDeps.durability` takes.
246
+ *
247
+ * Generic in the offset type, defaulted to `string` so an existing call site
248
+ * needs no change — see {@link SandboxRunDriverOptions.durability} for why
249
+ * hardcoding the default locked out branded-cursor backends.
250
+ */
251
+ durability: (runId: string) => StreamDurability<TOffset>
252
+ /**
253
+ * The out-of-band "did the agent reach its sentinel?" probe. INJECTED, because
254
+ * neither the delivery log nor this package can answer it — see the module doc.
255
+ * {@link probeRunExit} is the implementation an application wires in once it has
256
+ * resolved the run's `SandboxHandle`.
257
+ */
258
+ hasFinished: (record: RunRecord) => Promise<RunExitProbe>
259
+ /** Produce the run's remaining events. Called only once the claim is held. */
260
+ drive: (input: {
261
+ runId: string
262
+ threadId: string
263
+ signal: AbortSignal
264
+ }) => AsyncIterable<StreamChunk>
265
+ /** Sweep clock, passed rather than read so a sweep is reproducible. */
266
+ now: number
267
+ /** Detached-run TTL; `detachedSince <= now - ttl` expires, INCLUSIVELY. */
268
+ detachedRunTtlMs: number
269
+ /** Safety net per drive. Defaults to {@link DEFAULT_RUN_BUDGET_MS}. */
270
+ runBudgetMs?: number
271
+ /** Batch cap. Defaults to {@link DEFAULT_MAX_RUNS}. */
272
+ maxRuns?: number
273
+ /** Quiescence window; defaults to `DEFAULT_FENCE_QUIET_MS`. */
274
+ fenceQuietMs?: number
275
+ /**
276
+ * Tear the run's sandbox down. Called ONLY after the run reached a terminal
277
+ * status, and with the ORIGINALLY LISTED record — see {@link reapDetachedRuns}.
278
+ * `sandboxReclaimer` in `reclaim.ts` is the ready-made implementation.
279
+ */
280
+ reclaim?: (record: RunRecord) => Promise<void>
281
+ logger?: InternalLogger
282
+ }
283
+
284
+ async function* singleValue(value: string): AsyncIterable<string> {
285
+ yield value
286
+ }
287
+
288
+ /** Decode the base64 frame `journalExitProbeCommand` emits. */
289
+ async function decodeFrame(stdout: string): Promise<string> {
290
+ const decoder = new TextDecoder()
291
+ let text = ''
292
+ for await (const bytes of decodeBase64Stream(singleValue(stdout))) {
293
+ text += decoder.decode(bytes, { stream: true })
294
+ }
295
+ return text + decoder.decode()
296
+ }
297
+
298
+ /**
299
+ * Read the END of a run's journal and answer whether the agent reached its
300
+ * `{"__exit":N}` sentinel. Read-only: no append, no record write, no `close()`.
301
+ *
302
+ * This is the whole reason the reaper is safe. It is the ONLY way to learn that a
303
+ * detached run is over without driving it, because the delivery log stops growing
304
+ * the moment the viewer leaves while the journal does not.
305
+ *
306
+ * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it
307
+ * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or
308
+ * a frame the provider truncated must never be read as "the agent exited".
309
+ *
310
+ * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that
311
+ * does not exist yet is indistinguishable here from one with no sentinel, and both
312
+ * mean "do not touch this run".
313
+ */
314
+ export async function probeRunExit(input: {
315
+ handle: SandboxHandle
316
+ runId: string
317
+ /** Journal directory; defaults to `DEFAULT_JOURNAL_DIR`, as `journalPaths` does. */
318
+ dir?: string
319
+ /** Tail bytes to read. Defaults to {@link DEFAULT_EXIT_PROBE_BYTES}. */
320
+ maxBytes?: number
321
+ }): Promise<RunExitProbe> {
322
+ try {
323
+ const paths = journalPaths(input.runId, input.dir)
324
+ const result = await input.handle.process.exec(
325
+ journalExitProbeCommand(
326
+ paths,
327
+ input.maxBytes ?? DEFAULT_EXIT_PROBE_BYTES,
328
+ ),
329
+ )
330
+ // `paths` supplies the per-run sentinel nonce: without it a mid-flight
331
+ // agent that printed any JSON object carrying `__exit` would read as
332
+ // `'finished'` here, and the caller would drive and reclaim a LIVE run.
333
+ const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths)
334
+ return exitCode === null
335
+ ? { state: 'producing' }
336
+ : { state: 'finished', exitCode }
337
+ } catch (error) {
338
+ return { state: 'unknown', error }
339
+ }
340
+ }
341
+
342
+ /** Every outcome key present at zero, so a consumer can read any of them. */
343
+ function emptyOutcomes(): Record<ReapRunOutcome, number> {
344
+ return {
345
+ finalized: 0,
346
+ expired: 0,
347
+ producing: 0,
348
+ unknown: 0,
349
+ 'budget-exceeded': 0,
350
+ 'not-claimed': 0,
351
+ 'reclaim-failed': 0,
352
+ failed: 0,
353
+ }
354
+ }
355
+
356
+ /**
357
+ * Report through a consumer-supplied logger without letting it break the sweep.
358
+ * Mirrors `run.ts`'s `safeLog`: this module's totality must not be defeated by a
359
+ * sink that cannot serialize a thrown value.
360
+ */
361
+ function safeLog(
362
+ logger: InternalLogger | undefined,
363
+ level: 'errors' | 'sandbox',
364
+ message: string,
365
+ context: Record<string, unknown>,
366
+ ): void {
367
+ try {
368
+ if (level === 'errors') logger?.errors(message, context)
369
+ else logger?.sandbox(message, context)
370
+ } catch {
371
+ // Intentionally empty: there is no second channel to report on.
372
+ }
373
+ }
374
+
375
+ /** Resolved-once settings shared by every run in one sweep. */
376
+ interface ReapContext<TOffset extends string = string> {
377
+ options: ReapOptions<TOffset>
378
+ runBudgetMs: number
379
+ fenceQuietMs: number
380
+ /** Inclusive expiry cutoff: `detachedSince <= cutoff` is expired. */
381
+ cutoff: number
382
+ }
383
+
384
+ /** Whether a thrown value means "we do not own this run", which is normal. */
385
+ function isClaimRefusal(error: unknown): boolean {
386
+ return (
387
+ error instanceof RunClaimNotAcquiredError ||
388
+ error instanceof RunClaimLostError
389
+ )
390
+ }
391
+
392
+ /**
393
+ * Sweep ONE run. Never rejects: the caller folds the returned entry into the
394
+ * summary and moves on.
395
+ *
396
+ * The ORDER of the steps below is the contract, not an implementation detail:
397
+ *
398
+ * 1. **Classify expiry first**, because an expired run needs no probe — its
399
+ * outcome is terminal whether or not the agent finished, so a probe would only
400
+ * add a provider round-trip and a way to fail.
401
+ * 2. **Otherwise probe BEFORE touching anything.** `'producing'` and `'unknown'`
402
+ * return here, having made no claim, no append, no record write, and no
403
+ * `close()`. Driving past this point is the whole defect described in the
404
+ * module doc.
405
+ * 3. Claim, so two hosts never drive one run.
406
+ * 4. **Re-derive expiry from a record read INSIDE the lock**, and only then
407
+ * record the cancel. The listed record is stale by the time the claim is
408
+ * held, and the cancel is sticky.
409
+ * 5. Quiesce, so a predecessor still writing is observed rather than raced.
410
+ * 6. **Arm the run budget**, so it bounds the drive rather than the queue the
411
+ * two steps above stood in.
412
+ * 7. Pipe with BOTH authoritative seams fenced, mirroring `driver.ts`.
413
+ * 8. Reclaim, and ONLY once the record actually reached terminal.
414
+ */
415
+ async function reapOne<TOffset extends string>(
416
+ record: RunRecord,
417
+ ctx: ReapContext<TOffset>,
418
+ counters: { probed: number },
419
+ ): Promise<ReapRunEntry> {
420
+ const { runs, locks, logger } = ctx.options
421
+ const { runId, threadId } = record
422
+
423
+ try {
424
+ // INCLUSIVE, exactly as `RunStore.listReclaimable` documents its own cutoff:
425
+ // a run detached at precisely `now - ttlMs` IS expired. The two must agree,
426
+ // or a run would be listed as reclaimable and then classified as fresh on
427
+ // every single sweep, forever.
428
+ const expired =
429
+ record.detachedSince !== undefined && record.detachedSince <= ctx.cutoff
430
+
431
+ let exitCode: number | undefined
432
+ if (!expired) {
433
+ counters.probed += 1
434
+ const probe = await ctx.options.hasFinished(record)
435
+ if (probe.state !== 'finished') {
436
+ // THE LEAVE-ALONE PATH. Deliberately returns before `withRunClaim`, so
437
+ // not even `driverEpoch` moves — and above all `detachedSince` is left
438
+ // exactly as it was, since it is both this run's TTL evidence and the
439
+ // field the next sweep selects on.
440
+ safeLog(logger, 'sandbox', `reap: leaving run ${runId} alone`, {
441
+ runId,
442
+ state: probe.state,
443
+ ...(probe.state === 'unknown' && probe.error !== undefined
444
+ ? { error: probe.error }
445
+ : {}),
446
+ })
447
+ return {
448
+ runId,
449
+ outcome: probe.state,
450
+ ...(probe.state === 'unknown' && probe.error !== undefined
451
+ ? { error: probe.error }
452
+ : {}),
453
+ }
454
+ }
455
+ exitCode = probe.exitCode
456
+ }
457
+
458
+ // Armed INSIDE the claim, below. Read after it for the outcome, so it is
459
+ // hoisted here rather than declared in the callback.
460
+ let budget: AbortSignal | undefined
461
+ const final = await withRunClaim(
462
+ {
463
+ runs,
464
+ locks,
465
+ runId,
466
+ fenceQuietMs: ctx.fenceQuietMs,
467
+ ...(logger === undefined ? {} : { logger }),
468
+ },
469
+ async (claim) => {
470
+ if (expired) {
471
+ // RE-DERIVED FROM A RECORD READ INSIDE THE LOCK, never from the listed
472
+ // one. `stream-to-response.ts`'s `startRunDriver` CLEARS
473
+ // `detachedSince` when a real viewer attaches — deliberately stopping
474
+ // the TTL clock — and it takes this same per-run lock, so an
475
+ // expiry decided at listing time is stale by the time the claim is
476
+ // held. Cancelling on the stale value poisoned a now-live run:
477
+ // nothing in the tree ever clears `cancelRequested`, so on that
478
+ // viewer's next ORDINARY disconnect `middleware.ts`'s
479
+ // `wasCancelRequested` read skips the detach branch and destroys the
480
+ // sandbox of a healthy, actively-viewed run.
481
+ const current = await runs.get(runId)
482
+ if (current === null) {
483
+ throw new RunClaimNotAcquiredError(runId, 'unknown')
484
+ }
485
+ if (
486
+ current.detachedSince === undefined ||
487
+ current.detachedSince > ctx.cutoff
488
+ ) {
489
+ // The viewer came back. `'not-claimed'` already documents "a real
490
+ // viewer attaching mid-sweep is exactly this", and refusing here
491
+ // leaves the run as untouched as the leave-alone path does: no
492
+ // cancel, no append, no terminal record, no `close()`.
493
+ throw new RunClaimNotAcquiredError(runId, 'superseded')
494
+ }
495
+ // BEFORE the drive, never after — and never before the claim.
496
+ // `withSandbox`'s `onAbort` resolves the out-of-band cancel band from
497
+ // the record, so recording the intent first is what makes the teardown
498
+ // an explicit cancel that DESTROYS the sandbox rather than a second
499
+ // detach that re-arms `detachedSince` and leaves the run to be swept
500
+ // again forever. Recorded after the drive it is pure bookkeeping on a
501
+ // run that already tore down the wrong way. Recorded before the CLAIM
502
+ // it is an unfenced, sticky write on a record this host does not own,
503
+ // derived from a value the lock exists to make current.
504
+ await requestRunCancel(runs, runId)
505
+ }
506
+ // Before the first append, never after: `pipeToRunLog` snapshots to align.
507
+ await awaitLogQuiescence(
508
+ ctx.options.durability(runId),
509
+ ctx.fenceQuietMs,
510
+ )
511
+ // A safety net, not a mechanism (see the module doc). `AbortSignal.any`
512
+ // is this package's idiom for linking one — see
513
+ // `testkit/takeover-conformance.ts`.
514
+ //
515
+ // ARMED HERE, not before `withRunClaim`. Both the lock wait and the
516
+ // quiescence wait consume a timer started earlier: quiescence always
517
+ // sleeps at least one `fenceQuietMs` and may sleep six, and lock
518
+ // acquisition waits behind whoever holds it, unbounded. The effective
519
+ // budget was silently `runBudgetMs − fenceQuietMs − lockWait`, and once
520
+ // it went negative the claim was acquired with the timer already fired:
521
+ // `pipeToRunLog` hit its entry `signal.aborted` check before pulling one
522
+ // chunk, so a FINISHED agent's transcript was recorded `'aborted'` and
523
+ // its log closed — and a terminal record leaves `listReclaimable`
524
+ // forever, so that transcript was then unreplayable while `reclaim`
525
+ // destroyed the sandbox holding the only copy. The budget bounds the
526
+ // DRIVE, not the queue.
527
+ budget = AbortSignal.timeout(ctx.runBudgetMs)
528
+ // `claim.signal` is in the composed signal because losing the lease MUST
529
+ // stop the drive: a successor that took the run over is appending to the
530
+ // same log, and this drive continuing would double every chunk.
531
+ const signal = AbortSignal.any([claim.signal, budget])
532
+ return pipeToRunLog(ctx.options.drive({ runId, threadId, signal }), {
533
+ // BOTH seams, over the SAME claim, as `driver.ts` explains: fencing the
534
+ // log alone just moves the harm to "a dead host marks the successor's
535
+ // live run failed".
536
+ runs: fenceRunStore(runs, claim, {
537
+ ...(logger === undefined ? {} : { logger }),
538
+ }),
539
+ durability: (id) =>
540
+ fenceDurability(ctx.options.durability(id), claim, { runs }),
541
+ runId,
542
+ threadId,
543
+ signal,
544
+ ...(logger === undefined ? {} : { logger }),
545
+ })
546
+ },
547
+ )
548
+
549
+ const terminal = isTerminalRunStatus(final.status)
550
+ let outcome: ReapRunOutcome
551
+ // `&& !expired` is the whole subtlety. The budget is an ANOMALY only on the
552
+ // finalization path, where the probe already said the agent hit its sentinel
553
+ // and a replay that will not finish in 30s means the journal read, the
554
+ // translation, or the log is misbehaving. On the EXPIRY path there was no
555
+ // probe and the agent may well be mid-sentence: `requestRunCancel` writes a
556
+ // record field whose only reader is `withSandbox`'s `onAbort` (which runs
557
+ // after something else has already aborted), so the budget is the sole thing
558
+ // that ends the drive of a still-producing expired run. That is the designed
559
+ // path, not a misbehaving one, and reporting it as the anomaly made
560
+ // `'expired'` unreachable for exactly the runs the TTL exists to expire.
561
+ // `budget` is armed inside the claim, so reaching here means it was armed;
562
+ // `?? false` keeps the read total rather than asserting that.
563
+ if ((budget?.aborted ?? false) && !expired) {
564
+ outcome = 'budget-exceeded'
565
+ } else if (!terminal) {
566
+ // The terminal write was SUPPRESSED and `finish`'s re-read answered with a
567
+ // live record, which `fenceRunStore` only does when this host lost the claim
568
+ // to another one. That is the same fact as a refused claim, reported the
569
+ // same way rather than as a success that wrote nothing.
570
+ outcome = 'not-claimed'
571
+ } else {
572
+ outcome = expired ? 'expired' : 'finalized'
573
+ }
574
+
575
+ // CAPTURED BEFORE THE RECLAIM BLOCK, which may overwrite `outcome` with
576
+ // `'reclaim-failed'`. Conditioning the `terminalizedAnyway` spread on the
577
+ // post-reclaim `outcome` dropped the budget diagnostic from exactly the
578
+ // entries that need it most: a run that blew its budget AND then failed to
579
+ // reclaim reported neither fact but the leak, and an operator cannot
580
+ // diagnose a leak on a run whose replay was already misbehaving without
581
+ // knowing that it was.
582
+ const budgetAnomaly = outcome === 'budget-exceeded'
583
+
584
+ let reclaimError: unknown
585
+ if (terminal && ctx.options.reclaim !== undefined) {
586
+ try {
587
+ // `record`, NOT `final`. When the terminal `update` fails, `finish` returns
588
+ // a LOCALLY REBUILT record that carries only `runId`/`threadId`/`startedAt`
589
+ // plus the terminal patch — no `sandboxKey` — so `reclaimSandbox` would see
590
+ // `undefined`, answer `'no-sandbox-key'`, and the sandbox would leak
591
+ // silently on exactly the path where something already went wrong.
592
+ await ctx.options.reclaim(record)
593
+ } catch (error) {
594
+ // CAUGHT HERE rather than in the outer catch, which would report a bare
595
+ // `'failed'` with no `status` and no `exitCode`. `reclaimSandbox`
596
+ // deliberately does not guard `instances.get` (its contract is that the
597
+ // CALLER records the failure) and neither does `sandboxReclaimer`, so a
598
+ // throwing instance store landed there. By this point the record is
599
+ // terminal and the log closed, so the run is out of `listReclaimable`
600
+ // forever and no later sweep will retry: the sandbox leaks, and an
601
+ // operator reading `'failed'` cannot tell "transcript saved, sandbox NOT
602
+ // reclaimed" from "the sweep failed and the run was never finalized".
603
+ reclaimError = error
604
+ outcome = 'reclaim-failed'
605
+ safeLog(logger, 'errors', `reap: reclaiming run ${runId} failed`, {
606
+ runId,
607
+ status: final.status,
608
+ error,
609
+ })
610
+ }
611
+ }
612
+
613
+ return {
614
+ runId,
615
+ outcome,
616
+ status: final.status,
617
+ ...(exitCode === undefined ? {} : { exitCode }),
618
+ ...(budgetAnomaly ? { terminalizedAnyway: terminal } : {}),
619
+ ...(reclaimError === undefined ? {} : { error: reclaimError }),
620
+ }
621
+ } catch (error) {
622
+ if (isClaimRefusal(error)) {
623
+ safeLog(logger, 'sandbox', `reap: not driving run ${runId}`, {
624
+ runId,
625
+ error,
626
+ })
627
+ return { runId, outcome: 'not-claimed', error }
628
+ }
629
+ // Folded into the summary rather than rethrown: one bad run must not abandon
630
+ // the rest of the batch, and there is no caller to receive a rejection.
631
+ safeLog(logger, 'errors', `reap: sweeping run ${runId} failed`, {
632
+ runId,
633
+ error,
634
+ })
635
+ return { runId, outcome: 'failed', error }
636
+ }
637
+ }
638
+
639
+ /**
640
+ * Sweep the detached runs a `RunStore` surfaces, saving each finished run's
641
+ * transcript and reclaiming its sandbox.
642
+ *
643
+ * A plain async function with no timer and no daemon: call it from a cron, a
644
+ * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects
645
+ * — every failure is logged and counted in the returned {@link ReapResult}.
646
+ *
647
+ * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is
648
+ * every detached run, which is the candidate set for FINALIZATION (a run that hit
649
+ * its sentinel one second after the viewer left has an unsaved transcript and
650
+ * must not wait out the TTL), and expiry is then classified in-process against
651
+ * the same inclusive cutoff. Listing twice with two TTLs would cost a second
652
+ * store round-trip to compute a subset.
653
+ *
654
+ * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be
655
+ * reaped, which answers `{ considered: 0 }` plus one log line rather than
656
+ * throwing — the same graceful degrade every other optional-method call site in
657
+ * the repo does (`store.findActiveRun?.(threadId)`).
658
+ */
659
+ export async function reapDetachedRuns<TOffset extends string = string>(
660
+ options: ReapOptions<TOffset>,
661
+ ): Promise<ReapResult> {
662
+ const logger = options.logger
663
+ const outcomes = emptyOutcomes()
664
+ const entries: Array<ReapRunEntry> = []
665
+ const empty = (): ReapResult => ({
666
+ considered: 0,
667
+ probed: 0,
668
+ outcomes,
669
+ runs: entries,
670
+ })
671
+
672
+ const list = options.runs.listReclaimable?.bind(options.runs)
673
+ if (list === undefined) {
674
+ safeLog(
675
+ logger,
676
+ 'sandbox',
677
+ 'reap: the run store does not implement listReclaimable; nothing to sweep',
678
+ {},
679
+ )
680
+ return empty()
681
+ }
682
+
683
+ let candidates: Array<RunRecord>
684
+ try {
685
+ candidates = await list({ now: options.now, ttlMs: 0 })
686
+ } catch (error) {
687
+ safeLog(logger, 'errors', 'reap: listing reclaimable runs failed', {
688
+ error,
689
+ })
690
+ return empty()
691
+ }
692
+
693
+ // Capped so one invocation cannot outlive its platform's budget and be killed
694
+ // mid-drive. `slice` and not a `break`, so `considered` reports the batch the
695
+ // sweep actually took responsibility for.
696
+ const maxRuns = Math.max(0, Math.trunc(options.maxRuns ?? DEFAULT_MAX_RUNS))
697
+ const batch = candidates.slice(0, maxRuns)
698
+
699
+ const ctx: ReapContext<TOffset> = {
700
+ options,
701
+ runBudgetMs: options.runBudgetMs ?? DEFAULT_RUN_BUDGET_MS,
702
+ fenceQuietMs: options.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS,
703
+ cutoff: options.now - options.detachedRunTtlMs,
704
+ }
705
+ const counters = { probed: 0 }
706
+
707
+ // Sequential on purpose: each run costs a lock, a provider round-trip, and a
708
+ // full replay, and a cron invocation's budget is the scarce resource. Fanning
709
+ // out would multiply peak load against the provider for no throughput a
710
+ // subsequent tick cannot supply.
711
+ for (const record of batch) {
712
+ const entry = await reapOne(record, ctx, counters)
713
+ outcomes[entry.outcome] += 1
714
+ entries.push(entry)
715
+ }
716
+
717
+ return {
718
+ considered: batch.length,
719
+ probed: counters.probed,
720
+ outcomes,
721
+ runs: entries,
722
+ }
723
+ }