@tanstack/ai-sandbox 0.2.4 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
@@ -0,0 +1,238 @@
1
+ import { SandboxHandle } from './contracts.js';
2
+ import { InternalLogger } from '@tanstack/ai/adapter-internals';
3
+ import { LockStore } from '@tanstack/ai/locks';
4
+ import { RunRecord, RunStatus, RunStore, StreamChunk, StreamDurability } from '@tanstack/ai';
5
+ /**
6
+ * Safety net for a single run's drive. Not the mechanism that decides whether a
7
+ * run finished — see the module doc for why that design was rejected — so this is
8
+ * generous rather than tight: it only has to stop a drive that has genuinely
9
+ * wedged on a run the journal already said was over.
10
+ *
11
+ * On the expiry path it is not merely a net: it is what stops a still-producing
12
+ * agent, since nothing polls the cancel recorded before that drive. A caller that
13
+ * expires live agents may want a tighter value there than a finalization replay
14
+ * needs.
15
+ */
16
+ export declare const DEFAULT_RUN_BUDGET_MS = 30000;
17
+ /**
18
+ * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU
19
+ * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands
20
+ * would be killed mid-run rather than finishing 25 and returning; the next tick
21
+ * takes the next batch.
22
+ */
23
+ export declare const DEFAULT_MAX_RUNS = 25;
24
+ /** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */
25
+ export declare const DEFAULT_EXIT_PROBE_BYTES = 4096;
26
+ /**
27
+ * What the out-of-band probe learned about a detached run's agent.
28
+ *
29
+ * THREE ARMS, not a boolean, because "could not tell" must not be
30
+ * indistinguishable from "still working": both leave the run alone, but only one
31
+ * of them is a condition an operator should see. A two-valued probe would also
32
+ * invite the caller to treat a provider `exec` failure as "finished" and drive a
33
+ * live run — the exact defect this module exists to prevent.
34
+ */
35
+ export type RunExitProbe =
36
+ /** The `{"__exit":N}` sentinel is in the journal. The agent is over. */
37
+ {
38
+ state: 'finished';
39
+ exitCode: number;
40
+ }
41
+ /** No sentinel. The agent is mid-flight (or never started). LEAVE IT ALONE. */
42
+ | {
43
+ state: 'producing';
44
+ }
45
+ /** The probe could not answer — no sandbox, `exec` rejected, frame undecodable. */
46
+ | {
47
+ state: 'unknown';
48
+ error?: unknown;
49
+ };
50
+ /** What one sweep did to one run. */
51
+ export type ReapRunOutcome =
52
+ /**
53
+ * The probe saw `{"__exit":N}`, the run was driven to a terminal status, and
54
+ * its transcript is saved. The happy path.
55
+ */
56
+ 'finalized'
57
+ /**
58
+ * Past `detachedRunTtlMs`. Cancelled first, then driven to terminal. The probe
59
+ * is skipped: the outcome is terminal whether the agent finished or not.
60
+ *
61
+ * Reported even when {@link ReapOptions.runBudgetMs} is what ended the drive —
62
+ * on this path that is the mechanism rather than an anomaly, so `'expired'` is
63
+ * the truthful outcome. The run's own `status` distinguishes the two shapes:
64
+ * an agent that had already finished replays to `'completed'`, while one still
65
+ * producing when the budget fired is `'aborted'`.
66
+ */
67
+ | 'expired'
68
+ /**
69
+ * Still producing. `pipeToRunLog` was NEVER entered — nothing appended, no
70
+ * terminal record written, `close()` not called, `detachedSince` untouched.
71
+ */
72
+ | 'producing'
73
+ /** The probe could not answer. Left exactly as untouched as `'producing'`. */
74
+ | 'unknown'
75
+ /**
76
+ * ANOMALY. The drive outran {@link ReapOptions.runBudgetMs} on a run the
77
+ * journal already said was finished. The record IS terminal and the log IS
78
+ * closed (`pipeToRunLog` guarantees both), so this is a diagnostic, not a leak
79
+ * — but a finished run that would not replay in 30s means the journal read, the
80
+ * translation, or the log is misbehaving.
81
+ *
82
+ * FINALIZATION ONLY. An expired run that outran its budget reports `'expired'`:
83
+ * there was no probe on that path and the agent may legitimately still have been
84
+ * producing, so the budget firing is the designed stop, not a misbehaving replay.
85
+ */
86
+ | 'budget-exceeded'
87
+ /**
88
+ * Another host holds the claim, or held it and superseded us mid-drive. Normal:
89
+ * a real viewer attaching mid-sweep is exactly this. Also covers a run that
90
+ * reached terminal in another host's hands between the listing and the claim.
91
+ */
92
+ | 'not-claimed'
93
+ /**
94
+ * The transcript IS saved and the record IS terminal — only
95
+ * {@link ReapOptions.reclaim} threw, so the sandbox is still up.
96
+ *
97
+ * A DISTINCT outcome rather than `'failed'`, because the two need opposite
98
+ * operator responses and `'failed'` cannot express this one: it carries no
99
+ * `status` and no `exitCode`, so "transcript saved, sandbox NOT reclaimed"
100
+ * read identically to "the sweep failed and the run was never finalized".
101
+ *
102
+ * NOT RETRYABLE BY THE SWEEP. The record is terminal by now, so the run has
103
+ * left `listReclaimable` for good; the sandbox leaks until something else
104
+ * tears it down. This entry, with its `error`, is the only notice of that.
105
+ *
106
+ * OVERWRITES `'budget-exceeded'` when both happened, because the leak is what
107
+ * needs acting on — {@link ReapRunEntry.terminalizedAnyway} is what preserves
108
+ * the budget half of that pair.
109
+ *
110
+ * `sandboxReclaimer` REJECTS on its `'destroy-failed'` arm precisely so this
111
+ * outcome is reachable through the shipped reclaimer and not only through a
112
+ * custom one; see `SandboxReclaimFailedError` in `reclaim.ts`.
113
+ */
114
+ | 'reclaim-failed'
115
+ /** Something threw. Logged, recorded here, and the sweep continued. */
116
+ | 'failed';
117
+ /** One run's line in the sweep summary. */
118
+ export interface ReapRunEntry {
119
+ runId: string;
120
+ outcome: ReapRunOutcome;
121
+ /** The run's status after the sweep, when the run was driven. */
122
+ status?: RunStatus;
123
+ /** The agent's exit code, when the probe read one. */
124
+ exitCode?: number;
125
+ /**
126
+ * THE BUDGET ANOMALY MARKER, and the only field whose mere PRESENCE carries a
127
+ * fact: it is set if and only if the drive outran
128
+ * {@link ReapOptions.runBudgetMs} on the finalization path — the condition
129
+ * `'budget-exceeded'` names. Its value is whether the record nonetheless
130
+ * reached a terminal status, practically always `true` since `pipeToRunLog` is
131
+ * total; it is reported rather than assumed so an operator does not have to
132
+ * infer it.
133
+ *
134
+ * SURVIVES A FAILED RECLAIM. `reclaim` runs after the outcome is classified
135
+ * and overwrites it with `'reclaim-failed'`, which is the more urgent fact (a
136
+ * leaked sandbox nothing will retry) and so wins the single `outcome` slot.
137
+ * This field is therefore what keeps the budget anomaly on the entry: an
138
+ * operator seeing `'reclaim-failed'` WITH `terminalizedAnyway` present is
139
+ * looking at a run that blew its budget and then leaked, and needs both halves.
140
+ */
141
+ terminalizedAnyway?: boolean;
142
+ error?: unknown;
143
+ }
144
+ export interface ReapResult {
145
+ /** Runs in this batch — i.e. after the {@link ReapOptions.maxRuns} cap. */
146
+ considered: number;
147
+ /** Runs {@link ReapOptions.hasFinished} was actually called for. */
148
+ probed: number;
149
+ outcomes: Record<ReapRunOutcome, number>;
150
+ runs: Array<ReapRunEntry>;
151
+ }
152
+ export interface ReapOptions<TOffset extends string = string> {
153
+ runs: RunStore;
154
+ locks: LockStore;
155
+ /**
156
+ * Per-run event log factory, same shape `RunDeps.durability` takes.
157
+ *
158
+ * Generic in the offset type, defaulted to `string` so an existing call site
159
+ * needs no change — see {@link SandboxRunDriverOptions.durability} for why
160
+ * hardcoding the default locked out branded-cursor backends.
161
+ */
162
+ durability: (runId: string) => StreamDurability<TOffset>;
163
+ /**
164
+ * The out-of-band "did the agent reach its sentinel?" probe. INJECTED, because
165
+ * neither the delivery log nor this package can answer it — see the module doc.
166
+ * {@link probeRunExit} is the implementation an application wires in once it has
167
+ * resolved the run's `SandboxHandle`.
168
+ */
169
+ hasFinished: (record: RunRecord) => Promise<RunExitProbe>;
170
+ /** Produce the run's remaining events. Called only once the claim is held. */
171
+ drive: (input: {
172
+ runId: string;
173
+ threadId: string;
174
+ signal: AbortSignal;
175
+ }) => AsyncIterable<StreamChunk>;
176
+ /** Sweep clock, passed rather than read so a sweep is reproducible. */
177
+ now: number;
178
+ /** Detached-run TTL; `detachedSince <= now - ttl` expires, INCLUSIVELY. */
179
+ detachedRunTtlMs: number;
180
+ /** Safety net per drive. Defaults to {@link DEFAULT_RUN_BUDGET_MS}. */
181
+ runBudgetMs?: number;
182
+ /** Batch cap. Defaults to {@link DEFAULT_MAX_RUNS}. */
183
+ maxRuns?: number;
184
+ /** Quiescence window; defaults to `DEFAULT_FENCE_QUIET_MS`. */
185
+ fenceQuietMs?: number;
186
+ /**
187
+ * Tear the run's sandbox down. Called ONLY after the run reached a terminal
188
+ * status, and with the ORIGINALLY LISTED record — see {@link reapDetachedRuns}.
189
+ * `sandboxReclaimer` in `reclaim.ts` is the ready-made implementation.
190
+ */
191
+ reclaim?: (record: RunRecord) => Promise<void>;
192
+ logger?: InternalLogger;
193
+ }
194
+ /**
195
+ * Read the END of a run's journal and answer whether the agent reached its
196
+ * `{"__exit":N}` sentinel. Read-only: no append, no record write, no `close()`.
197
+ *
198
+ * This is the whole reason the reaper is safe. It is the ONLY way to learn that a
199
+ * detached run is over without driving it, because the delivery log stops growing
200
+ * the moment the viewer leaves while the journal does not.
201
+ *
202
+ * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it
203
+ * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or
204
+ * a frame the provider truncated must never be read as "the agent exited".
205
+ *
206
+ * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that
207
+ * does not exist yet is indistinguishable here from one with no sentinel, and both
208
+ * mean "do not touch this run".
209
+ */
210
+ export declare function probeRunExit(input: {
211
+ handle: SandboxHandle;
212
+ runId: string;
213
+ /** Journal directory; defaults to `DEFAULT_JOURNAL_DIR`, as `journalPaths` does. */
214
+ dir?: string;
215
+ /** Tail bytes to read. Defaults to {@link DEFAULT_EXIT_PROBE_BYTES}. */
216
+ maxBytes?: number;
217
+ }): Promise<RunExitProbe>;
218
+ /**
219
+ * Sweep the detached runs a `RunStore` surfaces, saving each finished run's
220
+ * transcript and reclaiming its sandbox.
221
+ *
222
+ * A plain async function with no timer and no daemon: call it from a cron, a
223
+ * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects
224
+ * — every failure is logged and counted in the returned {@link ReapResult}.
225
+ *
226
+ * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is
227
+ * every detached run, which is the candidate set for FINALIZATION (a run that hit
228
+ * its sentinel one second after the viewer left has an unsaved transcript and
229
+ * must not wait out the TTL), and expiry is then classified in-process against
230
+ * the same inclusive cutoff. Listing twice with two TTLs would cost a second
231
+ * store round-trip to compute a subset.
232
+ *
233
+ * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be
234
+ * reaped, which answers `{ considered: 0 }` plus one log line rather than
235
+ * throwing — the same graceful degrade every other optional-method call site in
236
+ * the repo does (`store.findActiveRun?.(threadId)`).
237
+ */
238
+ export declare function reapDetachedRuns<TOffset extends string = string>(options: ReapOptions<TOffset>): Promise<ReapResult>;
@@ -0,0 +1,355 @@
1
+ import { journalExitProbeCommand, journalPaths, parseJournalExit } from "./journal.js";
2
+ import { decodeBase64Stream } from "./journal-bytes.js";
3
+ import { RunClaimLostError, RunClaimNotAcquiredError, awaitLogQuiescence, fenceDurability, fenceRunStore, withRunClaim } from "./claim.js";
4
+ import { pipeToRunLog } from "./run.js";
5
+ import { isTerminalRunStatus, requestRunCancel } from "@tanstack/ai";
6
+ //#region src/reap.ts
7
+ /**
8
+ * The sweep `RunStore.listReclaimable` was always missing a consumer for: take a
9
+ * detached run whose viewer never came back, save its transcript, terminalize its
10
+ * record, and tear its sandbox down.
11
+ *
12
+ * THE ONE RULE THAT SHAPES EVERYTHING HERE: **never drive a run to find out
13
+ * whether it finished.**
14
+ *
15
+ * The obvious design — hand the run to `pipeToRunLog` under a short
16
+ * `runBudgetMs` and see whether it terminalizes — was measured and is broken.
17
+ * `pipeToRunLog` is total by construction: it ALWAYS writes a terminal status and
18
+ * ALWAYS calls `durability.close()`. Against a run that has not finished, all
19
+ * three producer shapes are destructive:
20
+ *
21
+ * | producer's reaction to the budget signal | stored status | `close()` |
22
+ * | ---------------------------------------- | ------------- | --------- |
23
+ * | ignores it and keeps producing | `aborted` | called |
24
+ * | returns on abort (the realistic `drive`) | `aborted` | called |
25
+ * | throws an AbortError | `failed` | called |
26
+ *
27
+ * The middle row USED to read `completed`, which was the fatal one: a signal-aware
28
+ * producer exits its loop NORMALLY, and `pipeToRunLog` only checked its signal
29
+ * per chunk, so a healthy mid-flight run was recorded as `'completed'` with a
30
+ * `finishedAt` — a false transcript. That gap is fixed (`run.ts` re-checks the
31
+ * signal after the loop), so the status is now honest on all three rows. The rule
32
+ * above is UNCHANGED, because the status was never the whole harm: every row
33
+ * writes a terminal record and closes a log that commit `5a1f821c9` deliberately
34
+ * leaves OPEN for takeover (ending every attached client's stream), and a terminal
35
+ * record drops out of `listReclaimable` forever, so TTL expiry can never reclaim
36
+ * that run's sandbox. A cost leak with no recovery path. There is therefore no
37
+ * "still running" outcome in {@link ReapRunOutcome}: it is unreachable by
38
+ * construction, not merely unlikely.
39
+ *
40
+ * So sentinel-reached is detected OUT OF BAND, through the in-sandbox journal
41
+ * ({@link probeRunExit}), and `pipeToRunLog` is entered only for a run already
42
+ * KNOWN to have finished, or for one whose TTL has expired (terminal either way).
43
+ * On the FINALIZATION path `runBudgetMs` therefore degrades from a load-bearing
44
+ * mechanism into a safety net whose expiry is a genuine anomaly — see
45
+ * `'budget-exceeded'`. On the EXPIRY path it stays load-bearing: nothing polls the
46
+ * cancel this module records, so the budget is what ends the drive of an expired
47
+ * run whose agent is still producing, and its expiry there is the designed path.
48
+ *
49
+ * WHY THE PROBE IS INJECTED (`ReapOptions.hasFinished`) rather than resolved
50
+ * here, exactly like `ReapOptions.reclaim`:
51
+ *
52
+ * - It cannot read `durability.snapshot()`. After a detach nothing appends to the
53
+ * delivery log — the host that would have appended is the host that left — so
54
+ * the log is frozen at the last delivered chunk while the JOURNAL keeps
55
+ * growing. The log can only ever say "no news".
56
+ * - It cannot resolve a `SandboxHandle` either. `SandboxInstanceStore` is
57
+ * `get`/`upsert`/`delete` with no `list` (see `reclaim.ts` for why that is
58
+ * deliberate), and only the application maps a `sandboxKey` to a live handle.
59
+ *
60
+ * NEVER REJECTS. This runs from a cron, an `alarm()`, or a `waitUntil` with
61
+ * nobody to catch it, so every per-run failure is logged and folded into
62
+ * {@link ReapResult} rather than escaping.
63
+ *
64
+ * NEVER CLEARS `detachedSince`. That field is what the reaper SELECTS on, and
65
+ * `packages/ai/src/stream-to-response.ts`'s `startRunDriver` clears it because a
66
+ * real viewer stopping the TTL clock is the opposite job. Its comment there names
67
+ * borrowing that path "the single most likely bug in this phase"; clearing the
68
+ * marker would reset the TTL on every sweep and a detached run would never
69
+ * expire.
70
+ */
71
+ /**
72
+ * Safety net for a single run's drive. Not the mechanism that decides whether a
73
+ * run finished — see the module doc for why that design was rejected — so this is
74
+ * generous rather than tight: it only has to stop a drive that has genuinely
75
+ * wedged on a run the journal already said was over.
76
+ *
77
+ * On the expiry path it is not merely a net: it is what stops a still-producing
78
+ * agent, since nothing polls the cancel recorded before that drive. A caller that
79
+ * expires live agents may want a tighter value there than a finalization replay
80
+ * needs.
81
+ */
82
+ var DEFAULT_RUN_BUDGET_MS = 3e4;
83
+ /**
84
+ * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU
85
+ * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands
86
+ * would be killed mid-run rather than finishing 25 and returning; the next tick
87
+ * takes the next batch.
88
+ */
89
+ var DEFAULT_MAX_RUNS = 25;
90
+ /** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */
91
+ var DEFAULT_EXIT_PROBE_BYTES = 4096;
92
+ async function* singleValue(value) {
93
+ yield value;
94
+ }
95
+ /** Decode the base64 frame `journalExitProbeCommand` emits. */
96
+ async function decodeFrame(stdout) {
97
+ const decoder = new TextDecoder();
98
+ let text = "";
99
+ for await (const bytes of decodeBase64Stream(singleValue(stdout))) text += decoder.decode(bytes, { stream: true });
100
+ return text + decoder.decode();
101
+ }
102
+ /**
103
+ * Read the END of a run's journal and answer whether the agent reached its
104
+ * `{"__exit":N}` sentinel. Read-only: no append, no record write, no `close()`.
105
+ *
106
+ * This is the whole reason the reaper is safe. It is the ONLY way to learn that a
107
+ * detached run is over without driving it, because the delivery log stops growing
108
+ * the moment the viewer leaves while the journal does not.
109
+ *
110
+ * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it
111
+ * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or
112
+ * a frame the provider truncated must never be read as "the agent exited".
113
+ *
114
+ * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that
115
+ * does not exist yet is indistinguishable here from one with no sentinel, and both
116
+ * mean "do not touch this run".
117
+ */
118
+ async function probeRunExit(input) {
119
+ try {
120
+ const paths = journalPaths(input.runId, input.dir);
121
+ const exitCode = parseJournalExit(await decodeFrame((await input.handle.process.exec(journalExitProbeCommand(paths, input.maxBytes ?? 4096))).stdout), paths);
122
+ return exitCode === null ? { state: "producing" } : {
123
+ state: "finished",
124
+ exitCode
125
+ };
126
+ } catch (error) {
127
+ return {
128
+ state: "unknown",
129
+ error
130
+ };
131
+ }
132
+ }
133
+ /** Every outcome key present at zero, so a consumer can read any of them. */
134
+ function emptyOutcomes() {
135
+ return {
136
+ finalized: 0,
137
+ expired: 0,
138
+ producing: 0,
139
+ unknown: 0,
140
+ "budget-exceeded": 0,
141
+ "not-claimed": 0,
142
+ "reclaim-failed": 0,
143
+ failed: 0
144
+ };
145
+ }
146
+ /**
147
+ * Report through a consumer-supplied logger without letting it break the sweep.
148
+ * Mirrors `run.ts`'s `safeLog`: this module's totality must not be defeated by a
149
+ * sink that cannot serialize a thrown value.
150
+ */
151
+ function safeLog(logger, level, message, context) {
152
+ try {
153
+ if (level === "errors") logger?.errors(message, context);
154
+ else logger?.sandbox(message, context);
155
+ } catch {}
156
+ }
157
+ /** Whether a thrown value means "we do not own this run", which is normal. */
158
+ function isClaimRefusal(error) {
159
+ return error instanceof RunClaimNotAcquiredError || error instanceof RunClaimLostError;
160
+ }
161
+ /**
162
+ * Sweep ONE run. Never rejects: the caller folds the returned entry into the
163
+ * summary and moves on.
164
+ *
165
+ * The ORDER of the steps below is the contract, not an implementation detail:
166
+ *
167
+ * 1. **Classify expiry first**, because an expired run needs no probe — its
168
+ * outcome is terminal whether or not the agent finished, so a probe would only
169
+ * add a provider round-trip and a way to fail.
170
+ * 2. **Otherwise probe BEFORE touching anything.** `'producing'` and `'unknown'`
171
+ * return here, having made no claim, no append, no record write, and no
172
+ * `close()`. Driving past this point is the whole defect described in the
173
+ * module doc.
174
+ * 3. Claim, so two hosts never drive one run.
175
+ * 4. **Re-derive expiry from a record read INSIDE the lock**, and only then
176
+ * record the cancel. The listed record is stale by the time the claim is
177
+ * held, and the cancel is sticky.
178
+ * 5. Quiesce, so a predecessor still writing is observed rather than raced.
179
+ * 6. **Arm the run budget**, so it bounds the drive rather than the queue the
180
+ * two steps above stood in.
181
+ * 7. Pipe with BOTH authoritative seams fenced, mirroring `driver.ts`.
182
+ * 8. Reclaim, and ONLY once the record actually reached terminal.
183
+ */
184
+ async function reapOne(record, ctx, counters) {
185
+ const { runs, locks, logger } = ctx.options;
186
+ const { runId, threadId } = record;
187
+ try {
188
+ const expired = record.detachedSince !== void 0 && record.detachedSince <= ctx.cutoff;
189
+ let exitCode;
190
+ if (!expired) {
191
+ counters.probed += 1;
192
+ const probe = await ctx.options.hasFinished(record);
193
+ if (probe.state !== "finished") {
194
+ safeLog(logger, "sandbox", `reap: leaving run ${runId} alone`, {
195
+ runId,
196
+ state: probe.state,
197
+ ...probe.state === "unknown" && probe.error !== void 0 ? { error: probe.error } : {}
198
+ });
199
+ return {
200
+ runId,
201
+ outcome: probe.state,
202
+ ...probe.state === "unknown" && probe.error !== void 0 ? { error: probe.error } : {}
203
+ };
204
+ }
205
+ exitCode = probe.exitCode;
206
+ }
207
+ let budget;
208
+ const final = await withRunClaim({
209
+ runs,
210
+ locks,
211
+ runId,
212
+ fenceQuietMs: ctx.fenceQuietMs,
213
+ ...logger === void 0 ? {} : { logger }
214
+ }, async (claim) => {
215
+ if (expired) {
216
+ const current = await runs.get(runId);
217
+ if (current === null) throw new RunClaimNotAcquiredError(runId, "unknown");
218
+ if (current.detachedSince === void 0 || current.detachedSince > ctx.cutoff) throw new RunClaimNotAcquiredError(runId, "superseded");
219
+ await requestRunCancel(runs, runId);
220
+ }
221
+ await awaitLogQuiescence(ctx.options.durability(runId), ctx.fenceQuietMs);
222
+ budget = AbortSignal.timeout(ctx.runBudgetMs);
223
+ const signal = AbortSignal.any([claim.signal, budget]);
224
+ return pipeToRunLog(ctx.options.drive({
225
+ runId,
226
+ threadId,
227
+ signal
228
+ }), {
229
+ runs: fenceRunStore(runs, claim, { ...logger === void 0 ? {} : { logger } }),
230
+ durability: (id) => fenceDurability(ctx.options.durability(id), claim, { runs }),
231
+ runId,
232
+ threadId,
233
+ signal,
234
+ ...logger === void 0 ? {} : { logger }
235
+ });
236
+ });
237
+ const terminal = isTerminalRunStatus(final.status);
238
+ let outcome;
239
+ if ((budget?.aborted ?? false) && !expired) outcome = "budget-exceeded";
240
+ else if (!terminal) outcome = "not-claimed";
241
+ else outcome = expired ? "expired" : "finalized";
242
+ const budgetAnomaly = outcome === "budget-exceeded";
243
+ let reclaimError;
244
+ if (terminal && ctx.options.reclaim !== void 0) try {
245
+ await ctx.options.reclaim(record);
246
+ } catch (error) {
247
+ reclaimError = error;
248
+ outcome = "reclaim-failed";
249
+ safeLog(logger, "errors", `reap: reclaiming run ${runId} failed`, {
250
+ runId,
251
+ status: final.status,
252
+ error
253
+ });
254
+ }
255
+ return {
256
+ runId,
257
+ outcome,
258
+ status: final.status,
259
+ ...exitCode === void 0 ? {} : { exitCode },
260
+ ...budgetAnomaly ? { terminalizedAnyway: terminal } : {},
261
+ ...reclaimError === void 0 ? {} : { error: reclaimError }
262
+ };
263
+ } catch (error) {
264
+ if (isClaimRefusal(error)) {
265
+ safeLog(logger, "sandbox", `reap: not driving run ${runId}`, {
266
+ runId,
267
+ error
268
+ });
269
+ return {
270
+ runId,
271
+ outcome: "not-claimed",
272
+ error
273
+ };
274
+ }
275
+ safeLog(logger, "errors", `reap: sweeping run ${runId} failed`, {
276
+ runId,
277
+ error
278
+ });
279
+ return {
280
+ runId,
281
+ outcome: "failed",
282
+ error
283
+ };
284
+ }
285
+ }
286
+ /**
287
+ * Sweep the detached runs a `RunStore` surfaces, saving each finished run's
288
+ * transcript and reclaiming its sandbox.
289
+ *
290
+ * A plain async function with no timer and no daemon: call it from a cron, a
291
+ * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects
292
+ * — every failure is logged and counted in the returned {@link ReapResult}.
293
+ *
294
+ * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is
295
+ * every detached run, which is the candidate set for FINALIZATION (a run that hit
296
+ * its sentinel one second after the viewer left has an unsaved transcript and
297
+ * must not wait out the TTL), and expiry is then classified in-process against
298
+ * the same inclusive cutoff. Listing twice with two TTLs would cost a second
299
+ * store round-trip to compute a subset.
300
+ *
301
+ * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be
302
+ * reaped, which answers `{ considered: 0 }` plus one log line rather than
303
+ * throwing — the same graceful degrade every other optional-method call site in
304
+ * the repo does (`store.findActiveRun?.(threadId)`).
305
+ */
306
+ async function reapDetachedRuns(options) {
307
+ const logger = options.logger;
308
+ const outcomes = emptyOutcomes();
309
+ const entries = [];
310
+ const empty = () => ({
311
+ considered: 0,
312
+ probed: 0,
313
+ outcomes,
314
+ runs: entries
315
+ });
316
+ const list = options.runs.listReclaimable?.bind(options.runs);
317
+ if (list === void 0) {
318
+ safeLog(logger, "sandbox", "reap: the run store does not implement listReclaimable; nothing to sweep", {});
319
+ return empty();
320
+ }
321
+ let candidates;
322
+ try {
323
+ candidates = await list({
324
+ now: options.now,
325
+ ttlMs: 0
326
+ });
327
+ } catch (error) {
328
+ safeLog(logger, "errors", "reap: listing reclaimable runs failed", { error });
329
+ return empty();
330
+ }
331
+ const maxRuns = Math.max(0, Math.trunc(options.maxRuns ?? 25));
332
+ const batch = candidates.slice(0, maxRuns);
333
+ const ctx = {
334
+ options,
335
+ runBudgetMs: options.runBudgetMs ?? 3e4,
336
+ fenceQuietMs: options.fenceQuietMs ?? 5e3,
337
+ cutoff: options.now - options.detachedRunTtlMs
338
+ };
339
+ const counters = { probed: 0 };
340
+ for (const record of batch) {
341
+ const entry = await reapOne(record, ctx, counters);
342
+ outcomes[entry.outcome] += 1;
343
+ entries.push(entry);
344
+ }
345
+ return {
346
+ considered: batch.length,
347
+ probed: counters.probed,
348
+ outcomes,
349
+ runs: entries
350
+ };
351
+ }
352
+ //#endregion
353
+ export { DEFAULT_EXIT_PROBE_BYTES, DEFAULT_MAX_RUNS, DEFAULT_RUN_BUDGET_MS, probeRunExit, reapDetachedRuns };
354
+
355
+ //# sourceMappingURL=reap.js.map