@tanstack/ai-sandbox 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
package/src/run.ts CHANGED
@@ -1,18 +1,37 @@
1
1
  /**
2
2
  * The "run driver" for the inverted/serverless sandbox model: pump a `chat()`
3
- * stream into a {@link RunEventLog} so a trigger can return immediately while a
4
- * durable orchestrator drives the run and clients tail from a cursor.
3
+ * stream into core's two durable seams — a {@link RunStore} for the run's
4
+ * lifecycle record and a {@link StreamDurability} for its event log — so a
5
+ * trigger can return immediately while a durable orchestrator drives the run
6
+ * and clients tail from an opaque offset.
5
7
  *
6
8
  * The key inversion vs. a classic request/response handler: there is no caller
7
- * holding the stream open, so nothing to throw an error *back to*. The log is
8
- * the only channel — every chunk (including a terminal {@link EventType.RUN_ERROR})
9
- * is persisted under a `seq`, and a thrown stream error is recorded as a
10
- * synthesized `RUN_ERROR` event plus the record's `error` field. Tailing clients
11
- * therefore always observe failures; {@link pipeToRunLog} never rejects.
9
+ * holding the stream open, so nothing to throw an error *back to*. The event log
10
+ * is the only channel — every chunk (including a terminal
11
+ * {@link EventType.RUN_ERROR}) is appended and assigned a resumable offset, and
12
+ * a thrown stream error is recorded as a synthesized `RUN_ERROR` event plus the
13
+ * record's `error` field. Tailing clients therefore always observe failures;
14
+ * {@link pipeToRunLog} never rejects.
15
+ *
16
+ * "Never rejects" is load-bearing rather than aspirational: {@link RunController}
17
+ * consumes the returned promise fire-and-forget, so a rejection would be an
18
+ * unhandled rejection (process-fatal on modern Node, instance-fatal inside a
19
+ * Durable Object) with nobody to report it to. Every store/log call is therefore
20
+ * individually guarded, and because absorbing a failure silently in the one
21
+ * module whose premise is that nobody is listening would make the failure
22
+ * invisible, each guard reports through the optional {@link RunDeps.logger}.
12
23
  */
13
24
  import { EventType } from '@tanstack/ai'
14
- import type { StreamChunk } from '@tanstack/ai'
15
- import type { RunError, RunEvent, RunEventLog, RunRecord } from './run-log'
25
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
26
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
27
+ import type {
28
+ RunError,
29
+ RunRecord,
30
+ RunStore,
31
+ StreamChunk,
32
+ StreamDurability,
33
+ TerminalRunStatus,
34
+ } from '@tanstack/ai'
16
35
 
17
36
  /** Whether a chunk is the terminal error event the chat engine emits. */
18
37
  function isRunErrorChunk(
@@ -21,94 +40,336 @@ function isRunErrorChunk(
21
40
  return chunk.type === EventType.RUN_ERROR
22
41
  }
23
42
 
24
- /** Pull `{ message, code }` off a RUN_ERROR chunk for the run record. */
25
- function runErrorFromChunk(
26
- chunk: StreamChunk & { message: string; code?: string },
27
- ): RunError {
28
- return chunk.code !== undefined
29
- ? { message: chunk.message, code: chunk.code }
30
- : { message: chunk.message }
43
+ /**
44
+ * Narrow a thrown value (or a `RUN_ERROR` chunk's payload) to the record's
45
+ * structured error, keeping the provider's `code` when it supplies one: a bare
46
+ * message is prose that changes between model versions, while `code` is what a
47
+ * consumer branches on to retry, escalate, or show specific UI.
48
+ */
49
+ function toRunError(error: unknown): RunError {
50
+ const payload = toRunErrorPayload(error)
51
+ return {
52
+ message: payload.message,
53
+ ...(payload.code === undefined ? {} : { code: payload.code }),
54
+ }
31
55
  }
32
56
 
33
- /** Render an unknown thrown value as a stable error message. */
34
- function messageOf(error: unknown): string {
35
- return error instanceof Error ? error.message : String(error)
57
+ /**
58
+ * Fold a secondary failure into the primary error, mirroring `combineFailures`
59
+ * in `packages/ai/src/stream-to-response.ts`: the primary cause stays first and
60
+ * keeps its `code`, and the phase that produced the secondary failure is named.
61
+ * The secondary must never *replace* the primary: the provider's error is what
62
+ * an operator needs, and a failure while recording it is the lesser fact.
63
+ */
64
+ function withSecondaryFailure(
65
+ primary: RunError,
66
+ secondary: unknown,
67
+ phase: string,
68
+ ): RunError {
69
+ return {
70
+ ...primary,
71
+ message: `${primary.message}; ${phase}: ${toRunError(secondary).message}`,
72
+ }
36
73
  }
37
74
 
38
75
  /** Build the synthetic RUN_ERROR chunk appended when the stream throws. */
39
- function syntheticRunError(message: string): StreamChunk {
40
- const chunk: { type: EventType.RUN_ERROR; message: string } = {
76
+ function syntheticRunError(error: RunError): StreamChunk {
77
+ const chunk: { type: EventType.RUN_ERROR; message: string; code?: string } = {
41
78
  type: EventType.RUN_ERROR,
42
- message,
79
+ message: error.message,
80
+ ...(error.code === undefined ? {} : { code: error.code }),
43
81
  }
44
82
  return chunk
45
83
  }
46
84
 
47
- export interface PipeToRunLogOptions {
48
- log: RunEventLog
85
+ /**
86
+ * The two durable seams a run driver needs: lifecycle record + event log.
87
+ *
88
+ * Generic in the log's offset type, and DEFAULTED to `string` so every existing
89
+ * call site keeps compiling unchanged. The parameter is not decoration: a
90
+ * backend that brands its cursors — `@tanstack/ai-durable-stream`'s
91
+ * `durableStream` returns `StreamDurability<DurableStreamOffset>` — is NOT
92
+ * assignable to `StreamDurability<string>`, because `read` takes an offset and
93
+ * is therefore contravariant in it. Hardcoding the default here made the
94
+ * production multi-host backend unusable without a cast (see
95
+ * `tests/offset-generics.test-d.ts`).
96
+ */
97
+ export interface RunDeps<TOffset extends string = string> {
98
+ /** Run lifecycle record (status, thread, timings). */
99
+ runs: RunStore
100
+ /**
101
+ * Per-run delivery-durable event log the run's chunks are appended to.
102
+ *
103
+ * A FACTORY, not an instance, and that is load-bearing rather than stylistic.
104
+ * A `StreamDurability` is bound to one run — `memoryStream(request)` resolves
105
+ * its `runId` from the request, and a backend adapter's offsets embed a cursor
106
+ * into one log. Holding a single instance made two failures reachable:
107
+ *
108
+ * - **Silent mis-binding at concurrency 1.** `start({ runId })` accepted an
109
+ * arbitrary id while the instance was bound to another, writing the record
110
+ * under one id and the events under another with no error raised. Resolving
111
+ * the log FROM the `runId` makes that unrepresentable.
112
+ * - **Cross-talk at concurrency > 1.** Parallel runs interleaved their chunks
113
+ * into one log, and whichever finished first called `close()` and
114
+ * terminalized every other run's stream.
115
+ *
116
+ * Called exactly once per run, at the start of {@link pipeToRunLog}. An
117
+ * implementation MUST return the same instance for the same `runId` within a
118
+ * process if it wants `snapshot()` to see its own appends.
119
+ */
120
+ durability: (runId: string) => StreamDurability<TOffset>
121
+ /**
122
+ * Optional sink for failures this driver absorbs rather than rejecting with.
123
+ * A detached run has no caller to receive an error, so without a logger a
124
+ * failing store or event log is invisible to an operator. Same
125
+ * `logger?.errors(...)` contract core uses in `stream-to-response.ts`.
126
+ */
127
+ logger?: InternalLogger
128
+ }
129
+
130
+ export interface PipeToRunLogOptions<
131
+ TOffset extends string = string,
132
+ > extends RunDeps<TOffset> {
49
133
  runId: string
50
- threadId?: string
134
+ threadId: string
51
135
  /** Abort consumption mid-stream; the run finishes as `aborted`. */
52
136
  signal?: AbortSignal
53
137
  }
54
138
 
139
+ /** Everything {@link finish} needs, including the fields it rebuilds a record from. */
140
+ interface FinishContext {
141
+ runs: RunStore
142
+ /**
143
+ * `close()` only — see {@link finish}. Narrowed to that one member rather than
144
+ * threading `TOffset` through here, because `close` is the sole method this
145
+ * context touches and it is offset-free, so a `Pick` accepts a log at ANY
146
+ * offset instantiation without making `finish` generic for nothing.
147
+ */
148
+ durability: Pick<StreamDurability, 'close'>
149
+ runId: string
150
+ threadId: string
151
+ startedAt: number
152
+ logger?: InternalLogger
153
+ }
154
+
155
+ /**
156
+ * Report through a consumer-supplied logger without letting it break the
157
+ * caller. Every logger call in this module sits inside a `catch` body, so an
158
+ * throwing sink would escape that body and defeat the totality the guards
159
+ * exist to provide. Swallowing here is deliberate: there is no second channel
160
+ * left to report a reporting failure on.
161
+ */
162
+ function safeLog(
163
+ logger: InternalLogger | undefined,
164
+ message: string,
165
+ context: Record<string, unknown>,
166
+ ): void {
167
+ try {
168
+ logger?.errors(message, context)
169
+ } catch {
170
+ // Intentionally empty: see above.
171
+ }
172
+ }
173
+
174
+ /**
175
+ * Record the terminal status, terminalize the event log, and answer with the
176
+ * run's final record.
177
+ *
178
+ * TOTAL BY CONSTRUCTION: every step is individually guarded, so this never
179
+ * throws and never rejects. Two consequences the guards buy:
180
+ *
181
+ * - `durability.close()` runs on EVERY exit path, including a failed `update`.
182
+ * Skipping it would wedge the record at `running` *and* park every live
183
+ * tailer forever, because a durability `read` only ends once the log closes.
184
+ * - The re-read of the record is best effort. An eventually-consistent or
185
+ * read-replica store may answer `null` for a run that was just driven, which
186
+ * must not turn a successful run into a rejection; the locally rebuilt record
187
+ * is returned instead. It is also preferred outright when `update` failed,
188
+ * since the store then still holds the stale `running` row.
189
+ *
190
+ * THE TERMINAL WRITE IS NOT GUARANTEED TO LAND, and this function deliberately
191
+ * does not check whether it did. Under `sandboxRunDriver` the `runs` handed in is
192
+ * `fenceRunStore`d (`src/claim.ts`), which SUPPRESSES a terminal write — resolving
193
+ * without writing — when the driver has lost its claim, because a host that no
194
+ * longer owns the run must not declare it over while the successor is streaming
195
+ * it. From here that is indistinguishable from a successful write, on purpose:
196
+ * the epoch belongs to the claim module, not to this generic driver, and the
197
+ * re-read below then answers with the successor's live record, which is the
198
+ * truthful thing to resolve with. A driver wired without a claim (a plain
199
+ * `pipeToRunLog` call) is unfenced and always writes.
200
+ */
201
+ async function finish(
202
+ ctx: FinishContext,
203
+ status: TerminalRunStatus,
204
+ error?: RunError,
205
+ ): Promise<RunRecord> {
206
+ // NOTE: every logger call below goes through `safeLog`. The logger is
207
+ // consumer-supplied and is handed arbitrary thrown values, so a sink that
208
+ // cannot serialize one (a circular payload, say) would otherwise throw from
209
+ // inside a `catch` body and escape, skipping `durability.close()` and leaving
210
+ // the run wedged at `'running'` with live tailers parked. The reporting
211
+ // channel must never be able to break the guarantee it exists to report on.
212
+ const { runs, durability, runId, logger } = ctx
213
+ const finishedAt = Date.now()
214
+ const patch = {
215
+ status,
216
+ finishedAt,
217
+ ...(error === undefined ? {} : { error }),
218
+ }
219
+ const local: RunRecord = {
220
+ runId,
221
+ threadId: ctx.threadId,
222
+ startedAt: ctx.startedAt,
223
+ ...patch,
224
+ }
225
+
226
+ let recorded = true
227
+ try {
228
+ await runs.update(runId, patch)
229
+ } catch (updateError) {
230
+ recorded = false
231
+ safeLog(logger, 'run: recording the terminal run record failed', {
232
+ runId,
233
+ status,
234
+ error: updateError,
235
+ })
236
+ }
237
+
238
+ try {
239
+ await durability.close()
240
+ } catch (closeError) {
241
+ safeLog(logger, 'run: closing the run event log failed', {
242
+ runId,
243
+ status,
244
+ error: closeError,
245
+ })
246
+ }
247
+
248
+ if (!recorded) return local
249
+
250
+ try {
251
+ const latest = await runs.get(runId)
252
+ if (latest !== null) return latest
253
+ safeLog(logger, 'run: record vanished before the terminal re-read', {
254
+ runId,
255
+ status,
256
+ })
257
+ } catch (getError) {
258
+ safeLog(logger, 'run: re-reading the terminal run record failed', {
259
+ runId,
260
+ status,
261
+ error: getError,
262
+ })
263
+ }
264
+ return local
265
+ }
266
+
55
267
  /**
56
268
  * Open the run, append every chunk from `stream`, and finish with the right
57
269
  * terminal status. Resolves with the final {@link RunRecord} and never rejects:
58
- * a thrown stream error is surfaced as a `RUN_ERROR` event + the record's
59
- * `error`, which is what tailing clients see.
270
+ * a thrown stream error is surfaced as a `RUN_ERROR` event plus the record's
271
+ * `error`, which is what tailing clients see. A store or event-log failure
272
+ * along the way is logged through {@link RunDeps.logger} and still terminalizes
273
+ * the run rather than escaping to a caller that does not exist.
60
274
  *
61
- * - normal completion → `finish('done')`
62
- * - a `RUN_ERROR` chunk → append it, then `finish('error', { message, code })`
63
- * - the stream throws → append a synthesized `RUN_ERROR`, then `finish('error')`
64
- * - `signal` aborts mid-stream → stop consuming, `finish('aborted')`
275
+ * - normal completion → `completed`
276
+ * - a `RUN_ERROR` chunk → append it, then `failed`
277
+ * - the stream throws → append a synthesized `RUN_ERROR`, then `failed`
278
+ * - `signal` aborts at ANY point before the stream ends → `aborted`, whether the
279
+ * producer keeps yielding, ends its stream, or is never asked for another
280
+ * chunk. An abort outranks a clean exit: the run did not complete.
65
281
  */
66
- export async function pipeToRunLog(
282
+ export async function pipeToRunLog<TOffset extends string = string>(
67
283
  stream: AsyncIterable<StreamChunk>,
68
- opts: PipeToRunLogOptions,
284
+ opts: PipeToRunLogOptions<TOffset>,
69
285
  ): Promise<RunRecord> {
70
- const { log, runId, threadId, signal } = opts
71
- await log.open(threadId !== undefined ? { runId, threadId } : { runId })
72
- if (signal?.aborted) {
73
- await log.finish(runId, 'aborted')
74
- return reread(log, runId)
286
+ const { runs, runId, threadId, signal, logger } = opts
287
+ // Resolved ONCE, from the runId being driven. Everything below — including
288
+ // `finish`'s `close()` — uses this one instance, so a factory that mints a
289
+ // fresh log per call cannot split one run across two logs, and the log can
290
+ // never belong to a run other than the one whose record is being written.
291
+ const durability = opts.durability(runId)
292
+ const ctx: FinishContext = {
293
+ runs,
294
+ durability,
295
+ runId,
296
+ threadId,
297
+ startedAt: Date.now(),
298
+ ...(logger === undefined ? {} : { logger }),
75
299
  }
76
300
 
77
301
  try {
302
+ // Inside the `try` so a store failure at creation is handled like any
303
+ // other: recorded as a failed run with a terminalized log, not rejected.
304
+ await runs.createOrResume({ runId, threadId, startedAt: ctx.startedAt })
305
+ if (signal?.aborted) return finish(ctx, 'aborted')
306
+
78
307
  for await (const chunk of stream) {
79
- if (signal?.aborted) {
80
- await log.finish(runId, 'aborted')
81
- return reread(log, runId)
82
- }
83
- await log.append(runId, chunk)
308
+ if (signal?.aborted) return finish(ctx, 'aborted')
309
+ await durability.append([chunk])
84
310
  if (isRunErrorChunk(chunk)) {
85
- await log.finish(runId, 'error', runErrorFromChunk(chunk))
86
- return reread(log, runId)
311
+ return finish(
312
+ ctx,
313
+ 'failed',
314
+ toRunError({ message: chunk.message, code: chunk.code }),
315
+ )
87
316
  }
88
317
  }
89
- } catch (error) {
318
+ } catch (streamError) {
90
319
  // Detached run: no caller to throw to. Record the failure in the log so
91
320
  // tailing clients observe it, then return — do NOT rethrow.
92
- const message = messageOf(error)
93
- await log.append(runId, syntheticRunError(message))
94
- await log.finish(runId, 'error', { message })
95
- return reread(log, runId)
321
+ let recorded = toRunError(streamError)
322
+ // Deliberately not "the stream failed": `runs.createOrResume` above is
323
+ // inside this `try`, so an operator reading a wedged run must not be told
324
+ // the provider stream broke when the store never let the run start.
325
+ safeLog(logger, 'run: the run failed before completing', {
326
+ runId,
327
+ error: streamError,
328
+ })
329
+ try {
330
+ await durability.append([syntheticRunError(recorded)])
331
+ } catch (appendError) {
332
+ // The recovery append is itself a failure path. It must not destroy the
333
+ // cause it was recording, so the provider's error stays primary and this
334
+ // secondary failure is merged in and logged separately. When the cause IS
335
+ // a lost claim, this append is refused too and the `finish` below writes
336
+ // nothing either — a fenced store suppresses the terminal record for the
337
+ // same reason the fenced log refused the chunk (see `finish`).
338
+ const phase = 'appending the synthesized RUN_ERROR failed'
339
+ safeLog(logger, `run: ${phase}`, { runId, error: appendError })
340
+ recorded = withSecondaryFailure(recorded, appendError, phase)
341
+ }
342
+ return finish(ctx, 'failed', recorded)
96
343
  }
97
344
 
98
- await log.finish(runId, 'done')
99
- return reread(log, runId)
100
- }
101
-
102
- /** Re-read the now-terminal record; the run was just driven, so it must exist. */
103
- async function reread(log: RunEventLog, runId: string): Promise<RunRecord> {
104
- const latest = await log.get(runId)
105
- if (!latest) throw new Error(`run: record for "${runId}" vanished mid-run`)
106
- return latest
345
+ // RE-CHECKED after the loop, and this is the common shape rather than the
346
+ // exotic one. The in-loop check only fires if the producer yields at least
347
+ // once MORE after the abort; two ways past it are routine:
348
+ //
349
+ // - The producer is signal-aware and reacts by ENDING its stream. `chat()`
350
+ // does exactly this, so the loop exits NORMALLY.
351
+ // - The abort lands BETWEEN two chunks, while the loop is suspended on a
352
+ // producer that then finishes on its own.
353
+ //
354
+ // Falling through to `'completed'` in either case is a false transcript, not
355
+ // a cosmetic mislabel: it was measured on the reaper's TTL-expiry path, where
356
+ // a run the reaper had force-expired — and whose sandbox it had already
357
+ // destroyed — was recorded as having completed successfully. Any caller whose
358
+ // producer ends its stream on abort reaches the same gap, a takeover that
359
+ // loses its claim mid-drive included, which is why the check belongs here and
360
+ // not in one caller.
361
+ //
362
+ // A producer that THREW on the abort is deliberately untouched: it returned
363
+ // from inside the `catch` above as `'failed'`, because a thrown value is a
364
+ // reported failure a tailing client must be shown, and this driver's log is
365
+ // that client's only channel.
366
+ if (signal?.aborted) return finish(ctx, 'aborted')
367
+ return finish(ctx, 'completed')
107
368
  }
108
369
 
109
370
  export interface RunControllerStartInput {
110
371
  runId: string
111
- threadId?: string
372
+ threadId: string
112
373
  stream: AsyncIterable<StreamChunk>
113
374
  /** Abort consumption mid-stream; the run finishes as `aborted`. */
114
375
  signal?: AbortSignal
@@ -121,15 +382,23 @@ export interface RunHandle {
121
382
  }
122
383
 
123
384
  /**
124
- * Thin orchestration helper over a {@link RunEventLog}: fire-and-track a run via
125
- * {@link pipeToRunLog}, tail it from a cursor, and `drain()` all in-flight runs
385
+ * Thin orchestration helper over {@link RunDeps}: fire-and-track a run via
386
+ * {@link pipeToRunLog}, tail one run by id, and `drain()` all in-flight runs
126
387
  * (e.g. inside a `ctx.waitUntil`). Holds no run state of its own beyond the set
127
388
  * of currently in-flight `done` promises.
389
+ *
390
+ * Safe for concurrent runs. {@link RunDeps.durability} is a per-run factory, so
391
+ * each run appends to its own log and no run's `close()` terminalizes another's.
392
+ * The identity trap this class used to document — `start({ runId })` writing the
393
+ * lifecycle record under one id and the events under another, silently and at
394
+ * concurrency 1 — is unrepresentable now that the log is resolved FROM the
395
+ * `runId`. Every method is keyed by run accordingly: `attach(runId, …)` and
396
+ * `status(runId)` no longer disagree about whether the surface is per-run.
128
397
  */
129
- export class RunController {
398
+ export class RunController<TOffset extends string = string> {
130
399
  private readonly inFlight = new Set<Promise<RunRecord>>()
131
400
 
132
- constructor(private readonly log: RunEventLog) {}
401
+ constructor(private readonly deps: RunDeps<TOffset>) {}
133
402
 
134
403
  /**
135
404
  * Kick off `pipeToRunLog` without awaiting it and return the `runId`
@@ -137,31 +406,52 @@ export class RunController {
137
406
  */
138
407
  start(input: RunControllerStartInput): RunHandle {
139
408
  const done = pipeToRunLog(input.stream, {
140
- log: this.log,
409
+ ...this.deps,
141
410
  runId: input.runId,
142
- ...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
411
+ threadId: input.threadId,
143
412
  ...(input.signal !== undefined ? { signal: input.signal } : {}),
144
413
  })
145
414
  this.inFlight.add(done)
146
- void done.finally(() => this.inFlight.delete(done))
415
+ // Two-argument `then`, deliberately NOT `.finally`: `.finally` returns a new
416
+ // promise that adopts any rejection, and discarding that promise would make
417
+ // the rejection unhandled (fatal on modern Node defaults, and it kills the
418
+ // instance inside a Durable Object). Handling both outcomes here means the
419
+ // derived promise always settles fulfilled, so nothing is left unhandled
420
+ // even if `pipeToRunLog`'s "never rejects" contract is ever broken.
421
+ const forget = (): void => void this.inFlight.delete(done)
422
+ void done.then(forget, forget)
147
423
  return { runId: input.runId, done }
148
424
  }
149
425
 
150
- /** Resumable client tail — replay from `fromSeq`, then live-tail to terminal. */
426
+ /**
427
+ * Resumable client tail for ONE run — replay from `fromOffset`, then
428
+ * live-tail. Takes `runId` because the log it reads is per-run; the old
429
+ * `attach(fromOffset)` signature advertised a multi-run surface the type could
430
+ * not deliver.
431
+ */
151
432
  attach(
152
433
  runId: string,
153
- opts?: { fromSeq?: number; signal?: AbortSignal },
154
- ): AsyncIterable<RunEvent> {
155
- return this.log.read(runId, opts)
434
+ fromOffset: TOffset,
435
+ signal?: AbortSignal,
436
+ ): AsyncIterable<{ offset: TOffset; chunk: StreamChunk }> {
437
+ return this.deps.durability(runId).read(fromOffset, signal)
156
438
  }
157
439
 
158
- /** Current run record, or null if the run is unknown. */
440
+ /** Current run record, or null when the run is unknown. */
159
441
  status(runId: string): Promise<RunRecord | null> {
160
- return this.log.get(runId)
442
+ return this.deps.runs.get(runId)
161
443
  }
162
444
 
163
- /** Await every currently in-flight run's `done` promise. */
445
+ /**
446
+ * Await every currently in-flight run's `done` promise.
447
+ *
448
+ * Uses `allSettled` rather than `all` because this is typically awaited
449
+ * inside a `ctx.waitUntil`: `all` would reject on the first failure, abandon
450
+ * the wait on every other run, and surface that rejection to the platform.
451
+ * Draining is about keeping the isolate alive until the runs settle; each
452
+ * run's own outcome is already recorded in its record and log.
453
+ */
164
454
  async drain(): Promise<void> {
165
- await Promise.all([...this.inFlight])
455
+ await Promise.allSettled([...this.inFlight])
166
456
  }
167
457
  }