@tanstack/ai-sandbox-cloudflare 0.3.12 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,6 +30,7 @@ import {
30
30
  withSandbox,
31
31
  } from '@tanstack/ai-sandbox'
32
32
  import { SandboxCoordinator, resolveBridgeOrigin } from './coordinator'
33
+ import { runWithCallbackActivity } from './coordinator-callbacks'
33
34
  import { timingSafeBearerEqualWeb } from './web-crypto'
34
35
  import type { StartRunInput } from './coordinator'
35
36
  import type {
@@ -233,21 +234,29 @@ export abstract class ChatSandboxCoordinator<
233
234
  ) {
234
235
  return new Response('unauthorized', { status: 401 })
235
236
  }
236
- let message: unknown
237
- try {
238
- message = await request.json()
239
- } catch {
240
- // A malformed body must still produce a valid JSON-RPC error so the agent's
241
- // MCP client can react, rather than an opaque DO 500 that can wedge the run.
242
- return this.jsonResponse({
243
- jsonrpc: '2.0',
244
- id: null,
245
- error: { code: -32700, message: 'Parse error' },
246
- })
247
- }
248
- const reply = await handleBridgeJsonRpc(bridge.core, message)
249
- // A notification (no id) yields null → MCP expects an empty 202 ack.
250
- if (reply === null) return new Response(null, { status: 202 })
251
- return this.jsonResponse(reply)
237
+ return runWithCallbackActivity(
238
+ this,
239
+ runId,
240
+ (id) => this.log.touch(id),
241
+ async () => {
242
+ let message: unknown
243
+ try {
244
+ message = await request.json()
245
+ } catch {
246
+ // A malformed body must still produce a valid JSON-RPC error so the
247
+ // agent's MCP client can react, rather than an opaque DO 500 that can
248
+ // wedge the run.
249
+ return this.jsonResponse({
250
+ jsonrpc: '2.0',
251
+ id: null,
252
+ error: { code: -32700, message: 'Parse error' },
253
+ })
254
+ }
255
+ const reply = await handleBridgeJsonRpc(bridge.core, message)
256
+ // A notification (no id) yields null → MCP expects an empty 202 ack.
257
+ if (reply === null) return new Response(null, { status: 202 })
258
+ return this.jsonResponse(reply)
259
+ },
260
+ )
252
261
  }
253
262
  }
@@ -34,6 +34,7 @@ import {
34
34
  } from '@tanstack/ai-sandbox'
35
35
  import { getSandbox } from '@cloudflare/sandbox'
36
36
  import { SandboxCoordinator, resolveBridgeOrigin } from './coordinator'
37
+ import { runWithCallbackActivity } from './coordinator-callbacks'
37
38
  import { timingSafeBearerEqualWeb } from './web-crypto'
38
39
  import type { StartRunInput } from './coordinator'
39
40
  import type { ContainerRunRequest, HarnessId } from './protocol'
@@ -409,29 +410,41 @@ export abstract class ContainerSandboxCoordinator<
409
410
  ) {
410
411
  return new Response('unauthorized', { status: 401 })
411
412
  }
412
- let payload: unknown
413
- try {
414
- payload = await request.json()
415
- } catch {
416
- return this.jsonResponse({ error: 'body must be valid JSON' }, 400)
417
- }
418
- if (!isToolExecRequest(payload)) {
419
- return this.jsonResponse({ error: 'body must be { name, args }' }, 400)
420
- }
421
- try {
422
- const result = await executeHostTool(
423
- state.hostTools,
424
- payload.name,
425
- payload.args,
426
- {
427
- ...(state.context !== undefined ? { context: state.context } : {}),
428
- signal: state.abort.signal,
429
- },
430
- )
431
- return this.jsonResponse({ result })
432
- } catch (error) {
433
- const message = error instanceof Error ? error.message : String(error)
434
- return this.jsonResponse({ error: message }, 500)
435
- }
413
+ return runWithCallbackActivity(
414
+ this,
415
+ runId,
416
+ (id) => this.log.touch(id),
417
+ async () => {
418
+ let payload: unknown
419
+ try {
420
+ payload = await request.json()
421
+ } catch {
422
+ return this.jsonResponse({ error: 'body must be valid JSON' }, 400)
423
+ }
424
+ if (!isToolExecRequest(payload)) {
425
+ return this.jsonResponse(
426
+ { error: 'body must be { name, args }' },
427
+ 400,
428
+ )
429
+ }
430
+ try {
431
+ const result = await executeHostTool(
432
+ state.hostTools,
433
+ payload.name,
434
+ payload.args,
435
+ {
436
+ ...(state.context !== undefined
437
+ ? { context: state.context }
438
+ : {}),
439
+ signal: state.abort.signal,
440
+ },
441
+ )
442
+ return this.jsonResponse({ result })
443
+ } catch (error) {
444
+ const message = error instanceof Error ? error.message : String(error)
445
+ return this.jsonResponse({ error: message }, 500)
446
+ }
447
+ },
448
+ )
436
449
  }
437
450
  }
@@ -0,0 +1,58 @@
1
+ type TouchRun = (runId: string) => Promise<void>
2
+
3
+ const inFlightCallbacks = new WeakMap<object, Map<string, number>>()
4
+
5
+ function increment(owner: object, runId: string): void {
6
+ let runs = inFlightCallbacks.get(owner)
7
+ if (!runs) {
8
+ runs = new Map()
9
+ inFlightCallbacks.set(owner, runs)
10
+ }
11
+ runs.set(runId, (runs.get(runId) ?? 0) + 1)
12
+ }
13
+
14
+ function decrement(owner: object, runId: string): void {
15
+ const runs = inFlightCallbacks.get(owner)
16
+ if (!runs) return
17
+ const count = runs.get(runId) ?? 0
18
+ if (count > 1) {
19
+ runs.set(runId, count - 1)
20
+ return
21
+ }
22
+ runs.delete(runId)
23
+ if (runs.size === 0) inFlightCallbacks.delete(owner)
24
+ }
25
+
26
+ export function hasInFlightCallback(owner: object, runId: string): boolean {
27
+ return (inFlightCallbacks.get(owner)?.get(runId) ?? 0) > 0
28
+ }
29
+
30
+ export async function runWithCallbackActivity<T>(
31
+ owner: object,
32
+ runId: string,
33
+ touch: TouchRun,
34
+ operation: () => Promise<T>,
35
+ ): Promise<T> {
36
+ increment(owner, runId)
37
+ try {
38
+ // Failure here proves liveness could not be persisted, so do not execute the
39
+ // callback operation.
40
+ await touch(runId)
41
+ try {
42
+ return await operation()
43
+ } finally {
44
+ // This is best-effort bookkeeping after an operation has produced its
45
+ // result or error. It must never replace that original outcome.
46
+ try {
47
+ await touch(runId)
48
+ } catch (error) {
49
+ console.error(
50
+ `[sandbox-coordinator] completion activity touch failed for run ${runId}:`,
51
+ error,
52
+ )
53
+ }
54
+ }
55
+ } finally {
56
+ decrement(owner, runId)
57
+ }
58
+ }
@@ -31,21 +31,30 @@ import { EventType, isTerminalRunStatus } from '@tanstack/ai'
31
31
  // migrated on read; see './run-log').
32
32
  import { RunController } from '@tanstack/ai-sandbox'
33
33
  import { runLogStore, runLogStream } from './durability'
34
+ import { hasInFlightCallback } from './coordinator-callbacks'
34
35
  import { DurableObjectRunEventLog } from './run-log-do'
35
36
  import type { ModelMessage, StreamChunk } from '@tanstack/ai'
36
37
  import type { RunLogRecord } from './run-log'
37
38
 
38
- /** Re-arm window for the liveness watchdog while a run is in flight (ms). */
39
+ /** Upper bound on the watchdog check interval while a run is in flight (ms). */
39
40
  const WATCHDOG_MS = 30_000
40
41
 
41
- /**
42
- * How long a non-terminal run may go without ANY new event before the watchdog
43
- * presumes the orchestrator driving it is dead (eviction that lost the
44
- * `waitUntil` promise, an uncaught fault, a hung container) and fails the run so
45
- * tailing clients stop waiting forever. Generous so a legitimately slow agent
46
- * step (a long tool call that emits no chunks) is not killed prematurely.
47
- */
48
- const WATCHDOG_STALL_MS = 5 * 60_000
42
+ /** Default permitted period without persisted run activity (ms). */
43
+ const DEFAULT_STALL_TIMEOUT_MS = 5 * 60_000
44
+
45
+ /** @internal Shared validation for direct subclasses and the eager factory path. */
46
+ export function normalizeStallTimeoutMs(
47
+ stallTimeoutMs: number | false | undefined,
48
+ ): number | false {
49
+ if (stallTimeoutMs === undefined) return DEFAULT_STALL_TIMEOUT_MS
50
+ if (stallTimeoutMs === false) return false
51
+ if (!Number.isSafeInteger(stallTimeoutMs) || stallTimeoutMs <= 0) {
52
+ throw new TypeError(
53
+ 'stallTimeoutMs must be a positive safe integer or false',
54
+ )
55
+ }
56
+ return stallTimeoutMs
57
+ }
49
58
 
50
59
  /** What the Worker hands the coordinator to start a run. */
51
60
  export interface StartRunInput {
@@ -106,9 +115,15 @@ export abstract class SandboxCoordinator<
106
115
  * running — double-delivering events and racing the persisted cursor.
107
116
  */
108
117
  private readonly pumping = new WeakSet<WebSocket>()
118
+ private readonly stallTimeoutMs: number | false
109
119
 
110
- constructor(ctx: DurableObjectState, env: TEnv) {
120
+ constructor(
121
+ ctx: DurableObjectState,
122
+ env: TEnv,
123
+ stallTimeoutMs?: number | false,
124
+ ) {
111
125
  super(ctx, env)
126
+ this.stallTimeoutMs = normalizeStallTimeoutMs(stallTimeoutMs)
112
127
  this.log = new DurableObjectRunEventLog(ctx.storage)
113
128
  this.controller = new RunController({
114
129
  runs: runLogStore(this.log),
@@ -154,7 +169,12 @@ export abstract class SandboxCoordinator<
154
169
 
155
170
  async startRun(input: StartRunInput): Promise<{ runId: string }> {
156
171
  const existing = await this.log.get(input.runId)
157
- if (existing) return { runId: input.runId } // idempotent re-trigger
172
+ if (existing) {
173
+ if (!isTerminalRunStatus(existing.status)) await this.armWatchdog()
174
+ return { runId: input.runId }
175
+ }
176
+
177
+ await this.armWatchdog()
158
178
 
159
179
  // Open the run BEFORE building the stream. `pipeToRunLog`'s never-rejects
160
180
  // guarantee only covers failures AFTER the stream is handed to it — a throw
@@ -189,7 +209,6 @@ export abstract class SandboxCoordinator<
189
209
  // running the settle hook.
190
210
  const settle = (): void => this.onRunSettled(input.runId)
191
211
  this.ctx.waitUntil(done.then(settle, settle))
192
- await this.ctx.storage.setAlarm(Date.now() + WATCHDOG_MS)
193
212
  return { runId: input.runId }
194
213
  }
195
214
 
@@ -320,41 +339,58 @@ export abstract class SandboxCoordinator<
320
339
  // ===========================================================================
321
340
 
322
341
  override async alarm(): Promise<void> {
342
+ // A previously configured alarm may still be delivered once after watchdogs
343
+ // are disabled. It must self-extinguish before reading storage or entering a
344
+ // catch path that could re-arm it. Deliberately do not call deleteAlarm().
345
+ if (this.stallTimeoutMs === false) return
346
+
323
347
  try {
324
348
  // Through the log (not a raw `rec:` list) so legacy records are migrated
325
349
  // on the way out — the storage layout is the log's private concern.
326
350
  const runs = await this.log.list()
327
- const now = Date.now()
351
+ const cutoff = Date.now() - this.stallTimeoutMs
328
352
  let active = false
329
353
  for (const record of runs) {
330
354
  if (isTerminalRunStatus(record.status)) continue
331
- if (now - record.updatedAt > WATCHDOG_STALL_MS) {
332
- // No progress for too long — the driver is presumed dead. Fail the run
333
- // so tailing clients stop waiting forever (the whole point of the
334
- // watchdog; without this a stuck run sits at `running` indefinitely).
335
- await this.failStalledRun(record.runId)
336
- } else {
355
+ if (
356
+ hasInFlightCallback(this, record.runId) ||
357
+ record.updatedAt >= cutoff
358
+ ) {
337
359
  active = true
360
+ continue
338
361
  }
362
+
363
+ // Re-check and terminalize atomically against the same strict cutoff.
364
+ // A callback touch or normal completion racing this alarm wins cleanly.
365
+ const finished = await this.failStalledRun(record.runId, cutoff)
366
+ if (finished) this.onRunSettled(record.runId)
367
+ else active = true
339
368
  }
340
- if (active) await this.ctx.storage.setAlarm(Date.now() + WATCHDOG_MS)
369
+ if (active) await this.armWatchdog()
341
370
  } catch (error) {
342
371
  // Never let the watchdog die silently: a transient storage error must not
343
372
  // permanently disable liveness detection. Re-arm and try again next tick.
344
373
  console.error('[sandbox-coordinator] watchdog alarm failed:', error)
345
- await this.ctx.storage.setAlarm(Date.now() + WATCHDOG_MS)
374
+ await this.armWatchdog()
346
375
  }
347
376
  }
348
377
 
349
- /** Mark a stalled (orchestrator-presumed-dead) run as a terminal error. */
350
- private async failStalledRun(runId: string): Promise<void> {
351
- const message = 'run watchdog: no progress; orchestrator presumed dead'
352
- try {
353
- await this.log.append(runId, { type: EventType.RUN_ERROR, message })
354
- } catch {
355
- // The run may have just reached terminal concurrently; finish is idempotent.
378
+ /** Arm without allowing a new run to postpone an earlier pending check. */
379
+ private async armWatchdog(): Promise<void> {
380
+ if (this.stallTimeoutMs === false) return
381
+ const next = Date.now() + Math.min(WATCHDOG_MS, this.stallTimeoutMs)
382
+ const pending = await this.ctx.storage.getAlarm()
383
+ if (pending === null || pending > next) {
384
+ await this.ctx.storage.setAlarm(next)
356
385
  }
357
- await this.log.finish(runId, 'failed', { message })
358
- this.onRunSettled(runId)
386
+ }
387
+
388
+ /** Mark a strictly stale run as a terminal error if it still qualifies. */
389
+ private failStalledRun(runId: string, cutoff: number): Promise<boolean> {
390
+ const message = 'run watchdog: no progress; orchestrator presumed dead'
391
+ return this.log.finishIfStale(runId, cutoff, {
392
+ type: EventType.RUN_ERROR,
393
+ message,
394
+ })
359
395
  }
360
396
  }
package/src/factory.ts CHANGED
@@ -45,7 +45,7 @@ import { cloudflareSandbox } from './provider'
45
45
  import { ChatSandboxCoordinator } from './chat-coordinator'
46
46
  import { ContainerSandboxCoordinator } from './container-coordinator'
47
47
  import { createSandboxAgentWorker } from './worker'
48
- import { resolvePreviewHost } from './coordinator'
48
+ import { normalizeStallTimeoutMs, resolvePreviewHost } from './coordinator'
49
49
  import type { ChatCoordinatorEnv, ChatRunConfig } from './chat-coordinator'
50
50
  import type {
51
51
  ContainerCoordinatorEnv,
@@ -83,6 +83,11 @@ export interface SandboxAgentEnv
83
83
  interface BaseAgentConfig<TEnv extends SandboxAgentEnv> {
84
84
  /** chat()-provided server tools, resolved per run (DO-drives: bridged over MCP). */
85
85
  tools?: (input: StartRunInput, env: TEnv) => Array<AnyTool>
86
+ /**
87
+ * Fail a run after this many milliseconds without persisted activity. Omitted
88
+ * defaults to five minutes; `false` disables the watchdog.
89
+ */
90
+ stallTimeoutMs?: number | false
86
91
  }
87
92
 
88
93
  /** DO-drives config: the DO runs `chat()` with the given adapter. */
@@ -188,11 +193,16 @@ function resolveCoordinator<TEnv extends SandboxAgentEnv>(
188
193
  export function createCloudflareSandboxAgent<
189
194
  TEnv extends SandboxAgentEnv = SandboxAgentEnv,
190
195
  >(config: CloudflareSandboxAgentConfig<TEnv>): CloudflareSandboxAgent<TEnv> {
196
+ const stallTimeoutMs = normalizeStallTimeoutMs(config.stallTimeoutMs)
191
197
  const worker = createSandboxAgentWorker<TEnv>(resolveCoordinator)
192
198
 
193
199
  if (config.mode === 'colocated') {
194
200
  const colocated = config
195
201
  class ConfiguredContainerCoordinator extends ContainerSandboxCoordinator<TEnv> {
202
+ constructor(ctx: DurableObjectState, env: TEnv) {
203
+ super(ctx, env, stallTimeoutMs)
204
+ }
205
+
196
206
  protected override config(input: StartRunInput): ContainerRunConfig {
197
207
  return {
198
208
  hostTools: colocated.tools?.(input, this.env) ?? [],
@@ -207,6 +217,10 @@ export function createCloudflareSandboxAgent<
207
217
 
208
218
  const doDrives = config
209
219
  class ConfiguredChatCoordinator extends ChatSandboxCoordinator<TEnv> {
220
+ constructor(ctx: DurableObjectState, env: TEnv) {
221
+ super(ctx, env, stallTimeoutMs)
222
+ }
223
+
210
224
  protected override config(input: StartRunInput): ChatRunConfig {
211
225
  const tools = doDrives.tools?.(input, this.env)
212
226
  return {
@@ -68,6 +68,94 @@ export const PREVIEW_GUIDANCE: string = [
68
68
  'Once it is listening, call `exposePreview` with that port, then share the URL.',
69
69
  ].join('\n')
70
70
 
71
+ const LOCAL_PROBE_TIMEOUT_MS = 5_000
72
+ // Fresh quick tunnels need a few seconds of DNS/edge propagation, hence retries.
73
+ const EDGE_PROBE_ATTEMPTS = 5
74
+ const EDGE_PROBE_BASE_DELAY_MS = 250
75
+ const EDGE_PROBE_FETCH_TIMEOUT_MS = 3_000
76
+
77
+ /**
78
+ * Probe the port INSIDE the sandbox via `containerFetch`. Returns the HTTP
79
+ * status when a listener answered (any response — 4xx/5xx included — proves one
80
+ * exists), or the failure symptom (a string) when nothing did.
81
+ */
82
+ async function localProbe(
83
+ sandbox: Sandbox,
84
+ port: number,
85
+ ): Promise<number | string> {
86
+ // A race instead of AbortSignal: signals don't serialize across the sandbox
87
+ // RPC boundary, and a lost in-flight probe response is harmless.
88
+ let timeoutId: ReturnType<typeof setTimeout> | undefined
89
+ const timeout = new Promise<never>((_, reject) => {
90
+ timeoutId = setTimeout(
91
+ () => reject(new Error(`no response within ${LOCAL_PROBE_TIMEOUT_MS}ms`)),
92
+ LOCAL_PROBE_TIMEOUT_MS,
93
+ )
94
+ })
95
+ try {
96
+ const res = await Promise.race([
97
+ sandbox.containerFetch('http://preview/', { method: 'HEAD' }, port),
98
+ timeout,
99
+ ])
100
+ return res.status
101
+ } catch (error) {
102
+ return error instanceof Error ? error.message : String(error)
103
+ } finally {
104
+ clearTimeout(timeoutId)
105
+ }
106
+ }
107
+
108
+ /**
109
+ * Probe a tunnel URL through the public edge with bounded retries. Only 502/530
110
+ * — Cloudflare's tunnel/origin-unreachable signatures — mark the URL as
111
+ * unreachable (401/403/404 prove the server answered, and `redirect: 'manual'`
112
+ * keeps a login redirect from probing some other site), and even those are
113
+ * trusted when the app answered the SAME status locally, so an app's own
114
+ * 502/530 never gets its healthy tunnel destroyed. The verdict split exists
115
+ * because only an OBSERVED non-matching 502/530 ('stale') is evidence that
116
+ * justifies destroying the tunnel; fetch exceptions ('unverified') prove
117
+ * nothing about it — the probe path itself may be what failed. A throw
118
+ * anywhere in the retry window therefore wins over an earlier 502/530: the
119
+ * window did not finish as HTTP probes, so we must not destroy.
120
+ */
121
+ async function edgeProbeFailure(
122
+ url: string,
123
+ localStatus: number,
124
+ ): Promise<{ verdict: 'stale' | 'unverified'; symptom: string } | null> {
125
+ let lastFailure = 'no response'
126
+ let verdict: 'stale' | 'unverified' = 'unverified'
127
+ let sawFetchException = false
128
+ for (let attempt = 0; attempt < EDGE_PROBE_ATTEMPTS; attempt += 1) {
129
+ if (attempt > 0) {
130
+ await new Promise((resolve) =>
131
+ setTimeout(resolve, EDGE_PROBE_BASE_DELAY_MS * 2 ** (attempt - 1)),
132
+ )
133
+ }
134
+ try {
135
+ const res = await fetch(url, {
136
+ method: 'HEAD',
137
+ redirect: 'manual',
138
+ signal: AbortSignal.timeout(EDGE_PROBE_FETCH_TIMEOUT_MS),
139
+ })
140
+ if (
141
+ (res.status !== 502 && res.status !== 530) ||
142
+ res.status === localStatus
143
+ ) {
144
+ return null
145
+ }
146
+ lastFailure = `HTTP ${res.status}`
147
+ verdict = 'stale'
148
+ } catch (error) {
149
+ sawFetchException = true
150
+ lastFailure = error instanceof Error ? error.message : String(error)
151
+ }
152
+ }
153
+ return {
154
+ verdict: sawFetchException ? 'unverified' : verdict,
155
+ symptom: lastFailure,
156
+ }
157
+ }
158
+
71
159
  /**
72
160
  * Build the `exposePreview` server tool for one run. Starting a tunnel is a
73
161
  * HOST-side call on the Sandbox DO stub, so an in-sandbox agent cannot make it from
@@ -100,11 +188,55 @@ export function exposePreviewTool(input: StartRunInput, env: PreviewToolEnv) {
100
188
  const sandbox = getSandbox(env.Sandbox, input.threadId, {
101
189
  transport: 'rpc',
102
190
  })
191
+ // Gate tunnel work on a live listener: a fresh tunnel to a dead port is still
192
+ // a dead preview, and the failure the agent can FIX is "start the server".
193
+ const local = await localProbe(sandbox, port)
194
+ if (typeof local === 'string') {
195
+ throw new Error(
196
+ `No server is listening on port ${port} inside the sandbox (${local}). Start the dev server (bound to 0.0.0.0:${port}) first, then retry exposePreview.`,
197
+ )
198
+ }
103
199
  // A Cloudflare quick tunnel (`*.trycloudflare.com`) run by `cloudflared` INSIDE
104
200
  // the sandbox: it bypasses the local Vite dev server's port entirely (so Vite
105
201
  // can't hijack the preview's asset requests) and needs no custom domain on a
106
202
  // deploy. `get(port)` is idempotent per port. See the Sandbox SDK `tunnels` API.
107
203
  const tunnel = await sandbox.tunnels.get(port)
108
- return { url: tunnel.url }
204
+ const edgeFailure = await edgeProbeFailure(tunnel.url, local)
205
+ if (edgeFailure === null) return { url: tunnel.url }
206
+ // Never destroy on 'unverified': the tunnel may be healthy with only the
207
+ // probe path broken, so destroying it could kill a working preview.
208
+ if (edgeFailure.verdict === 'unverified') {
209
+ throw new Error(
210
+ `Port ${port} is serving inside the sandbox, but the preview tunnel could not be verified from the edge (${edgeFailure.symptom}). The tunnel was left in place — retry exposePreview in a few seconds.`,
211
+ )
212
+ }
213
+ // Local server healthy but the edge kept answering 502/530: the cached tunnel
214
+ // record is suspect. Refresh, bounded to ONE so we never churn tunnels.
215
+ await sandbox.tunnels.destroy(port)
216
+ const fresh = await sandbox.tunnels.get(port)
217
+ const freshFailure = await edgeProbeFailure(fresh.url, local)
218
+ if (freshFailure === null) {
219
+ return {
220
+ url: fresh.url,
221
+ note: `The tunnel for port ${port} was stale, so it was replaced. Any previously shared preview URL for this port is dead — share this new URL instead.`,
222
+ }
223
+ }
224
+ // Don't leave a known-dead record in DO storage (the #992 failure mode).
225
+ if (freshFailure.verdict === 'stale') {
226
+ await sandbox.tunnels.destroy(port)
227
+ }
228
+ const [diagnosis, hint] =
229
+ freshFailure.verdict === 'stale'
230
+ ? [
231
+ 'its preview tunnel never became reachable',
232
+ 'Retry exposePreview, and if it keeps failing, restart the dev server and try again.',
233
+ ]
234
+ : [
235
+ 'the replacement preview tunnel could not be verified from the edge',
236
+ 'Retry exposePreview in a few seconds.',
237
+ ]
238
+ throw new Error(
239
+ `Port ${port} is serving inside the sandbox, but ${diagnosis} (old tunnel: ${edgeFailure.symptom}; replacement tunnel: ${freshFailure.symptom}). ${hint}`,
240
+ )
109
241
  })
110
242
  }
package/src/run-log-do.ts CHANGED
@@ -149,15 +149,79 @@ export class DurableObjectRunEventLog implements RunEventLog {
149
149
  this.wake(runId)
150
150
  }
151
151
 
152
+ async touch(runId: string): Promise<void> {
153
+ await this.storage.transaction(async (txn) => {
154
+ const stored = await txn.get<StoredRunRecord>(recKey(runId))
155
+ if (!stored) return
156
+ const { record, migrated } = migrateStoredRunRecord(stored)
157
+ if (isTerminalRunStatus(record.status)) {
158
+ if (migrated) await txn.put(recKey(runId), record)
159
+ return
160
+ }
161
+ await txn.put(recKey(runId), { ...record, updatedAt: Date.now() })
162
+ })
163
+ // Activity without a new event or status transition gives a tailing reader
164
+ // nothing to observe, so intentionally do not wake it.
165
+ }
166
+
167
+ async finishIfStale(
168
+ runId: string,
169
+ cutoff: number,
170
+ chunk: Extract<StreamChunk, { type: 'RUN_ERROR' }>,
171
+ ): Promise<boolean> {
172
+ const finished = await this.storage.transaction(async (txn) => {
173
+ const stored = await txn.get<StoredRunRecord>(recKey(runId))
174
+ if (!stored) return false
175
+ const { record, migrated } = migrateStoredRunRecord(stored)
176
+ if (isTerminalRunStatus(record.status) || record.updatedAt >= cutoff) {
177
+ if (migrated) await txn.put(recKey(runId), record)
178
+ return false
179
+ }
180
+
181
+ const now = Date.now()
182
+ const seq = record.lastSeq + 1
183
+ const next: RunLogRecord = {
184
+ ...record,
185
+ status: 'failed',
186
+ lastSeq: seq,
187
+ error: {
188
+ message: chunk.message,
189
+ ...(chunk.code !== undefined ? { code: chunk.code } : {}),
190
+ },
191
+ finishedAt: now,
192
+ updatedAt: now,
193
+ }
194
+ // Read the current activity clock and commit the terminal event + record
195
+ // in this one transaction. Keep wake-ups outside: transaction callbacks
196
+ // may be retried and must contain no external work.
197
+ await txn.put(evtKey(runId, seq), chunk)
198
+ await txn.put(recKey(runId), next)
199
+ return true
200
+ })
201
+ if (finished) this.wake(runId)
202
+ return finished
203
+ }
204
+
152
205
  async update(runId: string, patch: RunRecordPatch): Promise<void> {
153
- const record = await this.getRecord(runId)
154
- if (!record) return // unknown runId is a no-op
155
- const next: RunLogRecord = { ...record, ...patch, updatedAt: Date.now() }
156
- await this.storage.put(recKey(runId), next)
206
+ const updated = await this.storage.transaction(async (txn) => {
207
+ const stored = await txn.get<StoredRunRecord>(recKey(runId))
208
+ if (!stored) return false
209
+ const { record, migrated } = migrateStoredRunRecord(stored)
210
+ if (isTerminalRunStatus(record.status)) {
211
+ if (migrated) await txn.put(recKey(runId), record)
212
+ return false
213
+ }
214
+ await txn.put(recKey(runId), {
215
+ ...record,
216
+ ...patch,
217
+ updatedAt: Date.now(),
218
+ })
219
+ return true
220
+ })
157
221
  // A patch may terminalize the shared status field (core's driver writes its
158
- // terminal status through `RunStore.update`) — parked readers must see it
159
- // now, not a TAIL_POLL_MS later.
160
- this.wake(runId)
222
+ // terminal status through `RunStore.update`) — wake only after that update
223
+ // commits. A late update racing a terminal writer is an atomic no-op.
224
+ if (updated) this.wake(runId)
161
225
  }
162
226
 
163
227
  async get(runId: string): Promise<RunLogRecord | null> {