@tanstack/ai-sandbox-cloudflare 0.3.12 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/chat-coordinator.js +19 -16
- package/dist/esm/chat-coordinator.js.map +1 -1
- package/dist/esm/container-coordinator.js +20 -17
- package/dist/esm/container-coordinator.js.map +1 -1
- package/dist/esm/coordinator-callbacks.d.ts +4 -0
- package/dist/esm/coordinator-callbacks.js +45 -0
- package/dist/esm/coordinator-callbacks.js.map +1 -0
- package/dist/esm/coordinator.d.ts +7 -2
- package/dist/esm/coordinator.js +43 -28
- package/dist/esm/coordinator.js.map +1 -1
- package/dist/esm/factory.d.ts +5 -0
- package/dist/esm/factory.js +8 -1
- package/dist/esm/factory.js.map +1 -1
- package/dist/esm/preview-tool.js +77 -1
- package/dist/esm/preview-tool.js.map +1 -1
- package/dist/esm/run-log-do.d.ts +4 -0
- package/dist/esm/run-log-do.js +59 -9
- package/dist/esm/run-log-do.js.map +1 -1
- package/dist/esm/run-log.d.ts +9 -4
- package/dist/esm/run-log.js +24 -1
- package/dist/esm/run-log.js.map +1 -1
- package/package.json +5 -5
- package/src/chat-coordinator.ts +25 -16
- package/src/container-coordinator.ts +37 -24
- package/src/coordinator-callbacks.ts +58 -0
- package/src/coordinator.ts +66 -30
- package/src/factory.ts +15 -1
- package/src/preview-tool.ts +133 -1
- package/src/run-log-do.ts +71 -7
- package/src/run-log.ts +46 -5
package/src/chat-coordinator.ts
CHANGED
|
@@ -30,6 +30,7 @@ import {
|
|
|
30
30
|
withSandbox,
|
|
31
31
|
} from '@tanstack/ai-sandbox'
|
|
32
32
|
import { SandboxCoordinator, resolveBridgeOrigin } from './coordinator'
|
|
33
|
+
import { runWithCallbackActivity } from './coordinator-callbacks'
|
|
33
34
|
import { timingSafeBearerEqualWeb } from './web-crypto'
|
|
34
35
|
import type { StartRunInput } from './coordinator'
|
|
35
36
|
import type {
|
|
@@ -233,21 +234,29 @@ export abstract class ChatSandboxCoordinator<
|
|
|
233
234
|
) {
|
|
234
235
|
return new Response('unauthorized', { status: 401 })
|
|
235
236
|
}
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
237
|
+
return runWithCallbackActivity(
|
|
238
|
+
this,
|
|
239
|
+
runId,
|
|
240
|
+
(id) => this.log.touch(id),
|
|
241
|
+
async () => {
|
|
242
|
+
let message: unknown
|
|
243
|
+
try {
|
|
244
|
+
message = await request.json()
|
|
245
|
+
} catch {
|
|
246
|
+
// A malformed body must still produce a valid JSON-RPC error so the
|
|
247
|
+
// agent's MCP client can react, rather than an opaque DO 500 that can
|
|
248
|
+
// wedge the run.
|
|
249
|
+
return this.jsonResponse({
|
|
250
|
+
jsonrpc: '2.0',
|
|
251
|
+
id: null,
|
|
252
|
+
error: { code: -32700, message: 'Parse error' },
|
|
253
|
+
})
|
|
254
|
+
}
|
|
255
|
+
const reply = await handleBridgeJsonRpc(bridge.core, message)
|
|
256
|
+
// A notification (no id) yields null → MCP expects an empty 202 ack.
|
|
257
|
+
if (reply === null) return new Response(null, { status: 202 })
|
|
258
|
+
return this.jsonResponse(reply)
|
|
259
|
+
},
|
|
260
|
+
)
|
|
252
261
|
}
|
|
253
262
|
}
|
|
@@ -34,6 +34,7 @@ import {
|
|
|
34
34
|
} from '@tanstack/ai-sandbox'
|
|
35
35
|
import { getSandbox } from '@cloudflare/sandbox'
|
|
36
36
|
import { SandboxCoordinator, resolveBridgeOrigin } from './coordinator'
|
|
37
|
+
import { runWithCallbackActivity } from './coordinator-callbacks'
|
|
37
38
|
import { timingSafeBearerEqualWeb } from './web-crypto'
|
|
38
39
|
import type { StartRunInput } from './coordinator'
|
|
39
40
|
import type { ContainerRunRequest, HarnessId } from './protocol'
|
|
@@ -409,29 +410,41 @@ export abstract class ContainerSandboxCoordinator<
|
|
|
409
410
|
) {
|
|
410
411
|
return new Response('unauthorized', { status: 401 })
|
|
411
412
|
}
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
413
|
+
return runWithCallbackActivity(
|
|
414
|
+
this,
|
|
415
|
+
runId,
|
|
416
|
+
(id) => this.log.touch(id),
|
|
417
|
+
async () => {
|
|
418
|
+
let payload: unknown
|
|
419
|
+
try {
|
|
420
|
+
payload = await request.json()
|
|
421
|
+
} catch {
|
|
422
|
+
return this.jsonResponse({ error: 'body must be valid JSON' }, 400)
|
|
423
|
+
}
|
|
424
|
+
if (!isToolExecRequest(payload)) {
|
|
425
|
+
return this.jsonResponse(
|
|
426
|
+
{ error: 'body must be { name, args }' },
|
|
427
|
+
400,
|
|
428
|
+
)
|
|
429
|
+
}
|
|
430
|
+
try {
|
|
431
|
+
const result = await executeHostTool(
|
|
432
|
+
state.hostTools,
|
|
433
|
+
payload.name,
|
|
434
|
+
payload.args,
|
|
435
|
+
{
|
|
436
|
+
...(state.context !== undefined
|
|
437
|
+
? { context: state.context }
|
|
438
|
+
: {}),
|
|
439
|
+
signal: state.abort.signal,
|
|
440
|
+
},
|
|
441
|
+
)
|
|
442
|
+
return this.jsonResponse({ result })
|
|
443
|
+
} catch (error) {
|
|
444
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
445
|
+
return this.jsonResponse({ error: message }, 500)
|
|
446
|
+
}
|
|
447
|
+
},
|
|
448
|
+
)
|
|
436
449
|
}
|
|
437
450
|
}
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
type TouchRun = (runId: string) => Promise<void>
|
|
2
|
+
|
|
3
|
+
const inFlightCallbacks = new WeakMap<object, Map<string, number>>()
|
|
4
|
+
|
|
5
|
+
function increment(owner: object, runId: string): void {
|
|
6
|
+
let runs = inFlightCallbacks.get(owner)
|
|
7
|
+
if (!runs) {
|
|
8
|
+
runs = new Map()
|
|
9
|
+
inFlightCallbacks.set(owner, runs)
|
|
10
|
+
}
|
|
11
|
+
runs.set(runId, (runs.get(runId) ?? 0) + 1)
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
function decrement(owner: object, runId: string): void {
|
|
15
|
+
const runs = inFlightCallbacks.get(owner)
|
|
16
|
+
if (!runs) return
|
|
17
|
+
const count = runs.get(runId) ?? 0
|
|
18
|
+
if (count > 1) {
|
|
19
|
+
runs.set(runId, count - 1)
|
|
20
|
+
return
|
|
21
|
+
}
|
|
22
|
+
runs.delete(runId)
|
|
23
|
+
if (runs.size === 0) inFlightCallbacks.delete(owner)
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function hasInFlightCallback(owner: object, runId: string): boolean {
|
|
27
|
+
return (inFlightCallbacks.get(owner)?.get(runId) ?? 0) > 0
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export async function runWithCallbackActivity<T>(
|
|
31
|
+
owner: object,
|
|
32
|
+
runId: string,
|
|
33
|
+
touch: TouchRun,
|
|
34
|
+
operation: () => Promise<T>,
|
|
35
|
+
): Promise<T> {
|
|
36
|
+
increment(owner, runId)
|
|
37
|
+
try {
|
|
38
|
+
// Failure here proves liveness could not be persisted, so do not execute the
|
|
39
|
+
// callback operation.
|
|
40
|
+
await touch(runId)
|
|
41
|
+
try {
|
|
42
|
+
return await operation()
|
|
43
|
+
} finally {
|
|
44
|
+
// This is best-effort bookkeeping after an operation has produced its
|
|
45
|
+
// result or error. It must never replace that original outcome.
|
|
46
|
+
try {
|
|
47
|
+
await touch(runId)
|
|
48
|
+
} catch (error) {
|
|
49
|
+
console.error(
|
|
50
|
+
`[sandbox-coordinator] completion activity touch failed for run ${runId}:`,
|
|
51
|
+
error,
|
|
52
|
+
)
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
} finally {
|
|
56
|
+
decrement(owner, runId)
|
|
57
|
+
}
|
|
58
|
+
}
|
package/src/coordinator.ts
CHANGED
|
@@ -31,21 +31,30 @@ import { EventType, isTerminalRunStatus } from '@tanstack/ai'
|
|
|
31
31
|
// migrated on read; see './run-log').
|
|
32
32
|
import { RunController } from '@tanstack/ai-sandbox'
|
|
33
33
|
import { runLogStore, runLogStream } from './durability'
|
|
34
|
+
import { hasInFlightCallback } from './coordinator-callbacks'
|
|
34
35
|
import { DurableObjectRunEventLog } from './run-log-do'
|
|
35
36
|
import type { ModelMessage, StreamChunk } from '@tanstack/ai'
|
|
36
37
|
import type { RunLogRecord } from './run-log'
|
|
37
38
|
|
|
38
|
-
/**
|
|
39
|
+
/** Upper bound on the watchdog check interval while a run is in flight (ms). */
|
|
39
40
|
const WATCHDOG_MS = 30_000
|
|
40
41
|
|
|
41
|
-
/**
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
42
|
+
/** Default permitted period without persisted run activity (ms). */
|
|
43
|
+
const DEFAULT_STALL_TIMEOUT_MS = 5 * 60_000
|
|
44
|
+
|
|
45
|
+
/** @internal Shared validation for direct subclasses and the eager factory path. */
|
|
46
|
+
export function normalizeStallTimeoutMs(
|
|
47
|
+
stallTimeoutMs: number | false | undefined,
|
|
48
|
+
): number | false {
|
|
49
|
+
if (stallTimeoutMs === undefined) return DEFAULT_STALL_TIMEOUT_MS
|
|
50
|
+
if (stallTimeoutMs === false) return false
|
|
51
|
+
if (!Number.isSafeInteger(stallTimeoutMs) || stallTimeoutMs <= 0) {
|
|
52
|
+
throw new TypeError(
|
|
53
|
+
'stallTimeoutMs must be a positive safe integer or false',
|
|
54
|
+
)
|
|
55
|
+
}
|
|
56
|
+
return stallTimeoutMs
|
|
57
|
+
}
|
|
49
58
|
|
|
50
59
|
/** What the Worker hands the coordinator to start a run. */
|
|
51
60
|
export interface StartRunInput {
|
|
@@ -106,9 +115,15 @@ export abstract class SandboxCoordinator<
|
|
|
106
115
|
* running — double-delivering events and racing the persisted cursor.
|
|
107
116
|
*/
|
|
108
117
|
private readonly pumping = new WeakSet<WebSocket>()
|
|
118
|
+
private readonly stallTimeoutMs: number | false
|
|
109
119
|
|
|
110
|
-
constructor(
|
|
120
|
+
constructor(
|
|
121
|
+
ctx: DurableObjectState,
|
|
122
|
+
env: TEnv,
|
|
123
|
+
stallTimeoutMs?: number | false,
|
|
124
|
+
) {
|
|
111
125
|
super(ctx, env)
|
|
126
|
+
this.stallTimeoutMs = normalizeStallTimeoutMs(stallTimeoutMs)
|
|
112
127
|
this.log = new DurableObjectRunEventLog(ctx.storage)
|
|
113
128
|
this.controller = new RunController({
|
|
114
129
|
runs: runLogStore(this.log),
|
|
@@ -154,7 +169,12 @@ export abstract class SandboxCoordinator<
|
|
|
154
169
|
|
|
155
170
|
async startRun(input: StartRunInput): Promise<{ runId: string }> {
|
|
156
171
|
const existing = await this.log.get(input.runId)
|
|
157
|
-
if (existing)
|
|
172
|
+
if (existing) {
|
|
173
|
+
if (!isTerminalRunStatus(existing.status)) await this.armWatchdog()
|
|
174
|
+
return { runId: input.runId }
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
await this.armWatchdog()
|
|
158
178
|
|
|
159
179
|
// Open the run BEFORE building the stream. `pipeToRunLog`'s never-rejects
|
|
160
180
|
// guarantee only covers failures AFTER the stream is handed to it — a throw
|
|
@@ -189,7 +209,6 @@ export abstract class SandboxCoordinator<
|
|
|
189
209
|
// running the settle hook.
|
|
190
210
|
const settle = (): void => this.onRunSettled(input.runId)
|
|
191
211
|
this.ctx.waitUntil(done.then(settle, settle))
|
|
192
|
-
await this.ctx.storage.setAlarm(Date.now() + WATCHDOG_MS)
|
|
193
212
|
return { runId: input.runId }
|
|
194
213
|
}
|
|
195
214
|
|
|
@@ -320,41 +339,58 @@ export abstract class SandboxCoordinator<
|
|
|
320
339
|
// ===========================================================================
|
|
321
340
|
|
|
322
341
|
override async alarm(): Promise<void> {
|
|
342
|
+
// A previously configured alarm may still be delivered once after watchdogs
|
|
343
|
+
// are disabled. It must self-extinguish before reading storage or entering a
|
|
344
|
+
// catch path that could re-arm it. Deliberately do not call deleteAlarm().
|
|
345
|
+
if (this.stallTimeoutMs === false) return
|
|
346
|
+
|
|
323
347
|
try {
|
|
324
348
|
// Through the log (not a raw `rec:` list) so legacy records are migrated
|
|
325
349
|
// on the way out — the storage layout is the log's private concern.
|
|
326
350
|
const runs = await this.log.list()
|
|
327
|
-
const
|
|
351
|
+
const cutoff = Date.now() - this.stallTimeoutMs
|
|
328
352
|
let active = false
|
|
329
353
|
for (const record of runs) {
|
|
330
354
|
if (isTerminalRunStatus(record.status)) continue
|
|
331
|
-
if (
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
await this.failStalledRun(record.runId)
|
|
336
|
-
} else {
|
|
355
|
+
if (
|
|
356
|
+
hasInFlightCallback(this, record.runId) ||
|
|
357
|
+
record.updatedAt >= cutoff
|
|
358
|
+
) {
|
|
337
359
|
active = true
|
|
360
|
+
continue
|
|
338
361
|
}
|
|
362
|
+
|
|
363
|
+
// Re-check and terminalize atomically against the same strict cutoff.
|
|
364
|
+
// A callback touch or normal completion racing this alarm wins cleanly.
|
|
365
|
+
const finished = await this.failStalledRun(record.runId, cutoff)
|
|
366
|
+
if (finished) this.onRunSettled(record.runId)
|
|
367
|
+
else active = true
|
|
339
368
|
}
|
|
340
|
-
if (active) await this.
|
|
369
|
+
if (active) await this.armWatchdog()
|
|
341
370
|
} catch (error) {
|
|
342
371
|
// Never let the watchdog die silently: a transient storage error must not
|
|
343
372
|
// permanently disable liveness detection. Re-arm and try again next tick.
|
|
344
373
|
console.error('[sandbox-coordinator] watchdog alarm failed:', error)
|
|
345
|
-
await this.
|
|
374
|
+
await this.armWatchdog()
|
|
346
375
|
}
|
|
347
376
|
}
|
|
348
377
|
|
|
349
|
-
/**
|
|
350
|
-
private async
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
378
|
+
/** Arm without allowing a new run to postpone an earlier pending check. */
|
|
379
|
+
private async armWatchdog(): Promise<void> {
|
|
380
|
+
if (this.stallTimeoutMs === false) return
|
|
381
|
+
const next = Date.now() + Math.min(WATCHDOG_MS, this.stallTimeoutMs)
|
|
382
|
+
const pending = await this.ctx.storage.getAlarm()
|
|
383
|
+
if (pending === null || pending > next) {
|
|
384
|
+
await this.ctx.storage.setAlarm(next)
|
|
356
385
|
}
|
|
357
|
-
|
|
358
|
-
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** Mark a strictly stale run as a terminal error if it still qualifies. */
|
|
389
|
+
private failStalledRun(runId: string, cutoff: number): Promise<boolean> {
|
|
390
|
+
const message = 'run watchdog: no progress; orchestrator presumed dead'
|
|
391
|
+
return this.log.finishIfStale(runId, cutoff, {
|
|
392
|
+
type: EventType.RUN_ERROR,
|
|
393
|
+
message,
|
|
394
|
+
})
|
|
359
395
|
}
|
|
360
396
|
}
|
package/src/factory.ts
CHANGED
|
@@ -45,7 +45,7 @@ import { cloudflareSandbox } from './provider'
|
|
|
45
45
|
import { ChatSandboxCoordinator } from './chat-coordinator'
|
|
46
46
|
import { ContainerSandboxCoordinator } from './container-coordinator'
|
|
47
47
|
import { createSandboxAgentWorker } from './worker'
|
|
48
|
-
import { resolvePreviewHost } from './coordinator'
|
|
48
|
+
import { normalizeStallTimeoutMs, resolvePreviewHost } from './coordinator'
|
|
49
49
|
import type { ChatCoordinatorEnv, ChatRunConfig } from './chat-coordinator'
|
|
50
50
|
import type {
|
|
51
51
|
ContainerCoordinatorEnv,
|
|
@@ -83,6 +83,11 @@ export interface SandboxAgentEnv
|
|
|
83
83
|
interface BaseAgentConfig<TEnv extends SandboxAgentEnv> {
|
|
84
84
|
/** chat()-provided server tools, resolved per run (DO-drives: bridged over MCP). */
|
|
85
85
|
tools?: (input: StartRunInput, env: TEnv) => Array<AnyTool>
|
|
86
|
+
/**
|
|
87
|
+
* Fail a run after this many milliseconds without persisted activity. Omitted
|
|
88
|
+
* defaults to five minutes; `false` disables the watchdog.
|
|
89
|
+
*/
|
|
90
|
+
stallTimeoutMs?: number | false
|
|
86
91
|
}
|
|
87
92
|
|
|
88
93
|
/** DO-drives config: the DO runs `chat()` with the given adapter. */
|
|
@@ -188,11 +193,16 @@ function resolveCoordinator<TEnv extends SandboxAgentEnv>(
|
|
|
188
193
|
export function createCloudflareSandboxAgent<
|
|
189
194
|
TEnv extends SandboxAgentEnv = SandboxAgentEnv,
|
|
190
195
|
>(config: CloudflareSandboxAgentConfig<TEnv>): CloudflareSandboxAgent<TEnv> {
|
|
196
|
+
const stallTimeoutMs = normalizeStallTimeoutMs(config.stallTimeoutMs)
|
|
191
197
|
const worker = createSandboxAgentWorker<TEnv>(resolveCoordinator)
|
|
192
198
|
|
|
193
199
|
if (config.mode === 'colocated') {
|
|
194
200
|
const colocated = config
|
|
195
201
|
class ConfiguredContainerCoordinator extends ContainerSandboxCoordinator<TEnv> {
|
|
202
|
+
constructor(ctx: DurableObjectState, env: TEnv) {
|
|
203
|
+
super(ctx, env, stallTimeoutMs)
|
|
204
|
+
}
|
|
205
|
+
|
|
196
206
|
protected override config(input: StartRunInput): ContainerRunConfig {
|
|
197
207
|
return {
|
|
198
208
|
hostTools: colocated.tools?.(input, this.env) ?? [],
|
|
@@ -207,6 +217,10 @@ export function createCloudflareSandboxAgent<
|
|
|
207
217
|
|
|
208
218
|
const doDrives = config
|
|
209
219
|
class ConfiguredChatCoordinator extends ChatSandboxCoordinator<TEnv> {
|
|
220
|
+
constructor(ctx: DurableObjectState, env: TEnv) {
|
|
221
|
+
super(ctx, env, stallTimeoutMs)
|
|
222
|
+
}
|
|
223
|
+
|
|
210
224
|
protected override config(input: StartRunInput): ChatRunConfig {
|
|
211
225
|
const tools = doDrives.tools?.(input, this.env)
|
|
212
226
|
return {
|
package/src/preview-tool.ts
CHANGED
|
@@ -68,6 +68,94 @@ export const PREVIEW_GUIDANCE: string = [
|
|
|
68
68
|
'Once it is listening, call `exposePreview` with that port, then share the URL.',
|
|
69
69
|
].join('\n')
|
|
70
70
|
|
|
71
|
+
const LOCAL_PROBE_TIMEOUT_MS = 5_000
|
|
72
|
+
// Fresh quick tunnels need a few seconds of DNS/edge propagation, hence retries.
|
|
73
|
+
const EDGE_PROBE_ATTEMPTS = 5
|
|
74
|
+
const EDGE_PROBE_BASE_DELAY_MS = 250
|
|
75
|
+
const EDGE_PROBE_FETCH_TIMEOUT_MS = 3_000
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Probe the port INSIDE the sandbox via `containerFetch`. Returns the HTTP
|
|
79
|
+
* status when a listener answered (any response — 4xx/5xx included — proves one
|
|
80
|
+
* exists), or the failure symptom (a string) when nothing did.
|
|
81
|
+
*/
|
|
82
|
+
async function localProbe(
|
|
83
|
+
sandbox: Sandbox,
|
|
84
|
+
port: number,
|
|
85
|
+
): Promise<number | string> {
|
|
86
|
+
// A race instead of AbortSignal: signals don't serialize across the sandbox
|
|
87
|
+
// RPC boundary, and a lost in-flight probe response is harmless.
|
|
88
|
+
let timeoutId: ReturnType<typeof setTimeout> | undefined
|
|
89
|
+
const timeout = new Promise<never>((_, reject) => {
|
|
90
|
+
timeoutId = setTimeout(
|
|
91
|
+
() => reject(new Error(`no response within ${LOCAL_PROBE_TIMEOUT_MS}ms`)),
|
|
92
|
+
LOCAL_PROBE_TIMEOUT_MS,
|
|
93
|
+
)
|
|
94
|
+
})
|
|
95
|
+
try {
|
|
96
|
+
const res = await Promise.race([
|
|
97
|
+
sandbox.containerFetch('http://preview/', { method: 'HEAD' }, port),
|
|
98
|
+
timeout,
|
|
99
|
+
])
|
|
100
|
+
return res.status
|
|
101
|
+
} catch (error) {
|
|
102
|
+
return error instanceof Error ? error.message : String(error)
|
|
103
|
+
} finally {
|
|
104
|
+
clearTimeout(timeoutId)
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Probe a tunnel URL through the public edge with bounded retries. Only 502/530
|
|
110
|
+
* — Cloudflare's tunnel/origin-unreachable signatures — mark the URL as
|
|
111
|
+
* unreachable (401/403/404 prove the server answered, and `redirect: 'manual'`
|
|
112
|
+
* keeps a login redirect from probing some other site), and even those are
|
|
113
|
+
* trusted when the app answered the SAME status locally, so an app's own
|
|
114
|
+
* 502/530 never gets its healthy tunnel destroyed. The verdict split exists
|
|
115
|
+
* because only an OBSERVED non-matching 502/530 ('stale') is evidence that
|
|
116
|
+
* justifies destroying the tunnel; fetch exceptions ('unverified') prove
|
|
117
|
+
* nothing about it — the probe path itself may be what failed. A throw
|
|
118
|
+
* anywhere in the retry window therefore wins over an earlier 502/530: the
|
|
119
|
+
* window did not finish as HTTP probes, so we must not destroy.
|
|
120
|
+
*/
|
|
121
|
+
async function edgeProbeFailure(
|
|
122
|
+
url: string,
|
|
123
|
+
localStatus: number,
|
|
124
|
+
): Promise<{ verdict: 'stale' | 'unverified'; symptom: string } | null> {
|
|
125
|
+
let lastFailure = 'no response'
|
|
126
|
+
let verdict: 'stale' | 'unverified' = 'unverified'
|
|
127
|
+
let sawFetchException = false
|
|
128
|
+
for (let attempt = 0; attempt < EDGE_PROBE_ATTEMPTS; attempt += 1) {
|
|
129
|
+
if (attempt > 0) {
|
|
130
|
+
await new Promise((resolve) =>
|
|
131
|
+
setTimeout(resolve, EDGE_PROBE_BASE_DELAY_MS * 2 ** (attempt - 1)),
|
|
132
|
+
)
|
|
133
|
+
}
|
|
134
|
+
try {
|
|
135
|
+
const res = await fetch(url, {
|
|
136
|
+
method: 'HEAD',
|
|
137
|
+
redirect: 'manual',
|
|
138
|
+
signal: AbortSignal.timeout(EDGE_PROBE_FETCH_TIMEOUT_MS),
|
|
139
|
+
})
|
|
140
|
+
if (
|
|
141
|
+
(res.status !== 502 && res.status !== 530) ||
|
|
142
|
+
res.status === localStatus
|
|
143
|
+
) {
|
|
144
|
+
return null
|
|
145
|
+
}
|
|
146
|
+
lastFailure = `HTTP ${res.status}`
|
|
147
|
+
verdict = 'stale'
|
|
148
|
+
} catch (error) {
|
|
149
|
+
sawFetchException = true
|
|
150
|
+
lastFailure = error instanceof Error ? error.message : String(error)
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
return {
|
|
154
|
+
verdict: sawFetchException ? 'unverified' : verdict,
|
|
155
|
+
symptom: lastFailure,
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
71
159
|
/**
|
|
72
160
|
* Build the `exposePreview` server tool for one run. Starting a tunnel is a
|
|
73
161
|
* HOST-side call on the Sandbox DO stub, so an in-sandbox agent cannot make it from
|
|
@@ -100,11 +188,55 @@ export function exposePreviewTool(input: StartRunInput, env: PreviewToolEnv) {
|
|
|
100
188
|
const sandbox = getSandbox(env.Sandbox, input.threadId, {
|
|
101
189
|
transport: 'rpc',
|
|
102
190
|
})
|
|
191
|
+
// Gate tunnel work on a live listener: a fresh tunnel to a dead port is still
|
|
192
|
+
// a dead preview, and the failure the agent can FIX is "start the server".
|
|
193
|
+
const local = await localProbe(sandbox, port)
|
|
194
|
+
if (typeof local === 'string') {
|
|
195
|
+
throw new Error(
|
|
196
|
+
`No server is listening on port ${port} inside the sandbox (${local}). Start the dev server (bound to 0.0.0.0:${port}) first, then retry exposePreview.`,
|
|
197
|
+
)
|
|
198
|
+
}
|
|
103
199
|
// A Cloudflare quick tunnel (`*.trycloudflare.com`) run by `cloudflared` INSIDE
|
|
104
200
|
// the sandbox: it bypasses the local Vite dev server's port entirely (so Vite
|
|
105
201
|
// can't hijack the preview's asset requests) and needs no custom domain on a
|
|
106
202
|
// deploy. `get(port)` is idempotent per port. See the Sandbox SDK `tunnels` API.
|
|
107
203
|
const tunnel = await sandbox.tunnels.get(port)
|
|
108
|
-
|
|
204
|
+
const edgeFailure = await edgeProbeFailure(tunnel.url, local)
|
|
205
|
+
if (edgeFailure === null) return { url: tunnel.url }
|
|
206
|
+
// Never destroy on 'unverified': the tunnel may be healthy with only the
|
|
207
|
+
// probe path broken, so destroying it could kill a working preview.
|
|
208
|
+
if (edgeFailure.verdict === 'unverified') {
|
|
209
|
+
throw new Error(
|
|
210
|
+
`Port ${port} is serving inside the sandbox, but the preview tunnel could not be verified from the edge (${edgeFailure.symptom}). The tunnel was left in place — retry exposePreview in a few seconds.`,
|
|
211
|
+
)
|
|
212
|
+
}
|
|
213
|
+
// Local server healthy but the edge kept answering 502/530: the cached tunnel
|
|
214
|
+
// record is suspect. Refresh, bounded to ONE so we never churn tunnels.
|
|
215
|
+
await sandbox.tunnels.destroy(port)
|
|
216
|
+
const fresh = await sandbox.tunnels.get(port)
|
|
217
|
+
const freshFailure = await edgeProbeFailure(fresh.url, local)
|
|
218
|
+
if (freshFailure === null) {
|
|
219
|
+
return {
|
|
220
|
+
url: fresh.url,
|
|
221
|
+
note: `The tunnel for port ${port} was stale, so it was replaced. Any previously shared preview URL for this port is dead — share this new URL instead.`,
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
// Don't leave a known-dead record in DO storage (the #992 failure mode).
|
|
225
|
+
if (freshFailure.verdict === 'stale') {
|
|
226
|
+
await sandbox.tunnels.destroy(port)
|
|
227
|
+
}
|
|
228
|
+
const [diagnosis, hint] =
|
|
229
|
+
freshFailure.verdict === 'stale'
|
|
230
|
+
? [
|
|
231
|
+
'its preview tunnel never became reachable',
|
|
232
|
+
'Retry exposePreview, and if it keeps failing, restart the dev server and try again.',
|
|
233
|
+
]
|
|
234
|
+
: [
|
|
235
|
+
'the replacement preview tunnel could not be verified from the edge',
|
|
236
|
+
'Retry exposePreview in a few seconds.',
|
|
237
|
+
]
|
|
238
|
+
throw new Error(
|
|
239
|
+
`Port ${port} is serving inside the sandbox, but ${diagnosis} (old tunnel: ${edgeFailure.symptom}; replacement tunnel: ${freshFailure.symptom}). ${hint}`,
|
|
240
|
+
)
|
|
109
241
|
})
|
|
110
242
|
}
|
package/src/run-log-do.ts
CHANGED
|
@@ -149,15 +149,79 @@ export class DurableObjectRunEventLog implements RunEventLog {
|
|
|
149
149
|
this.wake(runId)
|
|
150
150
|
}
|
|
151
151
|
|
|
152
|
+
async touch(runId: string): Promise<void> {
|
|
153
|
+
await this.storage.transaction(async (txn) => {
|
|
154
|
+
const stored = await txn.get<StoredRunRecord>(recKey(runId))
|
|
155
|
+
if (!stored) return
|
|
156
|
+
const { record, migrated } = migrateStoredRunRecord(stored)
|
|
157
|
+
if (isTerminalRunStatus(record.status)) {
|
|
158
|
+
if (migrated) await txn.put(recKey(runId), record)
|
|
159
|
+
return
|
|
160
|
+
}
|
|
161
|
+
await txn.put(recKey(runId), { ...record, updatedAt: Date.now() })
|
|
162
|
+
})
|
|
163
|
+
// Activity without a new event or status transition gives a tailing reader
|
|
164
|
+
// nothing to observe, so intentionally do not wake it.
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
async finishIfStale(
|
|
168
|
+
runId: string,
|
|
169
|
+
cutoff: number,
|
|
170
|
+
chunk: Extract<StreamChunk, { type: 'RUN_ERROR' }>,
|
|
171
|
+
): Promise<boolean> {
|
|
172
|
+
const finished = await this.storage.transaction(async (txn) => {
|
|
173
|
+
const stored = await txn.get<StoredRunRecord>(recKey(runId))
|
|
174
|
+
if (!stored) return false
|
|
175
|
+
const { record, migrated } = migrateStoredRunRecord(stored)
|
|
176
|
+
if (isTerminalRunStatus(record.status) || record.updatedAt >= cutoff) {
|
|
177
|
+
if (migrated) await txn.put(recKey(runId), record)
|
|
178
|
+
return false
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const now = Date.now()
|
|
182
|
+
const seq = record.lastSeq + 1
|
|
183
|
+
const next: RunLogRecord = {
|
|
184
|
+
...record,
|
|
185
|
+
status: 'failed',
|
|
186
|
+
lastSeq: seq,
|
|
187
|
+
error: {
|
|
188
|
+
message: chunk.message,
|
|
189
|
+
...(chunk.code !== undefined ? { code: chunk.code } : {}),
|
|
190
|
+
},
|
|
191
|
+
finishedAt: now,
|
|
192
|
+
updatedAt: now,
|
|
193
|
+
}
|
|
194
|
+
// Read the current activity clock and commit the terminal event + record
|
|
195
|
+
// in this one transaction. Keep wake-ups outside: transaction callbacks
|
|
196
|
+
// may be retried and must contain no external work.
|
|
197
|
+
await txn.put(evtKey(runId, seq), chunk)
|
|
198
|
+
await txn.put(recKey(runId), next)
|
|
199
|
+
return true
|
|
200
|
+
})
|
|
201
|
+
if (finished) this.wake(runId)
|
|
202
|
+
return finished
|
|
203
|
+
}
|
|
204
|
+
|
|
152
205
|
async update(runId: string, patch: RunRecordPatch): Promise<void> {
|
|
153
|
-
const
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
206
|
+
const updated = await this.storage.transaction(async (txn) => {
|
|
207
|
+
const stored = await txn.get<StoredRunRecord>(recKey(runId))
|
|
208
|
+
if (!stored) return false
|
|
209
|
+
const { record, migrated } = migrateStoredRunRecord(stored)
|
|
210
|
+
if (isTerminalRunStatus(record.status)) {
|
|
211
|
+
if (migrated) await txn.put(recKey(runId), record)
|
|
212
|
+
return false
|
|
213
|
+
}
|
|
214
|
+
await txn.put(recKey(runId), {
|
|
215
|
+
...record,
|
|
216
|
+
...patch,
|
|
217
|
+
updatedAt: Date.now(),
|
|
218
|
+
})
|
|
219
|
+
return true
|
|
220
|
+
})
|
|
157
221
|
// A patch may terminalize the shared status field (core's driver writes its
|
|
158
|
-
// terminal status through `RunStore.update`) —
|
|
159
|
-
//
|
|
160
|
-
this.wake(runId)
|
|
222
|
+
// terminal status through `RunStore.update`) — wake only after that update
|
|
223
|
+
// commits. A late update racing a terminal writer is an atomic no-op.
|
|
224
|
+
if (updated) this.wake(runId)
|
|
161
225
|
}
|
|
162
226
|
|
|
163
227
|
async get(runId: string): Promise<RunLogRecord | null> {
|