@tanstack/ai-sandbox 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/agents-file.js +53 -34
- package/dist/esm/agents-file.js.map +1 -1
- package/dist/esm/align.d.ts +121 -0
- package/dist/esm/align.js +197 -0
- package/dist/esm/align.js.map +1 -0
- package/dist/esm/approvals.js +63 -29
- package/dist/esm/approvals.js.map +1 -1
- package/dist/esm/attach-preflight.d.ts +85 -0
- package/dist/esm/attach-preflight.js +189 -0
- package/dist/esm/attach-preflight.js.map +1 -0
- package/dist/esm/bootstrap.js +103 -117
- package/dist/esm/bootstrap.js.map +1 -1
- package/dist/esm/bridge-events.js +96 -71
- package/dist/esm/bridge-events.js.map +1 -1
- package/dist/esm/capabilities.d.ts +0 -5
- package/dist/esm/capabilities.js +32 -28
- package/dist/esm/capabilities.js.map +1 -1
- package/dist/esm/chunk-identity.d.ts +52 -0
- package/dist/esm/chunk-identity.js +102 -0
- package/dist/esm/chunk-identity.js.map +1 -0
- package/dist/esm/claim.d.ts +187 -0
- package/dist/esm/claim.js +349 -0
- package/dist/esm/claim.js.map +1 -0
- package/dist/esm/contracts.d.ts +13 -0
- package/dist/esm/driver.d.ts +83 -0
- package/dist/esm/driver.js +138 -0
- package/dist/esm/driver.js.map +1 -0
- package/dist/esm/durability.d.ts +263 -0
- package/dist/esm/durability.js +230 -0
- package/dist/esm/durability.js.map +1 -0
- package/dist/esm/errors.js +28 -24
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/file-diff.js +151 -135
- package/dist/esm/file-diff.js.map +1 -1
- package/dist/esm/git-exec.js +51 -62
- package/dist/esm/git-exec.js.map +1 -1
- package/dist/esm/harness-cwd.js +24 -19
- package/dist/esm/harness-cwd.js.map +1 -1
- package/dist/esm/index.d.ts +30 -8
- package/dist/esm/index.js +23 -91
- package/dist/esm/instance-store.d.ts +88 -0
- package/dist/esm/instance-store.js +67 -0
- package/dist/esm/instance-store.js.map +1 -0
- package/dist/esm/journal-bytes.d.ts +67 -0
- package/dist/esm/journal-bytes.js +110 -0
- package/dist/esm/journal-bytes.js.map +1 -0
- package/dist/esm/journal-reader.d.ts +66 -0
- package/dist/esm/journal-reader.js +228 -0
- package/dist/esm/journal-reader.js.map +1 -0
- package/dist/esm/journal-sweep.d.ts +113 -0
- package/dist/esm/journal-sweep.js +309 -0
- package/dist/esm/journal-sweep.js.map +1 -0
- package/dist/esm/journal.d.ts +542 -0
- package/dist/esm/journal.js +679 -0
- package/dist/esm/journal.js.map +1 -0
- package/dist/esm/key.js +36 -33
- package/dist/esm/key.js.map +1 -1
- package/dist/esm/middleware.d.ts +50 -2
- package/dist/esm/middleware.js +335 -208
- package/dist/esm/middleware.js.map +1 -1
- package/dist/esm/ngrok.js +75 -49
- package/dist/esm/ngrok.js.map +1 -1
- package/dist/esm/policy.js +43 -34
- package/dist/esm/policy.js.map +1 -1
- package/dist/esm/projection.js +16 -8
- package/dist/esm/projection.js.map +1 -1
- package/dist/esm/reap.d.ts +238 -0
- package/dist/esm/reap.js +355 -0
- package/dist/esm/reap.js.map +1 -0
- package/dist/esm/reclaim.d.ts +84 -0
- package/dist/esm/reclaim.js +106 -0
- package/dist/esm/reclaim.js.map +1 -0
- package/dist/esm/remote-tools.js +73 -62
- package/dist/esm/remote-tools.js.map +1 -1
- package/dist/esm/run.d.ts +93 -25
- package/dist/esm/run.js +274 -79
- package/dist/esm/run.js.map +1 -1
- package/dist/esm/runner.d.ts +119 -2
- package/dist/esm/runner.js +270 -51
- package/dist/esm/runner.js.map +1 -1
- package/dist/esm/sandbox.d.ts +3 -2
- package/dist/esm/sandbox.js +139 -123
- package/dist/esm/sandbox.js.map +1 -1
- package/dist/esm/secrets.js +39 -47
- package/dist/esm/secrets.js.map +1 -1
- package/dist/esm/setup-plan.js +22 -14
- package/dist/esm/setup-plan.js.map +1 -1
- package/dist/esm/shell.d.ts +8 -0
- package/dist/esm/shell.js +197 -158
- package/dist/esm/shell.js.map +1 -1
- package/dist/esm/testkit/conformance.d.ts +16 -0
- package/dist/esm/testkit/conformance.js +97 -0
- package/dist/esm/testkit/conformance.js.map +1 -0
- package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
- package/dist/esm/testkit/journal-conformance.d.ts +51 -0
- package/dist/esm/testkit/journal-conformance.js +378 -0
- package/dist/esm/testkit/journal-conformance.js.map +1 -0
- package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
- package/dist/esm/testkit/reaper-conformance.js +847 -0
- package/dist/esm/testkit/reaper-conformance.js.map +1 -0
- package/dist/esm/testkit/shell-spawn.d.ts +2 -0
- package/dist/esm/testkit/shell-spawn.js +60 -0
- package/dist/esm/testkit/shell-spawn.js.map +1 -0
- package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
- package/dist/esm/testkit/takeover-conformance.js +685 -0
- package/dist/esm/testkit/takeover-conformance.js.map +1 -0
- package/dist/esm/tool-bridge.js +227 -180
- package/dist/esm/tool-bridge.js.map +1 -1
- package/dist/esm/tool-history.d.ts +62 -0
- package/dist/esm/tool-history.js +171 -0
- package/dist/esm/tool-history.js.map +1 -0
- package/dist/esm/watch.js +310 -236
- package/dist/esm/watch.js.map +1 -1
- package/dist/esm/workspace.d.ts +1 -1
- package/dist/esm/workspace.js +49 -28
- package/dist/esm/workspace.js.map +1 -1
- package/package.json +16 -6
- package/skills/ai-sandbox/SKILL.md +658 -20
- package/src/align.ts +297 -0
- package/src/attach-preflight.ts +292 -0
- package/src/capabilities.ts +4 -13
- package/src/chunk-identity.ts +154 -0
- package/src/claim.ts +479 -0
- package/src/contracts.ts +13 -0
- package/src/driver.ts +205 -0
- package/src/durability.ts +380 -0
- package/src/index.ts +212 -27
- package/src/instance-store.ts +122 -0
- package/src/journal-bytes.ts +136 -0
- package/src/journal-reader.ts +359 -0
- package/src/journal-sweep.ts +406 -0
- package/src/journal.ts +875 -0
- package/src/middleware.ts +470 -30
- package/src/reap.ts +723 -0
- package/src/reclaim.ts +191 -0
- package/src/run.ts +365 -75
- package/src/runner.ts +347 -3
- package/src/sandbox.ts +38 -8
- package/src/shell.ts +106 -38
- package/src/testkit/conformance.ts +117 -0
- package/src/testkit/durable-run-fields-conformance.ts +147 -0
- package/src/testkit/journal-conformance.ts +676 -0
- package/src/testkit/reaper-conformance.ts +1201 -0
- package/src/testkit/shell-spawn.ts +67 -0
- package/src/testkit/takeover-conformance.ts +1040 -0
- package/src/tool-history.ts +245 -0
- package/src/workspace.ts +1 -1
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/run-log.d.ts +0 -81
- package/dist/esm/run-log.js +0 -107
- package/dist/esm/run-log.js.map +0 -1
- package/dist/esm/store.d.ts +0 -53
- package/dist/esm/store.js +0 -34
- package/dist/esm/store.js.map +0 -1
- package/src/run-log.ts +0 -224
- package/src/store.ts +0 -83
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic chunk identity — the prerequisite that makes journal replay
|
|
3
|
+
* exact.
|
|
4
|
+
*
|
|
5
|
+
* The design premise is that re-translating a journal prefix reproduces the
|
|
6
|
+
* chunks a previous host already delivered, so a successor can recognize and
|
|
7
|
+
* skip them (see `align.ts`, a later task). Two things in the default path
|
|
8
|
+
* break that premise:
|
|
9
|
+
*
|
|
10
|
+
* 1. `ChatAdapter.generateId()` — `packages/ai/src/activities/chat/adapter.ts:227`,
|
|
11
|
+
* read directly for this task — is:
|
|
12
|
+
*
|
|
13
|
+
* ```ts
|
|
14
|
+
* protected generateId(): string {
|
|
15
|
+
* return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
|
|
16
|
+
* }
|
|
17
|
+
* ```
|
|
18
|
+
*
|
|
19
|
+
* and every harness translator mints message ids through it (wired as
|
|
20
|
+
* `genId` in the Grok Build, Claude Code, and Codex text adapters). Both
|
|
21
|
+
* `Date.now()` and `Math.random()` are non-reproducible: replaying the same
|
|
22
|
+
* journal bytes through a second `generateId()` call produces a different
|
|
23
|
+
* id every time, so "same bytes ⇒ same chunks" is false on the journaled
|
|
24
|
+
* path today. {@link createRunScopedIdGen} replaces it with a run-scoped
|
|
25
|
+
* counter that has neither a clock nor randomness, so two generators built
|
|
26
|
+
* from the same `runId` always produce the same sequence.
|
|
27
|
+
* 2. Chunks also carry `timestamp: Date.now()`, which cannot be reproduced at
|
|
28
|
+
* all, deterministic id or not. {@link chunkFingerprint} therefore excludes
|
|
29
|
+
* exactly that field — nothing downstream keys on a chunk's timestamp, so
|
|
30
|
+
* leaving it wall-clock is safe, but every other field must participate in
|
|
31
|
+
* the comparison or a real divergence would go undetected.
|
|
32
|
+
*/
|
|
33
|
+
import type { StreamChunk } from '@tanstack/ai'
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* A deterministic id generator scoped to one run.
|
|
37
|
+
*
|
|
38
|
+
* Passed as the harness translators' `genId`, so translating the same journal
|
|
39
|
+
* prefix twice mints the same message ids. The counter is per-generator, so a
|
|
40
|
+
* replay must create a fresh one and start from the journal's first byte —
|
|
41
|
+
* which is exactly what the alignment step (a later task) assumes.
|
|
42
|
+
*
|
|
43
|
+
* No clock, no `Math.random`, no crypto: `next` is the only state, and it is
|
|
44
|
+
* seeded fresh for every call to this factory.
|
|
45
|
+
*/
|
|
46
|
+
export function createRunScopedIdGen(runId: string): () => string {
|
|
47
|
+
let next = 0
|
|
48
|
+
return () => {
|
|
49
|
+
const id = `${runId}-${next}`
|
|
50
|
+
next += 1
|
|
51
|
+
return id
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Fields excluded from a fingerprint because they are wall-clock and therefore
|
|
57
|
+
* unreproducible. Kept as an explicit set so adding one is a deliberate,
|
|
58
|
+
* reviewable act rather than a silent loosening of the comparison.
|
|
59
|
+
*/
|
|
60
|
+
const VOLATILE_FIELDS: ReadonlySet<string> = new Set(['timestamp'])
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* The conversation id every chunk carries. Excluded from
|
|
64
|
+
* {@link chunkFingerprintIgnoringThreadId} — and ONLY from that variant — so
|
|
65
|
+
* alignment can tell an id-only mismatch from a real content divergence.
|
|
66
|
+
*/
|
|
67
|
+
const THREAD_ID_FIELD = 'threadId'
|
|
68
|
+
|
|
69
|
+
const VOLATILE_AND_THREAD_ID: ReadonlySet<string> = new Set([
|
|
70
|
+
...VOLATILE_FIELDS,
|
|
71
|
+
THREAD_ID_FIELD,
|
|
72
|
+
])
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* `dropped` applies at the TOP LEVEL only (nested calls pass `undefined`).
|
|
76
|
+
* A `threadId` nested inside, say, a tool call's arguments is real content and
|
|
77
|
+
* must keep participating in the comparison.
|
|
78
|
+
*/
|
|
79
|
+
function stableStringify(
|
|
80
|
+
value: unknown,
|
|
81
|
+
dropped: ReadonlySet<string> | undefined,
|
|
82
|
+
): string {
|
|
83
|
+
if (value === null) return 'null'
|
|
84
|
+
if (Array.isArray(value)) {
|
|
85
|
+
return `[${value.map((item) => stableStringify(item, undefined)).join(',')}]`
|
|
86
|
+
}
|
|
87
|
+
if (typeof value === 'object') {
|
|
88
|
+
const record: Record<string, unknown> = value as Record<string, unknown>
|
|
89
|
+
const keys = Object.keys(record)
|
|
90
|
+
.filter((key) => dropped === undefined || !dropped.has(key))
|
|
91
|
+
.sort()
|
|
92
|
+
const parts = keys.map((key) => {
|
|
93
|
+
const entry = record[key]
|
|
94
|
+
const encoded =
|
|
95
|
+
entry === undefined
|
|
96
|
+
? '"__undefined__"'
|
|
97
|
+
: stableStringify(entry, undefined)
|
|
98
|
+
return `${JSON.stringify(key)}:${encoded}`
|
|
99
|
+
})
|
|
100
|
+
return `{${parts.join(',')}}`
|
|
101
|
+
}
|
|
102
|
+
const encoded = JSON.stringify(value)
|
|
103
|
+
return encoded === undefined ? 'null' : encoded
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* A stable, order-independent identity for a chunk, excluding wall-clock
|
|
108
|
+
* fields. Used to recognize the chunks a previous host already appended.
|
|
109
|
+
*
|
|
110
|
+
* - **Key-order independent**: object keys are sorted before stringifying, so
|
|
111
|
+
* a JSON round trip through the journal (which does not preserve key order)
|
|
112
|
+
* cannot spuriously diverge.
|
|
113
|
+
* - **Recurses into nested arrays and objects**: tool-call arguments are
|
|
114
|
+
* nested, and a shallow fingerprint would miss a changed argument.
|
|
115
|
+
* - **Excludes exactly `VOLATILE_FIELDS`** (`timestamp`) — everything else
|
|
116
|
+
* participates, including fields whose value is `undefined`.
|
|
117
|
+
* - **Distinguishes present-but-`undefined` from absent**: `undefined` is
|
|
118
|
+
* encoded as the sentinel string `"__undefined__"` rather than dropped, so
|
|
119
|
+
* `{a: undefined}` and `{}` do not collide. A translator emitting an
|
|
120
|
+
* explicit `undefined` is a different chunk shape and must fingerprint
|
|
121
|
+
* differently.
|
|
122
|
+
*/
|
|
123
|
+
export function chunkFingerprint(chunk: StreamChunk): string {
|
|
124
|
+
return stableStringify(chunk, VOLATILE_FIELDS)
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* {@link chunkFingerprint} with the chunk's own `threadId` also excluded.
|
|
129
|
+
*
|
|
130
|
+
* NOT an alternative identity — never use it to decide that two chunks are the
|
|
131
|
+
* same. Its single purpose is DIAGNOSIS: when a replay diverges from the stored
|
|
132
|
+
* log, comparing both fingerprints answers "did the agent behave differently, or
|
|
133
|
+
* did only the conversation id move?". Two chunks that match here but not under
|
|
134
|
+
* {@link chunkFingerprint} differ in `threadId` and nothing else, which is a
|
|
135
|
+
* misconfigured attach route rather than a determinism regression (see
|
|
136
|
+
* `JournalReplayThreadIdMismatchError` in `align.ts`).
|
|
137
|
+
*/
|
|
138
|
+
export function chunkFingerprintIgnoringThreadId(chunk: StreamChunk): string {
|
|
139
|
+
return stableStringify(chunk, VOLATILE_AND_THREAD_ID)
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* A chunk's own `threadId`, or `undefined` when it carries none.
|
|
144
|
+
*
|
|
145
|
+
* Reads the field structurally rather than narrowing on `chunk.type`: nearly
|
|
146
|
+
* every member of the `StreamChunk` union declares `threadId?: string`, and an
|
|
147
|
+
* exhaustive switch would have to be revisited for each new member while adding
|
|
148
|
+
* nothing — a chunk with no `threadId` is exactly the `undefined` case.
|
|
149
|
+
*/
|
|
150
|
+
export function chunkThreadId(chunk: StreamChunk): string | undefined {
|
|
151
|
+
const record: Record<string, unknown> = chunk as Record<string, unknown>
|
|
152
|
+
const value = record[THREAD_ID_FIELD]
|
|
153
|
+
return typeof value === 'string' ? value : undefined
|
|
154
|
+
}
|
package/src/claim.ts
ADDED
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The single-writer claim: what makes a takeover safe to attempt at all.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS MODULE EXISTS. `alignToStoredLog` decides where its appends start by
|
|
5
|
+
* reading `durability.snapshot()`, and `snapshot()` carries NO LOCK — core says
|
|
6
|
+
* so explicitly (`packages/ai/src/stream-durability.ts`: "a concurrent `append`
|
|
7
|
+
* may land immediately after the snapshot is taken"). If two hosts drive one
|
|
8
|
+
* run, both snapshot, both compute a "remainder", and both append it. The log
|
|
9
|
+
* then holds the same logical chunk twice under two different offsets, and the
|
|
10
|
+
* client CANNOT survive that: `ai-client`'s de-dup is keyed on the adapter's
|
|
11
|
+
* offset string, so a re-appended chunk looks new, and the stream processor
|
|
12
|
+
* applies text and tool-argument deltas unconditionally. The visible result is
|
|
13
|
+
* doubled message text and `{"a":1}{"a":1}` tool arguments.
|
|
14
|
+
*
|
|
15
|
+
* Takeover is by definition two hosts wanting one run, so nothing may read a
|
|
16
|
+
* journal for a run it has not claimed.
|
|
17
|
+
*
|
|
18
|
+
* THREE LAYERS, strongest first:
|
|
19
|
+
*
|
|
20
|
+
* 1. **The lease.** {@link withRunClaim} runs the whole drive inside
|
|
21
|
+
* `LockStore.withLock('run-driver:<runId>', …)`, so the snapshot and every
|
|
22
|
+
* append that follows are one critical section. A lease-backed lock aborts
|
|
23
|
+
* the callback signal the moment ownership is lost, and
|
|
24
|
+
* {@link fenceDurability} turns that into a thrown {@link RunClaimLostError}
|
|
25
|
+
* BEFORE the append reaches the log.
|
|
26
|
+
* 2. **The epoch.** Each successful claim bumps `RunRecord.driverEpoch`.
|
|
27
|
+
* {@link fenceDurability} re-reads it every
|
|
28
|
+
* {@link DEFAULT_EPOCH_RECHECK_APPENDS} appends and refuses to append once a
|
|
29
|
+
* higher epoch exists. This covers what a lease cannot: an
|
|
30
|
+
* `InMemoryLockStore`, whose signal is a fresh `AbortController().signal`
|
|
31
|
+
* that is never aborted, and any backend whose renewal is coarser than the
|
|
32
|
+
* run's append rate. Once EITHER fence has refused an append, the fence
|
|
33
|
+
* latches shut and every later append refuses without re-reading anything.
|
|
34
|
+
* 3. **Quiescence.** {@link awaitLogQuiescence} requires the stored log to stop
|
|
35
|
+
* growing before the successor appends anything, so a predecessor that is
|
|
36
|
+
* still writing is OBSERVED rather than raced.
|
|
37
|
+
*
|
|
38
|
+
* THE LOG IS NOT THE ONLY AUTHORITATIVE CHANNEL. A host that has lost its claim
|
|
39
|
+
* must not write authoritative facts about the run through ANY seam, and there
|
|
40
|
+
* are two: the event log and the run RECORD. Fencing only the log moves the harm
|
|
41
|
+
* rather than removing it — a superseded driver whose append was refused folds
|
|
42
|
+
* that refusal into a terminal `runs.update`, so the record reads `'failed'` for
|
|
43
|
+
* a run the successor is healthily streaming, and `isTerminalRunStatus` (which
|
|
44
|
+
* `findActiveRun`, the resume driver, and `reapDetachedRuns` all branch on) then
|
|
45
|
+
* answers `true` for a live run. {@link fenceRunStore} closes that seam; both
|
|
46
|
+
* fences share one per-claim latch so they can never disagree about whether the
|
|
47
|
+
* claim is still held.
|
|
48
|
+
*
|
|
49
|
+
* WHY THE EPOCH RE-CHECK COUNTS APPENDS, NOT MILLISECONDS. `pipeToRunLog`
|
|
50
|
+
* appends ONE chunk per call, so a time-based interval couples the fence's
|
|
51
|
+
* resolution to the run's chunk rate: at 500 chunks/sec a 2s interval lets a
|
|
52
|
+
* superseded driver write ~1000 chunks before it notices. A count gives a hard
|
|
53
|
+
* bound independent of rate — see {@link DEFAULT_EPOCH_RECHECK_APPENDS}.
|
|
54
|
+
*
|
|
55
|
+
* WHAT THIS IS NOT. It is not airtight fencing.
|
|
56
|
+
*
|
|
57
|
+
* - A predecessor paused (GC, VM suspend) for longer than the quiescence
|
|
58
|
+
* window, between its last fence check and its append landing at the backend,
|
|
59
|
+
* can still write one batch. Closing that requires a compare-and-set on the
|
|
60
|
+
* durability write; `StreamDurability.append` has no such parameter and this
|
|
61
|
+
* phase deliberately does not add one.
|
|
62
|
+
* - Layer 3 is only meaningful across PROCESSES. On a single-process
|
|
63
|
+
* `InMemoryLockStore` the two claims are serialized by the lock, not
|
|
64
|
+
* concurrent, so `awaitLogQuiescence` can never observe a predecessor still
|
|
65
|
+
* writing there — and consequently no unit test on that backend proves layer
|
|
66
|
+
* 3 does anything. What the tests do prove on that backend is layer 2.
|
|
67
|
+
*
|
|
68
|
+
* The mitigation for both is deployment-level: use a lease-backed distributed
|
|
69
|
+
* `LockStore`, and keep `fenceQuietMs` above the lease's renewal interval.
|
|
70
|
+
*/
|
|
71
|
+
import { isTerminalRunStatus } from '@tanstack/ai'
|
|
72
|
+
import type { LockStore } from '@tanstack/ai/locks'
|
|
73
|
+
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
74
|
+
import type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'
|
|
75
|
+
|
|
76
|
+
/** Quiescence window before a successor's first append. */
|
|
77
|
+
export const DEFAULT_FENCE_QUIET_MS = 5_000
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Appends a fenced log makes between `driverEpoch` re-reads.
|
|
81
|
+
*
|
|
82
|
+
* Deliberately a COUNT, not an interval: `pipeToRunLog` appends one chunk per
|
|
83
|
+
* call, so this bounds a superseded driver to at most 31 further chunk batches
|
|
84
|
+
* (the bump can land immediately after a check) regardless of how fast the run
|
|
85
|
+
* streams. At 500 chunks/sec that worst case is ~62ms of writes; at 5
|
|
86
|
+
* chunks/sec it is ~6s of writes — either way 31 chunks, never ~1000.
|
|
87
|
+
*
|
|
88
|
+
* The cost of a smaller number is one extra `RunStore.get` per 32 chunks.
|
|
89
|
+
*/
|
|
90
|
+
export const DEFAULT_EPOCH_RECHECK_APPENDS = 32
|
|
91
|
+
|
|
92
|
+
/** Probes {@link awaitLogQuiescence} makes before giving up. */
|
|
93
|
+
const MAX_QUIESCENCE_PROBES = 6
|
|
94
|
+
|
|
95
|
+
/** Lock key for a run's driver. Per-run, so two runs never serialize. */
|
|
96
|
+
export function runDriverLockKey(runId: string): string {
|
|
97
|
+
return `run-driver:${runId}`
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** The claim was never acquired, so the caller must not drive the run. */
|
|
101
|
+
export class RunClaimNotAcquiredError extends Error {
|
|
102
|
+
constructor(
|
|
103
|
+
readonly runId: string,
|
|
104
|
+
readonly reason: 'terminal' | 'unknown' | 'superseded',
|
|
105
|
+
) {
|
|
106
|
+
super(`run ${runId}: driver claim not acquired (${reason})`)
|
|
107
|
+
this.name = 'RunClaimNotAcquiredError'
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** The claim was held and has been superseded; stop writing immediately. */
|
|
112
|
+
export class RunClaimLostError extends Error {
|
|
113
|
+
constructor(
|
|
114
|
+
readonly runId: string,
|
|
115
|
+
readonly heldEpoch: number,
|
|
116
|
+
readonly observedEpoch: number | 'lease-lost',
|
|
117
|
+
) {
|
|
118
|
+
super(
|
|
119
|
+
`run ${runId}: driver claim lost (held epoch ${heldEpoch}, observed ${observedEpoch})`,
|
|
120
|
+
)
|
|
121
|
+
this.name = 'RunClaimLostError'
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** A held claim on one run. */
|
|
126
|
+
export interface RunClaim {
|
|
127
|
+
runId: string
|
|
128
|
+
/** This driver's fencing token; strictly greater than any predecessor's. */
|
|
129
|
+
epoch: number
|
|
130
|
+
/** Aborts when the lock can no longer guarantee ownership. */
|
|
131
|
+
signal: AbortSignal
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
export interface WithRunClaimOptions {
|
|
135
|
+
runs: RunStore
|
|
136
|
+
locks: LockStore
|
|
137
|
+
runId: string
|
|
138
|
+
/**
|
|
139
|
+
* Quiescence window for {@link awaitLogQuiescence}. Defaults to
|
|
140
|
+
* {@link DEFAULT_FENCE_QUIET_MS}.
|
|
141
|
+
*
|
|
142
|
+
* `withRunClaim` itself does not read this: it has no durability handle. It
|
|
143
|
+
* lives here so a caller assembling a drive passes ONE options object to
|
|
144
|
+
* `withRunClaim`, `awaitLogQuiescence`, and {@link fenceDurability} instead of
|
|
145
|
+
* three that can drift apart.
|
|
146
|
+
*/
|
|
147
|
+
fenceQuietMs?: number
|
|
148
|
+
/**
|
|
149
|
+
* Forwarded to {@link fenceDurability}. Defaults to
|
|
150
|
+
* {@link DEFAULT_EPOCH_RECHECK_APPENDS}. Same rationale as `fenceQuietMs`.
|
|
151
|
+
*/
|
|
152
|
+
epochRecheckAppends?: number
|
|
153
|
+
logger?: InternalLogger
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Claim exclusive driver rights on `runId` for the duration of `fn`.
|
|
158
|
+
*
|
|
159
|
+
* The ENTIRE body runs inside the lock, so a snapshot taken by `fn` and every
|
|
160
|
+
* append that follows it sit in one critical section.
|
|
161
|
+
*
|
|
162
|
+
* Rejects with {@link RunClaimNotAcquiredError} when the run is unknown or
|
|
163
|
+
* already terminal — a terminal run has nothing left to drive, and bumping its
|
|
164
|
+
* epoch would fence out nobody while confusing an operator reading the record.
|
|
165
|
+
*
|
|
166
|
+
* The epoch is bumped INSIDE the lock and only after those checks pass, so a
|
|
167
|
+
* refused claim leaves `driverEpoch` untouched.
|
|
168
|
+
*/
|
|
169
|
+
export async function withRunClaim<T>(
|
|
170
|
+
options: WithRunClaimOptions,
|
|
171
|
+
fn: (claim: RunClaim) => Promise<T>,
|
|
172
|
+
): Promise<T> {
|
|
173
|
+
const { runs, locks, runId, logger } = options
|
|
174
|
+
return locks.withLock(runDriverLockKey(runId), async (signal) => {
|
|
175
|
+
const record = await runs.get(runId)
|
|
176
|
+
if (record === null) {
|
|
177
|
+
throw new RunClaimNotAcquiredError(runId, 'unknown')
|
|
178
|
+
}
|
|
179
|
+
if (isTerminalRunStatus(record.status)) {
|
|
180
|
+
throw new RunClaimNotAcquiredError(runId, 'terminal')
|
|
181
|
+
}
|
|
182
|
+
const epoch = (record.driverEpoch ?? 0) + 1
|
|
183
|
+
await runs.update(runId, { driverEpoch: epoch })
|
|
184
|
+
logger?.sandbox(`run ${runId}: driver claim acquired at epoch ${epoch}`, {
|
|
185
|
+
runId,
|
|
186
|
+
epoch,
|
|
187
|
+
})
|
|
188
|
+
return fn({ runId, epoch, signal })
|
|
189
|
+
})
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Wait until the stored log stops growing, then answer how many entries it
|
|
194
|
+
* holds.
|
|
195
|
+
*
|
|
196
|
+
* Uses `snapshot()`, never `read()`: `read` tails and only resolves once the log
|
|
197
|
+
* is terminalized or the caller aborts, and a taken-over run's log is open by
|
|
198
|
+
* definition — the host that would have closed it is the host that died.
|
|
199
|
+
*
|
|
200
|
+
* Rejects rather than looping forever. A log that never quiesces means a
|
|
201
|
+
* predecessor is still actively writing, which is a condition to surface, not to
|
|
202
|
+
* append into.
|
|
203
|
+
*
|
|
204
|
+
* This only detects a CONCURRENT predecessor, which means it can only fire when
|
|
205
|
+
* the two drivers are in different processes. Within one process an
|
|
206
|
+
* `InMemoryLockStore` serializes the claims, so the predecessor has already
|
|
207
|
+
* stopped by the time the successor probes.
|
|
208
|
+
*/
|
|
209
|
+
export async function awaitLogQuiescence<TOffset extends string = string>(
|
|
210
|
+
durability: StreamDurability<TOffset>,
|
|
211
|
+
quietMs: number,
|
|
212
|
+
): Promise<number> {
|
|
213
|
+
let previous = (await durability.snapshot()).length
|
|
214
|
+
for (let probe = 0; probe < MAX_QUIESCENCE_PROBES; probe += 1) {
|
|
215
|
+
await sleep(quietMs)
|
|
216
|
+
const current = (await durability.snapshot()).length
|
|
217
|
+
if (current === previous) return current
|
|
218
|
+
previous = current
|
|
219
|
+
}
|
|
220
|
+
throw new Error(
|
|
221
|
+
`journal takeover: the event log never quiesced after ${MAX_QUIESCENCE_PROBES} probes (${previous} entries and still growing); another host is still driving this run`,
|
|
222
|
+
)
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
function sleep(ms: number): Promise<void> {
|
|
226
|
+
if (ms <= 0) return Promise.resolve()
|
|
227
|
+
return new Promise<void>((resolve) => setTimeout(resolve, ms))
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* The one-way "this claim is gone" flag, latched by the first refusal.
|
|
232
|
+
*
|
|
233
|
+
* Keyed by the claim rather than held in one wrapper's closure because a claim
|
|
234
|
+
* has TWO fenced seams — its log ({@link fenceDurability}) and its record
|
|
235
|
+
* ({@link fenceRunStore}) — and a latch per wrapper would let them disagree: a
|
|
236
|
+
* lease that flaps back to `aborted === false`, or an epoch read that fails,
|
|
237
|
+
* would re-open the fence that had not refused yet. Losing a claim is not
|
|
238
|
+
* transient, so one observation must close both.
|
|
239
|
+
*
|
|
240
|
+
* A `WeakMap` and not a field on {@link RunClaim} so the claim stays the plain
|
|
241
|
+
* data structure core's `RunDriverOptions.claim` types it as, and so the latch is
|
|
242
|
+
* collected with the claim.
|
|
243
|
+
*/
|
|
244
|
+
interface ClaimLatch {
|
|
245
|
+
/** `undefined` while the fence is open; otherwise the refusal to replay. */
|
|
246
|
+
lost: RunClaimLostError | undefined
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
const CLAIM_LATCHES = new WeakMap<RunClaim, ClaimLatch>()
|
|
250
|
+
|
|
251
|
+
function latchFor(claim: RunClaim): ClaimLatch {
|
|
252
|
+
const existing = CLAIM_LATCHES.get(claim)
|
|
253
|
+
if (existing !== undefined) return existing
|
|
254
|
+
const latch: ClaimLatch = { lost: undefined }
|
|
255
|
+
CLAIM_LATCHES.set(claim, latch)
|
|
256
|
+
return latch
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* The I/O-free half of the check: the latch and the lease. Synchronous on
|
|
261
|
+
* purpose — a fenced write must be refused BEFORE anything can half-land.
|
|
262
|
+
*/
|
|
263
|
+
function claimLostSynchronously(
|
|
264
|
+
claim: RunClaim,
|
|
265
|
+
latch: ClaimLatch,
|
|
266
|
+
): RunClaimLostError | undefined {
|
|
267
|
+
if (latch.lost !== undefined) return latch.lost
|
|
268
|
+
if (claim.signal.aborted) {
|
|
269
|
+
latch.lost = new RunClaimLostError(claim.runId, claim.epoch, 'lease-lost')
|
|
270
|
+
return latch.lost
|
|
271
|
+
}
|
|
272
|
+
return undefined
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
/**
|
|
276
|
+
* The other half: re-read `driverEpoch` and refuse once a successor exists.
|
|
277
|
+
*
|
|
278
|
+
* A store failure is NOT treated as loss. The lease is the primary fence and it
|
|
279
|
+
* has not fired, so fencing ourselves out on a store blip would kill a healthy
|
|
280
|
+
* driver — and, for the record fence, would suppress a legitimate terminal write
|
|
281
|
+
* and strand the run at `'running'`, which is worse than the write it prevents.
|
|
282
|
+
*/
|
|
283
|
+
async function claimLostByEpoch(
|
|
284
|
+
claim: RunClaim,
|
|
285
|
+
latch: ClaimLatch,
|
|
286
|
+
runs: RunStore,
|
|
287
|
+
): Promise<RunClaimLostError | undefined> {
|
|
288
|
+
let observed: number | undefined
|
|
289
|
+
try {
|
|
290
|
+
observed = (await runs.get(claim.runId))?.driverEpoch
|
|
291
|
+
} catch {
|
|
292
|
+
return undefined
|
|
293
|
+
}
|
|
294
|
+
if (observed !== undefined && observed > claim.epoch) {
|
|
295
|
+
latch.lost = new RunClaimLostError(claim.runId, claim.epoch, observed)
|
|
296
|
+
return latch.lost
|
|
297
|
+
}
|
|
298
|
+
return undefined
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/**
|
|
302
|
+
* Wrap a log so every `append` is fenced by `claim`.
|
|
303
|
+
*
|
|
304
|
+
* `append` is the ONLY fenced method, deliberately:
|
|
305
|
+
*
|
|
306
|
+
* - `close()` must never be fenced. It runs on every teardown path including
|
|
307
|
+
* the teardown caused by losing the claim, and a fenced `close` would leave
|
|
308
|
+
* the record wedged at `'running'` with every live tailer parked forever (a
|
|
309
|
+
* `read` only ends when the log closes).
|
|
310
|
+
* - `read` / `snapshot` / `resumeFrom` do not mutate, so a superseded host
|
|
311
|
+
* reading them is harmless.
|
|
312
|
+
*
|
|
313
|
+
* The lease check is synchronous and happens before any I/O, so a fenced append
|
|
314
|
+
* cannot half-land. The epoch re-check is throttled to `epochRecheckAppends`
|
|
315
|
+
* because it costs a store read and the append path is hot.
|
|
316
|
+
*
|
|
317
|
+
* ONE REFUSAL CLOSES THE FENCE FOR GOOD. The first `append` that is refused —
|
|
318
|
+
* for EITHER cause, lost lease or moved epoch — latches this wrapper shut, and
|
|
319
|
+
* every later `append` refuses immediately without consulting the throttle and
|
|
320
|
+
* without a store read. This is not a nicety:
|
|
321
|
+
*
|
|
322
|
+
* - Losing a claim is not transient. Epochs only move forward and a lease is
|
|
323
|
+
* never handed back, so a wrapper that has refused once can never legitimately
|
|
324
|
+
* append again. Re-deciding per append can only produce a WRONG answer.
|
|
325
|
+
* - The throttle makes that wrong answer reachable. A refusal consumes the
|
|
326
|
+
* re-read budget, so the very next append rides a fresh throttle window and is
|
|
327
|
+
* NOT re-checked. `pipeToRunLog`'s recovery path appends a `RUN_ERROR` right
|
|
328
|
+
* after the refusal it is recovering from, and that log belongs to the
|
|
329
|
+
* SUCCESSOR: a terminal `RUN_ERROR` from a dead host would fail the stream for
|
|
330
|
+
* every client attached to the live, healthy run.
|
|
331
|
+
* - It is also strictly cheaper: a latched boolean replaces a store read.
|
|
332
|
+
*
|
|
333
|
+
* The latch deliberately does NOT extend to `close()` — see above.
|
|
334
|
+
*
|
|
335
|
+
* PASSES THE OFFSET TYPE THROUGH, rather than collapsing it to `string`. The
|
|
336
|
+
* fence sits mid-chain between a caller's log and `pipeToRunLog`, so widening
|
|
337
|
+
* here would reintroduce the branded-offset wall one layer in: a
|
|
338
|
+
* `StreamDurability<DurableStreamOffset>` would go in and a
|
|
339
|
+
* `StreamDurability<string>` would come out, which is not assignable back to
|
|
340
|
+
* the caller's own type.
|
|
341
|
+
*/
|
|
342
|
+
export function fenceDurability<TOffset extends string = string>(
|
|
343
|
+
durability: StreamDurability<TOffset>,
|
|
344
|
+
claim: RunClaim,
|
|
345
|
+
options: { runs: RunStore; epochRecheckAppends?: number },
|
|
346
|
+
): StreamDurability<TOffset> {
|
|
347
|
+
const recheckAppends = Math.max(
|
|
348
|
+
1,
|
|
349
|
+
Math.trunc(options.epochRecheckAppends ?? DEFAULT_EPOCH_RECHECK_APPENDS),
|
|
350
|
+
)
|
|
351
|
+
// Seeded at the threshold so the FIRST append always re-reads the epoch: a
|
|
352
|
+
// successor may have claimed between this fence being built and its first
|
|
353
|
+
// write.
|
|
354
|
+
let appendsSinceEpochRead = recheckAppends
|
|
355
|
+
// Latched by the FIRST refusal and never cleared, and SHARED with this claim's
|
|
356
|
+
// record fence so the two seams cannot disagree.
|
|
357
|
+
const latch = latchFor(claim)
|
|
358
|
+
|
|
359
|
+
async function assertHeld(): Promise<void> {
|
|
360
|
+
// Layer 1 plus the latch: no I/O, so nothing has been written yet, and once
|
|
361
|
+
// refused no throttle and no store read can let a later append through.
|
|
362
|
+
const synchronous = claimLostSynchronously(claim, latch)
|
|
363
|
+
if (synchronous !== undefined) throw synchronous
|
|
364
|
+
if (appendsSinceEpochRead < recheckAppends) {
|
|
365
|
+
appendsSinceEpochRead += 1
|
|
366
|
+
return
|
|
367
|
+
}
|
|
368
|
+
appendsSinceEpochRead = 1
|
|
369
|
+
// Layer 2, throttled because it costs a store read.
|
|
370
|
+
const byEpoch = await claimLostByEpoch(claim, latch, options.runs)
|
|
371
|
+
if (byEpoch !== undefined) throw byEpoch
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
return {
|
|
375
|
+
resumeFrom: () => durability.resumeFrom(),
|
|
376
|
+
append: async (chunks: Array<StreamChunk>) => {
|
|
377
|
+
await assertHeld()
|
|
378
|
+
return durability.append(chunks)
|
|
379
|
+
},
|
|
380
|
+
read: (offset, signal) => durability.read(offset, signal),
|
|
381
|
+
close: () => durability.close(),
|
|
382
|
+
snapshot: () => durability.snapshot(),
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* Wrap a run store so a TERMINAL record write is fenced by `claim`.
|
|
388
|
+
*
|
|
389
|
+
* The record is the run's other authoritative channel, and the same rule applies
|
|
390
|
+
* to it: a host that has lost its claim must not state that the run is over. It
|
|
391
|
+
* reaches this seam by the most ordinary route — `pipeToRunLog` catches the
|
|
392
|
+
* `RunClaimLostError` its refused append threw, folds it in, and calls
|
|
393
|
+
* `finish(ctx, 'failed', …)` — so fencing the log alone only moves where the harm
|
|
394
|
+
* surfaces. `'completed'` and `'aborted'` arrive the same way (an empty stream
|
|
395
|
+
* that never appended; a lease loss that aborts `claim.signal`, which
|
|
396
|
+
* `pipeToRunLog` reads as an abort before it appends anything), which is why the
|
|
397
|
+
* gate is {@link isTerminalRunStatus} and not "did an append refuse".
|
|
398
|
+
*
|
|
399
|
+
* SUPPRESSED, NOT ATTEMPTED-AND-SWALLOWED, and not thrown either. `update`
|
|
400
|
+
* resolves without writing. `pipeToRunLog` must not reject — `RunController.start`
|
|
401
|
+
* consumes its promise fire-and-forget — and a rejection here would additionally
|
|
402
|
+
* make `finish` report the run through the local rebuilt record as if the store
|
|
403
|
+
* had broken, which is a different and false fact.
|
|
404
|
+
*
|
|
405
|
+
* WHAT IS *NOT* FENCED, deliberately:
|
|
406
|
+
*
|
|
407
|
+
* - **`close()`** is not on this seam at all, and must stay off it: see
|
|
408
|
+
* {@link fenceDurability}. A wedged `'running'` record with tailers parked
|
|
409
|
+
* forever is worse than the write being prevented.
|
|
410
|
+
* - **Non-terminal writes pass through**, including `detachedSince` and
|
|
411
|
+
* `sandboxKey` written by a superseded host. They are stale, but staleness is
|
|
412
|
+
* not the harm being fixed: none of them can make a live run look finished, so
|
|
413
|
+
* none can mislead `isTerminalRunStatus`, `findActiveRun`, or the reaper. They
|
|
414
|
+
* are also self-healing — the successor owns those fields and overwrites them —
|
|
415
|
+
* whereas over-suppressing strands a record: `createOrResume` is how the row
|
|
416
|
+
* comes into existence at all, and refusing a non-terminal write on a
|
|
417
|
+
* mis-observed loss would leave a run with no record to recover from. Suppress
|
|
418
|
+
* the writes that assert an outcome; let bookkeeping through.
|
|
419
|
+
* - **Reads** (`get`, `listByThread`, `listReclaimable`, `findActiveRun`) do not
|
|
420
|
+
* mutate, so a superseded host reading them is harmless. `finish`'s terminal
|
|
421
|
+
* re-read therefore still works and answers with the SUCCESSOR's live record,
|
|
422
|
+
* which is the truthful thing to resolve with.
|
|
423
|
+
* - **Another run's record.** The fence knows about `claim.runId` only; a write
|
|
424
|
+
* aimed elsewhere is not this claim's to judge.
|
|
425
|
+
*
|
|
426
|
+
* The OPTIONAL methods (`listByThread`, `listReclaimable`) are forwarded only
|
|
427
|
+
* when the wrapped store actually has them: consumers feature-detect
|
|
428
|
+
* (`store.listReclaimable?.(…)`), so materializing one that delegates to a
|
|
429
|
+
* missing method would turn a graceful degrade into a `TypeError`.
|
|
430
|
+
* `findActiveRun` is required on the contract, so it forwards unconditionally.
|
|
431
|
+
*/
|
|
432
|
+
export function fenceRunStore(
|
|
433
|
+
runs: RunStore,
|
|
434
|
+
claim: RunClaim,
|
|
435
|
+
options: { logger?: InternalLogger } = {},
|
|
436
|
+
): RunStore {
|
|
437
|
+
const latch = latchFor(claim)
|
|
438
|
+
// Bound, not merely captured: the store may be a class instance
|
|
439
|
+
// (`InMemoryRunStore`), whose methods need their receiver.
|
|
440
|
+
const listByThread = runs.listByThread?.bind(runs)
|
|
441
|
+
const listReclaimable = runs.listReclaimable?.bind(runs)
|
|
442
|
+
|
|
443
|
+
return {
|
|
444
|
+
createOrResume: (input) => runs.createOrResume(input),
|
|
445
|
+
get: (runId) => runs.get(runId),
|
|
446
|
+
findActiveRun: (threadId) => runs.findActiveRun(threadId),
|
|
447
|
+
update: async (runId, patch) => {
|
|
448
|
+
const status = patch.status
|
|
449
|
+
if (
|
|
450
|
+
runId !== claim.runId ||
|
|
451
|
+
status === undefined ||
|
|
452
|
+
!isTerminalRunStatus(status)
|
|
453
|
+
) {
|
|
454
|
+
return runs.update(runId, patch)
|
|
455
|
+
}
|
|
456
|
+
const lost =
|
|
457
|
+
claimLostSynchronously(claim, latch) ??
|
|
458
|
+
// Unthrottled, unlike the append path: a terminal write happens once per
|
|
459
|
+
// run, so one store read is not a hot cost — and it is the read that
|
|
460
|
+
// catches a superseded driver whose stream ended without ever appending.
|
|
461
|
+
(await claimLostByEpoch(claim, latch, runs))
|
|
462
|
+
if (lost === undefined) return runs.update(runId, patch)
|
|
463
|
+
// Absorbing this silently would make it invisible: a detached run has no
|
|
464
|
+
// caller to report to. The logger is consumer-supplied, so a throwing sink
|
|
465
|
+
// must not turn a suppression into a rejection.
|
|
466
|
+
try {
|
|
467
|
+
options.logger?.sandbox(
|
|
468
|
+
`run ${runId}: suppressed a terminal '${status}' record write from a superseded driver`,
|
|
469
|
+
{ runId, status, heldEpoch: claim.epoch, error: lost },
|
|
470
|
+
)
|
|
471
|
+
} catch {
|
|
472
|
+
// Intentionally empty: there is no second channel to report on.
|
|
473
|
+
}
|
|
474
|
+
return undefined
|
|
475
|
+
},
|
|
476
|
+
...(listByThread === undefined ? {} : { listByThread }),
|
|
477
|
+
...(listReclaimable === undefined ? {} : { listReclaimable }),
|
|
478
|
+
}
|
|
479
|
+
}
|
package/src/contracts.ts
CHANGED
|
@@ -36,6 +36,19 @@ export interface SandboxCapabilities {
|
|
|
36
36
|
* file + shell redirection.
|
|
37
37
|
*/
|
|
38
38
|
writableStdin: boolean
|
|
39
|
+
/**
|
|
40
|
+
* A spawned process can be forcibly terminated via {@link SpawnHandle.kill}
|
|
41
|
+
* and aborted mid-flight via the {@link ProcessOptions.signal} passed to
|
|
42
|
+
* {@link SandboxProcess.spawn}. `true` for host/Docker; some edge providers
|
|
43
|
+
* (e.g. Cloudflare) implement `kill()` as a no-op and drop the abort signal
|
|
44
|
+
* entirely, so a long-running follower process (e.g. `tail -f`) started
|
|
45
|
+
* there can never be stopped by the caller — only polled and abandoned.
|
|
46
|
+
* Callers MUST branch on this before relying on `kill`/abort to reclaim a
|
|
47
|
+
* background process: a bring-your-own provider that omits it would
|
|
48
|
+
* otherwise be silently treated as killable, leaking an unstoppable process
|
|
49
|
+
* inside the sandbox.
|
|
50
|
+
*/
|
|
51
|
+
killableProcesses: boolean
|
|
39
52
|
/** Capture/restore filesystem snapshots via {@link SandboxHandle.snapshot}. */
|
|
40
53
|
snapshots: boolean
|
|
41
54
|
/** Declarative network egress allow/deny policy. */
|