@preventive/triage 1.0.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/common/save-error-reason.ts +53 -0
- package/common/utf8.d.ts +13 -0
- package/common/utf8.js +57 -0
- package/out/brotli-fallback.js +3 -0
- package/out/client-sync.js +15 -0
- package/out/graph.js +4 -0
- package/out/icon-maskable.svg +5 -0
- package/out/icon.svg +5 -0
- package/out/index.html +78 -0
- package/out/manifest.webmanifest +30 -0
- package/out/prism.js +14 -0
- package/out/terminal.js +39 -0
- package/out/view.css +1 -0
- package/out/view.html +12 -0
- package/out/view.js +138 -0
- package/package.json +129 -0
- package/server/auth.ts +99 -0
- package/server/config.example.json +3 -0
- package/server/config.ts +196 -0
- package/server/db-neon.ts +374 -0
- package/server/db-revision-sql.ts +152 -0
- package/server/db-stmt.ts +53 -0
- package/server/db.ts +577 -0
- package/server/http.ts +142 -0
- package/server/hub.ts +98 -0
- package/server/index.ts +353 -0
- package/server/lifecycle.ts +177 -0
- package/server/neon-driver.ts +26 -0
- package/server/objstore/blob-fs.ts +164 -0
- package/server/objstore/blob-vercel.ts +508 -0
- package/server/objstore/blob.ts +169 -0
- package/server/objstore/fs.ts +67 -0
- package/server/objstore/handlers.ts +235 -0
- package/server/objstore/init.ts +118 -0
- package/server/objstore/reaper.ts +199 -0
- package/server/objstore/rest.ts +484 -0
- package/server/objstore/sign.ts +164 -0
- package/server/objstore/store-neon.ts +351 -0
- package/server/objstore/store.ts +799 -0
- package/server/objstore/tokens.ts +168 -0
- package/server/origin.ts +68 -0
- package/server/peer.ts +38 -0
- package/server/sign.ts +231 -0
- package/server/static.ts +374 -0
- package/server/sync-handlers.ts +311 -0
- package/server/util.ts +27 -0
- package/server/validation.ts +36 -0
- package/server/ws-server.ts +245 -0
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
// Filesystem operations for the v1.objstore module — paths +
|
|
2
|
+
// async durable-write primitives. Post-startup callers go through
|
|
3
|
+
// `fs/promises` so a slow disk doesn't block the event loop and
|
|
4
|
+
// stall unrelated requests (WS heartbeats, other workspaces).
|
|
5
|
+
//
|
|
6
|
+
// `liveFilePath` / `stagingFilePath` are the canonical layout —
|
|
7
|
+
// the reaper derives the same shape. The live blob is CONTENT-
|
|
8
|
+
// ADDRESSED: its filename is the content hash, not the resourceTag,
|
|
9
|
+
// so a hash names exactly one immutable byte-string. The base64url
|
|
10
|
+
// alphabet for tag / contentHash / stagingId means no `..` traversal
|
|
11
|
+
// is reachable; validators in store.ts gate inputs at the wire
|
|
12
|
+
// boundary.
|
|
13
|
+
|
|
14
|
+
import { mkdir, open, rename, unlink } from 'node:fs/promises'
|
|
15
|
+
import { basename, dirname, join } from 'node:path'
|
|
16
|
+
import { errMsg } from '../util.ts'
|
|
17
|
+
|
|
18
|
+
export function liveFilePath(dir: string, tag: string, contentHash: string): string {
|
|
19
|
+
return join(dir, tag, `${contentHash}.bin`)
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function stagingFilePath(dir: string, tag: string, stagingId: string): string {
|
|
23
|
+
return join(dir, tag, '.staging', `${stagingId}.bin`)
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export async function ensureStagingDir(root: string, tag: string): Promise<void> {
|
|
27
|
+
await mkdir(join(root, tag, '.staging'), { recursive: true })
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// fsync(staging) → rename(staging → live) → fsync(parent dir).
|
|
31
|
+
// Order matters: a crash post-rename pre-DB-write leaves a stranded
|
|
32
|
+
// committed-name file (reaper-cleaned), never the inverse. Any
|
|
33
|
+
// failure along the way (ENOENT on the staging file after a racing
|
|
34
|
+
// abort, EACCES, ENOSPC, EIO) → `false`, caller routes through
|
|
35
|
+
// abortPut. Directory fsync isn't supported on every filesystem;
|
|
36
|
+
// the inner catch silently no-ops there. PR #4 review.
|
|
37
|
+
export async function durableRenameStagedToLive(stagingPath: string, livePath: string): Promise<boolean> {
|
|
38
|
+
try {
|
|
39
|
+
let fh = await open(stagingPath, 'r')
|
|
40
|
+
try { await fh.sync() } finally { await fh.close() }
|
|
41
|
+
await rename(stagingPath, livePath)
|
|
42
|
+
try {
|
|
43
|
+
fh = await open(dirname(livePath), 'r')
|
|
44
|
+
try { await fh.sync() } finally { await fh.close() }
|
|
45
|
+
} catch { /* not all FS layers support directory fsync */ }
|
|
46
|
+
return true
|
|
47
|
+
} catch {
|
|
48
|
+
return false
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// Tolerate ENOENT (expected race against the reaper / abort) but
|
|
53
|
+
// log everything else (EACCES/EROFS/EBUSY/EIO) — silently dropping
|
|
54
|
+
// those would let deletes "succeed" while files pile up on disk.
|
|
55
|
+
// Don't propagate: the DB-side row drop already committed by the
|
|
56
|
+
// time callers reach here, and the reaper picks up the stranded
|
|
57
|
+
// file on its next sweep. PR #4 review.
|
|
58
|
+
export async function unlinkIfExists(filePath: string): Promise<void> {
|
|
59
|
+
try { await unlink(filePath) } catch (err: unknown) {
|
|
60
|
+
const code = (err as NodeJS.ErrnoException)?.code
|
|
61
|
+
if (code === 'ENOENT') return
|
|
62
|
+
// Log the basename only — the full path contains the workspace
|
|
63
|
+
// tag (Ed25519 public key) and shouldn't go to operator logs
|
|
64
|
+
// verbatim. PR #4 review H3.
|
|
65
|
+
console.warn(`unlink …/${basename(filePath)} failed:`, code ?? errMsg(err))
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
// v1.objstore WS message handlers — control plane only. Byte
|
|
2
|
+
// transfer lives on the REST plane (rest.ts); these handlers mint
|
|
3
|
+
// the bearer tokens the REST handler validates. Each WS message
|
|
4
|
+
// that authorises byte transfer requires an Ed25519 signature
|
|
5
|
+
// against the workspaceTag, then a short-TTL HMAC token bound to
|
|
6
|
+
// the (tag, resource, version-or-stagingId, length) tuple.
|
|
7
|
+
//
|
|
8
|
+
// Operations that mutate (begin / commit / delete) take NO in-process
|
|
9
|
+
// lock: commit correctness is the atomic version-CAS in commitPut,
|
|
10
|
+
// begin's prev_version check is advisory, and delete is a precondition-
|
|
11
|
+
// checked single-row drop (see store.ts for the full lock-free
|
|
12
|
+
// rationale).
|
|
13
|
+
//
|
|
14
|
+
// Broadcasts to peers subscribed via `workspace-subscribe` reuse
|
|
15
|
+
// the existing subscriber map: `objstore-deleted` is emitted from
|
|
16
|
+
// handleDelete here; `objstore-put` is emitted from the REST PUT
|
|
17
|
+
// handler after commit.
|
|
18
|
+
|
|
19
|
+
import type { WebSocket } from 'ws'
|
|
20
|
+
import {
|
|
21
|
+
type Handle,
|
|
22
|
+
MAX_CONTENT_LENGTH,
|
|
23
|
+
type ObjectRow,
|
|
24
|
+
beginPut,
|
|
25
|
+
deleteObject,
|
|
26
|
+
getLive,
|
|
27
|
+
isValidContentHash,
|
|
28
|
+
isValidSignature,
|
|
29
|
+
isValidTag,
|
|
30
|
+
objectMetaWire,
|
|
31
|
+
} from './store.ts'
|
|
32
|
+
import {
|
|
33
|
+
type ObjstoreDeleteMsg,
|
|
34
|
+
type ObjstoreFetchMsg,
|
|
35
|
+
type ObjstorePutBeginMsg,
|
|
36
|
+
verifyObjstoreDeleteSig,
|
|
37
|
+
verifyObjstoreFetchSig,
|
|
38
|
+
verifyObjstorePutSig,
|
|
39
|
+
} from './sign.ts'
|
|
40
|
+
import { type TokenSecret, mintGetToken, mintPutToken } from './tokens.ts'
|
|
41
|
+
import { debugTag } from '../util.ts'
|
|
42
|
+
|
|
43
|
+
export type ObjstoreDeps = {
|
|
44
|
+
handle: Handle
|
|
45
|
+
secret: TokenSecret
|
|
46
|
+
send: (socket: WebSocket, msg: object) => void
|
|
47
|
+
broadcast: (tag: string, msg: object, except: WebSocket | null) => void
|
|
48
|
+
getNonce: (socket: WebSocket) => string | undefined
|
|
49
|
+
debug: boolean
|
|
50
|
+
// Auth gate for the FIRST put-begin against a never-before-seen
|
|
51
|
+
// workspace tag. See ObjstoreInitDeps for rationale. Run AFTER
|
|
52
|
+
// sig verify so `unauthorized` only reaches legitimate signers.
|
|
53
|
+
// Absent → no gating (no-config default).
|
|
54
|
+
authGate?: (socket: WebSocket, workspaceTag: string) => Promise<boolean>
|
|
55
|
+
sendUnauthorized?: (socket: WebSocket, ctx: { kind: 'gated'; workspaceTag: string; resourceTag: string }) => void
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function urlPathFor(tag: string, resourceTag: string): string {
|
|
59
|
+
return `/api/objstore/${tag}/${resourceTag}`
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Shared gate every objstore handler runs after its message-specific
|
|
63
|
+
// field checks: fetch the socket's challenge nonce, verify the signed
|
|
64
|
+
// message against it, then re-confirm the socket is still OPEN — the
|
|
65
|
+
// close handler may have fired during the verify await, and attaching
|
|
66
|
+
// to / replying on a closed socket is the half-handshake leak case
|
|
67
|
+
// (PR #4 review F4). Returns true when the caller may proceed.
|
|
68
|
+
// Centralising the post-await readyState recheck keeps that
|
|
69
|
+
// easy-to-forget invariant in one auditable place.
|
|
70
|
+
async function verified<M extends { workspaceTag?: unknown }>(
|
|
71
|
+
deps: ObjstoreDeps, socket: WebSocket, msg: M, label: string,
|
|
72
|
+
verify: (m: M, nonce: string) => Promise<boolean>,
|
|
73
|
+
): Promise<boolean> {
|
|
74
|
+
const nonce = deps.getNonce(socket)
|
|
75
|
+
if (typeof nonce !== 'string') return false
|
|
76
|
+
if (!await verify(msg, nonce)) {
|
|
77
|
+
if (deps.debug) console.warn(`reject objstore-${label}: bad sig`, debugTag(msg.workspaceTag as string))
|
|
78
|
+
return false
|
|
79
|
+
}
|
|
80
|
+
return socket.readyState === socket.OPEN
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
async function handlePutBegin(deps: ObjstoreDeps, socket: WebSocket, msg: ObjstorePutBeginMsg): Promise<void> {
|
|
84
|
+
if (!isValidTag(msg.workspaceTag) || !isValidTag(msg.resourceTag)) return
|
|
85
|
+
if (!isValidContentHash(msg.contentHash)) return
|
|
86
|
+
if (!isValidSignature(msg.signature)) return
|
|
87
|
+
// `Number.isSafeInteger` rather than `typeof === 'number'`: a NaN
|
|
88
|
+
// or non-finite value would pass typeof + comparisons (NaN <
|
|
89
|
+
// MAX_CONTENT_LENGTH is false), reaching sig verify only for the
|
|
90
|
+
// signature to fail. Cheaper to reject up-front, and consistent
|
|
91
|
+
// with `verifyObjstorePutSig`'s `isSafeNonNegativeInt` gate.
|
|
92
|
+
if (!Number.isSafeInteger(msg.expectedLength) || (msg.expectedLength as number) < 0 || (msg.expectedLength as number) > MAX_CONTENT_LENGTH) return
|
|
93
|
+
// Symmetric with `handleDelete`'s prevVersion gate (line 116) and
|
|
94
|
+
// `verifyObjstorePutSig`'s `isSafeIntOrNull` (sign.ts:119). Without
|
|
95
|
+
// this, a non-safe-integer `prevVersion` (NaN, 2^53+1, ...) would
|
|
96
|
+
// pass the typeof check below and reach sig verify, burning a
|
|
97
|
+
// hash + Ed25519 round-trip on a guaranteed-fail input. Input-
|
|
98
|
+
// validation audit `server/objstore/handlers.ts:76`.
|
|
99
|
+
if (msg.prevVersion != null && (typeof msg.prevVersion !== 'number' || !Number.isSafeInteger(msg.prevVersion))) return
|
|
100
|
+
if (!await verified(deps, socket, msg, 'put-begin', verifyObjstorePutSig)) return
|
|
101
|
+
const tag = msg.workspaceTag
|
|
102
|
+
const resourceTag = msg.resourceTag
|
|
103
|
+
// Auth gate for the FIRST action against a never-before-seen
|
|
104
|
+
// workspace tag (no rows in workspace_revision AND none in
|
|
105
|
+
// workspace_object). Mirrors handleSave in server/index.ts; runs
|
|
106
|
+
// AFTER sig verify so `unauthorized` only reaches a legitimate
|
|
107
|
+
// signer. The gate is config-driven (server/config.json
|
|
108
|
+
// `password`) and is a no-op when no password is configured.
|
|
109
|
+
if (deps.authGate && deps.sendUnauthorized && await deps.authGate(socket, tag)) {
|
|
110
|
+
if (socket.readyState !== socket.OPEN) return
|
|
111
|
+
if (deps.debug) console.warn(`reject objstore-put-begin: unauthorized (new workspace ${debugTag(tag)})`)
|
|
112
|
+
deps.sendUnauthorized(socket, { kind: 'gated', workspaceTag: tag, resourceTag })
|
|
113
|
+
return
|
|
114
|
+
}
|
|
115
|
+
const prevVersion = typeof msg.prevVersion === 'number' ? msg.prevVersion : null
|
|
116
|
+
// `verifyObjstorePutSig` already enforced the prevVersion/prevIncarnation
|
|
117
|
+
// pairing (null-iff-null + valid id shape), so this narrows safely.
|
|
118
|
+
const prevIncarnation = typeof msg.prevIncarnation === 'string' ? msg.prevIncarnation : null
|
|
119
|
+
// No lock: beginPut's prev check is advisory (a fast-fail so the
|
|
120
|
+
// client rebases before uploading). The authoritative precondition is
|
|
121
|
+
// commitPut's version+incarnation-CAS, which stays correct no matter
|
|
122
|
+
// what races between this begin and that commit.
|
|
123
|
+
const result = await beginPut(deps.handle, {
|
|
124
|
+
workspaceTag: tag, resourceTag, prevVersion, prevIncarnation,
|
|
125
|
+
expectedLength: msg.expectedLength as number,
|
|
126
|
+
contentHash: msg.contentHash as string,
|
|
127
|
+
signature: msg.signature as string,
|
|
128
|
+
})
|
|
129
|
+
if (!result.ok) {
|
|
130
|
+
// `workspace-full` is the per-workspace 100-resource cap; goes
|
|
131
|
+
// out as a typed error so the client can distinguish a quota
|
|
132
|
+
// refusal from a `conflict` (which is a version-precondition
|
|
133
|
+
// mismatch that the client can fix by re-reading + rebasing).
|
|
134
|
+
if (result.reason === 'workspace-full') {
|
|
135
|
+
deps.send(socket, { type: 'objstore-put-error', workspaceTag: tag, resourceTag, reason: 'workspace-full' })
|
|
136
|
+
return
|
|
137
|
+
}
|
|
138
|
+
deps.send(socket, conflictReply('put', tag, resourceTag, result.conflict))
|
|
139
|
+
return
|
|
140
|
+
}
|
|
141
|
+
const { token, exp } = mintPutToken(deps.secret, tag, resourceTag, result.stagingId, msg.expectedLength as number)
|
|
142
|
+
deps.send(socket, {
|
|
143
|
+
type: 'objstore-put-token',
|
|
144
|
+
workspaceTag: tag, resourceTag,
|
|
145
|
+
stagingId: result.stagingId,
|
|
146
|
+
urlPath: urlPathFor(tag, resourceTag),
|
|
147
|
+
token, expiresAt: exp,
|
|
148
|
+
})
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// `current` is the live `ObjectRow`, which carries the server-only
|
|
152
|
+
// `putAt` debug column. Project it through `objectMetaWire` before it
|
|
153
|
+
// reaches the wire so the conflict frame matches the fetch/list/PUT
|
|
154
|
+
// frames and never leaks `putAt`. Typed `ObjectRow | null` so the
|
|
155
|
+
// projection contract is checked rather than relying on `object`.
|
|
156
|
+
function conflictReply(action: 'put' | 'delete', tag: string, resourceTag: string, current: ObjectRow | null): object {
|
|
157
|
+
return current
|
|
158
|
+
? { type: 'objstore-conflict', action, workspaceTag: tag, resourceTag, current: objectMetaWire(current) }
|
|
159
|
+
: { type: 'objstore-conflict', action, workspaceTag: tag, resourceTag }
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function handleDelete(deps: ObjstoreDeps, socket: WebSocket, msg: ObjstoreDeleteMsg): Promise<void> {
|
|
163
|
+
if (!isValidTag(msg.workspaceTag) || !isValidTag(msg.resourceTag) || !isValidSignature(msg.signature)) return
|
|
164
|
+
if (msg.prevVersion != null && (typeof msg.prevVersion !== 'number' || !Number.isSafeInteger(msg.prevVersion))) return
|
|
165
|
+
if (!await verified(deps, socket, msg, 'delete', verifyObjstoreDeleteSig)) return
|
|
166
|
+
const tag = msg.workspaceTag
|
|
167
|
+
const resourceTag = msg.resourceTag
|
|
168
|
+
const prev = typeof msg.prevVersion === 'number' ? msg.prevVersion : null
|
|
169
|
+
const prevIncarnation = typeof msg.prevIncarnation === 'string' ? msg.prevIncarnation : null
|
|
170
|
+
// No lock: deleteObject is a precondition-checked version-CAS drop.
|
|
171
|
+
// A concurrent commit OR delete races that CAS (not a shared blob —
|
|
172
|
+
// the live blob is content-addressed + GC'd by the reaper, never
|
|
173
|
+
// unlinked here): exactly one op wins, the loser gets conflict /
|
|
174
|
+
// not-found (and never broadcasts). See the deleteObject doc in
|
|
175
|
+
// store.ts.
|
|
176
|
+
const result = await deleteObject(deps.handle, tag, resourceTag, prev, prevIncarnation)
|
|
177
|
+
if (!result.ok) {
|
|
178
|
+
if (result.reason === 'conflict') deps.send(socket, conflictReply('delete', tag, resourceTag, result.conflict ?? null))
|
|
179
|
+
else deps.send(socket, { type: 'objstore-delete-error', workspaceTag: tag, resourceTag, reason: result.reason })
|
|
180
|
+
return
|
|
181
|
+
}
|
|
182
|
+
deps.send(socket, { type: 'objstore-deleted-ack', workspaceTag: tag, resourceTag, deletedVersion: result.deletedVersion })
|
|
183
|
+
if (result.deletedVersion === 0) return // sentinel: nothing to broadcast
|
|
184
|
+
// Broadcast to ALL subscribers (including the originator). The
|
|
185
|
+
// REST PUT path's `objstore-put` broadcast at rest.ts:308 already
|
|
186
|
+
// uses `except: null`, so a session that listens via `onPut`
|
|
187
|
+
// observes its own PUTs as echo events. The same symmetry on
|
|
188
|
+
// `onDeleted` lets `session.onDeleted` fire for the session's
|
|
189
|
+
// own deletes — pinned by `tests/objstore-client-races.test.js`.
|
|
190
|
+
deps.broadcast(tag, { type: 'objstore-deleted', workspaceTag: tag, resourceTag, version: result.deletedVersion }, null)
|
|
191
|
+
if (deps.debug) console.log(`objstore delete → ${debugTag(tag)}/${resourceTag.slice(0, 8)}…`)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
async function handleFetch(deps: ObjstoreDeps, socket: WebSocket, msg: ObjstoreFetchMsg): Promise<void> {
|
|
195
|
+
if (!isValidTag(msg.workspaceTag) || !isValidTag(msg.resourceTag) || !isValidSignature(msg.signature)) return
|
|
196
|
+
if (!await verified(deps, socket, msg, 'fetch', verifyObjstoreFetchSig)) return
|
|
197
|
+
const tag = msg.workspaceTag
|
|
198
|
+
const resourceTag = msg.resourceTag
|
|
199
|
+
// Direct (workspace_tag, resource_tag) lookup — `listLive(...).find()`
|
|
200
|
+
// is O(n) per fetch and gets expensive for workspaces with many
|
|
201
|
+
// resources. PR #4 review.
|
|
202
|
+
const row = await getLive(deps.handle, tag, resourceTag)
|
|
203
|
+
if (!row) {
|
|
204
|
+
deps.send(socket, { type: 'objstore-fetch-not-found', workspaceTag: tag, resourceTag })
|
|
205
|
+
return
|
|
206
|
+
}
|
|
207
|
+
const { token, exp } = mintGetToken(deps.secret, tag, resourceTag, row.version, row.incarnation)
|
|
208
|
+
deps.send(socket, {
|
|
209
|
+
type: 'objstore-fetch-token',
|
|
210
|
+
workspaceTag: tag,
|
|
211
|
+
...objectMetaWire(row),
|
|
212
|
+
urlPath: urlPathFor(tag, resourceTag),
|
|
213
|
+
token, expiresAt: exp,
|
|
214
|
+
})
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Staging rows abandoned by a disconnected socket are picked up by
|
|
218
|
+
// the reaper's TTL pass within `STAGING_TTL_MS_DEFAULT` — no per-
|
|
219
|
+
// socket bookkeeping is needed on disconnect today. If that ever
|
|
220
|
+
// changes, wire a `cleanupSocket(socket)` into both this bundle and
|
|
221
|
+
// server/index.ts's close handler.
|
|
222
|
+
|
|
223
|
+
export type ObjstoreHandlers = {
|
|
224
|
+
handlePutBegin: (s: WebSocket, m: ObjstorePutBeginMsg) => Promise<void>
|
|
225
|
+
handleDelete: (s: WebSocket, m: ObjstoreDeleteMsg) => Promise<void>
|
|
226
|
+
handleFetch: (s: WebSocket, m: ObjstoreFetchMsg) => Promise<void>
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
export function createObjstoreHandlers(deps: ObjstoreDeps): ObjstoreHandlers {
|
|
230
|
+
return {
|
|
231
|
+
handlePutBegin: (s, m) => handlePutBegin(deps, s, m),
|
|
232
|
+
handleDelete: (s, m) => handleDelete(deps, s, m),
|
|
233
|
+
handleFetch: (s, m) => handleFetch(deps, s, m),
|
|
234
|
+
}
|
|
235
|
+
}
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// Wire-time wrapper around the v1.objstore module — takes a pre-
|
|
2
|
+
// opened storage handle (SQLite or Neon, opened one level up
|
|
3
|
+
// alongside the workspace_revision handle), mints a per-process
|
|
4
|
+
// bearer-token secret, kicks off the orphan reaper, and returns
|
|
5
|
+
// the WS handlers + REST deps + a teardown callback the index.ts
|
|
6
|
+
// shutdown path calls before closing the shared DB handle.
|
|
7
|
+
|
|
8
|
+
import type { WebSocket } from 'ws'
|
|
9
|
+
import type { Handle } from './store.ts'
|
|
10
|
+
import { reapOrphans } from './reaper.ts'
|
|
11
|
+
import { type ObjstoreHandlers, createObjstoreHandlers } from './handlers.ts'
|
|
12
|
+
import { type ObjstoreRestDeps } from './rest.ts'
|
|
13
|
+
import { type TokenSecret, newTokenSecret } from './tokens.ts'
|
|
14
|
+
import { errStack } from '../util.ts'
|
|
15
|
+
|
|
16
|
+
export type ObjstoreInitDeps = {
|
|
17
|
+
// Pre-opened storage handle, sharing the SQLite connection with
|
|
18
|
+
// the workspace_revision handle in server/db.ts.
|
|
19
|
+
handle: Handle
|
|
20
|
+
reapIntervalMs: number
|
|
21
|
+
send: (socket: WebSocket, msg: object) => void
|
|
22
|
+
broadcast: (tag: string, msg: object, except: WebSocket | null) => void
|
|
23
|
+
getNonce: (socket: WebSocket) => string | undefined
|
|
24
|
+
debug: boolean
|
|
25
|
+
// Auth gate for the FIRST objstore-put-begin against a workspace
|
|
26
|
+
// tag that doesn't yet exist on the server. Returns `true` to
|
|
27
|
+
// DENY (unauthorized), `false` to allow. Called AFTER sig verify
|
|
28
|
+
// so the resulting `unauthorized` frame only reaches a legitimate
|
|
29
|
+
// signer. Falsy/missing → no gating (the no-config default).
|
|
30
|
+
authGate?: (socket: WebSocket, workspaceTag: string) => Promise<boolean>
|
|
31
|
+
sendUnauthorized?: (socket: WebSocket, ctx: { kind: 'gated'; workspaceTag: string; resourceTag: string }) => void
|
|
32
|
+
// HMAC secret for REST bearer tokens. Optional — falls back to a
|
|
33
|
+
// fresh process-local secret if omitted (the historical single-
|
|
34
|
+
// process behaviour). Multi-replica deployments MUST supply a
|
|
35
|
+
// shared secret here so a token minted on one replica's WS plane
|
|
36
|
+
// validates on another replica's REST plane (the WS-to-REST hop
|
|
37
|
+
// is not load-balancer-pinned). See server/index.ts boot logic
|
|
38
|
+
// for the env var (`OBJSTORE_TOKEN_SECRET`).
|
|
39
|
+
tokenSecret?: TokenSecret
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export type ObjstoreInit = {
|
|
43
|
+
handlers: ObjstoreHandlers
|
|
44
|
+
restDeps: ObjstoreRestDeps
|
|
45
|
+
startupReap: Promise<void>
|
|
46
|
+
// Cancels the periodic timer AND resolves only after any
|
|
47
|
+
// currently-in-flight sweep has finished, so shutdown can safely
|
|
48
|
+
// close the DB without an outstanding readdir / unlink touching
|
|
49
|
+
// a closed handle. PR #4 review.
|
|
50
|
+
stopReaper: () => Promise<void>
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export function initObjstore(deps: ObjstoreInitDeps): ObjstoreInit {
|
|
54
|
+
const handle = deps.handle
|
|
55
|
+
const secret = deps.tokenSecret ?? newTokenSecret()
|
|
56
|
+
const handlers = createObjstoreHandlers({
|
|
57
|
+
handle, secret,
|
|
58
|
+
send: deps.send, broadcast: deps.broadcast,
|
|
59
|
+
getNonce: deps.getNonce, debug: deps.debug,
|
|
60
|
+
...(deps.authGate ? { authGate: deps.authGate } : {}),
|
|
61
|
+
...(deps.sendUnauthorized ? { sendUnauthorized: deps.sendUnauthorized } : {}),
|
|
62
|
+
})
|
|
63
|
+
const restDeps: ObjstoreRestDeps = {
|
|
64
|
+
handle, secret, broadcast: deps.broadcast, debug: deps.debug,
|
|
65
|
+
}
|
|
66
|
+
// Re-entrancy guard for periodic + startup sweeps. Kicking the
|
|
67
|
+
// startup sweep through the same `enqueueSweep` path means the
|
|
68
|
+
// interval ticking before the startup pass finishes naturally
|
|
69
|
+
// short-circuits — no risk of two `reapOrphans()` racing
|
|
70
|
+
// readdir / unlink against the same dirs. PR #4 review.
|
|
71
|
+
let inFlight: Promise<void> | null = null
|
|
72
|
+
function enqueueSweep(): Promise<void> {
|
|
73
|
+
if (inFlight) return inFlight
|
|
74
|
+
// Log reaper failures unconditionally (not gated on `debug`) — a
|
|
75
|
+
// failed sweep means stranded files / staging rows that never get
|
|
76
|
+
// cleaned, and operators need to see it. Inner catch makes the
|
|
77
|
+
// promise resolve so callers (startupReap awaiter + setInterval)
|
|
78
|
+
// don't have to handle rejection.
|
|
79
|
+
const p = reapOrphans(handle).catch((err) => {
|
|
80
|
+
console.warn('objstore reaper error:', errStack(err))
|
|
81
|
+
}).finally(() => { if (inFlight === p) inFlight = null })
|
|
82
|
+
inFlight = p
|
|
83
|
+
return p
|
|
84
|
+
}
|
|
85
|
+
// Caller awaits this before accepting traffic so a fresh boot
|
|
86
|
+
// can't hand out list / fetch / put-begin against a tag whose
|
|
87
|
+
// on-disk state still has stranded files from a prior crash.
|
|
88
|
+
const startupReap = enqueueSweep()
|
|
89
|
+
// Jittered start of the periodic timer. Multi-replica deploys
|
|
90
|
+
// (Neon + Vercel Blob) commonly boot N replicas in tight lock-
|
|
91
|
+
// step (deploy rollout, cluster restart) and would otherwise
|
|
92
|
+
// sync every replica's reaper at the same wall-clock tick,
|
|
93
|
+
// hammering the DB + blob store with N×readdir+lock-acquire
|
|
94
|
+
// bursts. A random first-interval delay deconcurrencies the
|
|
95
|
+
// cluster without changing the long-term sweep cadence.
|
|
96
|
+
// Jitter range is 0…1× reapIntervalMs (i.e., the next sweep
|
|
97
|
+
// happens at [interval, 2×interval] after boot); subsequent
|
|
98
|
+
// sweeps stay at exactly `reapIntervalMs` apart.
|
|
99
|
+
let reapTimer: ReturnType<typeof setInterval> | null = null
|
|
100
|
+
const jitterMs = Math.floor(Math.random() * deps.reapIntervalMs)
|
|
101
|
+
const firstTimer = setTimeout(() => {
|
|
102
|
+
enqueueSweep()
|
|
103
|
+
reapTimer = setInterval(enqueueSweep, deps.reapIntervalMs)
|
|
104
|
+
reapTimer.unref?.()
|
|
105
|
+
}, deps.reapIntervalMs + jitterMs)
|
|
106
|
+
firstTimer.unref?.()
|
|
107
|
+
return {
|
|
108
|
+
handlers, restDeps, startupReap,
|
|
109
|
+
stopReaper: async () => {
|
|
110
|
+
clearTimeout(firstTimer)
|
|
111
|
+
if (reapTimer) clearInterval(reapTimer)
|
|
112
|
+
// Drain whichever sweep is currently running (startup or
|
|
113
|
+
// periodic) — either could be mid-readdir/unlink at SIGTERM
|
|
114
|
+
// time and would otherwise outlive `handle.close()`.
|
|
115
|
+
if (inFlight) await inFlight.catch(() => {})
|
|
116
|
+
},
|
|
117
|
+
}
|
|
118
|
+
}
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
// Orphan reaper / GC for the v1.objstore byte plane. Two passes:
|
|
2
|
+
// 1. Unreferenced live blobs. Live blobs are content-addressed
|
|
3
|
+
// (`${tag}/${contentHash}.bin`). For each workspace, build the
|
|
4
|
+
// set of content hashes the live table references, list the live
|
|
5
|
+
// blobs, and GC any blob whose hash is in NO live row AND whose
|
|
6
|
+
// age (now − last-modified) is past the grace window. The grace
|
|
7
|
+
// window (one `STAGING_TTL_MS_DEFAULT`) is what makes the
|
|
8
|
+
// reaper-vs-commit race safe without a lock: a just-promoted but
|
|
9
|
+
// not-yet-referenced blob (commit between promote and CAS) is
|
|
10
|
+
// younger than the grace, so it's never reaped out from under an
|
|
11
|
+
// in-flight commit. Covers the "DELETE dropped the row" case
|
|
12
|
+
// (the blob is now unreferenced) and the "commit crashed
|
|
13
|
+
// mid-promotion" case (a stranded, unreferenced blob).
|
|
14
|
+
// 2. Stale staging. Rows in workspace_object_staging older than
|
|
15
|
+
// `stagingTtlMs` → drop row + unlink blob. Staging blobs with no
|
|
16
|
+
// row → unlink (catches a row drop that crashed before the
|
|
17
|
+
// blob delete).
|
|
18
|
+
//
|
|
19
|
+
// Backend-agnostic: every storage operation goes through
|
|
20
|
+
// `handle.blob.*`. The FS backend (blob-fs.ts) implements list/
|
|
21
|
+
// unlink against the local filesystem; the Vercel backend (blob-
|
|
22
|
+
// vercel.ts) against the SDK's list / del. The reaper takes NO lock.
|
|
23
|
+
// Pass 1's blob GC is made race-safe by content-addressing + the age
|
|
24
|
+
// grace window + a live-reference re-read just before each unlink.
|
|
25
|
+
// Pass 2's stale-staging sweep is made race-safe by an atomic
|
|
26
|
+
// conditional delete (`deleteStagingIfStale`) whose `begun_at`
|
|
27
|
+
// predicate can't match a row a concurrent upload just refreshed.
|
|
28
|
+
|
|
29
|
+
import { type Handle, STAGING_TTL_MS_DEFAULT, isValidContentHash, isValidStagingId, isValidTag } from './store.ts'
|
|
30
|
+
|
|
31
|
+
type StagingRow = {
|
|
32
|
+
workspace_tag: string
|
|
33
|
+
resource_tag: string
|
|
34
|
+
staging_id: string
|
|
35
|
+
begun_at: number
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// GC one workspace's live blobs against the set of hashes the live
|
|
39
|
+
// table references. A blob is unlinked IFF: its hash parses as a valid
|
|
40
|
+
// content hash (defense against operator-seeded / foreign files), it's
|
|
41
|
+
// older than the grace window (measured from the listing's
|
|
42
|
+
// last-modified time), AND — re-read immediately before the unlink —
|
|
43
|
+
// no live row references it. Shared by both GC passes
|
|
44
|
+
// (reapUnreferencedForTag's per-tag sweep and reapOrphans'
|
|
45
|
+
// whole-workspace straggler sweep).
|
|
46
|
+
//
|
|
47
|
+
// Two layers of race protection, neither of which is a lock:
|
|
48
|
+
// - Age grace window. A blob a freshly-promoted-but-not-yet-CAS'd
|
|
49
|
+
// commit just wrote is younger than the grace, so it survives this
|
|
50
|
+
// sweep entirely (its listing mtime is recent). This is the
|
|
51
|
+
// primary guard for the commit's promote→CAS window.
|
|
52
|
+
// - Live-set re-read just before unlink. A commit whose CAS landed
|
|
53
|
+
// after our initial per-tag snapshot but before this unlink is
|
|
54
|
+
// caught here — its row now references the hash, so we skip. The
|
|
55
|
+
// test is "no live row references this hash" (not "this resource")
|
|
56
|
+
// because the blob path carries only the hash — the reaper lists
|
|
57
|
+
// blobs by hash and can't know which resource wrote one. (Hashes
|
|
58
|
+
// are effectively unique per PUT — a random nonce per encrypt makes
|
|
59
|
+
// each ciphertext unique — so this is really "is this hash still
|
|
60
|
+
// some current version's", not a dedup/sharing check.)
|
|
61
|
+
// A lock would not help here even if one existed: the commit path is
|
|
62
|
+
// keyed on the resourceTag while a blob is named by its content hash,
|
|
63
|
+
// so the two can't share a key. The grace window + re-read are the
|
|
64
|
+
// actual safety net. init.ts runs one sweep at a time per process,
|
|
65
|
+
// but a multi-replica deploy genuinely runs reapers concurrently;
|
|
66
|
+
// they stay safe via idempotent unlink + the per-blob grace window +
|
|
67
|
+
// live-set re-read, not via mutual exclusion.
|
|
68
|
+
async function gcBlobIfUnreferenced(
|
|
69
|
+
handle: Handle, tag: string, hash: string, modifiedMs: number, now: number, grace: number,
|
|
70
|
+
): Promise<void> {
|
|
71
|
+
if (!isValidContentHash(hash)) return
|
|
72
|
+
if (now - modifiedMs < grace) return
|
|
73
|
+
const refs = await liveHashSet(handle, tag)
|
|
74
|
+
if (refs.has(hash)) return
|
|
75
|
+
await handle.blob.unlinkLive(tag, hash)
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// The set of content hashes referenced by the workspace's live rows.
|
|
79
|
+
async function liveHashSet(handle: Handle, tag: string): Promise<Set<string>> {
|
|
80
|
+
const rows = await handle.selectLive.all(tag)
|
|
81
|
+
return new Set(rows.map((r) => r.content_hash))
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Sweep one workspace's live-blob listing against the referenced-hash
|
|
85
|
+
// set. Anything unreferenced AND past the grace window → GC. The
|
|
86
|
+
// snapshot we read up front can race a concurrent commit; the
|
|
87
|
+
// reference re-read inside `gcBlobIfUnreferenced` (plus the grace
|
|
88
|
+
// window) ensures we never unlink a blob a live row names.
|
|
89
|
+
async function reapUnreferencedForTag(handle: Handle, tag: string, now: number, grace: number): Promise<void> {
|
|
90
|
+
if (!isValidTag(tag)) return
|
|
91
|
+
const blobs = await handle.blob.listLiveBlobs(tag)
|
|
92
|
+
if (blobs.length === 0) return
|
|
93
|
+
const referenced = await liveHashSet(handle, tag)
|
|
94
|
+
for (const { hash, modifiedMs } of blobs) {
|
|
95
|
+
if (!isValidContentHash(hash)) continue
|
|
96
|
+
// Referenced in our snapshot → skip the grace + re-read path
|
|
97
|
+
// entirely; only unreferenced blobs need it.
|
|
98
|
+
if (referenced.has(hash)) continue
|
|
99
|
+
await gcBlobIfUnreferenced(handle, tag, hash, modifiedMs, now, grace)
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// Drop staging rows older than the TTL and unlink their on-storage
|
|
104
|
+
// blobs. Every row field that flows into a backend call is re-
|
|
105
|
+
// validated — DB tampering / a future migration that introduces an
|
|
106
|
+
// unsanitised column shouldn't be able to trick the reaper into
|
|
107
|
+
// touching unintended keys. PR #4 review.
|
|
108
|
+
async function reapStaleStagingRows(handle: Handle, now: number, stagingTtlMs: number): Promise<void> {
|
|
109
|
+
// Push the staleness filter into SQL so the
|
|
110
|
+
// `workspace_object_staging_begun_at_idx` index handles the scan.
|
|
111
|
+
// The snapshot is O(stale-rows) cluster-wide. DB-layout audit
|
|
112
|
+
// `server/objstore/store.ts`.
|
|
113
|
+
const staleBefore = now - stagingTtlMs
|
|
114
|
+
const staging = await handle.listAllStaging.all(staleBefore) as StagingRow[]
|
|
115
|
+
for (const s of staging) {
|
|
116
|
+
if (!isValidTag(s.workspace_tag) || !isValidTag(s.resource_tag) || !isValidStagingId(s.staging_id)) {
|
|
117
|
+
// Truncate fields — the full workspace_tag is an Ed25519 public
|
|
118
|
+
// key and shouldn't land in operator logs verbatim. PR #4
|
|
119
|
+
// review H3.
|
|
120
|
+
console.warn(`reaper: skipping malformed staging row tag=${String(s.workspace_tag).slice(0, 12)}… res=${String(s.resource_tag).slice(0, 8)}… sid=${String(s.staging_id).slice(0, 8)}…`)
|
|
121
|
+
continue
|
|
122
|
+
}
|
|
123
|
+
// ATOMIC conditional delete (replaces the old in-lock begun_at
|
|
124
|
+
// re-read, PR #4 "F1"). The delete fires only if `begun_at` is
|
|
125
|
+
// STILL older than the SAME `staleBefore` we snapshotted with — so
|
|
126
|
+
// a concurrent REST PUT that finished its body and called
|
|
127
|
+
// `refreshStagingBegunAt` (bumping begun_at to ~now, well after
|
|
128
|
+
// `staleBefore`) makes the predicate fail: the row isn't deleted,
|
|
129
|
+
// `deleteStagingIfStale` returns undefined, and we skip the unlink,
|
|
130
|
+
// leaving the row for that PUT's commit. No lock, no TOCTOU window
|
|
131
|
+
// between a read and a delete — the CAS is the whole check.
|
|
132
|
+
const deleted = await handle.deleteStagingIfStale.get(s.workspace_tag, s.resource_tag, s.staging_id, staleBefore)
|
|
133
|
+
if (!deleted) continue
|
|
134
|
+
// The row was genuinely stale and we removed it; now drop its
|
|
135
|
+
// bytes. Row-first, then unlink — symmetric with deleteObject in
|
|
136
|
+
// store.ts. If a crash lands between them the bytes outlive the
|
|
137
|
+
// row, and a later sweep's orphan-staging-file pass (or this
|
|
138
|
+
// pass's idempotent unlink of an absent blob) self-heals.
|
|
139
|
+
await handle.blob.unlinkStaging(s.workspace_tag, s.staging_id)
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// Sweep staging blobs whose row is missing — happens when a commit
|
|
144
|
+
// dropped the row but crashed before the unlink (Vercel: between
|
|
145
|
+
// the post-copy `del(staging)` and the DB delete; FS: not possible
|
|
146
|
+
// since rename removes the staging file), or after
|
|
147
|
+
// reapStaleStagingRows already nuked the row but the unlink failed.
|
|
148
|
+
// Per-blob row lookup (vs a snapshot at the caller) closes the race
|
|
149
|
+
// where a concurrent beginPut between snapshot-time and unlink-time
|
|
150
|
+
// would have its blob unlinked while the row was already in DB. The
|
|
151
|
+
// per-blob SELECT is sub-ms and sids are 16-byte random — a
|
|
152
|
+
// same-sid beginPut in the microsecond between our SELECT and
|
|
153
|
+
// unlink is 1/2^128. Also handles the malformed-row case: a row
|
|
154
|
+
// with valid (workspace_tag, staging_id) but malformed resource_tag
|
|
155
|
+
// still pins its blob. PR #4 review H1.
|
|
156
|
+
async function reapOrphanedStagingFiles(handle: Handle, tag: string): Promise<void> {
|
|
157
|
+
if (!isValidTag(tag)) return
|
|
158
|
+
const entries = await handle.blob.listStagingIds(tag)
|
|
159
|
+
if (entries.length === 0) return
|
|
160
|
+
for (const stagingId of entries) {
|
|
161
|
+
// Same on-disk-foreign-file guard as reapUnreferencedForTag.
|
|
162
|
+
if (!isValidStagingId(stagingId)) continue
|
|
163
|
+
if (await handle.selectStagingByWsSid.get(tag, stagingId)) continue
|
|
164
|
+
await handle.blob.unlinkStaging(tag, stagingId)
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
export async function reapOrphans(handle: Handle, stagingTtlMs: number = STAGING_TTL_MS_DEFAULT): Promise<void> {
|
|
169
|
+
const now = Date.now()
|
|
170
|
+
// The GC grace window reuses the staging TTL: a live blob is only
|
|
171
|
+
// eligible for unlink once it's unreferenced AND older than this.
|
|
172
|
+
const grace = stagingTtlMs
|
|
173
|
+
// Pass 1: tags the live table knows about — GC unreferenced live
|
|
174
|
+
// blobs (past the grace window) against the referenced-hash set.
|
|
175
|
+
const liveTagsRows = await handle.listLiveTags.all()
|
|
176
|
+
const liveTags = liveTagsRows.map((r) => r.workspace_tag)
|
|
177
|
+
for (const tag of liveTags) await reapUnreferencedForTag(handle, tag, now, grace)
|
|
178
|
+
// Whole-workspace deletes leave residue (dirs / blob-prefixes) the
|
|
179
|
+
// live table no longer lists. Walk the backend's top-level workspace
|
|
180
|
+
// listing to find them; for each straggler tag, GC its unreferenced
|
|
181
|
+
// blobs the same way (the referenced-hash set for a fully-deleted
|
|
182
|
+
// workspace is empty, so every past-grace blob is collected, while
|
|
183
|
+
// the grace window + live-set re-read still protect a racing
|
|
184
|
+
// put-begin → commit on a tag not in our `liveTags` snapshot).
|
|
185
|
+
const topLevel = await handle.blob.listWorkspaceTags()
|
|
186
|
+
const liveSet = new Set(liveTags)
|
|
187
|
+
for (const tag of topLevel) {
|
|
188
|
+
if (liveSet.has(tag) || !isValidTag(tag)) continue
|
|
189
|
+
await reapUnreferencedForTag(handle, tag, now, grace)
|
|
190
|
+
}
|
|
191
|
+
// Pass 2: stale staging rows + orphan staging blobs. The orphan
|
|
192
|
+
// sweep does per-blob row lookups (no caller-side snapshot), so a
|
|
193
|
+
// beginPut that lands between our list and our unlink has its
|
|
194
|
+
// row found by the fresh SELECT. PR #4 review H1.
|
|
195
|
+
await reapStaleStagingRows(handle, now, stagingTtlMs)
|
|
196
|
+
for (const tag of topLevel) {
|
|
197
|
+
await reapOrphanedStagingFiles(handle, tag)
|
|
198
|
+
}
|
|
199
|
+
}
|