@preventive/triage 1.0.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/LICENSE +21 -0
  2. package/common/save-error-reason.ts +53 -0
  3. package/common/utf8.d.ts +13 -0
  4. package/common/utf8.js +57 -0
  5. package/out/brotli-fallback.js +3 -0
  6. package/out/client-sync.js +15 -0
  7. package/out/graph.js +4 -0
  8. package/out/icon-maskable.svg +5 -0
  9. package/out/icon.svg +5 -0
  10. package/out/index.html +78 -0
  11. package/out/manifest.webmanifest +30 -0
  12. package/out/prism.js +14 -0
  13. package/out/terminal.js +39 -0
  14. package/out/view.css +1 -0
  15. package/out/view.html +12 -0
  16. package/out/view.js +138 -0
  17. package/package.json +129 -0
  18. package/server/auth.ts +99 -0
  19. package/server/config.example.json +3 -0
  20. package/server/config.ts +196 -0
  21. package/server/db-neon.ts +374 -0
  22. package/server/db-revision-sql.ts +152 -0
  23. package/server/db-stmt.ts +53 -0
  24. package/server/db.ts +577 -0
  25. package/server/http.ts +142 -0
  26. package/server/hub.ts +98 -0
  27. package/server/index.ts +353 -0
  28. package/server/lifecycle.ts +177 -0
  29. package/server/neon-driver.ts +26 -0
  30. package/server/objstore/blob-fs.ts +164 -0
  31. package/server/objstore/blob-vercel.ts +508 -0
  32. package/server/objstore/blob.ts +169 -0
  33. package/server/objstore/fs.ts +67 -0
  34. package/server/objstore/handlers.ts +235 -0
  35. package/server/objstore/init.ts +118 -0
  36. package/server/objstore/reaper.ts +199 -0
  37. package/server/objstore/rest.ts +484 -0
  38. package/server/objstore/sign.ts +164 -0
  39. package/server/objstore/store-neon.ts +351 -0
  40. package/server/objstore/store.ts +799 -0
  41. package/server/objstore/tokens.ts +168 -0
  42. package/server/origin.ts +68 -0
  43. package/server/peer.ts +38 -0
  44. package/server/sign.ts +231 -0
  45. package/server/static.ts +374 -0
  46. package/server/sync-handlers.ts +311 -0
  47. package/server/util.ts +27 -0
  48. package/server/validation.ts +36 -0
  49. package/server/ws-server.ts +245 -0
@@ -0,0 +1,799 @@
1
+ // SQLite + filesystem-backed object store for the v1.objstore
2
+ // protocol extension. Sibling of `server/db.ts`; shares the
3
+ // underlying DatabaseSync handle but keeps its own tables.
4
+ //
5
+ // workspace_object — one row per LIVE resource (no
6
+ // tombstones; new subscribers must
7
+ // not learn a deleted resource ever
8
+ // existed).
9
+ // workspace_object_staging — one row per IN-FLIGHT upload, so a
10
+ // restart between begin and commit
11
+ // can be reaped (see ./reaper.ts).
12
+ // Bytes live OUTSIDE sqlite at
13
+ // ${OBJSTORE_DIR}/${workspaceTag}/[.staging/]${id}.bin
14
+ // keeping the WAL out of the multi-MB bundle path. Live blobs are
15
+ // CONTENT-ADDRESSED (`${tag}/${contentHash}.bin`): the hash names
16
+ // exactly one immutable byte-string, so two racing commits write to
17
+ // DIFFERENT addresses and the live row's `content_hash` literally
18
+ // names its blob file (a metadata-vs-bytes desync is impossible).
19
+ // Commit is therefore a plain version compare-and-set on the row —
20
+ // no distributed lock. Commit/delete order is asymmetric so a crash
21
+ // at the worst moment leaves at most a STRANDED FILE (reaper-cleaned,
22
+ // once unreferenced AND past the GC grace window), never a row
23
+ // pointing at nothing:
24
+ // PUT commit: fsync(staging) → rename → fsync(parent) → DB CAS
25
+ // DELETE: DB row drop (the reaper GCs the unreferenced blob)
26
+
27
+ import { DatabaseSync } from 'node:sqlite'
28
+ import { mkdirSync } from 'node:fs'
29
+ import { type BlobBackend } from './blob.ts'
30
+ import { openFsBlobBackend } from './blob-fs.ts'
31
+ import { stagingFilePath } from './fs.ts'
32
+ import { type AllStmt, type GetStmt, type RunStmt, wrapAll, wrapGet, wrapRun } from '../db-stmt.ts'
33
+ import { errMsg, randomId } from '../util.ts'
34
+
35
+ // Default 1h, comfortably over a 50 MiB upload on a slow line. The
36
+ // reaper walks the staging table on this cadence; rows older than
37
+ // the TTL are dropped and their on-disk files unlinked.
38
+ export const STAGING_TTL_MS_DEFAULT = 60 * 60 * 1000
39
+
40
+ // `CHECK (version >= 0)` / `CHECK (content_length >= 0)` /
41
+ // `CHECK (expected_length >= 0)` / `CHECK (prev_version IS NULL OR
42
+ // prev_version >= 0)` are value-domain guards. STRICT (the table
43
+ // markers below) enforces each column's TYPE — an INTEGER stays an
44
+ // INTEGER — but NOT its value range: a manual `UPDATE workspace_object
45
+ // SET version = -1` is a perfectly valid integer that STRICT accepts,
46
+ // which then round-trips through `num()` (it only rejects
47
+ // non-safe-integers) and corrupts the commitPut version-monotonicity
48
+ // arithmetic. The CHECKs close that value-domain gap, mirroring the
49
+ // Neon schema's identical constraints (see `store-neon.ts`).
50
+ const SCHEMA = `
51
+ CREATE TABLE IF NOT EXISTS workspace_object (
52
+ workspace_tag TEXT NOT NULL,
53
+ resource_tag TEXT NOT NULL,
54
+ version INTEGER NOT NULL CHECK (version >= 0),
55
+ incarnation TEXT NOT NULL,
56
+ content_hash TEXT NOT NULL,
57
+ content_length INTEGER NOT NULL CHECK (content_length >= 0),
58
+ signature TEXT NOT NULL,
59
+ put_at INTEGER NOT NULL,
60
+ PRIMARY KEY (workspace_tag, resource_tag)
61
+ ) STRICT;
62
+
63
+ CREATE TABLE IF NOT EXISTS workspace_object_staging (
64
+ workspace_tag TEXT NOT NULL,
65
+ resource_tag TEXT NOT NULL,
66
+ staging_id TEXT NOT NULL,
67
+ prev_version INTEGER CHECK (prev_version IS NULL OR prev_version >= 0),
68
+ prev_incarnation TEXT,
69
+ expected_length INTEGER NOT NULL CHECK (expected_length >= 0),
70
+ content_hash TEXT NOT NULL,
71
+ signature TEXT NOT NULL,
72
+ begun_at INTEGER NOT NULL,
73
+ PRIMARY KEY (workspace_tag, resource_tag, staging_id)
74
+ ) STRICT;
75
+
76
+ CREATE INDEX IF NOT EXISTS workspace_object_staging_begun_at_idx
77
+ ON workspace_object_staging (begun_at);
78
+ `
79
+
80
+ // One LIVE row, exactly the shape the `workspace-subscribed` ack's
81
+ // `resources` array carries on the wire, minus `keyframe`-style
82
+ // server-only flags. `put_at` is a
83
+ // debug aid the wire format doesn't include — operators can inspect
84
+ // it via the DB but the server never volunteers it.
85
+ export type ObjectRow = {
86
+ resourceTag: string
87
+ version: number
88
+ // Random id minted on each first-write (insertLiveIfAbsent) and held
89
+ // constant across version bumps within a lineage. Delete drops the
90
+ // row, so a recreate mints a FRESH incarnation — this is what lets
91
+ // the commit CAS tell a stale `prev` (from a deleted incarnation)
92
+ // apart from a recreated one at the same version number.
93
+ incarnation: string
94
+ contentHash: string
95
+ contentLength: number
96
+ signature: string
97
+ putAt: number
98
+ }
99
+
100
+ // Input to `beginPut`. `prevVersion` is the precondition version the
101
+ // client thinks the server holds; mismatch means the resource raced
102
+ // and the client must rebase before retrying.
103
+ export type BeginPutInput = {
104
+ workspaceTag: string
105
+ resourceTag: string
106
+ prevVersion: number | null
107
+ // The incarnation the client believes is live. Null iff prevVersion
108
+ // is null (first-write precondition). Travels with prevVersion as an
109
+ // inseparable pair — a numeric prevVersion always carries one.
110
+ prevIncarnation: string | null
111
+ expectedLength: number
112
+ contentHash: string
113
+ signature: string
114
+ }
115
+
116
+ // `conflict` echoes the live row so the wire layer can include it in
117
+ // `objstore-conflict`; `accepted` hands back the staging id the REST
118
+ // PUT will reference. `workspace-full` is the per-workspace resource-
119
+ // count cap rejection — see `MAX_RESOURCES_PER_WORKSPACE`.
120
+ //
121
+ // `filePath` is set only for the FS-backed handle (where it's the
122
+ // absolute on-disk staging path); the Vercel-Blob handle omits it
123
+ // since "path" isn't a meaningful concept against a remote object
124
+ // store. Production code (rest.ts) never reads this field — the
125
+ // REST layer goes through `handle.blob.openStagingWriter(tag, sid)`.
126
+ // Tests for the FS path use it as a convenience to write fixture
127
+ // bytes directly to the staging slot.
128
+ export type BeginPutResult =
129
+ | { ok: true; stagingId: string; filePath?: string }
130
+ | { ok: false; reason: 'conflict'; conflict: ObjectRow | null }
131
+ | { ok: false; reason: 'workspace-full' }
132
+
133
+ export type CommitPutInput = {
134
+ workspaceTag: string
135
+ resourceTag: string
136
+ stagingId: string
137
+ // Optional: the storage-side byte count the caller already
138
+ // verified after the upload landed. When provided, commitPut
139
+ // skips the otherwise-redundant `statStaging` round-trip — for
140
+ // the Vercel backend that's one fewer HTTP HEAD per PUT.
141
+ // Caller must ONLY pass a value it observed for THIS stagingId
142
+ // after its upload finished (i.e. the REST PUT path's post-upload
143
+ // `statStaging`). Safe without a lock because staging ids are
144
+ // random — no other request writes this blob — and the sole writer
145
+ // (this PUT) has already completed. Tests that drive commitPut
146
+ // directly without the REST layer should omit this and let
147
+ // commitPut stat for itself.
148
+ observedSize?: number
149
+ }
150
+
151
+ export type CommitPutResult =
152
+ | { ok: true; row: ObjectRow }
153
+ | { ok: false; reason: 'no-staging' | 'size-mismatch' | 'io-error' | 'conflict'; conflict?: ObjectRow }
154
+
155
+ export type DeleteResult =
156
+ | { ok: true; deletedVersion: number }
157
+ | { ok: false; reason: 'not-found' | 'conflict'; conflict?: ObjectRow }
158
+
159
+ // Async statement shapes are shared with server/db.ts via
160
+ // ../db-stmt.ts — same `.get/.all/.run` → Promise contract across
161
+ // both planes.
162
+
163
+ // Row shape coming back from SELECTs (snake_case columns). The
164
+ // public `ObjectRow` is camelCased by `rowFromDb` at the call site.
165
+ type DbRow = {
166
+ resource_tag: string; version: number; incarnation: string; content_hash: string; content_length: number
167
+ signature: string; put_at: number
168
+ }
169
+
170
+ // Pre-prepared statements + the byte-plane backend. Held for process
171
+ // lifetime, closed from `shutdown()`.
172
+ //
173
+ // The objstore plane takes NO in-process mutex. Correctness rests
174
+ // entirely on three lock-free mechanisms:
175
+ // - the atomic version compare-and-set on commit
176
+ // (`insertLiveIfAbsent` / `updateLiveCAS`): N racing commits →
177
+ // exactly one wins, the losers get `conflict`;
178
+ // - content-addressed live blobs (`${tag}/${contentHash}.bin`):
179
+ // immutable, so a re-upload writes a DIFFERENT address and an
180
+ // in-flight GET never sees torn bytes;
181
+ // - the reaper's age grace window + an atomic conditional staging
182
+ // delete (`deleteStagingIfStale`) so the stale-staging sweep
183
+ // can't race an upload that just finished.
184
+ // These hold both within a single process AND across replicas — the
185
+ // old per-(tag, resourceTag) in-process mutex added nothing the CAS +
186
+ // content-addressing didn't already give, so it was removed.
187
+ export type Handle = {
188
+ // SQLite-only: the underlying `DatabaseSync`. Unset on the Neon
189
+ // backend (see ./store-neon.ts). Test-only fixture SQL routes
190
+ // through `handle.db.prepare(...)` and is therefore SQLite-coupled
191
+ // by construction.
192
+ db?: DatabaseSync
193
+ // Byte-plane backend (local FS or Vercel Blob). All bytes-side
194
+ // operations go through this — there is no direct fs.* call in
195
+ // store / rest / reaper. Selected at boot in server/index.ts.
196
+ blob: BlobBackend
197
+ // Storage root for the FS backend — set only when `blob` was
198
+ // constructed from `openFsBlobBackend(dir)`. Production code
199
+ // never touches this; it's a back-channel for tests that compute
200
+ // canonical paths via `stagingFilePath(handle.dir, …)`. The
201
+ // Vercel-backed Handle omits it.
202
+ dir?: string
203
+ insertStaging: RunStmt<[string, string, string, number | null, string | null, number, string, string, number]>
204
+ selectStaging: GetStmt<[string, string, string], {
205
+ prev_version: number | null
206
+ prev_incarnation: string | null
207
+ expected_length: number
208
+ content_hash: string
209
+ signature: string
210
+ begun_at: number
211
+ }>
212
+ selectStagingByWsSid: GetStmt<[string, string], unknown>
213
+ refreshStagingBegunAt: RunStmt<[number, string, string, string]>
214
+ deleteStaging: RunStmt<[string, string, string]>
215
+ // Atomic conditional staging delete used by the reaper's stale-row
216
+ // sweep. `deleteStagingIfStale(tag, res, sid, staleBefore)` deletes
217
+ // the row IFF its `begun_at < staleBefore`, returning `{ ok: 1 }`
218
+ // when a row was actually removed and `undefined` otherwise. This is
219
+ // the lock-free replacement for the old in-lock begun_at re-read
220
+ // (PR #4 "F1"): a slow PUT that finishes and calls
221
+ // `refreshStagingBegunAt` bumps `begun_at` fresh, so a concurrent
222
+ // reaper's conditional delete simply doesn't match (its predicate
223
+ // fails atomically) and the row survives for the commit. Mirrors the
224
+ // `insertLiveIfAbsent` RETURNING pattern.
225
+ deleteStagingIfStale: GetStmt<[string, string, string, number], { ok: number }>
226
+ selectLive: AllStmt<[string], DbRow>
227
+ selectLiveOne: GetStmt<[string, string], DbRow>
228
+ // Version-CAS commit primitives. Exactly one of the two runs per
229
+ // commit, picked by whether the staging row had a `prev_version`:
230
+ //
231
+ // `insertLiveIfAbsent(tag, res, contentHash, contentLength,
232
+ // signature, putAt)` — the prev_version == null (first-write)
233
+ // path. Inserts the row at version 1 IF ABSENT
234
+ // (`ON CONFLICT (tag,res) DO NOTHING RETURNING 1`). Returns
235
+ // `{ ok: 1 }` if we won the insert; undefined if a racer already
236
+ // created the row (caller → conflict + re-read).
237
+ //
238
+ // `updateLiveCAS(tag, res, nextVersion, contentHash, contentLength,
239
+ // signature, putAt, expectedVersion)` — the re-upload path
240
+ // (prev_version == v). Bumps the row to `nextVersion`
241
+ // `WHERE tag AND resource AND version = expectedVersion
242
+ // RETURNING 1`. Returns `{ ok: 1 }` if our CAS matched the live
243
+ // version; undefined if a racer bumped it first (caller →
244
+ // conflict + re-read). Exactly one racer wins; the loser rebases.
245
+ insertLiveIfAbsent: GetStmt<[string, string, string, string, number, string, number], { ok: number }>
246
+ updateLiveCAS: GetStmt<[string, string, number, string, number, string, number, number, string], { ok: number }>
247
+ // Version-CAS delete: `deleteLiveCAS(tag, res, expectedVersion)` drops
248
+ // the row only while its version still matches the precondition
249
+ // deleteObject read, returning `{ ok: 1 }` iff it removed a row.
250
+ // Without it, a stale delete could destroy a row a concurrent commit
251
+ // just bumped (lost update). Symmetric with `updateLiveCAS`.
252
+ deleteLiveCAS: GetStmt<[string, string, number, string], { ok: number }>
253
+ // `[staleBefore]` — only rows whose `begun_at < staleBefore` are
254
+ // returned. The reaper passes `Date.now() - stagingTtlMs` so the
255
+ // index `workspace_object_staging_begun_at_idx` is used and the
256
+ // sweep is O(stale-rows) instead of O(in-flight-uploads-cluster-
257
+ // wide). DB-layout audit `server/objstore/store.ts:312`.
258
+ listAllStaging: AllStmt<[number], { workspace_tag: string; resource_tag: string; staging_id: string; begun_at: number }>
259
+ listLiveTags: AllStmt<[], { workspace_tag: string }>
260
+ countLive: GetStmt<[string], { c: number }>
261
+ }
262
+
263
+ // Narrowing alias for the SQLite-backed Handle: `db` is guaranteed
264
+ // to be set. `openObjstore` returns this so call sites (production
265
+ // shutdown plumbing in `server/index.ts` + the entire SQLite-only
266
+ // test suite in `tests/server-objstore.test.js`) can reach
267
+ // `handle.db.prepare(...)` without an optional-chain or non-null
268
+ // assertion. A Neon-backed Handle (`openNeonObjstore`) keeps the
269
+ // wider `db?: DatabaseSync` shape; routing a Neon Handle into a
270
+ // SQLite-coupled call site is a type error at compile time.
271
+ export type SqliteHandle = Handle & { db: DatabaseSync }
272
+
273
+ // Per-workspace resource cap. Caps the live row count for a single
274
+ // workspace_tag so a holder of the seed (until per-account GitHub-auth
275
+ // quotas land) can't grow `workspace_object` without bound. Enforced
276
+ // at `beginPut` time for NEW resources only — re-uploads of an existing
277
+ // resourceTag (a new version of the same row) don't change the count
278
+ // and are allowed regardless. The count + insert are not atomic, so
279
+ // transient over-shoot under high concurrency across DIFFERENT
280
+ // resources is bounded by `(parallel new-resource begins - 1)` and is
281
+ // accepted (the cap is a soft policy bound, not a security invariant).
282
+ // This was already the contract under the old per-resource lock —
283
+ // that lock keyed on (tag, resourceTag), so concurrent NEW-resource
284
+ // begins held DIFFERENT locks and raced the count anyway.
285
+ export const MAX_RESOURCES_PER_WORKSPACE = 100
286
+
287
+ // Per-upload byte cap, shared by the WS plane (rejects oversize
288
+ // `expectedLength` in `objstore-put-begin`) and the REST plane (gates
289
+ // the PUT body via Content-Length + post-upload stat). Single source
290
+ // of truth so the two planes can't drift apart on a future bump.
291
+ export const MAX_CONTENT_LENGTH = 100 * 1024 * 1024
292
+
293
+ const TAG_RE = /^[\w-]+$/u
294
+ const CONTENT_HASH_RE = /^[\w-]{43}$/u // 32 raw bytes → 43 b64url chars (no padding)
295
+ const SIG_RE = /^[\w-]{86}$/u // 64 raw bytes → 86 b64url chars (no padding)
296
+ const STAGING_ID_RE = /^[\w-]{22}$/u // 16 raw bytes → 22 b64url chars (no padding)
297
+ const MAX_TAG_LEN = 256
298
+
299
+ // Strict shape gate. The ed25519 signature on every PUT/DELETE binds
300
+ // these fields, so a malformed value here means either a buggy
301
+ // client or someone fuzzing the relay — drop without inserting.
302
+ export function isValidTag(s: unknown): s is string {
303
+ return typeof s === 'string' && s.length > 0 && s.length <= MAX_TAG_LEN && TAG_RE.test(s)
304
+ }
305
+ export function isValidContentHash(s: unknown): s is string {
306
+ return typeof s === 'string' && CONTENT_HASH_RE.test(s)
307
+ }
308
+ export function isValidSignature(s: unknown): s is string {
309
+ return typeof s === 'string' && SIG_RE.test(s)
310
+ }
311
+ // `randomId()` (16 random bytes → base64url) produces exactly this
312
+ // shape (22 chars, base64url alphabet, no padding). Validated on
313
+ // the reaper-side path constructions so a tampered or migrated
314
+ // row whose `staging_id` somehow contains separators / `..` can't
315
+ // trick the reaper into unlinking outside `OBJSTORE_DIR`. PR #4
316
+ // review.
317
+ export function isValidStagingId(s: unknown): s is string {
318
+ return typeof s === 'string' && STAGING_ID_RE.test(s)
319
+ }
320
+ // Incarnation ids are minted by `randomId()` (same 16-byte base64url
321
+ // shape as staging ids), so they share the wire-shape gate. Used by
322
+ // the sig verifiers to reject a malformed client-supplied
323
+ // `prevIncarnation` before it reaches the CAS.
324
+ export function isValidIncarnation(s: unknown): s is string {
325
+ return typeof s === 'string' && STAGING_ID_RE.test(s)
326
+ }
327
+
328
+ function rowFromDb(r: DbRow): ObjectRow {
329
+ return {
330
+ resourceTag: r.resource_tag, version: r.version, incarnation: r.incarnation, contentHash: r.content_hash,
331
+ contentLength: r.content_length, signature: r.signature, putAt: r.put_at,
332
+ }
333
+ }
334
+
335
+ // The live-row fields every objstore wire frame carries (list result,
336
+ // fetch token, PUT broadcast). `putAt` is a server-only debug column
337
+ // the wire never includes. One projection so the emit sites
338
+ // (sync-handlers.ts subscribe-ack `resources`, handlers.ts handleFetch,
339
+ // rest.ts PUT broadcast) can't drift on the shape.
340
+ export type ObjectMetaWire = {
341
+ resourceTag: string; version: number; incarnation: string; contentHash: string; contentLength: number; signature: string
342
+ }
343
+ export function objectMetaWire(row: ObjectRow): ObjectMetaWire {
344
+ return {
345
+ resourceTag: row.resourceTag, version: row.version, incarnation: row.incarnation, contentHash: row.contentHash,
346
+ contentLength: row.contentLength, signature: row.signature,
347
+ }
348
+ }
349
+
350
+ // Convenience signature for the SQLite + local-FS pairing — the
351
+ // only pairing the SQLite DB plane supports (single-process). The
352
+ // second argument is the FS root path passed to the FS BlobBackend;
353
+ // kept as a string (rather than an opaque BlobBackend) so the
354
+ // existing test corpus (`openObjstore(db, objDir)`) doesn't have to
355
+ // thread a backend constructor through every fixture.
356
+ export function openObjstore(db: DatabaseSync, dir: string): SqliteHandle {
357
+ // Ensure the root storage directory exists. The server defaults
358
+ // this to `dirname(DB_PATH)/objstore`; an operator-supplied path
359
+ // with parents that don't exist also gets created here. Eager
360
+ // (vs lazy-on-first-beginPut) so the reaper's startup sweep over
361
+ // an empty root doesn't ENOENT.
362
+ mkdirSync(dir, { recursive: true })
363
+ db.exec(SCHEMA)
364
+ // Fail-loud on a pre-existing non-STRICT table — same rationale as
365
+ // server/db.ts: `CREATE TABLE IF NOT EXISTS … STRICT` doesn't
366
+ // upgrade an existing non-STRICT table, and dropping strict type
367
+ // affinity opens an operator-attack path. PR #4 review F3.
368
+ for (const name of ['workspace_object', 'workspace_object_staging']) {
369
+ const meta = db.prepare(`SELECT strict FROM pragma_table_list WHERE schema = 'main' AND name = ?`).get(name) as { strict: number } | undefined
370
+ if (meta && meta.strict !== 1) throw new Error(`${name} is non-STRICT — migrate before booting`)
371
+ }
372
+ const blob = openFsBlobBackend(dir)
373
+ // No `close` method on the returned Handle: the underlying
374
+ // `DatabaseSync` is owned by the caller (in production, the
375
+ // workspace_revision handle in `server/db.ts`, which closes it
376
+ // from `shutdown()`). Exposing `close()` here was misleading —
377
+ // a callsite reading `await objstoreHandle.close()` would
378
+ // reasonably assume it closes something, when in practice it
379
+ // either no-op'd (production) or left the connection open
380
+ // (tests construct their own DB and `db.close()` separately).
381
+ return {
382
+ db,
383
+ blob,
384
+ dir,
385
+ insertStaging: wrapRun(db.prepare(`
386
+ INSERT INTO workspace_object_staging
387
+ (workspace_tag, resource_tag, staging_id, prev_version, prev_incarnation,
388
+ expected_length, content_hash, signature, begun_at)
389
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
390
+ `)),
391
+ selectStaging: wrapGet(db.prepare(`
392
+ SELECT prev_version, prev_incarnation, expected_length, content_hash, signature, begun_at
393
+ FROM workspace_object_staging
394
+ WHERE workspace_tag = ? AND resource_tag = ? AND staging_id = ?
395
+ `)),
396
+ // Reaper orphan-file sweep: lookup by (ws, sid) only, no resource_tag
397
+ // (the staging filename doesn't carry it). PR #4 review H1.
398
+ selectStagingByWsSid: wrapGet(db.prepare(
399
+ `SELECT 1 FROM workspace_object_staging WHERE workspace_tag = ? AND staging_id = ?`,
400
+ )),
401
+ // Restamp `begun_at` post-upload so TTL counts from upload-done.
402
+ // PR #4 review H4.
403
+ refreshStagingBegunAt: wrapRun(db.prepare(
404
+ `UPDATE workspace_object_staging SET begun_at = ? WHERE workspace_tag = ? AND resource_tag = ? AND staging_id = ?`,
405
+ )),
406
+ deleteStaging: wrapRun(db.prepare(`
407
+ DELETE FROM workspace_object_staging
408
+ WHERE workspace_tag = ? AND resource_tag = ? AND staging_id = ?
409
+ `)),
410
+ // Conditional stale-row delete for the reaper. Atomic CAS on
411
+ // `begun_at`: drops the row only if it's still older than
412
+ // `staleBefore` (bind 4), so a concurrent `refreshStagingBegunAt`
413
+ // that bumped `begun_at` fresh makes the predicate fail and the
414
+ // RETURNING comes back empty → caller skips. Bind order:
415
+ // (tag, res, sid, staleBefore).
416
+ deleteStagingIfStale: wrapGet(db.prepare(`
417
+ DELETE FROM workspace_object_staging
418
+ WHERE workspace_tag = ? AND resource_tag = ? AND staging_id = ? AND begun_at < ?
419
+ RETURNING 1 AS ok
420
+ `)),
421
+ selectLive: wrapAll(db.prepare(`
422
+ SELECT resource_tag, version, incarnation, content_hash, content_length,
423
+ signature, put_at
424
+ FROM workspace_object
425
+ WHERE workspace_tag = ?
426
+ ORDER BY resource_tag ASC
427
+ `)),
428
+ selectLiveOne: wrapGet(db.prepare(`
429
+ SELECT resource_tag, version, incarnation, content_hash, content_length,
430
+ signature, put_at
431
+ FROM workspace_object
432
+ WHERE workspace_tag = ? AND resource_tag = ?
433
+ `)),
434
+ // First-write CAS: insert the live row at version 1 IF ABSENT.
435
+ // `ON CONFLICT … DO NOTHING RETURNING 1` returns a row only when
436
+ // OUR insert won — a racing first-write commit that landed first
437
+ // (same prev_version == null precondition) makes this a no-op and
438
+ // RETURNING comes back empty → caller maps to conflict. Bind
439
+ // order: (tag, res, hash, len, sig, put_at); version is the
440
+ // literal 1.
441
+ insertLiveIfAbsent: wrapGet(db.prepare(`
442
+ INSERT INTO workspace_object
443
+ (workspace_tag, resource_tag, version, incarnation, content_hash, content_length,
444
+ signature, put_at)
445
+ VALUES (?, ?, 1, ?, ?, ?, ?, ?)
446
+ ON CONFLICT (workspace_tag, resource_tag) DO NOTHING
447
+ RETURNING 1 AS ok
448
+ `)),
449
+ // Re-upload CAS: bump the row to `?3` (next version) only when the
450
+ // live version still equals `?8` (the version we read as our
451
+ // precondition). `RETURNING 1` comes back only when the WHERE
452
+ // matched — exactly one of N racing re-uploads against the same
453
+ // base version wins; the losers get an empty result → conflict +
454
+ // rebase. Bind order: (tag, res, nextVersion, hash, len, sig,
455
+ // put_at, expectedVersion).
456
+ updateLiveCAS: wrapGet(db.prepare(`
457
+ UPDATE workspace_object
458
+ SET version = ?3,
459
+ content_hash = ?4,
460
+ content_length = ?5,
461
+ signature = ?6,
462
+ put_at = ?7
463
+ WHERE workspace_tag = ?1 AND resource_tag = ?2 AND version = ?8 AND incarnation = ?9
464
+ RETURNING 1 AS ok
465
+ `)),
466
+ // Version-conditional drop for deleteObject: removes the row only
467
+ // if its version still equals the precondition. RETURNING tells us
468
+ // whether we won; 0 rows → a racing commit/delete moved it → the
469
+ // caller re-reads and returns conflict / not-found. Bind order:
470
+ // (tag, res, expectedVersion).
471
+ deleteLiveCAS: wrapGet(db.prepare(`
472
+ DELETE FROM workspace_object
473
+ WHERE workspace_tag = ? AND resource_tag = ? AND version = ? AND incarnation = ?
474
+ RETURNING 1 AS ok
475
+ `)),
476
+ listAllStaging: wrapAll(db.prepare(`
477
+ SELECT workspace_tag, resource_tag, staging_id, begun_at
478
+ FROM workspace_object_staging
479
+ WHERE begun_at < ?
480
+ `)),
481
+ listLiveTags: wrapAll(db.prepare(`
482
+ SELECT DISTINCT workspace_tag FROM workspace_object
483
+ `)),
484
+ countLive: wrapGet(db.prepare(`
485
+ SELECT COUNT(*) AS c FROM workspace_object WHERE workspace_tag = ?
486
+ `)),
487
+ }
488
+ }
489
+
490
+ export async function getLive(handle: Handle, tag: string, resourceTag: string): Promise<ObjectRow | null> {
491
+ const row = await handle.selectLiveOne.get(tag, resourceTag)
492
+ return row ? rowFromDb(row) : null
493
+ }
494
+
495
+ export async function listLive(handle: Handle, tag: string): Promise<ObjectRow[]> {
496
+ const rows = await handle.selectLive.all(tag)
497
+ return rows.map(rowFromDb)
498
+ }
499
+
500
+ // Mints a staging id, validates the prev_version precondition,
501
+ // inserts the staging row. Async — `ensureWorkspace` is genuinely
502
+ // async on the FS backend (mkdir) and a no-op on the Vercel backend;
503
+ // the DB calls are async-shaped wrappers around the sync
504
+ // `node:sqlite` driver. No lock: the prev_version check here is
505
+ // ADVISORY (a fast-fail so the client rebases before uploading) — the
506
+ // authoritative precondition is commitPut's version-CAS, which is
507
+ // atomic against concurrent commits regardless of what happens
508
+ // between this begin and that commit.
509
+ //
510
+ // Returns the stagingId the REST PUT layer pairs with the bytes; the
511
+ // optional `filePath` is set only for the FS backend (test seam, see
512
+ // BeginPutResult).
513
+ export async function beginPut(handle: Handle, input: BeginPutInput): Promise<BeginPutResult> {
514
+ const live = await getLive(handle, input.workspaceTag, input.resourceTag)
515
+ const liveVersion = live?.version ?? null
516
+ const liveIncarnation = live?.incarnation ?? null
517
+ // Advisory tuple check: both version AND incarnation must match the
518
+ // precondition. A stale `prev` whose version happens to align with a
519
+ // recreated incarnation (the cross-incarnation overwrite) is rejected
520
+ // here on the incarnation mismatch. Authoritative re-check is the CAS
521
+ // in commitPut.
522
+ if (liveVersion !== input.prevVersion || liveIncarnation !== input.prevIncarnation) {
523
+ return { ok: false, reason: 'conflict', conflict: live }
524
+ }
525
+ // Per-workspace resource cap. Only enforced for NEW resources —
526
+ // re-uploads of an existing resourceTag (live != null) don't
527
+ // change the count, so they're always allowed. Not atomic with the
528
+ // insert below (see MAX_RESOURCES_PER_WORKSPACE) — a soft policy
529
+ // bound, accepted to over-shoot under concurrent NEW-resource begins.
530
+ if (!live) {
531
+ const count = await handle.countLive.get(input.workspaceTag) as { c: number } | undefined
532
+ if ((count?.c ?? 0) >= MAX_RESOURCES_PER_WORKSPACE) {
533
+ return { ok: false, reason: 'workspace-full' }
534
+ }
535
+ }
536
+ const stagingId = randomId()
537
+ await handle.blob.ensureWorkspace(input.workspaceTag)
538
+ await handle.insertStaging.run(
539
+ input.workspaceTag,
540
+ input.resourceTag,
541
+ stagingId,
542
+ input.prevVersion,
543
+ input.prevIncarnation,
544
+ input.expectedLength,
545
+ input.contentHash,
546
+ input.signature,
547
+ Date.now(),
548
+ )
549
+ return handle.dir === undefined
550
+ ? { ok: true, stagingId }
551
+ : { ok: true, stagingId, filePath: stagingFilePath(handle.dir, input.workspaceTag, stagingId) }
552
+ }
553
+
554
+ // Validates the on-disk staged size, promotes the staging blob to its
555
+ // content-addressed live path, then commits via an atomic version
556
+ // compare-and-set on the live row: a first-write inserts at version 1
557
+ // IF ABSENT; a re-upload bumps the version IFF it still matches the
558
+ // precondition we read. Exactly one of N racing commits wins the CAS;
559
+ // the losers get `conflict` (with the current live row) and rebase.
560
+ // Because each PUT is content-addressed at its OWN hash (distinct PUTs
561
+ // get distinct hashes — random nonce per encrypt), N racing commits
562
+ // promote to N DIFFERENT immutable paths: no promote clobbers another's
563
+ // bytes, and a loser's blob is just left unreferenced for the GC. There
564
+ // is no metadata-vs-bytes desync to guard against. The
565
+ // CAS provides the commit's atomicity both within a process and across
566
+ // replicas, so NO in-process lock is taken. A crash between the
567
+ // promote and the CAS leaves the staging blob/row intact alongside (at
568
+ // most) a stranded, unreferenced live blob — the reaper's stale-staging
569
+ // sweep cleans the row, and the GC reaps the unreferenced live blob
570
+ // once it's past the grace window, matching the "stranded state,
571
+ // reaper-cleaned, never row-pointing-at-nothing" crash-safety contract.
572
+ //
573
+ // ACCEPTED TRADEOFF (lock removal): with no lock, the reaper no longer
574
+ // WAITS for an in-flight upload on this key. An upload taking >1h FROM
575
+ // BEGIN (i.e. exceeding the staging TTL during the body) can have its
576
+ // staging row reaped mid-flight by `deleteStagingIfStale`; this commit
577
+ // then sees no staging row and returns `no-staging` → REST 410. The
578
+ // previous lock made the reaper block on any in-flight upload
579
+ // (unbounded). Sub-1h uploads are unaffected: `begun_at` (set at begin)
580
+ // stays within the TTL through the body, so the conditional delete
581
+ // can't match, and the after-body `refreshStagingBegunAt` re-extends
582
+ // the TTL to cover this commit step. This matches the staging TTL's
583
+ // documented intent ("1h, comfortably over a 50 MiB upload on a slow
584
+ // line").
585
+ export async function commitPut(handle: Handle, input: CommitPutInput): Promise<CommitPutResult> {
586
+ const staging = await handle.selectStaging.get(input.workspaceTag, input.resourceTag, input.stagingId)
587
+ if (!staging) return { ok: false, reason: 'no-staging' }
588
+ let stagedSize: number | null
589
+ // statStaging failure here is a server-side issue (staging file
590
+ // was unlinked by a racing abort / reaper, EACCES, EIO, backend
591
+ // unreachable, …) — not a client length-mismatch. Route through
592
+ // `io-error` so the REST layer returns 5xx, not 400. PR #4 review.
593
+ //
594
+ // The REST PUT layer already statted the staging blob post-upload
595
+ // and threads the result in via `observedSize` — skipping the
596
+ // round-trip saves one Vercel HEAD per PUT. The staging blob can't
597
+ // have been resized between that stat and here: staging ids are
598
+ // 16-byte random, so no other request targets this blob, and the
599
+ // sole writer (this PUT's upload pipeline) has already finished
600
+ // before the REST stat ran. The only other actor that touches a
601
+ // staging blob is the reaper, which UNLINKS (it never resizes); a
602
+ // racing reaper unlink surfaces below as statStaging→io-error or a
603
+ // promote failure, not a wrong size. WS / test paths that omit
604
+ // `observedSize` fall through to the explicit stat.
605
+ if (input.observedSize === undefined) {
606
+ try { stagedSize = await handle.blob.statStaging(input.workspaceTag, input.stagingId) }
607
+ catch { return { ok: false, reason: 'io-error' } }
608
+ if (stagedSize == null) return { ok: false, reason: 'io-error' }
609
+ } else {
610
+ stagedSize = input.observedSize
611
+ }
612
+ // Truncation invariant: a partial upload (received < declared, or a
613
+ // mid-stream abort that left a short staging file) MUST NEVER be
614
+ // promoted to live. The REST layer already gates on
615
+ // `received !== declared` before reaching here, but commitPut re-
616
+ // stats as the last line of defense — if the storage-side size
617
+ // doesn't match what the signature committed to, we bail BEFORE the
618
+ // promotion. A client that retries the upload under the same
619
+ // resourceTag gets a fresh stagingId + fresh staging slot (staging
620
+ // ids are random, so the retry never shares a blob with the
621
+ // truncated original); the truncated original is untouched by the
622
+ // retry's promote. The live blob's bytes are therefore always a
623
+ // complete signed payload.
624
+ if (stagedSize !== staging.expected_length) return { ok: false, reason: 'size-mismatch' }
625
+ // Cheap early-out conflict check: a concurrent commit / delete may
626
+ // have raced past us between begin and now. The authoritative test
627
+ // is the CAS below (it's atomic against concurrent writers); this
628
+ // read just lets us skip the promote when we already know we've
629
+ // lost, and gives us the current row for the conflict result.
630
+ const live = await getLive(handle, input.workspaceTag, input.resourceTag)
631
+ const liveVersion = live?.version ?? null
632
+ const liveIncarnation = live?.incarnation ?? null
633
+ if (liveVersion !== staging.prev_version || liveIncarnation !== staging.prev_incarnation) {
634
+ // Don't unlink the staging blob here — the caller routes
635
+ // through abortPut to clean up consistently.
636
+ return { ok: false, reason: 'conflict', ...(live ? { conflict: live } : {}) }
637
+ }
638
+ // Promote to the CONTENT-ADDRESSED live path `${tag}/${hash}.bin`.
639
+ // `promoteStagingToLive` returns false on any backend error (FS:
640
+ // EACCES / ENOSPC / EIO / a racing abort that already unlinked
641
+ // the staging file; Vercel: copy failure). 'io-error' is mapped
642
+ // to HTTP 500 by the REST layer — it's a server-side fault, not
643
+ // a client-fixable one. Because the destination path IS the content
644
+ // hash, any write to it is byte-identical BY CONSTRUCTION, so a
645
+ // retried or racing promote to the same path is an idempotent
646
+ // rewrite, never a clobber. (Distinct PUTs get distinct hashes — a
647
+ // fresh random nonce per encrypt makes each ciphertext unique — so
648
+ // concurrent commits to the same resource write to DIFFERENT paths.)
649
+ if (!await handle.blob.promoteStagingToLive(input.workspaceTag, input.stagingId, staging.content_hash)) {
650
+ return { ok: false, reason: 'io-error' }
651
+ }
652
+ const nextVersion = (liveVersion ?? 0) + 1
653
+ // First-write mints a fresh incarnation; a re-upload preserves the
654
+ // matched one (updateLiveCAS doesn't touch the column). The pre-check
655
+ // above guarantees `prev_incarnation` is a real string on the
656
+ // re-upload path (live exists and its non-null incarnation equals it).
657
+ const freshIncarnation = randomId()
658
+ const committedIncarnation = staging.prev_version == null ? freshIncarnation : staging.prev_incarnation!
659
+ const putAt = Date.now()
660
+ // Version-CAS in try/catch so a Neon transient (5xx, network
661
+ // hiccup, connection-pool exhaustion) doesn't bypass the abortPut
662
+ // ladder by throwing out of commitPut. Without this guard, a thrown
663
+ // rejection skips the REST layer's `if (!r.ok) abortPut` branch and
664
+ // bubbles to handleRest's outer catch — the live blob is already
665
+ // promoted, the staging blob + row stay, and the client sees a 500.
666
+ // The reaper reconciles (stale-staging sweep + unreferenced-blob
667
+ // GC) but the surface is a 500 the caller can retry.
668
+ let won: { ok: number } | undefined
669
+ try {
670
+ if (staging.prev_version == null) {
671
+ // First write: insert at version 1 IF ABSENT. A racing
672
+ // first-write that landed first occupies the slot → our insert
673
+ // is a no-op → empty RETURNING → conflict.
674
+ won = await handle.insertLiveIfAbsent.get(
675
+ input.workspaceTag,
676
+ input.resourceTag,
677
+ freshIncarnation,
678
+ staging.content_hash,
679
+ staging.expected_length,
680
+ staging.signature,
681
+ putAt,
682
+ )
683
+ } else {
684
+ // Re-upload: bump version IFF the live version still equals our
685
+ // precondition. Exactly one of N racers against the same base
686
+ // version wins; the losers' CAS matches no row → conflict.
687
+ won = await handle.updateLiveCAS.get(
688
+ input.workspaceTag,
689
+ input.resourceTag,
690
+ nextVersion,
691
+ staging.content_hash,
692
+ staging.expected_length,
693
+ staging.signature,
694
+ putAt,
695
+ staging.prev_version,
696
+ committedIncarnation,
697
+ )
698
+ }
699
+ } catch (err) {
700
+ console.warn('commitPut version-CAS failed:', errMsg(err))
701
+ // Caller's `if (!r.ok) abortPut` cleans the staging side. The
702
+ // just-promoted live blob is unreferenced; the reaper's GC unlinks
703
+ // it once it's past the grace window.
704
+ return { ok: false, reason: 'io-error' }
705
+ }
706
+ if (!won) {
707
+ // A racer won the CAS between our pre-check `getLive` and the
708
+ // write. Re-read the live row so the caller can surface the
709
+ // current version in the conflict (the client rebases off it).
710
+ // Our just-promoted blob is now unreferenced — the winner's row
711
+ // names a different hash (distinct PUTs get distinct hashes), so the
712
+ // reaper's GC reclaims our blob once it's past the grace window. No
713
+ // desync — just a conflict to rebase.
714
+ const current = await getLive(handle, input.workspaceTag, input.resourceTag)
715
+ return { ok: false, reason: 'conflict', ...(current ? { conflict: current } : {}) }
716
+ }
717
+ // Staging cleanup AFTER the CAS so a crash between the promote and
718
+ // the CAS leaves the staging blob intact and the live row at its
719
+ // prior value — a state a retry can re-commit from. The alternative
720
+ // ordering (cleanup before the DB write) would drop the staging
721
+ // bytes a failed/retried commit still needs.
722
+ //
723
+ // FS backend: unlinkStaging is a no-op here because `rename`
724
+ // already removed the source file; the tolerant ENOENT path
725
+ // handles it. Vercel backend: actually deletes the staging blob.
726
+ await handle.blob.unlinkStaging(input.workspaceTag, input.stagingId)
727
+ await handle.deleteStaging.run(input.workspaceTag, input.resourceTag, input.stagingId)
728
+ return {
729
+ ok: true,
730
+ row: {
731
+ resourceTag: input.resourceTag,
732
+ version: nextVersion,
733
+ incarnation: committedIncarnation,
734
+ contentHash: staging.content_hash,
735
+ contentLength: staging.expected_length,
736
+ signature: staging.signature,
737
+ putAt,
738
+ },
739
+ }
740
+ }
741
+
742
+ // Idempotent — DELETE row is a no-op when already gone, not-found
743
+ // unlink is ignored. Routes through `handle.blob.unlinkStaging` so
744
+ // the FS backend uses async unlink (no event-loop stall on a slow
745
+ // disk) and the Vercel backend issues an HTTP DELETE.
746
+ export async function abortPut(handle: Handle, tag: string, resourceTag: string, stagingId: string): Promise<void> {
747
+ await handle.blob.unlinkStaging(tag, stagingId)
748
+ await handle.deleteStaging.run(tag, resourceTag, stagingId)
749
+ }
750
+
751
+ // `prevVersion = null` + missing row = already-deleted-or-never-
752
+ // existed; treat as success so retried DELETEs are idempotent. The
753
+ // `deletedVersion = 0` sentinel tells the broadcast path to skip.
754
+ //
755
+ // No lock: the drop is a version-CAS (`deleteLiveCAS` — DELETE WHERE
756
+ // version = prev), so every race resolves to exactly one winner, the
757
+ // same as the old lock did, with no lost update:
758
+ // - delete vs. a concurrent COMMIT on the same resource: whichever
759
+ // CAS lands first wins. A stale delete can NOT remove a row the
760
+ // commit just bumped — the `WHERE version = prev` no longer matches,
761
+ // so the delete gets `conflict` (re-read sees the bumped version).
762
+ // If the delete wins, the commit's `updateLiveCAS` matches no row →
763
+ // `conflict`. (The earlier draft used an UNCONDITIONAL delete here,
764
+ // which — without the lock — let `getLive` read v1, a commit bump
765
+ // to v2 slip in, and the delete then destroy v2: a lost update. The
766
+ // version-CAS closes that.)
767
+ // - two concurrent deletes with the same prevVersion: one CAS removes
768
+ // the row; the other matches no row → re-read → `not-found`.
769
+ //
770
+ // On success we ONLY drop the live row — we do NOT unlink the live
771
+ // blob. Blob reclamation is deferred to the reaper's grace-window GC
772
+ // (unlinks once no live row references the hash AND it's older than the
773
+ // grace window) rather than done inline, so the drop stays lock-free
774
+ // and can't race two things: a concurrent commit's promote→CAS window
775
+ // (a just-promoted blob isn't referenced yet — the age grace protects
776
+ // it) and an in-flight GET still streaming the bytes. NOT because the
777
+ // hash might be shared — distinct PUTs get distinct hashes (random
778
+ // nonce per encrypt → unique ciphertext), so the hash↔row mapping is
779
+ // effectively 1:1; this delete simply orphans the blob for the GC.
780
+ export async function deleteObject(
781
+ handle: Handle, tag: string, resourceTag: string, prevVersion: number | null, prevIncarnation: string | null,
782
+ ): Promise<DeleteResult> {
783
+ const live = await getLive(handle, tag, resourceTag)
784
+ if (!live) {
785
+ if (prevVersion == null) return { ok: true, deletedVersion: 0 }
786
+ return { ok: false, reason: 'not-found' }
787
+ }
788
+ if (live.version !== prevVersion || live.incarnation !== prevIncarnation) return { ok: false, reason: 'conflict', conflict: live }
789
+ // Version+incarnation-conditional drop: only delete while the row is
790
+ // STILL the exact (version, incarnation) we just read, so neither a
791
+ // commit bump NOR a delete+recreate landing between the read above and
792
+ // here can have its row removed by this stale delete.
793
+ const removed = await handle.deleteLiveCAS.get(tag, resourceTag, live.version, live.incarnation)
794
+ if (removed) return { ok: true, deletedVersion: live.version }
795
+ // A racing commit/delete moved or removed the row after our read.
796
+ const current = await getLive(handle, tag, resourceTag)
797
+ if (!current) return { ok: false, reason: 'not-found' }
798
+ return { ok: false, reason: 'conflict', conflict: current }
799
+ }