@preventive/triage 1.0.0-alpha.2 → 1.0.0-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -119,21 +119,19 @@ type VercelBlobSdk = {
119
119
 
120
120
  // Recognise "blob is gone" errors uniformly across read/write/delete
121
121
  // paths so callers can treat them as success (delete) or
122
- // not-found (read). The SDK exposes BlobNotFoundError as a class
123
- // with `.name === 'BlobNotFoundError'`; checking the name string
124
- // avoids importing the class at the top level (which would force
125
- // the optional peer dep to resolve).
122
+ // not-found (read). The SDK exposes BlobNotFoundError as a class with
123
+ // `.name === 'BlobNotFoundError'`; checking the name string avoids
124
+ // importing the class at the top level (which would force the optional
125
+ // peer dep to resolve).
126
126
  //
127
- // Class-name check ONLY. The SDK's internal mapper translates every
128
- // API `not_found` code into BlobNotFoundError-by-name; a bare-404
129
- // transport leak doesn't reach here. A prior version of this
130
- // function had a `/does not exist|\b404\b/` fallback that
131
- // DANGEROUSLY matched BlobStoreNotFoundError's message "This store
132
- // does not exist." — a config fault (revoked token, deleted store)
133
- // would silently surface as every-blob-missing across reads and
134
- // unlinks, masking the fatal misconfiguration. The tight name check
135
- // lets BlobStoreNotFoundError / other classes propagate as real
136
- // exceptions.
127
+ // Class-name check ONLY — the SDK's internal mapper turns every API
128
+ // `not_found` into BlobNotFoundError-by-name, so a bare-404 transport
129
+ // leak doesn't reach here. A broader `/does not exist|\b404\b/` match
130
+ // is DANGEROUS: it also matches BlobStoreNotFoundError's "This store
131
+ // does not exist.", so a config fault (revoked token, deleted store)
132
+ // would silently surface as every-blob-missing across reads/unlinks,
133
+ // masking the fatal misconfiguration. The tight name check lets
134
+ // BlobStoreNotFoundError / other classes propagate as real exceptions.
137
135
  function isNotFound(err: unknown): boolean {
138
136
  if (err == null || typeof err !== 'object') return false
139
137
  const name = (err as { name?: unknown }).name
@@ -242,20 +240,17 @@ function buildOpenStagingWriter(sdk: VercelBlobSdk, token: string): BlobBackend[
242
240
  abortSignal: ac.signal,
243
241
  })
244
242
  // Defuse a possible unhandled-rejection if abort() is called
245
- // BEFORE finalize() (the REST layer's error path). Attach a
246
- // detached `.catch` on the original promise so an early
247
- // rejection has a handler; finalize() awaits `putPromise`
248
- // directly, which still re-throws the original rejection
249
- // (the .catch returns a separate chain that doesn't replace
250
- // putPromise's state).
243
+ // BEFORE finalize() (the REST error path). The detached `.catch`
244
+ // gives an early rejection a handler; finalize() still awaits
245
+ // `putPromise` directly and re-throws the original rejection (the
246
+ // .catch is a separate chain, not a replacement of putPromise).
251
247
  putPromise.catch(() => {})
252
248
  return {
253
249
  writable: pt,
254
250
  // Await the upload's completion. After pipeline(req, counter,
255
- // pt) resolves, pt has emitted 'end' on the read side and
256
- // `put` is finalising the last multipart part. Awaiting here
257
- // gives us the same "bytes durable" guarantee that
258
- // pipeline-to-WriteStream gives the FS backend.
251
+ // pt) resolves, pt has emitted 'end' and `put` is finalising the
252
+ // last multipart part. Awaiting here gives the same "bytes
253
+ // durable" guarantee pipeline-to-WriteStream gives the FS backend.
259
254
  finalize: async () => { await putPromise },
260
255
  // Await the SDK's put-promise settlement (rejected via the
261
256
  // AbortController). Without this await, a slow upload that
@@ -315,11 +310,11 @@ function buildPromoteStagingToLive(sdk: VercelBlobSdk, token: string): BlobBacke
315
310
  //
316
311
  // `allowOverwrite: true` because the content-addressed live
317
312
  // pathname `${tag}/${contentHash}.bin` can be (re)written by a
318
- // retried or racing promote of the same blob. The destination
319
- // bytes are identical by construction (the path IS the hash), so
320
- // the overwrite is idempotent — never a clobber of DIFFERENT
321
- // bytes. Without the flag, the SDK sends `x-allow-overwrite: 0`
322
- // and Vercel rejects any such re-promote with BlobAccessError.
313
+ // retried or racing promote of the same blob. Destination bytes
314
+ // are identical by construction (the path IS the hash), so the
315
+ // overwrite is idempotent — never a clobber of DIFFERENT bytes.
316
+ // Without the flag the SDK sends `x-allow-overwrite: 0` and
317
+ // Vercel rejects the re-promote with BlobAccessError.
323
318
  await sdk.copy(from, to, {
324
319
  access: 'private',
325
320
  allowOverwrite: true,
@@ -356,38 +351,55 @@ function buildOpenLiveReader(sdk: VercelBlobSdk, token: string): BlobBackend['op
356
351
  // origin truth. Origin fetch is the right default for a store
357
352
  // where freshness > latency.
358
353
  try { res = await sdk.get(path, { access: 'private', useCache: false, token }) } catch (err) {
359
- if (isNotFound(err)) return { ok: false, reason: 'not-found' }
354
+ // A missing blob HERE is never "the resource doesn't exist" — the
355
+ // REST layer (rest.ts openLiveSnapshot) already confirmed a live row
356
+ // whose (version, incarnation) matches the GET token before calling
357
+ // us. So BlobNotFoundError means the bytes for a still-live row are
358
+ // momentarily gone: the reaper GC'd a hash a racing version-bump just
359
+ // unreferenced, or Vercel's read-after-write / edge propagation hasn't
360
+ // caught up to a freshly-promoted private blob. That is the documented
361
+ // `unavailable` (HTTP 503) transient — reconciled by reaper /
362
+ // propagation, retried by the client — NOT a 404. Returning
363
+ // `not-found` would emit a 404 the FS backend never emits for the
364
+ // same condition (blob-fs.ts maps ENOENT → `unavailable`), telling
365
+ // the client the resource is gone for good when it should refetch.
366
+ // See server/README.md's GET status table.
367
+ if (isNotFound(err)) return { ok: false, reason: 'unavailable', detail: 'vercel-get-not-found' }
360
368
  throw err
361
369
  }
362
- if (res == null) return { ok: false, reason: 'not-found' }
370
+ // SDK returned null (no blob) — same "bytes missing for a live row"
371
+ // transient as the BlobNotFoundError branch above → `unavailable`, not
372
+ // `not-found`.
373
+ if (res == null) return { ok: false, reason: 'unavailable', detail: 'vercel-get-null' }
363
374
  // statusCode 304 doesn't reach here in practice — the REST
364
375
  // GET layer doesn't pass If-None-Match — but a future call
365
376
  // site could. Treat as unavailable rather than streaming a
366
377
  // null body.
367
378
  if (res.statusCode !== 200 || res.stream == null) {
368
- return { ok: false, reason: 'unavailable' }
379
+ return { ok: false, reason: 'unavailable', detail: `vercel-get-status-${res.statusCode}` }
369
380
  }
370
- // `@vercel/blob@2.x`'s streaming `get()` for private blobs
371
- // returns the body but does NOT populate `blob.size` and does
372
- // NOT pass a `content-length` header through (verified
373
- // empirically: get.size=0, content-length-hdr=null while
374
- // head.size reports the true byte count). The REST layer
375
- // depends on a size to set `content-length` on its response
376
- // and for the integrity check against the DB row, so when
377
- // get() leaves it 0/null we fall back to a head() lookup.
378
- // Two round-trips per private read on Vercel until the SDK is
379
- // fixed — small price vs. the alternative of 503ing every read.
381
+ // `@vercel/blob@2.x`'s streaming `get()` for private blobs returns
382
+ // the body but does NOT populate `blob.size` nor pass a
383
+ // `content-length` header (verified empirically: get.size=0,
384
+ // content-length-hdr=null, while head.size reports the true count).
385
+ // The REST layer needs a size to set `content-length` and for the
386
+ // integrity check against the DB row, so on 0/null we fall back to
387
+ // head(). Two round-trips per private read until the SDK is fixed —
388
+ // small price vs. 503ing every read.
380
389
  let size: number | null | undefined = res.blob?.size
381
390
  if (size == null || size === 0) {
382
391
  try {
383
392
  const h = await sdk.head(path, { token })
384
393
  size = (h as { size?: number })?.size
385
394
  } catch (headErr) {
386
- if (isNotFound(headErr)) return { ok: false, reason: 'not-found' }
395
+ // Blob vanished between get() and the head() size fallback (a
396
+ // racing reaper GC) — still the "live row present, bytes gone"
397
+ // transient, so `unavailable` (503), matching the get() path above.
398
+ if (isNotFound(headErr)) return { ok: false, reason: 'unavailable', detail: 'vercel-head-not-found' }
387
399
  throw headErr
388
400
  }
389
401
  }
390
- if (size == null) return { ok: false, reason: 'unavailable' }
402
+ if (size == null) return { ok: false, reason: 'unavailable', detail: 'vercel-no-size' }
391
403
  // SDK returns a web ReadableStream<Uint8Array>; the REST layer
392
404
  // expects a Node Readable for pipeline(). Convert via
393
405
  // Readable.fromWeb — built-in and zero-copy where possible.
@@ -83,13 +83,27 @@ export type LiveReader = {
83
83
  close(): Promise<void>
84
84
  }
85
85
 
86
- // `not-found` maps to HTTP 404 (the live blob is gone or never
87
- // existed); `unavailable` maps to HTTP 503 (transient backend issue
88
- // the reaper will eventually sort out). The REST layer uses this
89
- // discrimination to set the right status code.
86
+ // A missing blob is always `unavailable` (→ HTTP 503), never a 404. The
87
+ // byte plane has NO view of the metadata row, so it can't decide whether
88
+ // a resource "doesn't exist" — only whether specific bytes are present
89
+ // right now. The authoritative "this resource/version is gone" 404 is the
90
+ // REST layer's call, made from the live row BEFORE it opens a reader
91
+ // (rest.ts openLiveSnapshot). By the time `openLiveReader` runs the row is
92
+ // already confirmed, so an absent blob means a transient bytes/metadata
93
+ // desync the reaper (or store propagation) reconciles — exactly the
94
+ // `unavailable`/503 contract, which the client retries. Both backends MUST
95
+ // map a missing blob to `unavailable` (FS: ENOENT; Vercel: BlobNotFoundError
96
+ // / null get()). No 404-mapping variant exists here so that bug can't recur.
97
+ //
98
+ // `detail` is a short, NON-SENSITIVE machine tag for the specific cause
99
+ // (e.g. 'vercel-get-not-found', 'fs-enoent', 'vercel-no-size'). Every
100
+ // byte-side failure collapses to the same 503 on the wire, so a permanent
101
+ // loss (reaper GC'd the bytes) and a transient read fault are otherwise
102
+ // indistinguishable — the REST layer logs `detail` so an operator can tell
103
+ // them apart. Purely diagnostic; the REST status is unchanged.
90
104
  export type OpenLiveResult =
91
105
  | { ok: true; reader: LiveReader }
92
- | { ok: false; reason: 'not-found' | 'unavailable' }
106
+ | { ok: false; reason: 'unavailable'; detail?: string }
93
107
 
94
108
  export type BlobBackend = {
95
109
  // Per-workspace setup. FS creates the on-disk staging directory;
@@ -128,9 +142,10 @@ export type BlobBackend = {
128
142
  promoteStagingToLive(tag: string, stagingId: string, contentHash: string): Promise<boolean>
129
143
 
130
144
  // Open a streaming reader for the content-addressed live blob.
131
- // `not-found` lets the REST layer return 404; `unavailable` returns
132
- // 503 for a transient state (file/blob missing while the row still
133
- // exists — reaper will reconcile on the next sweep).
145
+ // Called only after the REST layer has confirmed the live row, so a
146
+ // missing blob is the transient "row present, bytes gone" state →
147
+ // `unavailable` (HTTP 503), which the reaper reconciles and the client
148
+ // retries. Never a 404 from here — see OpenLiveResult above.
134
149
  openLiveReader(tag: string, contentHash: string): Promise<OpenLiveResult>
135
150
 
136
151
  // Idempotent deletes. Backends MUST tolerate "already gone" as
@@ -68,11 +68,11 @@ function urlPathFor(tag: string, resourceTag: string): string {
68
68
  // Shared gate every objstore handler runs after its message-specific
69
69
  // field checks: fetch the socket's challenge nonce, verify the signed
70
70
  // message against it, then re-confirm the socket is still OPEN — the
71
- // close handler may have fired during the verify await, and attaching
72
- // to / replying on a closed socket is the half-handshake leak case
73
- // (PR #4 review F4). Returns true when the caller may proceed.
74
- // Centralising the post-await readyState recheck keeps that
75
- // easy-to-forget invariant in one auditable place.
71
+ // close handler may have fired during the verify await, and replying
72
+ // on a closed socket is the half-handshake leak case (PR #4 review
73
+ // F4). Centralising the post-await readyState recheck keeps that
74
+ // invariant in one auditable place. Returns true when the caller may
75
+ // proceed.
76
76
  async function verified<M extends { workspaceTag?: unknown }>(
77
77
  deps: ObjstoreDeps, socket: WebSocket, msg: M, label: string,
78
78
  verify: (m: M, nonce: string) => Promise<boolean>,
@@ -96,12 +96,10 @@ async function handlePutBegin(deps: ObjstoreDeps, socket: WebSocket, msg: Objsto
96
96
  // signature to fail. Cheaper to reject up-front, and consistent
97
97
  // with `verifyObjstorePutSig`'s `isSafeNonNegativeInt` gate.
98
98
  if (!Number.isSafeInteger(msg.expectedLength) || (msg.expectedLength as number) < 0 || (msg.expectedLength as number) > MAX_CONTENT_LENGTH) return
99
- // Symmetric with `handleDelete`'s prevVersion gate (line 116) and
100
- // `verifyObjstorePutSig`'s `isSafeIntOrNull` (sign.ts:119). Without
101
- // this, a non-safe-integer `prevVersion` (NaN, 2^53+1, ...) would
102
- // pass the typeof check below and reach sig verify, burning a
103
- // hash + Ed25519 round-trip on a guaranteed-fail input. Input-
104
- // validation audit `server/objstore/handlers.ts:76`.
99
+ // Same up-front reject for `prevVersion` (symmetric with handleDelete
100
+ // and `verifyObjstorePutSig`'s `isSafeIntOrNull`): a non-safe-integer
101
+ // (NaN, 2^53+1, ...) would pass the typeof check below and reach sig
102
+ // verify, burning a hash + Ed25519 round-trip on a guaranteed fail.
105
103
  if (msg.prevVersion != null && (typeof msg.prevVersion !== 'number' || !Number.isSafeInteger(msg.prevVersion))) return
106
104
  if (!await verified(deps, socket, msg, 'put-begin', verifyObjstorePutSig)) return
107
105
  const tag = msg.workspaceTag
@@ -110,8 +108,8 @@ async function handlePutBegin(deps: ObjstoreDeps, socket: WebSocket, msg: Objsto
110
108
  // workspace tag (no rows in workspace_revision AND none in
111
109
  // workspace_object). Mirrors handleSave in server/index.ts; runs
112
110
  // AFTER sig verify so `unauthorized` only reaches a legitimate
113
- // signer. The gate is config-driven (server/config.json
114
- // `password`) and is a no-op when no password is configured.
111
+ // signer. Config-driven (server/config.json `password`), no-op when
112
+ // unconfigured.
115
113
  if (deps.authGate && deps.sendUnauthorized && await deps.authGate(socket, tag)) {
116
114
  if (socket.readyState !== socket.OPEN) return
117
115
  if (deps.debug) console.warn(`reject objstore-put-begin: unauthorized (new workspace ${debugTag(tag)})`)
@@ -56,6 +56,16 @@ export type ObjstoreInit = {
56
56
  }
57
57
 
58
58
  export function initObjstore(deps: ObjstoreInitDeps): ObjstoreInit {
59
+ // Fail loud on a lopsided auth config. The put-begin gate in
60
+ // handlers.ts only fires when BOTH authGate and sendUnauthorized are
61
+ // present (it needs the reporter to emit the `unauthorized` frame),
62
+ // so wiring authGate WITHOUT sendUnauthorized silently fails OPEN —
63
+ // unauthenticated first-writes to unknown workspaces would be accepted
64
+ // despite the operator's intent to gate them. Reject at boot rather
65
+ // than regress access control silently.
66
+ if (deps.authGate && !deps.sendUnauthorized) {
67
+ throw new Error('initObjstore: authGate requires sendUnauthorized (the put-begin gate needs it to emit the unauthorized frame)')
68
+ }
59
69
  const handle = deps.handle
60
70
  const secret = deps.tokenSecret ?? newTokenSecret()
61
71
  const handlers = createObjstoreHandlers({
@@ -93,15 +103,14 @@ export function initObjstore(deps: ObjstoreInitDeps): ObjstoreInit {
93
103
  // on-disk state still has stranded files from a prior crash.
94
104
  const startupReap = enqueueSweep()
95
105
  // Jittered start of the periodic timer. Multi-replica deploys
96
- // (Neon + Vercel Blob) commonly boot N replicas in tight lock-
97
- // step (deploy rollout, cluster restart) and would otherwise
98
- // sync every replica's reaper at the same wall-clock tick,
99
- // hammering the DB + blob store with N×readdir+lock-acquire
100
- // bursts. A random first-interval delay deconcurrencies the
101
- // cluster without changing the long-term sweep cadence.
102
- // Jitter range is 0…1× reapIntervalMs (i.e., the next sweep
103
- // happens at [interval, 2×interval] after boot); subsequent
104
- // sweeps stay at exactly `reapIntervalMs` apart.
106
+ // (Neon + Vercel Blob) commonly boot N replicas in tight lock-step
107
+ // (deploy rollout, cluster restart), which would otherwise sync every
108
+ // replica's reaper to the same wall-clock tick and hammer the DB +
109
+ // blob store with N×readdir+lock-acquire bursts. A random
110
+ // first-interval delay spreads the cluster out without changing the
111
+ // long-term cadence. Jitter is 0…1× reapIntervalMs (first sweep at
112
+ // [interval, 2×interval] after boot); subsequent sweeps stay exactly
113
+ // `reapIntervalMs` apart.
105
114
  let reapTimer: ReturnType<typeof setInterval> | null = null
106
115
  const jitterMs = Math.floor(Math.random() * deps.reapIntervalMs)
107
116
  const firstTimer = setTimeout(() => {
@@ -27,6 +27,7 @@
27
27
  // predicate can't match a row a concurrent upload just refreshed.
28
28
 
29
29
  import { type Handle, STAGING_TTL_MS_DEFAULT, isValidContentHash, isValidStagingId, isValidTag } from './store.ts'
30
+ import { debugId, debugTag } from '../util.ts'
30
31
 
31
32
  type StagingRow = {
32
33
  workspace_tag: string
@@ -67,12 +68,19 @@ type StagingRow = {
67
68
  // live-set re-read, not via mutual exclusion.
68
69
  async function gcBlobIfUnreferenced(
69
70
  handle: Handle, tag: string, hash: string, modifiedMs: number, now: number, grace: number,
70
- ): Promise<void> {
71
- if (!isValidContentHash(hash)) return
72
- if (now - modifiedMs < grace) return
71
+ ): Promise<boolean> {
72
+ if (!isValidContentHash(hash)) return false
73
+ if (now - modifiedMs < grace) return false
73
74
  const refs = await liveHashSet(handle, tag)
74
- if (refs.has(hash)) return
75
+ if (refs.has(hash)) return false
75
76
  await handle.blob.unlinkLive(tag, hash)
77
+ // Log EVERY live-blob deletion unconditionally (not behind `debug`):
78
+ // this is the only record that the GC removed bytes. A handful per
79
+ // sweep is normal (superseded versions aging out); a burst across many
80
+ // workspaces is the smoking gun for the "all uploaded data went
81
+ // missing" failure — pair it with the sweep summary in `reapOrphans`.
82
+ console.warn(`objstore-reaper: GC live blob ${debugTag(tag)}/${debugId(hash)} (unreferenced, age ${Math.round((now - modifiedMs) / 1000)}s ≥ grace ${Math.round(grace / 1000)}s)`)
83
+ return true
76
84
  }
77
85
 
78
86
  // The set of content hashes referenced by the workspace's live rows.
@@ -86,18 +94,22 @@ async function liveHashSet(handle: Handle, tag: string): Promise<Set<string>> {
86
94
  // snapshot we read up front can race a concurrent commit; the
87
95
  // reference re-read inside `gcBlobIfUnreferenced` (plus the grace
88
96
  // window) ensures we never unlink a blob a live row names.
89
- async function reapUnreferencedForTag(handle: Handle, tag: string, now: number, grace: number): Promise<void> {
90
- if (!isValidTag(tag)) return
97
+ // Returns the number of live blobs GC'd for this tag, so `reapOrphans`
98
+ // can surface a sweep-wide total (the headline signal for mass loss).
99
+ async function reapUnreferencedForTag(handle: Handle, tag: string, now: number, grace: number): Promise<number> {
100
+ if (!isValidTag(tag)) return 0
91
101
  const blobs = await handle.blob.listLiveBlobs(tag)
92
- if (blobs.length === 0) return
102
+ if (blobs.length === 0) return 0
93
103
  const referenced = await liveHashSet(handle, tag)
104
+ let gc = 0
94
105
  for (const { hash, modifiedMs } of blobs) {
95
106
  if (!isValidContentHash(hash)) continue
96
107
  // Referenced in our snapshot → skip the grace + re-read path
97
108
  // entirely; only unreferenced blobs need it.
98
109
  if (referenced.has(hash)) continue
99
- await gcBlobIfUnreferenced(handle, tag, hash, modifiedMs, now, grace)
110
+ if (await gcBlobIfUnreferenced(handle, tag, hash, modifiedMs, now, grace)) gc++
100
111
  }
112
+ return gc
101
113
  }
102
114
 
103
115
  // Drop staging rows older than the TTL and unlink their on-storage
@@ -172,9 +184,10 @@ export async function reapOrphans(handle: Handle, stagingTtlMs: number = STAGING
172
184
  const grace = stagingTtlMs
173
185
  // Pass 1: tags the live table knows about — GC unreferenced live
174
186
  // blobs (past the grace window) against the referenced-hash set.
187
+ let liveGc = 0
175
188
  const liveTagsRows = await handle.listLiveTags.all()
176
189
  const liveTags = liveTagsRows.map((r) => r.workspace_tag)
177
- for (const tag of liveTags) await reapUnreferencedForTag(handle, tag, now, grace)
190
+ for (const tag of liveTags) liveGc += await reapUnreferencedForTag(handle, tag, now, grace)
178
191
  // Whole-workspace deletes leave residue (dirs / blob-prefixes) the
179
192
  // live table no longer lists. Walk the backend's top-level workspace
180
193
  // listing to find them; for each straggler tag, GC its unreferenced
@@ -186,7 +199,14 @@ export async function reapOrphans(handle: Handle, stagingTtlMs: number = STAGING
186
199
  const liveSet = new Set(liveTags)
187
200
  for (const tag of topLevel) {
188
201
  if (liveSet.has(tag) || !isValidTag(tag)) continue
189
- await reapUnreferencedForTag(handle, tag, now, grace)
202
+ liveGc += await reapUnreferencedForTag(handle, tag, now, grace)
203
+ }
204
+ // Sweep-wide total. A nonzero count means the GC deleted live bytes
205
+ // this pass — logged unconditionally so "all data went missing"
206
+ // leaves an obvious server-side trail (a large count over a short
207
+ // window is the signature). Per-blob lines above carry which/why.
208
+ if (liveGc > 0) {
209
+ console.warn(`objstore-reaper: swept ${liveTags.length} live + ${topLevel.length} store tag(s); GC'd ${liveGc} live blob(s)`)
190
210
  }
191
211
  // Pass 2: stale staging rows + orphan staging blobs. The orphan
192
212
  // sweep does per-blob row lookups (no caller-side snapshot), so a