@preventive/triage 1.0.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/LICENSE +21 -0
  2. package/common/save-error-reason.ts +53 -0
  3. package/common/utf8.d.ts +13 -0
  4. package/common/utf8.js +57 -0
  5. package/out/brotli-fallback.js +3 -0
  6. package/out/client-sync.js +15 -0
  7. package/out/graph.js +4 -0
  8. package/out/icon-maskable.svg +5 -0
  9. package/out/icon.svg +5 -0
  10. package/out/index.html +78 -0
  11. package/out/manifest.webmanifest +30 -0
  12. package/out/prism.js +14 -0
  13. package/out/terminal.js +39 -0
  14. package/out/view.css +1 -0
  15. package/out/view.html +12 -0
  16. package/out/view.js +138 -0
  17. package/package.json +129 -0
  18. package/server/auth.ts +99 -0
  19. package/server/config.example.json +3 -0
  20. package/server/config.ts +196 -0
  21. package/server/db-neon.ts +374 -0
  22. package/server/db-revision-sql.ts +152 -0
  23. package/server/db-stmt.ts +53 -0
  24. package/server/db.ts +577 -0
  25. package/server/http.ts +142 -0
  26. package/server/hub.ts +98 -0
  27. package/server/index.ts +353 -0
  28. package/server/lifecycle.ts +177 -0
  29. package/server/neon-driver.ts +26 -0
  30. package/server/objstore/blob-fs.ts +164 -0
  31. package/server/objstore/blob-vercel.ts +508 -0
  32. package/server/objstore/blob.ts +169 -0
  33. package/server/objstore/fs.ts +67 -0
  34. package/server/objstore/handlers.ts +235 -0
  35. package/server/objstore/init.ts +118 -0
  36. package/server/objstore/reaper.ts +199 -0
  37. package/server/objstore/rest.ts +484 -0
  38. package/server/objstore/sign.ts +164 -0
  39. package/server/objstore/store-neon.ts +351 -0
  40. package/server/objstore/store.ts +799 -0
  41. package/server/objstore/tokens.ts +168 -0
  42. package/server/origin.ts +68 -0
  43. package/server/peer.ts +38 -0
  44. package/server/sign.ts +231 -0
  45. package/server/static.ts +374 -0
  46. package/server/sync-handlers.ts +311 -0
  47. package/server/util.ts +27 -0
  48. package/server/validation.ts +36 -0
  49. package/server/ws-server.ts +245 -0
@@ -0,0 +1,374 @@
1
+ // Static-file plane: serve the production UI bundle that `build.js
2
+ // build` writes to `out/`. The directory is enumerated once at boot,
3
+ // every regular file's bytes are slurped into the `files` map keyed
4
+ // by basename, and the handler answers GET/HEAD against that fixed
5
+ // whitelist. The only on-disk paths we ever open are
6
+ // `join(staticDir, dirent.name)` from `readdirSync` at boot — the
7
+ // request URL is only used as a Map key, never joined with the
8
+ // filesystem — so the handler has no path-traversal surface at all:
9
+ // `..`, percent-encoded slashes, nested subpaths and absolute-form
10
+ // URIs all just produce a key that isn't in the map and 404.
11
+ //
12
+ // Compression: every entry whose extension is in COMPRESSIBLE gets
13
+ // pre-computed brotli + gzip at boot. The handler picks brotli over
14
+ // gzip over identity based on the request's Accept-Encoding header;
15
+ // the response carries `Content-Encoding` and `Vary:
16
+ // Accept-Encoding` so intermediate caches bucket per encoding.
17
+ //
18
+ // Revalidation: every entry carries an ETag derived from a SHA-256
19
+ // of the identity bytes. Conditional GET (`If-None-Match`) returns
20
+ // 304 without the body. The ETag is shared across encodings — the
21
+ // `Vary` header tells caches to keep separate buckets per encoding,
22
+ // and a revalidation arrives under the same Accept-Encoding the
23
+ // cached entry was stored against, so the single ETag suffices.
24
+ //
25
+ // Preload extraction: at boot, HTML files have their
26
+ // `<link rel="preload">` / `<link rel="modulepreload">` tags lifted
27
+ // into a `Link` response header (RFC 8288) and dropped from the
28
+ // served body. The header arrives with the response headers — before
29
+ // HTML body parsing — so the browser can start the sub-resource
30
+ // fetches a round-trip earlier than the in-body tag would allow. The
31
+ // stripped body is what ETag + compression are computed against, so
32
+ // `out/index.html` on disk and the bytes we serve diverge by exactly
33
+ // the lifted tags.
34
+
35
+ import type { IncomingMessage as HttpRequest, ServerResponse } from 'node:http'
36
+ import { Buffer } from 'node:buffer'
37
+ import { readFileSync, readdirSync } from 'node:fs'
38
+ import { createHash } from 'node:crypto'
39
+ import { brotliCompressSync, gzipSync, constants as zlibConstants } from 'node:zlib'
40
+ import { extname, join } from 'node:path'
41
+
42
+ // MIME types for extensions the production UI bundle currently
43
+ // emits. Anything else falls through to `application/octet-stream`
44
+ // — a future image / font asset would still be served, just without
45
+ // a precise type. Extend if you add an extension; keep the set
46
+ // minimal so unknown extensions are visible at deploy time.
47
+ const CONTENT_TYPE: Record<string, string> = {
48
+ '.html': 'text/html; charset=utf-8',
49
+ '.css': 'text/css; charset=utf-8',
50
+ '.js': 'application/javascript; charset=utf-8',
51
+ '.svg': 'image/svg+xml',
52
+ '.webmanifest': 'application/manifest+json',
53
+ }
54
+ // Only types that actually shrink under brotli/gzip. The current
55
+ // build emits no binary assets — a future PNG / WOFF2 would gain
56
+ // nothing from re-compressing (they're already entropy-encoded) and
57
+ // would waste boot CPU + memory. Kept in lock-step with
58
+ // CONTENT_TYPE: every entry here MUST be a text-shaped MIME above.
59
+ const COMPRESSIBLE = new Set(['.html', '.css', '.js', '.svg', '.webmanifest'])
60
+
61
+ // Defensive headers attached to every static response. `nosniff` is
62
+ // the load-bearing one: without it a browser may MIME-sniff the
63
+ // `application/octet-stream` fallback (buildEntry) or a directly-
64
+ // navigated `image/svg+xml` icon into something script-executable.
65
+ // `no-referrer` keeps URLs of this (potentially sensitive) viewer out
66
+ // of outbound Referer headers, and `same-origin` CORP stops other
67
+ // origins embedding these bytes. CSP is deliberately NOT here: the
68
+ // HTML documents carry their own per-page `<meta http-equiv>` policy
69
+ // (and the three pages differ), so a blanket header CSP would just AND
70
+ // against the meta one and risk breaking a page — non-document assets
71
+ // get their own flat CSP via `StaticEntry.csp` instead.
72
+ const SECURITY_HEADERS = {
73
+ 'x-content-type-options': 'nosniff',
74
+ 'referrer-policy': 'no-referrer',
75
+ 'cross-origin-resource-policy': 'same-origin',
76
+ } as const
77
+
78
+ type StaticEntry = {
79
+ type: string
80
+ etag: string
81
+ identity: Buffer
82
+ gzip: Buffer | null
83
+ br: Buffer | null
84
+ link: string | null
85
+ csp: string | null
86
+ }
87
+
88
+ export type StaticHandler = (req: HttpRequest, res: ServerResponse) => boolean
89
+
90
+ // Build the static-file map and return a handler. The returned
91
+ // function answers true when it consumed the request (200 / 304),
92
+ // false when the request wasn't a static-file GET/HEAD — the
93
+ // caller falls through to its next route. Missing `staticDir`
94
+ // (pre-build case) logs a warning and returns a handler that always
95
+ // answers false; the API/WS planes are unaffected.
96
+ export function loadStatic(staticDir: string): StaticHandler {
97
+ const files = readStaticFiles(staticDir)
98
+ return function handleStatic(req: HttpRequest, res: ServerResponse): boolean {
99
+ if (req.method !== 'GET' && req.method !== 'HEAD') return false
100
+ if (typeof req.url !== 'string') return false
101
+ const path = req.url.split('?', 1)[0]!
102
+ // `/` aliases to `index.html`. Every other path is the leading-
103
+ // slash-stripped URL used as a literal Map key — no filesystem
104
+ // join, no path normalisation. Absolute-form URIs (`http://…`)
105
+ // don't start with `/` and fall through to null → 404.
106
+ const name = path === '/' ? 'index.html' : path.startsWith('/') ? path.slice(1) : null
107
+ const entry = name == null ? undefined : files.get(name)
108
+ if (!entry) return false
109
+ const ifNoneMatch = req.headers['if-none-match']
110
+ const { body, encoding } = pickEncoding(req.headers['accept-encoding'], entry)
111
+ if (typeof ifNoneMatch === 'string' && matchesETag(ifNoneMatch, entry.etag)) {
112
+ // 304 echoes the variant headers that would have accompanied
113
+ // the 200 for the same request (RFC 9111 §4.3.4): the cached
114
+ // entity already has a `Content-Encoding` matching this
115
+ // negotiation, so re-asserting it lets a strict shared cache
116
+ // bind the freshening to the right variant under Vary.
117
+ const headers: Record<string, string | number> = { ...SECURITY_HEADERS, 'etag': entry.etag, 'cache-control': 'no-cache', 'vary': 'accept-encoding' }
118
+ if (encoding != null) headers['content-encoding'] = encoding
119
+ if (entry.link != null) headers['link'] = entry.link
120
+ if (entry.csp != null) headers['content-security-policy'] = entry.csp
121
+ res.writeHead(304, headers)
122
+ res.end()
123
+ return true
124
+ }
125
+ const headers: Record<string, string | number> = {
126
+ ...SECURITY_HEADERS,
127
+ 'content-type': entry.type,
128
+ 'content-length': body.byteLength,
129
+ 'etag': entry.etag,
130
+ 'cache-control': 'no-cache',
131
+ 'vary': 'accept-encoding',
132
+ }
133
+ if (encoding != null) headers['content-encoding'] = encoding
134
+ if (entry.link != null) headers['link'] = entry.link
135
+ if (entry.csp != null) headers['content-security-policy'] = entry.csp
136
+ res.writeHead(200, headers)
137
+ // HEAD shares the GET response headers but never carries a body.
138
+ // Node won't drop it for us — caller must skip the write.
139
+ if (req.method === 'HEAD') res.end()
140
+ else res.end(body)
141
+ return true
142
+ }
143
+ }
144
+
145
+ function readStaticFiles(staticDir: string): ReadonlyMap<string, StaticEntry> {
146
+ const files = new Map<string, StaticEntry>()
147
+ let entries
148
+ try { entries = readdirSync(staticDir, { withFileTypes: true }) } catch (err) {
149
+ // Missing `out/` is the pre-build case — operator hasn't run
150
+ // `node --run build` yet. Log once and skip; the API/WS planes
151
+ // are still functional, just no UI is served.
152
+ if ((err as NodeJS.ErrnoException)?.code === 'ENOENT') {
153
+ console.warn(`static: ${staticDir} missing — UI not served (run \`node --run build\`)`)
154
+ return files
155
+ }
156
+ throw err
157
+ }
158
+ for (const entry of entries) {
159
+ // `isFile()` filters subdirectories, symlinks, FIFOs etc. The
160
+ // build emits a flat tree; dropping anything else keeps the
161
+ // surface minimal even if a stray non-file sneaks in.
162
+ if (!entry.isFile()) continue
163
+ files.set(entry.name, buildEntry(staticDir, entry.name))
164
+ }
165
+ return files
166
+ }
167
+
168
+ function buildEntry(staticDir: string, name: string): StaticEntry {
169
+ const ext = extname(name)
170
+ const raw = readFileSync(join(staticDir, name))
171
+ const type = CONTENT_TYPE[ext] ?? 'application/octet-stream'
172
+ // HTML: lift `<link rel="(module)preload" …>` into a Link header
173
+ // and drop the tags from the served body so the bytes ship without
174
+ // the now-redundant in-body hint. ETag + compression run against
175
+ // the stripped body — the on-disk file and the served body diverge
176
+ // by exactly the lifted tags.
177
+ let identity = raw
178
+ let link: string | null = null
179
+ if (ext === '.html') {
180
+ const extracted = extractPreloadLinks(raw.toString('utf8'))
181
+ link = extracted.link
182
+ if (link != null) identity = Buffer.from(extracted.html, 'utf8')
183
+ }
184
+ // 16 bytes of SHA-256 (32 hex chars) — plenty for revalidation
185
+ // collision-resistance against the ~tens of files we load.
186
+ const etag = `"${createHash('sha256').update(identity).digest('hex').slice(0, 32)}"`
187
+ const isCompressible = COMPRESSIBLE.has(ext)
188
+ // Brotli quality 11 is the maximum — slow at compress time but
189
+ // we pay it once at boot and ship the smaller bytes on every
190
+ // request thereafter. text mode tells the encoder we're working
191
+ // on UTF-8 text (every COMPRESSIBLE entry is text-shaped).
192
+ const br = isCompressible ? brotliCompressSync(identity, {
193
+ params: {
194
+ [zlibConstants.BROTLI_PARAM_QUALITY]: 11,
195
+ [zlibConstants.BROTLI_PARAM_MODE]: zlibConstants.BROTLI_MODE_TEXT,
196
+ },
197
+ }) : null
198
+ const gzip = isCompressible ? gzipSync(identity, { level: 9 }) : null
199
+ // HTML documents self-protect with a per-page `<meta http-equiv>`
200
+ // CSP, so we don't set a header CSP for them (it'd AND against the
201
+ // meta policy). Every other asset carries no in-body policy — a
202
+ // directly-navigated SVG icon or the octet-stream fallback would run
203
+ // with none — so pin them to `default-src 'none'`: it neutralises
204
+ // inline script and external fetches if such a resource is ever
205
+ // loaded as a top-level document, and is inert when the asset is
206
+ // fetched as a sub-resource (CSP only governs the document context).
207
+ const csp = ext === '.html' ? null : "default-src 'none'"
208
+ return { type, etag, identity, gzip, br, link, csp }
209
+ }
210
+
211
+ // Walk every `<link …>` in `html` that isn't inside an HTML comment
212
+ // or inside `<script>` / `<noscript>` raw-text content. For
213
+ // `rel="preload"` / `rel="modulepreload"`, accumulate an RFC 8288
214
+ // Link-header form and strip the tag (plus its trailing whitespace)
215
+ // from the body. Every other rel value, and every `<link>` inside a
216
+ // skipped region, is preserved verbatim. Returns the (possibly
217
+ // modified) HTML and the comma-joined Link header value, or
218
+ // `link: null` if no preload tags were found.
219
+ function extractPreloadLinks(html: string): { html: string, link: string | null } {
220
+ const links: string[] = []
221
+ // Skip ranges = char offsets inside comments / script bodies /
222
+ // noscript bodies. The build's `<noscript>` fallback (and any
223
+ // commented-out preload hint, or a `<link>` mentioned as a JS
224
+ // string inside `<script>`) MUST NOT get lifted into the response
225
+ // headers — it'd start fetches the page author explicitly didn't
226
+ // want, and on a malformed build could leak attacker-influenced
227
+ // bytes into the header. Heuristic, not a real HTML parser: the
228
+ // build emits clean, predictable tags; this just keeps the lift
229
+ // from over-reaching on the kinds of surrounding content the
230
+ // codebase actually produces.
231
+ const skip = findSkipRanges(html)
232
+ // `[^>]*` is greedy but stops at `>`, so it spans multi-line tag
233
+ // attribute soup, including newlines inside the attr list. Trailing
234
+ // `\s*` is part of `match` in both branches — when we strip a
235
+ // preload tag the line break goes with it (body collapses cleanly);
236
+ // when we keep a non-preload tag the whitespace is restored via
237
+ // `return match` (HTML layout preserved).
238
+ const tagRe = /<link\b[^>]*>\s*/giu
239
+ const stripped = html.replace(tagRe, (match, offset: number) => {
240
+ if (inAnyRange(offset, skip)) return match
241
+ const attrs = parseLinkAttrs(match)
242
+ const rel = (attrs['rel'] ?? '').toLowerCase()
243
+ if (rel !== 'preload' && rel !== 'modulepreload') return match
244
+ const href = attrs['href']
245
+ if (!href) return match
246
+ // Defence in depth against a future template emitting an attr
247
+ // value that splits the header (`\r\n` injects new headers;
248
+ // `,`/`;`/`"` confuse the Link parser; `<`/`>` terminate the
249
+ // URI-Reference early). The build emits clean values today, so
250
+ // this only fires under a malformed source — and bailing out
251
+ // (return match) is the safe choice: the tag stays in the body
252
+ // as an in-HTML preload hint, just no header lift.
253
+ if (isUnsafeAttr(href) || isUnsafeAttr(attrs['as'] ?? '') || isUnsafeAttr(attrs['type'] ?? '') || isUnsafeAttr(attrs['crossorigin'] ?? '')) return match
254
+ // Quote every param value uniformly. `rel`/`as`/`crossorigin`
255
+ // values happen to be valid RFC 7230 tokens today (preload,
256
+ // style, anonymous, …) and would parse unquoted, but the
257
+ // asymmetry with `type="…"` reads as accidental and a future
258
+ // value containing whitespace would silently split the header.
259
+ let value = `<${normalizeHref(href)}>; rel="${rel}"`
260
+ if (rel === 'preload' && attrs['as']) value += `; as="${attrs['as']}"`
261
+ if (attrs['type']) value += `; type="${attrs['type']}"`
262
+ // `'crossorigin' in attrs` distinguishes "attribute present" from
263
+ // "attribute missing"; the boolean form (no `=`) and the valued
264
+ // form (`crossorigin="anonymous"`) emit differently per RFC 8288.
265
+ if ('crossorigin' in attrs) value += attrs['crossorigin'] ? `; crossorigin="${attrs['crossorigin']}"` : '; crossorigin'
266
+ links.push(value)
267
+ return ''
268
+ })
269
+ return { html: stripped, link: links.length === 0 ? null : links.join(', ') }
270
+ }
271
+
272
+ function isUnsafeAttr(s: string): boolean {
273
+ return /[\r\n",;<>]/u.test(s)
274
+ }
275
+
276
+ // Find character ranges in `html` whose contents are NOT real HTML
277
+ // content: comment bodies and script / noscript raw-text bodies. A
278
+ // `<link>` whose match offset falls inside any range is preserved
279
+ // verbatim (no header lift). Heuristic: a real HTML parser is
280
+ // overkill here — the build emits clean, well-formed markup; this is
281
+ // defence against the build (or a hand edit) accidentally mentioning
282
+ // `<link rel="preload">` somewhere it isn't meant to fire.
283
+ function findSkipRanges(html: string): Array<[number, number]> {
284
+ const ranges: Array<[number, number]> = []
285
+ const patterns = [
286
+ /<!--[\s\S]*?-->/gu,
287
+ /<script\b[^>]*>[\s\S]*?<\/script\s*>/giu,
288
+ /<noscript\b[^>]*>[\s\S]*?<\/noscript\s*>/giu,
289
+ ]
290
+ for (const re of patterns) {
291
+ for (const m of html.matchAll(re)) ranges.push([m.index!, m.index! + m[0].length])
292
+ }
293
+ return ranges
294
+ }
295
+
296
+ function inAnyRange(offset: number, ranges: ReadonlyArray<readonly [number, number]>): boolean {
297
+ for (const [start, end] of ranges) {
298
+ if (offset >= start && offset < end) return true
299
+ }
300
+ return false
301
+ }
302
+
303
+ // Minimal HTML attribute parser scoped to a single `<link …>` tag.
304
+ // Handles double-quoted, single-quoted, and bare attribute values,
305
+ // plus boolean attributes (e.g. `crossorigin`). Lowercases names so
306
+ // downstream comparisons can use canonical keys. Not a general HTML
307
+ // parser — the build emits well-formed tags and that's the only
308
+ // surface we run on.
309
+ function parseLinkAttrs(tag: string): Record<string, string> {
310
+ const attrs: Record<string, string> = {}
311
+ const inner = tag.replace(/^<link\b/iu, '').replace(/\/?>\s*$/u, '')
312
+ const re = /([a-z_][\w-]*)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|(\S+)))?/giu
313
+ for (const m of inner.matchAll(re)) {
314
+ attrs[m[1]!.toLowerCase()] = m[2] ?? m[3] ?? m[4] ?? ''
315
+ }
316
+ return attrs
317
+ }
318
+
319
+ // Normalise `href` to a root-relative URL. The build emits
320
+ // `./foo.css`-style same-dir hrefs; Link header URIs resolve against
321
+ // the response URL so `./foo.css` would also work, but root-relative
322
+ // reads correctly regardless of which page the header rides on.
323
+ function normalizeHref(href: string): string {
324
+ const stripped = href.startsWith('./') ? href.slice(2) : href
325
+ return stripped.startsWith('/') ? stripped : `/${stripped}`
326
+ }
327
+
328
+ // Prefer brotli over gzip over identity. Both compressed variants
329
+ // are pre-computed; runtime cost is one Accept-Encoding parse +
330
+ // a Buffer reference pick. Node surfaces repeated `Accept-Encoding`
331
+ // headers as `string[]` (it's not on the list-fold whitelist) — join
332
+ // before parsing so the array case doesn't silently fall through to
333
+ // identity and miss compression.
334
+ function pickEncoding(accept: string | string[] | undefined, entry: StaticEntry): { body: Buffer, encoding: string | null } {
335
+ const header = Array.isArray(accept) ? accept.join(',') : accept
336
+ if (typeof header === 'string') {
337
+ if (entry.br && acceptsEncoding(header, 'br')) return { body: entry.br, encoding: 'br' }
338
+ if (entry.gzip && acceptsEncoding(header, 'gzip')) return { body: entry.gzip, encoding: 'gzip' }
339
+ }
340
+ return { body: entry.identity, encoding: null }
341
+ }
342
+
343
+ // `Accept-Encoding: <name>[;q=N], …`. Returns true if `encoding`
344
+ // (or `*`) is listed with a positive q-value. We don't sort by
345
+ // q-value — Accept-Encoding q-values are effectively never used in
346
+ // the wild to invert the brotli-then-gzip preference, and a wrong
347
+ // answer here just trades a few percent of bytes, never correctness.
348
+ function acceptsEncoding(header: string, encoding: string): boolean {
349
+ for (const part of header.split(',')) {
350
+ const tokens = part.trim().split(';').map((s) => s.trim())
351
+ const name = tokens[0]!.toLowerCase()
352
+ if (name !== encoding && name !== '*') continue
353
+ let q = 1
354
+ for (const t of tokens.slice(1)) {
355
+ const m = /^q=(\d*\.?\d+)$/iu.exec(t)
356
+ if (m) q = Number(m[1])
357
+ }
358
+ if (q > 0) return true
359
+ }
360
+ return false
361
+ }
362
+
363
+ // `If-None-Match: "…", "…", *`. Strong or weak (`W/`-prefixed)
364
+ // validators both match — for a static asset the distinction is
365
+ // academic (we never serve a semantically-equivalent variant), and
366
+ // matching both keeps the revalidation path working under proxies
367
+ // that downgrade the validator.
368
+ function matchesETag(header: string, etag: string): boolean {
369
+ for (const part of header.split(',')) {
370
+ const trimmed = part.trim()
371
+ if (trimmed === '*' || trimmed === etag || trimmed === `W/${etag}`) return true
372
+ }
373
+ return false
374
+ }