@preventive/triage 1.0.0-alpha.2 → 1.0.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. package/api/reap.ts +17 -0
  2. package/cli.js +6 -0
  3. package/client/finding-link.js +305 -0
  4. package/client/linked-findings.d.ts +1 -0
  5. package/client/linked-findings.js +111 -0
  6. package/common/bundle-metadata.d.ts +10 -0
  7. package/common/bundle-metadata.js +177 -0
  8. package/common/bundle-reasons.d.ts +2 -0
  9. package/common/bundle-reasons.js +21 -0
  10. package/common/bundle-sources.d.ts +3 -0
  11. package/common/bundle-sources.js +284 -0
  12. package/common/bundle-stats.js +41 -0
  13. package/common/bundle-tabs.js +1 -0
  14. package/common/code-language.js +36 -0
  15. package/common/default-scan-models.ts +30 -0
  16. package/common/finding-id.js +47 -0
  17. package/common/github-pr.ts +56 -0
  18. package/common/managed/comments.ts +34 -0
  19. package/common/managed/permissions.ts +35 -0
  20. package/common/managed/report-content.ts +42 -0
  21. package/common/managed/report-filter.ts +108 -0
  22. package/common/managed/roles.ts +28 -0
  23. package/common/managed/routes.d.ts +2 -0
  24. package/common/managed/routes.js +121 -0
  25. package/common/managed/scan-models.ts +6 -0
  26. package/common/managed/triage.ts +83 -0
  27. package/common/save-error-reason.ts +20 -7
  28. package/common/scan-server.ts +13 -0
  29. package/common/server-info.ts +33 -0
  30. package/common/utf8.d.ts +3 -0
  31. package/common/utf8.js +45 -0
  32. package/out/brotli-fallback.js +3 -3
  33. package/out/client-managed-import.js +81 -0
  34. package/out/client-managed.js +110 -0
  35. package/out/client-sync.js +16 -13
  36. package/out/graph.js +30 -4
  37. package/out/index.html +55 -8
  38. package/out/prism.js +2 -2
  39. package/out/stasis.svg +45 -0
  40. package/out/terminal.js +273 -39
  41. package/out/view.css +1 -1
  42. package/out/view.js +198 -62
  43. package/package.json +179 -55
  44. package/report/index.js +254 -0
  45. package/report/src/finding-id.js +80 -0
  46. package/report/src/finding.js +312 -0
  47. package/report/src/labels.js +33 -0
  48. package/report/src/md-structure.js +471 -0
  49. package/report/src/md-text.js +167 -0
  50. package/report/src/meta.js +76 -0
  51. package/report/src/parse-codex.js +147 -0
  52. package/report/src/parse-deepsec.js +197 -0
  53. package/report/src/parse-deepview-fields.js +375 -0
  54. package/report/src/parse-deepview-md.js +185 -0
  55. package/report/src/parse-md-id.js +137 -0
  56. package/report/src/parse-md.js +322 -0
  57. package/report/src/parse-piolium-id.js +79 -0
  58. package/report/src/parse-piolium-rows.js +131 -0
  59. package/report/src/parse-piolium-tokens.js +175 -0
  60. package/report/src/parse-piolium.js +400 -0
  61. package/report/src/security.js +63 -0
  62. package/report/src/utf8.js +21 -0
  63. package/report/src/write-md-finding.js +273 -0
  64. package/report/src/write-md.js +291 -0
  65. package/server-common/database-config.ts +16 -0
  66. package/server-common/initialize.ts +18 -0
  67. package/server-common/npm-advisories.ts +101 -0
  68. package/{server → server-common}/origin.ts +5 -5
  69. package/server-common/reap.ts +48 -0
  70. package/server-common/scan-config.ts +19 -0
  71. package/server-common/standalone.ts +29 -0
  72. package/server-common/storage-log.ts +34 -0
  73. package/server-common/vercel-blob.ts +110 -0
  74. package/server-e2e/app.ts +485 -0
  75. package/{server → server-e2e}/auth.ts +5 -1
  76. package/{server → server-e2e}/bus-receiver.ts +9 -8
  77. package/{server → server-e2e}/cli.js +9 -4
  78. package/{server → server-e2e}/config.ts +54 -39
  79. package/{server → server-e2e}/db-neon.ts +2 -2
  80. package/{server → server-e2e}/db-revision-sql.ts +7 -10
  81. package/{server → server-e2e}/db-stmt.ts +2 -2
  82. package/{server → server-e2e}/db.ts +96 -135
  83. package/{server → server-e2e}/http.ts +110 -12
  84. package/{server → server-e2e}/hub.ts +44 -14
  85. package/server-e2e/index.ts +17 -0
  86. package/server-e2e/lifecycle.ts +95 -0
  87. package/{server → server-e2e}/neon-driver.ts +2 -2
  88. package/{server → server-e2e}/npm-proxy.ts +11 -144
  89. package/{server → server-e2e}/objstore/blob-fs.ts +6 -8
  90. package/{server → server-e2e}/objstore/blob-vercel.ts +49 -141
  91. package/{server → server-e2e}/objstore/blob.ts +24 -9
  92. package/server-e2e/objstore/fetch-mint-guard.ts +74 -0
  93. package/{server → server-e2e}/objstore/handlers.ts +19 -20
  94. package/{server → server-e2e}/objstore/init.ts +53 -27
  95. package/{server → server-e2e}/objstore/reaper.ts +31 -11
  96. package/server-e2e/objstore/rest-deny.ts +28 -0
  97. package/server-e2e/objstore/rest-mint.ts +224 -0
  98. package/{server → server-e2e}/objstore/rest.ts +110 -93
  99. package/{server → server-e2e}/objstore/sign.ts +105 -0
  100. package/{server → server-e2e}/objstore/store-neon.ts +10 -14
  101. package/{server → server-e2e}/objstore/store.ts +98 -118
  102. package/{server → server-e2e}/objstore/tokens.ts +9 -12
  103. package/{server → server-e2e}/peer.ts +7 -9
  104. package/{server → server-e2e}/pubsub.ts +29 -36
  105. package/{server → server-e2e}/sign.ts +12 -14
  106. package/{server → server-e2e}/sse-server.ts +105 -73
  107. package/{server → server-e2e}/sse-session.ts +30 -16
  108. package/{server → server-e2e}/static.ts +36 -29
  109. package/server-e2e/sync-handlers.ts +408 -0
  110. package/{server → server-e2e}/util.ts +9 -0
  111. package/{server → server-e2e}/ws-server.ts +29 -23
  112. package/server-managed/activity.ts +231 -0
  113. package/server-managed/avatar-store.ts +51 -0
  114. package/server-managed/blob-store.ts +66 -0
  115. package/server-managed/blob-vercel.ts +125 -0
  116. package/server-managed/brotli.ts +10 -0
  117. package/server-managed/bundle-cache.ts +185 -0
  118. package/server-managed/bundle-catalog.ts +29 -0
  119. package/server-managed/bundle-store.ts +28 -0
  120. package/server-managed/bundle-summary-cache.ts +97 -0
  121. package/server-managed/bundle.ts +39 -0
  122. package/server-managed/cache-storage.ts +40 -0
  123. package/server-managed/cli.js +13 -0
  124. package/server-managed/combined.ts +46 -0
  125. package/server-managed/comments.ts +151 -0
  126. package/server-managed/config.ts +144 -0
  127. package/server-managed/content-access.ts +15 -0
  128. package/server-managed/crypto.ts +25 -0
  129. package/server-managed/db-methods.ts +1336 -0
  130. package/server-managed/db-neon.ts +171 -0
  131. package/server-managed/db-schema.ts +203 -0
  132. package/server-managed/db-table-names.ts +22 -0
  133. package/server-managed/db.ts +109 -0
  134. package/server-managed/github-app.ts +332 -0
  135. package/server-managed/github-metadata.ts +65 -0
  136. package/server-managed/github-oauth.ts +215 -0
  137. package/server-managed/github-pulls.ts +115 -0
  138. package/server-managed/http-response.ts +18 -0
  139. package/server-managed/http.ts +2043 -0
  140. package/server-managed/import-triage.ts +48 -0
  141. package/server-managed/index.ts +135 -0
  142. package/server-managed/public-workspace.ts +150 -0
  143. package/server-managed/repo-path.ts +21 -0
  144. package/server-managed/report-migration.ts +35 -0
  145. package/server-managed/report-query.ts +4 -0
  146. package/server-managed/report-response.ts +16 -0
  147. package/server-managed/report-sources.ts +154 -0
  148. package/server-managed/repository-discovery.ts +82 -0
  149. package/server-managed/repository-policy.ts +27 -0
  150. package/server-managed/session.ts +78 -0
  151. package/server-managed/slugs.ts +38 -0
  152. package/server-managed/sql-postgres.ts +30 -0
  153. package/server-managed/sql.ts +61 -0
  154. package/server-managed/static.ts +28 -0
  155. package/server-managed/storage.ts +34 -0
  156. package/server-managed/team-catalog.ts +7 -0
  157. package/server-managed/team-feed.ts +128 -0
  158. package/server-managed/team-reports.ts +156 -0
  159. package/server-managed/triage-response.ts +16 -0
  160. package/server-managed/uploads.ts +47 -0
  161. package/server-managed/workspace-shares.ts +150 -0
  162. package/server.ts +50 -0
  163. package/server/index.ts +0 -481
  164. package/server/lifecycle.ts +0 -204
  165. package/server/sync-handlers.ts +0 -327
  166. /package/{server → server-e2e}/config.example.json +0 -0
  167. /package/{server → server-e2e}/objstore/fs.ts +0 -0
  168. /package/{server → server-e2e}/validation.ts +0 -0
package/api/reap.ts ADDED
@@ -0,0 +1,17 @@
1
+ // E2e-only Vercel deployment: no persistent relay is running in this function.
2
+ // The managed deployment routes /api/reap through its top-level app instead.
3
+ import { createReapHandler } from '../server-common/reap.ts'
4
+ import { databaseUrls } from '../server-common/database-config.ts'
5
+ import { reapOrphans } from '../server-e2e/objstore/reaper.ts'
6
+ import { openNeonObjstore } from '../server-e2e/objstore/store-neon.ts'
7
+ import { openVercelBlobBackend } from '../server-e2e/objstore/blob-vercel.ts'
8
+
9
+ const handler = createReapHandler({ e2e: async () => {
10
+ const url = databaseUrls().e2e
11
+ const token = process.env['BLOB_READ_WRITE_TOKEN']
12
+ if (!url || !token) throw new Error('E2e cron requires DATABASE_URL or E2E_DATABASE_URL, plus BLOB_READ_WRITE_TOKEN')
13
+ const blob = await openVercelBlobBackend({ token })
14
+ // Stateless Neon HTTP handle: nothing to close, no relay listener or timer.
15
+ await reapOrphans(await openNeonObjstore(url, blob))
16
+ } })
17
+ export default handler
package/cli.js ADDED
@@ -0,0 +1,6 @@
1
+ #!/usr/bin/env node
2
+ import './strip-types-loader.js'
3
+
4
+ // Load TypeScript only after the stripping hook is registered.
5
+ const { start } = await import('./server.ts')
6
+ await start()
@@ -0,0 +1,305 @@
1
+ // Deep links to a single finding — pure logic side. Encodes the
2
+ // finding's id plus a "where does it live" hint into the URL hash
3
+ // (`#finding=<id>&v=<hints>`) and parses the same back on the receiving
4
+ // side, so "look at THIS issue" is a link you can paste into a chat
5
+ // instead of a screenshot plus directions.
6
+ //
7
+ // Sibling of `workspace-share-link.js`, and deliberately unlike it in
8
+ // one respect: the id here is NOT a secret. A share link carries the
9
+ // workspace's private key and is therefore encrypted; a finding id is
10
+ // meaningless without the report the recipient must already have, so it
11
+ // rides in the clear and the link stays legible in a chat client.
12
+ //
13
+ // The two LOCATION hints are a different matter. A report filename
14
+ // ("acme-corp-pentest-2024.json") and a workspace id are the sender's
15
+ // own metadata, and a URL is the most-forwarded, most-logged, most-
16
+ // screenshotted string there is — so they ship as 3-byte digests
17
+ // (4 base64url chars) rather than plaintext. That is a fingerprint, not
18
+ // a sealed box: 24 bits only narrows the field, and anyone holding a
19
+ // candidate name can confirm it by hashing. What it does buy is that
20
+ // the name itself never appears in the URL, and a recipient who does
21
+ // NOT hold the report learns nothing from the link but its length.
22
+ //
23
+ // Cheap on the receiving side too: matching a hint means hashing the
24
+ // names already on disk, not reading a single file (see
25
+ // `finding-locate.js`).
26
+ //
27
+ // The hint is a HINT, not an address. The receiver tries the named report
28
+ // or workspace, then a scan of everything stored locally (see
29
+ // `ui/view/finding-link-nav.js`) — so a link built in a workspace still
30
+ // lands for a recipient who only has the single report, and a stale
31
+ // hint costs a scan rather than the finding.
32
+ //
33
+ // Param names live in the same `&`-joined fragment namespace as
34
+ // `share=`, and the parser has the same shape as `extractShareEncoded`
35
+ // — a fragment can therefore only ever be one kind of link, and adding
36
+ // a fourth param later doesn't disturb either.
37
+
38
+ import { encodeUtf8 } from '../common/utf8.js'
39
+ import { managedRoutePath, parseManagedRoute } from '../common/managed/routes.js'
40
+ import { MAX_FINDING_ID_LENGTH, isLinkableFindingId } from '../common/finding-id.js'
41
+ export { isLinkableFindingId } from '../common/finding-id.js'
42
+
43
+ // ── location hints ───────────────────────────────────────────────────
44
+
45
+ // 3 bytes → exactly 4 base64url characters, no padding. Sized to keep
46
+ // the fragment short while still being selective enough that a hint
47
+ // usually names one report out of a user's collection: with 16.7M
48
+ // values, a library of 100 reports has a ~0.03% chance of any collision
49
+ // at all, and a collision costs only a wrong first guess (the receiver
50
+ // re-checks that the finding is actually there, then scans).
51
+ const HINT_BYTES = 3
52
+ const HINT_CHARS = 4
53
+ const HINT_RE = /^[\w-]{4}$/u
54
+
55
+ // Both hints travel as ONE `v=` parameter: 6 bytes, 8 base64url
56
+ // characters, report half first. Workspace-only links use `w` followed
57
+ // by their 3-byte / 4-character hint. Purely a length decision — two named
58
+ // params spend 20 characters (`&report=aB3-&ws=x_9Z`) to carry 8
59
+ // characters of payload, and a link is something people paste into
60
+ // chat, tickets and commit messages.
61
+ //
62
+ // Concatenating the two 4-character tokens IS the base64 of the 6-byte
63
+ // buffer: 3 bytes is exactly one base64 group, so nothing carries
64
+ // across the halves and either framing produces identical text. That's
65
+ // what lets the packing be a string join and the unpacking a slice,
66
+ // with no encode / decode round trip.
67
+ //
68
+ // `AAAA` (3 zero bytes) marks a half that isn't known — a single-file
69
+ // view has no workspace, and a report whose hint hasn't been hashed yet
70
+ // omits its own. A genuine digest of `AAAA` therefore reads as absent,
71
+ // once per 16.7M values per half; the cost is that the receiver skips
72
+ // straight to the scan, which is where an unmatched hint lands anyway.
73
+ const COMBINED_RE = /^[\w-]{8}$/u
74
+ const NO_HINT = 'AAAA'
75
+
76
+ // Domain-separated per kind so the same string can't produce a hint
77
+ // that matches in the other namespace — a workspace whose id happened
78
+ // to equal a report's filename would otherwise cross-resolve. Same
79
+ // convention as `workspace-id.js`'s derivation domain.
80
+ const HINT_DOMAINS = {
81
+ report: 'deepview/finding-link/report/v1\n',
82
+ workspace: 'deepview/finding-link/workspace/v1\n',
83
+ }
84
+
85
+ // `${kind}\0${value}` → token. Memoised because the LINK BUILDER must
86
+ // be synchronous: the Link button writes to the clipboard inside the
87
+ // click handler, and Safari drops the clipboard grant if the write
88
+ // happens after an await. Ingest primes this cache for every report it
89
+ // loads (and `switchToWorkspace` for the workspace), so by the time a
90
+ // card is on screen its hint is already here. A cold entry is not a
91
+ // failure — the link is simply built without that hint, and the
92
+ // receiver falls back to the scan.
93
+ const hintCache = new Map()
94
+
95
+ function hintKey(kind, value) { return `${kind}\0${value}` }
96
+
97
+ // Whether a value is a well-formed hint token, i.e. something
98
+ // `computeLinkHint` could have produced.
99
+ export function isLinkHint(value) {
100
+ return typeof value === 'string' && HINT_RE.test(value)
101
+ }
102
+
103
+ // Derive (and memoise) the hint token for a report filename or a
104
+ // workspace id. Resolves to null — never throws — when the kind is
105
+ // unknown, the value is empty, or `crypto.subtle` is unavailable (some
106
+ // `file://` setups): a missing hint degrades the link, it doesn't break
107
+ // it, so the caller shouldn't have to guard.
108
+ export async function computeLinkHint(kind, value) {
109
+ const domain = HINT_DOMAINS[kind]
110
+ if (!domain || typeof value !== 'string' || !value) return null
111
+ const key = hintKey(kind, value)
112
+ const cached = hintCache.get(key)
113
+ if (cached !== undefined) return cached
114
+ let token
115
+ try {
116
+ const digest = await crypto.subtle.digest('SHA-256', encodeUtf8(domain + value))
117
+ token = new Uint8Array(digest).slice(0, HINT_BYTES)
118
+ .toBase64({ alphabet: 'base64url', omitPadding: true })
119
+ } catch {
120
+ // Not cached: an environment without crypto.subtle stays broken,
121
+ // but a one-off failure (a lone surrogate in a filename tripping
122
+ // encodeUtf8) shouldn't poison the entry forever.
123
+ return null
124
+ }
125
+ hintCache.set(key, token)
126
+ return token
127
+ }
128
+
129
+ // Synchronous companion — the memoised token, or null if it hasn't been
130
+ // computed yet. The link builder uses this; see `hintCache` above for
131
+ // why it can't just await.
132
+ export function knownLinkHint(kind, value) {
133
+ if (typeof value !== 'string' || !value) return null
134
+ return hintCache.get(hintKey(kind, value)) ?? null
135
+ }
136
+
137
+ // Pack both hints into the `v=` value, or null when neither is known
138
+ // (in which case the param is omitted entirely rather than shipping
139
+ // eight characters of "nothing"). Throws on a non-empty value that
140
+ // isn't a token, so a caller passing a raw filename finds out
141
+ // immediately instead of shipping a hint-less link.
142
+ function packLinkHints(report, workspace) {
143
+ const halves = []
144
+ for (const [what, hint] of [['report', report], ['workspace', workspace]]) {
145
+ if (hint === null || hint === undefined || hint === '') { halves.push(NO_HINT); continue }
146
+ if (!isLinkHint(hint)) {
147
+ throw new TypeError(`encodeFindingRef: ${what} must be a link hint token, not a name`)
148
+ }
149
+ halves.push(hint)
150
+ }
151
+ if (halves.every((h) => h === NO_HINT)) return null
152
+ // Workspace → finding needs only the workspace half. Report links
153
+ // retain their original six-byte / eight-character encoding.
154
+ if (halves[0] === NO_HINT) return `w${halves[1]}`
155
+ return halves.join('')
156
+ }
157
+
158
+ // Split a validated `v=` value back into its two halves, mapping the
159
+ // `AAAA` filler to null.
160
+ function unpackLinkHints(value) {
161
+ const halves = [value.slice(0, HINT_CHARS), value.slice(HINT_CHARS)]
162
+ .map((h) => (h === NO_HINT ? null : h))
163
+ return { report: halves[0], workspace: halves[1] }
164
+ }
165
+
166
+ // ── fragment codec ───────────────────────────────────────────────────
167
+
168
+ // Build the fragment body (no leading '#') for a finding reference.
169
+ // `report` / `workspace` are hint TOKENS (from `computeLinkHint` /
170
+ // `knownLinkHint`), not names; they leave as the single packed `v=`
171
+ // param: `w` + four characters for a workspace, eight for a report with its
172
+ // optional workspace. Omitted when neither is known.
173
+ //
174
+ // The id is percent-encoded: usually a uuid with nothing to escape, but
175
+ // the codex importer's finding-URL ids carry `/`, `:`, `?` and —
176
+ // decisively — `&` and `=`, which would otherwise split into phantom
177
+ // params on the way back. The packed hints need no encoding (base64url
178
+ // is already fragment-safe) and are emitted verbatim.
179
+ export function encodeFindingRef({ id, report, workspace } = {}) {
180
+ if (!isLinkableFindingId(id)) {
181
+ throw new TypeError('encodeFindingRef: a persistent finding id is required')
182
+ }
183
+ const parts = [`finding=${encodeURIComponent(id)}`]
184
+ const packed = packLinkHints(report, workspace)
185
+ if (packed) parts.push(`v=${packed}`)
186
+ return parts.join('&')
187
+ }
188
+
189
+ // Full shareable URL for the current page origin + pathname, mirroring
190
+ // `buildShareUrl`. `location.search` is dropped for the same reason it
191
+ // is there: the target page takes no query params, so dragging the
192
+ // sender's current `?foo=bar` into the recipient's URL would be a
193
+ // surprising leak.
194
+ export function buildFindingUrl(ref, pathname) {
195
+ const encoded = encodeFindingRef(ref)
196
+ if (typeof location === 'undefined') return `${pathname ?? ''}#${encoded}`
197
+ return `${location.origin}${pathname ?? location.pathname}#${encoded}`
198
+ }
199
+
200
+ // Extract `{ id, report, workspace }` from a hash string (or
201
+ // `location.hash` when called with no argument), unpacking `v=` back
202
+ // into the two hint tokens. Returns null when the fragment carries no
203
+ // usable `finding=` param — a share link, an in-page anchor, an empty
204
+ // hash, or a payload that fails validation.
205
+ //
206
+ // HASH ONLY, matching `extractShareEncoded`. The id isn't secret, so the
207
+ // Referer-leak argument that motivates it there doesn't apply; the
208
+ // reason is uniformity — one place in the app reads deep links, and a
209
+ // caller that reached for `?finding=` would quietly bypass it.
210
+ //
211
+ // A malformed `v=` drops both hints rather than the whole ref: the id is
212
+ // what identifies the finding, and the receiver's scan can still land
213
+ // it. That also covers a link from an older build, whose separate
214
+ // `report=` / `ws=` params simply go unread. An un-decodable ID, on the
215
+ // other hand, fails the whole parse — there is nothing left to look up.
216
+ export function extractFindingRef(hash) {
217
+ const raw = typeof hash === 'string' ? hash : (typeof location === 'undefined' ? '' : location.hash)
218
+ if (!raw) return null
219
+ const stripped = raw.replace(/^#/u, '')
220
+ if (!stripped) return null
221
+ const found = { id: null, report: null, workspace: null }
222
+ for (const part of stripped.split('&')) {
223
+ const eq = part.indexOf('=')
224
+ if (eq < 0) continue
225
+ const key = part.slice(0, eq)
226
+ const rawValue = part.slice(eq + 1)
227
+ if (!rawValue) continue
228
+ // The packed hints are one workspace token or a token pair — validated
229
+ // verbatim rather than decoded, since base64url has nothing
230
+ // `encodeURIComponent` would have touched. Anything else is a link
231
+ // from another era or a mangled paste; drop it and let the scan
232
+ // take over.
233
+ if (key === 'v') {
234
+ if (rawValue.length === 5 && rawValue[0] === 'w' && isLinkHint(rawValue.slice(1))) {
235
+ const workspace = rawValue.slice(1)
236
+ Object.assign(found, { report: null, workspace: workspace === NO_HINT ? null : workspace })
237
+ continue
238
+ }
239
+ if (!COMBINED_RE.test(rawValue)) continue
240
+ Object.assign(found, unpackLinkHints(rawValue))
241
+ continue
242
+ }
243
+ if (key !== 'finding') continue
244
+ // Cheap length bound BEFORE decoding: percent-encoding expands at
245
+ // most 3:1, so nothing under this cap can decode to something over
246
+ // MAX_FINDING_ID_LENGTH, and a megabyte fragment never reaches
247
+ // decodeURIComponent. `isLinkableFindingId` re-checks the decoded
248
+ // length, so this is a guard, not the rule.
249
+ if (rawValue.length > MAX_FINDING_ID_LENGTH * 3) continue
250
+ let value
251
+ // decodeURIComponent throws URIError on a truncated / invalid escape
252
+ // (`%`, `%zz`) — common when a chat client mangles a pasted link.
253
+ try { value = decodeURIComponent(rawValue) } catch { continue }
254
+ if (isLinkableFindingId(value)) found.id = value
255
+ }
256
+ if (!found.id) return null
257
+ return found
258
+ }
259
+
260
+ // Recognise a finding deep link pasted into free text — the reverse of
261
+ // `buildFindingUrl`, used to linkify comments (see `parseCommentRefs` in
262
+ // `ui/view/format.js`). Returns `{ id, fragment, path }` with the E2E fragment
263
+ // re-emitted canonically and a managed destination URL, or null
264
+ // for anything that isn't one of OUR links.
265
+ //
266
+ // "Ours" means the CURRENT host, scheme included. A finding id resolves
267
+ // only against the reader's available reports, so a link to some other
268
+ // deployment couldn't be followed usefully even if it were offered — and
269
+ // a same-name look-alike on another host is exactly what a linkifier
270
+ // must not present as the real thing. Credentials are refused for the
271
+ // same reason (`https://user@triage.space/…` reads as ours but isn't
272
+ // something the app ever emits).
273
+ //
274
+ // Managed team/report paths are preserved to identify the intended copy.
275
+ // E2E entry links resolve from `/` in managed mode, without inheriting
276
+ // the current report. E2E consumers use just the fragment to stay on their
277
+ // current deployment, including `/index.html` and subpaths.
278
+ //
279
+ // The anti-mutation guard is the one from `githubRefToken`: `new URL`
280
+ // silently rewrites its input (resolving `..`, lower-casing, punycoding
281
+ // IDN homographs, dropping a default port), any of which can let a
282
+ // look-alike round up into a passing link. A candidate that isn't
283
+ // already canonical is refused rather than linkified. Every link this
284
+ // app emits round-trips unchanged, since `buildFindingUrl` builds from
285
+ // `location` itself.
286
+ export function parseFindingUrl(candidate, { managed = false } = {}) {
287
+ if (typeof location === 'undefined') return null
288
+ let u
289
+ try { u = new URL(candidate) } catch { return null }
290
+ if (u.href !== candidate) return null
291
+ if (u.protocol !== location.protocol) return null
292
+ if (u.host !== location.host) return null
293
+ if (u.username || u.password) return null
294
+ const route = parseManagedRoute(u)
295
+ if (route?.finding) {
296
+ return { id: route.finding.id, fragment: encodeFindingRef(route.finding), path: managedRoutePath(route) }
297
+ }
298
+ // Only E2E entry pages carry finding hashes. Unknown managed page URLs
299
+ // must not turn into a search for another accessible copy of the finding.
300
+ if (managed && route?.view !== 'home' && !/\/(?:index|view)\.html$/u.test(u.pathname)) return null
301
+ const ref = extractFindingRef(u.hash)
302
+ if (!ref) return null
303
+ const fragment = encodeFindingRef(ref)
304
+ return { id: ref.id, fragment, path: `/#${fragment}` }
305
+ }
@@ -0,0 +1 @@
1
+ export function parseLinkedFindings(content: string): { groups: string[][]; skipped: number } | null
@@ -0,0 +1,111 @@
1
+ // The "links" file — a small JSON document that says which findings
2
+ // are the SAME finding, reported twice:
3
+ //
4
+ // [[{"id":"a"},{"id":"b"}], [{"id":"c"},{"id":"d"},{"id":"e"}]]
5
+ //
6
+ // One inner array per link. Every finding named in it is a duplicate
7
+ // of every other one in that array, and the file says nothing else:
8
+ // no severity, no title, no file, no prose. That is the whole format.
9
+ //
10
+ // It is NOT a report, and this module deliberately sits outside
11
+ // `report/` because of it. A report carries findings; a links file
12
+ // carries only ids of findings that must already live in reports the
13
+ // reader holds. So it never reaches `ingestReport` / `state.reports`,
14
+ // it has its own view (`ui/view/render-links.js`) that points at the
15
+ // findings. It also supplies the "Duplicates:" row on finding cards
16
+ // and groups explicitly linked rows in workspace App view.
17
+ //
18
+ // Pure — no storage, no DOM, no app state. `linked-findings-index.js`
19
+ // beside this file is the OPFS-wide store built on top; everything
20
+ // here is text in, data out, so the recognition rules are testable on
21
+ // their own.
22
+
23
+ import { isLinkableFindingId } from './finding-link.js'
24
+
25
+ // The `source` marker a links file is filed under — the same slot a
26
+ // report's producer ('deepsec', 'piolium', …) occupies in the counts
27
+ // cache, so `getKind(name)` answers "what kind of file is this" for
28
+ // links and reports alike and the sidebar can bucket both from one
29
+ // lookup. No report format can collide with it: this value comes from
30
+ // here, never from a document's own content.
31
+ export const LINKS_KIND = 'links'
32
+
33
+ // Is this one entry of an inner array — `{"id": "…"}` — as the format
34
+ // spells it? Extra fields are allowed and ignored: an exporter that
35
+ // writes the title or the report alongside the id is writing a
36
+ // superset of this format, not a different one.
37
+ function entryId(entry) {
38
+ if (!entry || typeof entry !== 'object' || Array.isArray(entry)) return null
39
+ const { id } = entry
40
+ return typeof id === 'string' && id.length > 0 ? id : null
41
+ }
42
+
43
+ // Recognise `content` as a links file and normalise it.
44
+ //
45
+ // Returns `{ groups, skipped }` — `groups` is one `string[]` of finding
46
+ // ids per link, and `skipped` counts ids the app could never follow (a
47
+ // session-local numeric id, a control character, something absurdly
48
+ // long — `isLinkableFindingId` draws the same line the per-finding Link
49
+ // button draws). Returns null when the text isn't this format at all.
50
+ //
51
+ // Recognition is on SHAPE, and it is strict, because this parser runs
52
+ // against every dropped file before the report readers get their turn:
53
+ // a top-level array, holding only arrays, holding only objects with a
54
+ // string `id`. Nothing else this app reads is a bare JSON array, so
55
+ // nothing else can be mistaken for one — and an empty top-level array
56
+ // is refused rather than claimed, since `[]` is every empty JSON list
57
+ // in the world, not distinctively a links file.
58
+ //
59
+ // Normalisation drops what carries no information: ids repeated inside
60
+ // one link, ids that can't be linked, and then any link left naming
61
+ // fewer than two findings — a "link" to a single finding links nothing.
62
+ // A file whose every link normalises away is still a links file; it
63
+ // just has no links, which its view says plainly.
64
+ export function parseLinkedFindings(content) {
65
+ let data
66
+ try { data = JSON.parse(content) } catch { return null }
67
+ if (!Array.isArray(data) || data.length === 0) return null
68
+ const groups = []
69
+ let skipped = 0
70
+ for (const raw of data) {
71
+ if (!Array.isArray(raw)) return null
72
+ const ids = new Set()
73
+ for (const entry of raw) {
74
+ const id = entryId(entry)
75
+ if (id === null) return null
76
+ if (isLinkableFindingId(id)) ids.add(id)
77
+ else skipped++
78
+ }
79
+ if (ids.size > 1) groups.push([...ids])
80
+ }
81
+ return { groups, skipped }
82
+ }
83
+
84
+ // How many findings the links in `groups` name, counted once each — a
85
+ // finding named by two links is one linked finding. What the sidebar
86
+ // badge and the view's header report alongside the link count.
87
+ export function countLinkedIds(groups) {
88
+ const ids = new Set()
89
+ for (const group of groups) for (const id of group) ids.add(id)
90
+ return ids.size
91
+ }
92
+
93
+ // Fold links into `index` (a `Map<id, Set<id>>`): every id in a link
94
+ // gains every OTHER id in it as a duplicate. Called once per links
95
+ // file, so an id linked by two files ends up with the union of what
96
+ // both said about it.
97
+ //
98
+ // Union, NOT transitive closure. If one file links a↔b and another
99
+ // links b↔c, then b has two duplicates and a has one — a and c were
100
+ // never said to be the same finding, and inferring it would put a
101
+ // claim in the reader's card that no file they dropped ever made.
102
+ export function collectDuplicates(groups, index) {
103
+ for (const group of groups) {
104
+ for (const id of group) {
105
+ let set = index.get(id)
106
+ if (!set) index.set(id, set = new Set())
107
+ for (const other of group) if (other !== id) set.add(other)
108
+ }
109
+ }
110
+ return index
111
+ }
@@ -0,0 +1,10 @@
1
+ export const BUNDLE_METADATA_VERSION: number
2
+ export interface BundleIdentity { integrity: string; kind: string | null; size: number }
3
+ export interface BundleDetails extends BundleIdentity { bundle?: unknown; json?: unknown }
4
+ export function parseBundleContents(text: string, identity: BundleIdentity): BundleDetails
5
+ export interface BundleMetadata extends Record<string, unknown> {
6
+ files: [string, number | null, string | null, number | null][]
7
+ codeStats: { files: number; lines: number; bytes: number }
8
+ }
9
+ export function createBundleMetadata(details: BundleDetails): Promise<BundleMetadata>
10
+ export function createBundleSummary(details: BundleDetails, metadata?: BundleMetadata): { files: number; codeFiles: number; lines: number }
@@ -0,0 +1,177 @@
1
+ import { bundleCodeStats } from './bundle-stats.js'
2
+ import { Bundle } from '@exodus/stasis-core/bundle'
3
+ import { computeFileHash } from '../report/index.js'
4
+ import { utf8ByteLength } from './utf8.js'
5
+ import { bundleFileSizes, bundleSourcesAsMap, bundleUnsizedFiles } from './bundle-sources.js'
6
+
7
+ // Version 2 sizes every file by its bytes, resources included. A version 1
8
+ // index sized by what the source map held when it was written: before the
9
+ // map asked each entry's format, that took a directory capture for a file
10
+ // and a base64 resource for its spelling; after, it left every resource
11
+ // out. Either way its sizes are not file sizes, so it is read for its
12
+ // hashes alone and rebuilt on open.
13
+ export const BUNDLE_METADATA_VERSION = 2
14
+ const INDEX_VERSION = BUNDLE_METADATA_VERSION
15
+ const hashJobs = new WeakMap()
16
+ const SOURCE_TABS = new Set(['terminal', 'code', 'search', 'compare'])
17
+
18
+ export function bundleNeedsSources(tab, sourceFile = null) {
19
+ return Boolean(sourceFile) || SOURCE_TABS.has(tab)
20
+ }
21
+
22
+ // Bound outstanding WebCrypto jobs: serial awaits incur one task round-trip
23
+ // per file, while an unbounded Promise.all holds every encoded body at once.
24
+ export function computeBundleFileHashes(details) {
25
+ if (details.fileHashes) return Promise.resolve(details.fileHashes)
26
+ if (hashJobs.has(details)) return hashJobs.get(details)
27
+ const job = (async () => {
28
+ const entries = [...bundleSourcesAsMap(details)]
29
+ const result = new Map()
30
+ for (let i = 0; i < entries.length; i += 32) {
31
+ const hashes = await Promise.all(entries.slice(i, i + 32).map(async ([file, content]) => [file, await computeFileHash(content)]))
32
+ for (const [file, hash] of hashes) result.set(file, hash)
33
+ }
34
+ details.fileHashes = result
35
+ return result
36
+ })()
37
+ hashJobs.set(details, job)
38
+ job.catch(() => hashJobs.delete(details))
39
+ return job
40
+ }
41
+
42
+ // Count source lines without charging a trailing newline as an extra line.
43
+ // The source map intentionally excludes non-text resources, so a null entry
44
+ // remains the metadata marker for images, fonts, and other binary payloads.
45
+ export function bundleSourceLineCount(content) {
46
+ if (typeof content !== 'string' || content.length === 0) return 0
47
+ const breaks = content.match(/\r\n|\r|\n/gu)?.length ?? 0
48
+ return breaks + (/[\r\n]$/u.test(content) ? 0 : 1)
49
+ }
50
+
51
+ function mapObject(value) {
52
+ return value instanceof Map ? Object.fromEntries([...value].map(([key, v]) => [key, mapObject(v)])) : value
53
+ }
54
+
55
+ // A private, versioned derivative of the bundle, not a Stasis lock. Source
56
+ // bodies are omitted; file sizes and hashes keep metadata views self-contained.
57
+ // A row is `[path, bytes, hash, lines]`: a resource has bytes but no hash or
58
+ // lines, since it is no source; a directory capture has none of the three.
59
+ // `unsized` names the mounted files with no bytes to give (a base64 spelling
60
+ // that does not decode), which a null size alone cannot tell from no file.
61
+ export async function createBundleMetadata(details) {
62
+ const hashes = await computeBundleFileHashes(details)
63
+ const sizes = bundleFileSizes(details)
64
+ const sourceLines = details.lineCounts ??= new Map([...bundleSourcesAsMap(details)]
65
+ .map(([path, content]) => [path, bundleSourceLineCount(content)]))
66
+ const result = { version: INDEX_VERSION, integrity: details.integrity, kind: details.kind, size: details.size, codeStats: bundleCodeStats(sourceLines, sizes),
67
+ files: [...sizes].map(([path, size]) => [path, size, hashes.get(path) ?? null, sourceLines.get(path) ?? null]) }
68
+ const unsized = bundleUnsizedFiles(details)
69
+ if (unsized.size > 0) result.unsized = [...unsized]
70
+ if (details.kind === 'sourcemap') {
71
+ const { version, file, sourceRoot, names, sources = [], sourcesContent = [] } = details.json
72
+ result.json = { version, file, sourceRoot, sources }
73
+ // Sourcemaps can repeat a path with different/absent content. Keep the
74
+ // Overview's positional inventory, alongside the path-keyed graph index.
75
+ result.sourceSizes = sources.map((_, i) => typeof sourcesContent[i] === 'string' ? utf8ByteLength(sourcesContent[i]) : null)
76
+ result.namesCount = names?.length ?? null
77
+ } else {
78
+ const b = details.bundle
79
+ const bundle = { version: b.version, config: b.config, formats: mapObject(b.formats), imports: mapObject(b.imports), reason: b.reason, executable: [...b.executable] }
80
+ if (b.version === 0) {
81
+ bundle.sources = Object.fromEntries([...sizes.keys()].map((path) => [path, null]))
82
+ } else {
83
+ const modules = {}, sources = {}
84
+ for (const [dir, info] of b.modules) {
85
+ const target = dir.split('/').includes('node_modules') ? modules : sources
86
+ Object.defineProperty(target, dir, { enumerable: true, value: {
87
+ name: info.name, version: info.version, ecosystem: info.ecosystem,
88
+ files: Object.fromEntries(Object.keys(info.files).map((path) => [path, null])),
89
+ } })
90
+ }
91
+ bundle.modules = modules
92
+ if (b.config.scope === 'full') { bundle.sources = sources; bundle.entries = [...b.entries] }
93
+ }
94
+ result.bundle = bundle
95
+ }
96
+ return result
97
+ }
98
+
99
+ // Catalogs need counts without the per-file inventory. Match scan inputs:
100
+ // textual resources count as files (with zero LoC); binary resources do not
101
+ // count as Code inputs, and missing source bodies/directories are not files.
102
+ export function createBundleSummary(details, metadata) {
103
+ const files = (metadata?.files ?? [...bundleFileSizes(details)]).filter(([, size]) => size !== null)
104
+ const formats = details.kind === 'stasis' ? details.bundle.formats : null
105
+ const lines = metadata?.codeStats.lines ?? [...bundleSourcesAsMap(details).values()].reduce((sum, source) => sum + bundleSourceLineCount(source), 0)
106
+ return { files: files.length, lines,
107
+ codeFiles: files.filter(([path]) => !['resource:base64', 'directory'].includes(formats?.get(path))).length }
108
+ }
109
+
110
+ // `stale` marks an index this version did not write: its hashes still
111
+ // answer report lookups, but its sizes and line counts are not trusted,
112
+ // and an open rebuilds it.
113
+ export function parseBundleMetadata(data, integrity) {
114
+ if ((data?.version !== 1 && data?.version !== INDEX_VERSION) || data.integrity !== integrity || !['stasis', 'sourcemap'].includes(data.kind)
115
+ || !Number.isSafeInteger(data.size) || data.size < 0 || !Array.isArray(data.files)) throw new Error('Invalid bundle metadata')
116
+ const stale = data.version !== INDEX_VERSION
117
+ const fileHashes = new Map(), fileSizes = new Map(), lineCounts = new Map()
118
+ for (const row of data.files) {
119
+ // Version 1 rows predate the line count, and gave every sized entry a hash.
120
+ if (!Array.isArray(row) || (row.length !== 4 && !(stale && row.length === 3))) throw new Error('Invalid bundle metadata file')
121
+ const [path, size, hash, lines] = row
122
+ if (typeof path !== 'string' || fileSizes.has(path) || (size !== null && (!Number.isSafeInteger(size) || size < 0))
123
+ || (hash !== null && (typeof hash !== 'string' || !/^sha512-[A-Za-z0-9+/]{86}==$/u.test(hash)))
124
+ || (size === null && hash !== null) || (stale && size !== null && hash === null)
125
+ || (lines !== undefined && lines !== null && (!Number.isSafeInteger(lines) || lines < 0))) throw new Error('Invalid bundle metadata file')
126
+ fileSizes.set(path, size)
127
+ if (hash !== null) fileHashes.set(path, hash)
128
+ if (lines !== undefined && lines !== null) lineCounts.set(path, lines)
129
+ }
130
+ // Only a sizeless row can name a mounted file with no size, and only once.
131
+ const unsized = new Set(data.unsized ?? [])
132
+ if ((data.unsized !== undefined && (stale || !Array.isArray(data.unsized) || unsized.size !== data.unsized.length))
133
+ || [...unsized].some((path) => typeof path !== 'string' || !fileSizes.has(path) || fileSizes.get(path) !== null)) throw new Error('Invalid bundle metadata')
134
+ const details = { integrity, kind: data.kind, size: data.size, metadataOnly: true, fileSizes, fileHashes, lineCounts, unsizedFiles: unsized, stale }
135
+ if (data.kind === 'stasis') {
136
+ details.bundle = Bundle.parse(JSON.stringify(data.bundle))
137
+ const paths = details.bundle.sources
138
+ if (paths.size !== fileSizes.size || [...paths.keys()].some((path) => !fileSizes.has(path))) throw new Error('Invalid bundle metadata inventory')
139
+ // A file is hashed exactly when it is source: a resource has bytes and
140
+ // no hash. And only a base64 resource can be mounted without a size.
141
+ // Both are checked against the formats, which only the bundle carries.
142
+ const formats = details.bundle.formats
143
+ if (!stale && [...fileSizes].some(([path, size]) => fileHashes.has(path) !== (size !== null && !Bundle.isResourceFormat(formats.get(path))))) {
144
+ throw new Error('Invalid bundle metadata inventory')
145
+ }
146
+ if ([...unsized].some((path) => formats.get(path) !== 'resource:base64')) throw new Error('Invalid bundle metadata inventory')
147
+ } else {
148
+ if (!data.json || typeof data.json !== 'object' || ![null, undefined].includes(data.namesCount) && (!Number.isSafeInteger(data.namesCount) || data.namesCount < 0)) throw new Error('Invalid sourcemap metadata')
149
+ if (!Array.isArray(data.json.sources) || data.json.sources.some((path) => !fileSizes.has(path))
150
+ || !Array.isArray(data.sourceSizes) || data.sourceSizes.length !== data.json.sources.length
151
+ || data.sourceSizes.some((size) => size !== null && (!Number.isSafeInteger(size) || size < 0))
152
+ || [...fileSizes].some(([path, size]) => fileHashes.has(path) !== (size !== null)) || unsized.size > 0) throw new Error('Invalid sourcemap inventory')
153
+ details.json = data.json
154
+ details.sourceSizes = data.sourceSizes
155
+ details.namesCount = data.namesCount
156
+ }
157
+ if (data.codeStats === undefined) details.codeStats = bundleCodeStats(lineCounts, fileSizes)
158
+ else {
159
+ const stats = data.codeStats
160
+ const validCount = value => Number.isSafeInteger(value) && value >= 0
161
+ if (!stats || ![stats.files, stats.lines, stats.bytes].every(validCount) || !Array.isArray(stats.languages)
162
+ || stats.languages.some(row => !row || typeof row.key !== 'string' || typeof row.label !== 'string'
163
+ || ![row.files, row.lines, row.bytes].every(validCount))) throw new Error('Invalid bundle code stats')
164
+ details.codeStats = stats
165
+ }
166
+ return details
167
+ }
168
+
169
+ // HTTP content encoding handles compression before this shared JSON parser.
170
+ export function parseBundleContents(text, { integrity, kind, size }) {
171
+ if (kind === 'stasis') return { integrity, kind, size, bundle: Bundle.parse(text) }
172
+ if (kind !== 'sourcemap') throw new Error('Unsupported bundle format')
173
+ const json = JSON.parse(text)
174
+ if (!Array.isArray(json?.sources) || json.sources.some(path => typeof path !== 'string')
175
+ || (json.sourcesContent !== undefined && (!Array.isArray(json.sourcesContent) || json.sourcesContent.some(content => content !== null && typeof content !== 'string')))) throw new Error('Invalid sourcemap')
176
+ return { integrity, kind, size, json }
177
+ }
@@ -0,0 +1,2 @@
1
+ import type { BundleDetails } from './bundle-metadata.js'
2
+ export function bundleReasons(details: BundleDetails, sourcePaths?: Iterable<string>): Map<string, Set<string>>
@@ -0,0 +1,21 @@
1
+ import { bundleFileSizes } from './bundle-sources.js'
2
+
3
+ // Reason metadata attributes files to consumers (run, build plugins, etc.).
4
+ // It is informational and can be absent on older/single-consumer bundles.
5
+ export function bundleReasons(details, sourcePaths) {
6
+ const reasons = new Map()
7
+ const raw = details?.kind === 'stasis' ? details.bundle?.reason : null
8
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return reasons
9
+ const paths = new Set(sourcePaths ?? [...bundleFileSizes(details)].filter(([, size]) => size != null).map(([path]) => path))
10
+ for (const [reason, files] of Object.entries(raw).toSorted(([a], [b]) => a.localeCompare(b))) {
11
+ if (!reason || !Array.isArray(files)) continue
12
+ const present = new Set(files.filter((file) => typeof file === 'string' && paths.has(file)))
13
+ if (present.size > 0) reasons.set(reason, present)
14
+ }
15
+ // A reason is only useful as a filter if it changes the set of files.
16
+ // Keep all named options when at least one differs, otherwise hide the
17
+ // selector (and discard any stale selection) in every visualization.
18
+ if (![...reasons.values()].some((files) => files.size < paths.size)) reasons.clear()
19
+ return reasons
20
+ }
21
+