@preventive/triage 1.0.0-alpha.13 → 1.0.0-alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/out/brotli-fallback.js +3 -3
  2. package/out/client-admin.js +2 -2
  3. package/out/client-sync.js +14 -14
  4. package/out/graph.js +30 -5
  5. package/out/index.html +47 -7
  6. package/out/prism.js +2 -2
  7. package/out/stasis.svg +45 -0
  8. package/out/terminal.js +255 -43
  9. package/out/view.css +1 -1
  10. package/out/view.js +147 -110
  11. package/package.json +27 -4
  12. package/report/index.js +253 -0
  13. package/report/src/finding-id.js +80 -0
  14. package/report/src/finding.js +300 -0
  15. package/report/src/labels.js +33 -0
  16. package/report/src/md-structure.js +471 -0
  17. package/report/src/md-text.js +167 -0
  18. package/report/src/meta.js +51 -0
  19. package/report/src/parse-codex.js +147 -0
  20. package/report/src/parse-deepsec.js +197 -0
  21. package/report/src/parse-deepview-fields.js +375 -0
  22. package/report/src/parse-deepview-md.js +185 -0
  23. package/report/src/parse-md-id.js +137 -0
  24. package/report/src/parse-md.js +253 -0
  25. package/report/src/parse-piolium-id.js +79 -0
  26. package/report/src/parse-piolium-rows.js +131 -0
  27. package/report/src/parse-piolium-tokens.js +175 -0
  28. package/report/src/parse-piolium.js +400 -0
  29. package/report/src/utf8.js +21 -0
  30. package/report/src/write-md-finding.js +273 -0
  31. package/report/src/write-md.js +291 -0
  32. package/server-e2e/bus-receiver.ts +1 -0
  33. package/server-e2e/hub.ts +37 -6
  34. package/server-e2e/index.ts +3 -2
  35. package/server-e2e/objstore/handlers.ts +6 -5
  36. package/server-e2e/objstore/init.ts +1 -1
  37. package/server-e2e/objstore/rest-mint.ts +2 -2
  38. package/server-e2e/objstore/rest.ts +2 -2
  39. package/server-e2e/pubsub.ts +8 -5
  40. package/server-e2e/sync-handlers.ts +54 -28
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@preventive/triage",
3
- "version": "1.0.0-alpha.13",
3
+ "version": "1.0.0-alpha.15",
4
4
  "description": "Client & relay server for triaging of automated reports",
5
5
  "license": "MIT",
6
6
  "author": {
@@ -30,6 +30,7 @@
30
30
  },
31
31
  "exports": {
32
32
  "./reap": "./api/reap.ts",
33
+ "./report": "./report/index.js",
33
34
  "./server": "./server-e2e/index.ts",
34
35
  "./strip-types-loader": "./strip-types-loader.js",
35
36
  "./package.json": "./package.json"
@@ -80,6 +81,26 @@
80
81
  "server-e2e/ws-server.ts",
81
82
  "server-e2e/config.example.json",
82
83
  "server-common/origin.ts",
84
+ "report/index.js",
85
+ "report/src/finding-id.js",
86
+ "report/src/finding.js",
87
+ "report/src/labels.js",
88
+ "report/src/md-structure.js",
89
+ "report/src/md-text.js",
90
+ "report/src/meta.js",
91
+ "report/src/parse-codex.js",
92
+ "report/src/parse-deepsec.js",
93
+ "report/src/parse-deepview-fields.js",
94
+ "report/src/parse-deepview-md.js",
95
+ "report/src/parse-md-id.js",
96
+ "report/src/parse-md.js",
97
+ "report/src/parse-piolium-id.js",
98
+ "report/src/parse-piolium-rows.js",
99
+ "report/src/parse-piolium-tokens.js",
100
+ "report/src/parse-piolium.js",
101
+ "report/src/utf8.js",
102
+ "report/src/write-md-finding.js",
103
+ "report/src/write-md.js",
83
104
  "common/save-error-reason.ts",
84
105
  "common/server-info.ts",
85
106
  "common/utf8.d.ts",
@@ -94,6 +115,7 @@
94
115
  "out/index.html",
95
116
  "out/manifest.webmanifest",
96
117
  "out/prism.js",
118
+ "out/stasis.svg",
97
119
  "out/terminal.js",
98
120
  "out/view.css",
99
121
  "out/view.html",
@@ -102,7 +124,7 @@
102
124
  ],
103
125
  "scripts": {
104
126
  "lint": "node --run lint:js && node --run lint:css && node --run lint:ts",
105
- "lint:js": "oxlint -c .oxlintrc.json --ignore-pattern=ui/logs.js ./tests/ ./common/ ./server-common/ ./server-e2e/ ./server-managed/ ./client/ ./ui/ ./api/",
127
+ "lint:js": "oxlint -c .oxlintrc.json --ignore-pattern=ui/logs.js ./tests/ ./common/ ./report/ ./server-common/ ./server-e2e/ ./server-managed/ ./client/ ./ui/ ./api/",
106
128
  "lint:css": "stylelint 'ui/**/*.css'",
107
129
  "lint:ts": "tsc --noEmit",
108
130
  "prepack": "node --run build",
@@ -110,13 +132,14 @@
110
132
  "serve": "node build.js serve",
111
133
  "server": "node server-e2e/index.ts",
112
134
  "server-managed": "node server-managed/index.ts",
113
- "test": "node --js-base-64 --experimental-test-module-mocks --test-timeout=120000 --test ./tests/*.test.js"
135
+ "test": "node --js-base-64 --experimental-test-module-mocks --test-timeout=120000 --test ./tests/*.test.js ./report/tests/*.test.js"
114
136
  },
115
137
  "devDependencies": {
116
138
  "@electric-sql/pglite": "^0.5.5",
117
139
  "@exodus/stasis-core": "^1.0.0-beta.4",
118
140
  "@noble/ciphers": "^2.3.0",
119
- "@preventive/terminal": "^1.3.0",
141
+ "@preventive/diff": "^1.0.0",
142
+ "@preventive/terminal": "^1.12.0",
120
143
  "@rray/frontend": "^1.0.0",
121
144
  "@stylistic/stylelint-plugin": "^5.3.0",
122
145
  "@types/node": "^26.2.0",
@@ -0,0 +1,253 @@
1
+ // The report library — one door to every report format this project
2
+ // reads, and to the one it writes.
3
+ //
4
+ // A "report" is whatever an analyzer wrote: the JSON dump this
5
+ // project's own analyzer emits, or one of the shapes other tools
6
+ // produce — DeepSec and Piolium markdown, Claude Security markdown,
7
+ // Codex CSV — or the markdown document this library itself writes
8
+ // (write-md.js), read back by parse-deepview-md.js. The parsers beside
9
+ // this file each recognise exactly one of those and know nothing about
10
+ // each other; this module is the dispatch over them, so "which formats
11
+ // do we read, and in what order" is answered in one place instead of
12
+ // once per call site.
13
+ //
14
+ // import { loadFindings, writeMarkdown } from '../report/index.js'
15
+ // const report = await loadFindings(text)
16
+ // // → { format, data, findings: [ … with ids ] } | null
17
+ // const md = writeMarkdown({ title, groups: report.findings.map((f) => [f]) })
18
+ //
19
+ // Three entry points for reading, in rising order of how much they do:
20
+ //
21
+ // detectFormat — name the format, parse nothing further
22
+ // readReport — the parsed report, or the reason it isn't one
23
+ // loadFindings — parsed, flattened, and every finding carrying an id
24
+ //
25
+ // `analyzeReport` is `readReport` for a file list — entry count and
26
+ // producer — `reportEntries` is a report's entry list whichever of the
27
+ // two names it goes under (`findings` or `groups`), for a caller that
28
+ // has to keep the grouping rather than flatten it, and
29
+ // `backfillFindingIds` is the id step on its own, for a caller that
30
+ // has to interleave something with it.
31
+ //
32
+ // And one for writing: `writeMarkdown` takes findings — the parsers'
33
+ // own objects, grouped as the viewer groups them — with whatever the
34
+ // caller knows about the selection and the reader's annotations, and
35
+ // writes the markdown document the Download button saves (write-md.js
36
+ // for the document, write-md-finding.js for one finding, labels.js for
37
+ // the words). Both directions read a finding through finding.js, so
38
+ // what a parser produced and what the writer prints agree on what a
39
+ // finding IS — and the document is itself a report the readers above
40
+ // take (parse-deepview-md.js), so what was written can be loaded
41
+ // again, with the ids and fields it left with.
42
+ //
43
+ // Codex is the one format the content doesn't name: its export is a
44
+ // CSV, and a CSV is a container — one row per finding across several
45
+ // scans — rather than a report. `detectFormat` recognises it by the
46
+ // FILENAME (`.csv`) when given one; the readers are single-report and
47
+ // don't take it. A codex export goes through `parseCodexCsvToScans`,
48
+ // which splits it into one JSON-shaped report per scan, and each of
49
+ // those reads through the readers like any other JSON report.
50
+ //
51
+ // THIS FILE IS THE WHOLE SURFACE. The modules live in `src/` and the
52
+ // package exports one path — `@preventive/report`, this file — so
53
+ // everything a caller may hold is named here, in one list, and
54
+ // everything else is free to move, split or be renamed without
55
+ // breaking anyone. A consumer that wants `fenceRanges` or
56
+ // `findingTitle` imports it from here beside `loadFindings`; there is
57
+ // no second, deeper way in, inside this repo or out of it.
58
+ //
59
+ // That is a deliberate trade against the old shape, where every module
60
+ // was its own entry point. What it costs is the ability to reach past
61
+ // this list; what it buys is that the list IS the contract. Nothing is
62
+ // pulled in that a caller doesn't use: every module here is
63
+ // side-effect-free (`sideEffects: false` in package.json — the whole
64
+ // file is declarations), so a bundler drops what a caller never names,
65
+ // and `ui/view/format.js` still rides its lazily-loaded chunk.
66
+ //
67
+ // This directory is its own package (see package.json beside this
68
+ // file) and imports nothing outside itself: no DOM, no app state, no
69
+ // storage, nothing from the rest of the repo. Text in, data out — and
70
+ // data in, text out. That is what makes it reusable outside the
71
+ // viewer — the analyzer stamps its ids with the same `findingId` the
72
+ // viewer derives them with, so both sides agree on what a finding IS
73
+ // — and `node --test` in this directory runs its suite with nothing
74
+ // else installed — and nothing outside this directory reaches into it,
75
+ // tests included: every test of this library lives in `report/tests/`.
76
+ // Those come through the door like any other caller, except where they
77
+ // exercise an internal this file doesn't export; those name `../src/`,
78
+ // which is what they are testing.
79
+
80
+ import { parseDeepsecFindings } from './src/parse-deepsec.js'
81
+ import { parseDeepviewMarkdown } from './src/parse-deepview-md.js'
82
+ import { parseMarkdownFindings } from './src/parse-md.js'
83
+ import { parsePioliumFindings } from './src/parse-piolium.js'
84
+ import { deriveFindingId } from './src/finding-id.js'
85
+
86
+ // The rest of the surface, so a consumer needs one import: the codex
87
+ // splitter, the id helpers the analyzer shares with the viewer, and the
88
+ // run-meta projection a caller applies to the findings it loads.
89
+ export { parseCodexCsvToScans } from './src/parse-codex.js'
90
+ export { computeFileHash, deriveFindingId, findingId } from './src/finding-id.js'
91
+ export { META_FIELDS, inheritReportMeta, reportRepoGithub } from './src/meta.js'
92
+ // The writing side: the document writer, and the label tables it
93
+ // spells the app's enumerations with, for the viewer's surfaces that
94
+ // describe the same things in prose.
95
+ export { writeMarkdown } from './src/write-md.js'
96
+ export { COLOR_LABELS, SEVERITY_LABELS, SOURCE_LABELS, TRIAGE_LABELS, severityLabel } from './src/labels.js'
97
+
98
+ // Reading a finding: what a finding IS, asked of one. The card, the
99
+ // row, the filters and the writer all ask the same questions of the
100
+ // same object — which tier does this display under, what is its name,
101
+ // where does it sit, what did the pass say about it — and they ask
102
+ // them here, so a parser's output and every surface that renders it
103
+ // can't drift on the answers.
104
+ export {
105
+ REVALIDATE_KINDS, SEVERITIES, SEVERITY_ORDER, correctedVariants, descriptionSections,
106
+ displayedSeverity, effectiveSeverity, evidenceNote, findingDisplayName, findingTitle,
107
+ firstLine, hasSeverityCorrection, isAppFinding, locationLabel, prettyModel, revalidateKindOf,
108
+ runMetaLine, splitDescription, stripExportMarker, titledDescription,
109
+ } from './src/finding.js'
110
+
111
+ // The structural-markdown helpers, for a caller rendering the prose a
112
+ // parser handed back: where the fences are (so a `## ` inside a
113
+ // snippet stays in the snippet), and the escapes markdown puts on a
114
+ // name. The viewer's own markdown rendering (ui/view/format.js,
115
+ // export-view-chunks.js) reads the document's shape with these rather
116
+ // than keeping a second, subtly different set.
117
+ export { fenceRanges, inFence, unescapeMd } from './src/md-structure.js'
118
+ export { isHttpUrl } from './src/md-text.js'
119
+
120
+ // The markdown chain, in dispatch order: tightest guard first. This
121
+ // library's own document opens on a marker line no other format has;
122
+ // DeepSec keys off `## SEVERITY (n)` and Piolium off its `# Security
123
+ // Audit Report` / `## Technical Findings Detail` headings, while
124
+ // parse-md accepts any `# Title` document — so it has to stay last or
125
+ // it would swallow the others. Each returns the standard `{ type,
126
+ // findings, … }` shape, or null when the text isn't its format.
127
+ //
128
+ // `format` is this library's name for the document's producer. For the
129
+ // three foreign markdown formats it matches the `source` marker the
130
+ // parser stamps on what it returns, which is what the viewer reads for
131
+ // its header label. 'deepview-md' is the exception that proves the
132
+ // rule: the document is this library's, but its findings came from
133
+ // whichever producer the document names, and THAT is the `source` it
134
+ // carries back (none for the analyzer's own runs). 'json' has no
135
+ // marker (the analyzer's own dump carries `type` instead).
136
+ const MARKDOWN_FORMATS = [
137
+ ['deepview-md', parseDeepviewMarkdown],
138
+ ['deepsec', parseDeepsecFindings],
139
+ ['piolium', parsePioliumFindings],
140
+ ['claude-security', parseMarkdownFindings],
141
+ ]
142
+
143
+ // A report's entries: `findings`, or `groups` for a pre-deduplicated
144
+ // dump. Each entry is one finding or a Finding[] group. Null when the
145
+ // document carries neither as an array — which is how a JSON file that
146
+ // isn't a report at all (or a report with a malformed list) is told
147
+ // apart from an empty one.
148
+ //
149
+ // Exported because a report is two shapes and only one of them is
150
+ // called `findings`: a caller reading `data.findings` alone sees an
151
+ // empty report wherever the entries are groups — which is every
152
+ // deduplicated dump, and every export of a view that merged a finding
153
+ // reported twice (parse-deepview-md.js writes `groups` for exactly
154
+ // those). `loadFindings` is the answer for a caller that wants the
155
+ // member findings; this is the one for a caller that has to keep the
156
+ // grouping, as the viewer's ingest does.
157
+ export function reportEntries(data) {
158
+ if (Array.isArray(data?.findings)) return data.findings
159
+ if (Array.isArray(data?.groups)) return data.groups
160
+ return null
161
+ }
162
+
163
+ // Which producer wrote `content` — 'json' / 'deepview-md' / 'deepsec' /
164
+ // 'piolium' / 'claude-security' / 'codex', or null when nothing
165
+ // recognises it.
166
+ //
167
+ // `filename` is optional and decides only codex: a `.csv` is a codex
168
+ // export, and the content is not consulted for it (nothing in a CSV's
169
+ // text says whose it is, and no other format arrives as one). Every
170
+ // other format is named from the content alone, so a `.md` holding a
171
+ // JSON dump is 'json'. Case-insensitive on the extension; strip any
172
+ // download-duplicate suffix (`report (1).csv`) before calling if the
173
+ // name can carry one after the extension.
174
+ export function detectFormat(content, filename) {
175
+ if (typeof filename === 'string' && /\.csv$/iu.test(filename)) return 'codex'
176
+ return readReport(content).format
177
+ }
178
+
179
+ // Parse `content` in whichever format it turns out to be. JSON first —
180
+ // the analyzer's native dump is the common case and the only format
181
+ // with a cheap, total test — then the markdown chain when `JSON.parse`
182
+ // throws. A JSON document counts as a report only when it carries a
183
+ // `findings` (or `groups`) array: anything else that parses is some
184
+ // other JSON file, not an empty report.
185
+ //
186
+ // Returns `{ data, format, reason }`: the parsed report and its
187
+ // format, or `data: null` with `reason` saying why in one sentence —
188
+ // which a caller reporting "this file isn't a report" can show as is.
189
+ // The usual cause is a truncated or malformed JSON dump rather than an
190
+ // unknown format, so the JSON error rides along in that sentence.
191
+ export function readReport(content) {
192
+ let jsonError
193
+ try {
194
+ const data = JSON.parse(content)
195
+ if (reportEntries(data)) return { data, format: 'json', reason: null }
196
+ return { data: null, format: null, reason: 'JSON, but not a report: no findings array' }
197
+ } catch (err) {
198
+ jsonError = err
199
+ }
200
+ for (const [format, parse] of MARKDOWN_FORMATS) {
201
+ const data = parse(content)
202
+ if (data) return { data, format, reason: null }
203
+ }
204
+ return {
205
+ data: null,
206
+ format: null,
207
+ reason: `Not JSON, and not a recognized markdown format. (JSON error: ${jsonError.message})`,
208
+ }
209
+ }
210
+
211
+ // How many entries `content` holds and who produced it, without
212
+ // flattening anything or deriving a single id — what a file list wants
213
+ // for a badge next to a name. `count` is ENTRIES, not findings: an
214
+ // entry is either one finding or a pre-deduplicated group of them, and
215
+ // the entry count is what a user sees as rows.
216
+ export function analyzeReport(content) {
217
+ const { data } = readReport(content)
218
+ if (!data) return { count: 0, recognized: false }
219
+ return { count: reportEntries(data).length, source: data.source, recognized: true }
220
+ }
221
+
222
+ // Entries → member findings. A group contributes its members; falsy
223
+ // and non-object entries (a malformed list's stray strings and nulls)
224
+ // are dropped rather than handed on as findings.
225
+ function flattenFindings(entries) {
226
+ return entries.flat().filter((f) => f && typeof f === 'object')
227
+ }
228
+
229
+ // Fill in `f.id` for any finding that lacks one, deriving it from the
230
+ // same fingerprint the analyzer stamps. Mutates in place. Findings
231
+ // whose id can't be derived (a host without crypto.subtle) are left
232
+ // untouched. Batched via Promise.all — sequential awaits would
233
+ // serialise hundreds of crypto.subtle.digest calls for no reason.
234
+ export async function backfillFindingIds(findings) {
235
+ const idLess = findings.filter((f) => !f.id)
236
+ if (idLess.length === 0) return
237
+ const derived = await Promise.all(idLess.map(deriveFindingId))
238
+ idLess.forEach((f, i) => { if (derived[i]) f.id = derived[i] })
239
+ }
240
+
241
+ // Recognise, flatten, and give every finding an id — the whole read
242
+ // path in one call. Returns `{ format, data, findings }`, or null when
243
+ // nothing recognises the text. The findings are the parser's own
244
+ // objects (not copies), so a caller that means to keep them can, and
245
+ // one projecting run meta onto them has `inheritReportMeta` and `data`
246
+ // to hand.
247
+ export async function loadFindings(content) {
248
+ const { data, format } = readReport(content)
249
+ if (!data) return null
250
+ const findings = flattenFindings(reportEntries(data))
251
+ await backfillFindingIds(findings)
252
+ return { format, data, findings }
253
+ }
@@ -0,0 +1,80 @@
1
+ // Finding ids, stamped by the analyzer onto its JSON output and filled
2
+ // in by the viewer for findings that arrive without one. Web Crypto is
3
+ // the common surface — `crypto.subtle` exists in modern Node and in
4
+ // secure browser contexts — so one implementation runs in both.
5
+ //
6
+ // Two reports from the same source give a finding the same id; an edit
7
+ // to its description or its source invalidates it.
8
+
9
+ import { encodeUtf8 } from './utf8.js'
10
+
11
+ function toHex(bytes) {
12
+ return Array.from(bytes, (b) => b.toString(16).padStart(2, '0')).join('')
13
+ }
14
+
15
+ // The file-content hash the JSON output format uses. sha512 because the
16
+ // id below hashes a string that already includes it, so a collision here
17
+ // would propagate into an id collision. Padded base64 with the SRI-style
18
+ // tag. `btoa` rather than `Uint8Array#toBase64`, still flagged in Node.
19
+ export async function computeFileHash(source) {
20
+ const bytes = typeof source === 'string' ? encodeUtf8(source) : source
21
+ const digest = new Uint8Array(await crypto.subtle.digest('SHA-512', bytes))
22
+ return `sha512-${btoa(String.fromCodePoint(...digest))}`
23
+ }
24
+
25
+ // A fingerprint hashed into a v4-shaped UUID: derived, not random, but
26
+ // the shape lets a downstream tool treat it as an opaque id.
27
+ async function fingerprintToId(fingerprint) {
28
+ const bytes = encodeUtf8(JSON.stringify(fingerprint))
29
+ const digest = await crypto.subtle.digest('SHA-256', bytes)
30
+ const u = new Uint8Array(digest, 0, 16)
31
+ // version 4: 0100xxxx
32
+ u[6] = (u[6] & 0x0f) | 0x40
33
+ // variant 1: 10xxxxxx
34
+ u[8] = (u[8] & 0x3f) | 0x80
35
+ const hex = toHex(u)
36
+ return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`
37
+ }
38
+
39
+ // Stable per-finding id from the (severity, description, fileHash) triple
40
+ // the analyzer emits. fileHash being undefined is fine — JSON.stringify
41
+ // drops undefined keys, so a finding with no hash keys off the pair and
42
+ // re-runs over the same source yield the same ids.
43
+ export function findingId(severity, description, fileHash) {
44
+ return fingerprintToId({ severity, description, fileHash })
45
+ }
46
+
47
+ // An id derived from a finding, on the first discriminator it carries.
48
+ // null when `crypto.subtle` is unavailable (some `file://` setups), so
49
+ // the caller can fall back to a session-local id — the UI still works,
50
+ // without persistent triage on those findings.
51
+ //
52
+ // In order:
53
+ // - _idBasis — a FROZEN fingerprint a parser stamped, used verbatim;
54
+ // it exists so a change to the rendered description
55
+ // can't re-key stored triage (parse-md-id.js).
56
+ // - fileHash — as `findingId` above.
57
+ // - location — a markdown import's url, also stable.
58
+ // - file/line — last resort for a JSON finding with neither: not what
59
+ // the spec prescribes, but better than collapsing two
60
+ // unrelated findings onto one id.
61
+ export async function deriveFindingId(f) {
62
+ if (typeof crypto?.subtle?.digest !== 'function') return null
63
+ const fingerprint = fingerprintOf(f)
64
+ try {
65
+ return await fingerprintToId(fingerprint)
66
+ } catch {
67
+ return null
68
+ }
69
+ }
70
+
71
+ // The choice above as the object that gets hashed. Key order is part of
72
+ // the id — JSON.stringify keeps insertion order — so every shape lists
73
+ // severity and description first.
74
+ function fingerprintOf(f) {
75
+ if (f._idBasis) return f._idBasis
76
+ const { severity, description } = f
77
+ if (f.fileHash) return { severity, description, fileHash: f.fileHash }
78
+ if (f.location) return { severity, description, location: f.location }
79
+ return { severity, description, file: f.file, line: f.line }
80
+ }