@preventive/triage 1.0.0-alpha.13 → 1.0.0-alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/out/brotli-fallback.js +3 -3
  2. package/out/client-admin.js +2 -2
  3. package/out/client-sync.js +14 -14
  4. package/out/graph.js +30 -5
  5. package/out/index.html +47 -7
  6. package/out/prism.js +2 -2
  7. package/out/stasis.svg +45 -0
  8. package/out/terminal.js +255 -43
  9. package/out/view.css +1 -1
  10. package/out/view.js +147 -110
  11. package/package.json +27 -4
  12. package/report/index.js +253 -0
  13. package/report/src/finding-id.js +80 -0
  14. package/report/src/finding.js +300 -0
  15. package/report/src/labels.js +33 -0
  16. package/report/src/md-structure.js +471 -0
  17. package/report/src/md-text.js +167 -0
  18. package/report/src/meta.js +51 -0
  19. package/report/src/parse-codex.js +147 -0
  20. package/report/src/parse-deepsec.js +197 -0
  21. package/report/src/parse-deepview-fields.js +375 -0
  22. package/report/src/parse-deepview-md.js +185 -0
  23. package/report/src/parse-md-id.js +137 -0
  24. package/report/src/parse-md.js +253 -0
  25. package/report/src/parse-piolium-id.js +79 -0
  26. package/report/src/parse-piolium-rows.js +131 -0
  27. package/report/src/parse-piolium-tokens.js +175 -0
  28. package/report/src/parse-piolium.js +400 -0
  29. package/report/src/utf8.js +21 -0
  30. package/report/src/write-md-finding.js +273 -0
  31. package/report/src/write-md.js +291 -0
  32. package/server-e2e/bus-receiver.ts +1 -0
  33. package/server-e2e/hub.ts +37 -6
  34. package/server-e2e/index.ts +3 -2
  35. package/server-e2e/objstore/handlers.ts +6 -5
  36. package/server-e2e/objstore/init.ts +1 -1
  37. package/server-e2e/objstore/rest-mint.ts +2 -2
  38. package/server-e2e/objstore/rest.ts +2 -2
  39. package/server-e2e/pubsub.ts +8 -5
  40. package/server-e2e/sync-handlers.ts +54 -28
@@ -0,0 +1,51 @@
1
+ // Run-level meta — the fields at the top of a report describing the run
2
+ // that produced it. Both the report view (ui/view/ingest.js) and the
3
+ // OPFS-wide index (client/bundle-finding-index.js) project the header
4
+ // onto the findings, so every consumer reads run meta off a finding
5
+ // without asking whether the file was one run or a deduplicated dump.
6
+ export const META_FIELDS = ['type', 'model', 'think', 'effort', 'exportsMode']
7
+
8
+ // Each run-meta field the finding doesn't specify, filled in place from
9
+ // the report header. Per-field, not all-or-nothing: `deduplicate` stamps
10
+ // `model` per finding while the rest stay run-level, so a finding
11
+ // carrying only its own `model` still needs the header's `type`. Null
12
+ // counts as unspecified — a report is JSON, where `"type": null` can't be
13
+ // told from an omitted key. Source-marked reports opt out wholesale:
14
+ // each is one analyzer, and its report-level `type` is the product's
15
+ // category rather than a run descriptor.
16
+ export function inheritReportMeta(finding, data) {
17
+ if (data.source) return
18
+ for (const key of META_FIELDS) {
19
+ if (finding[key] == null && data[key] != null) finding[key] = data[key]
20
+ }
21
+ }
22
+
23
+ // `"repo": { "github": "owner/name" }` at the top of a native dump, the
24
+ // repository the run covered. NOT inherited onto findings the way
25
+ // META_FIELDS are: a finding's own `repo.github` names the upstream of
26
+ // the file IT sits in — a dependency's repo under `node_modules/` — so
27
+ // stamping the report's over it would mislabel every dependency finding.
28
+ //
29
+ // Takes the slug or a github.com URL, with or without scheme, `.git` or
30
+ // a trailing `/tree/main`, and normalises both to the slug. Anything
31
+ // else is null: a value the link builders would splice into a broken URL
32
+ // is worse than none.
33
+ const GITHUB_URL_RE = /^(?:https?:\/\/)?(?:www\.)?github\.com\/([^/?#]+)\/([^/?#]+?)(?:\.git)?(?:[/?#].*)?$/iu
34
+ const SLUG_RE = /^[\w.-]+\/[\w.-]+$/u
35
+
36
+ // Is this value that slug already? Asked wherever a report hands over
37
+ // something that may be one: `repo.github` below, a Piolium preamble's
38
+ // `**Target:**`, a finding's repo on its way to a link.
39
+ export function isRepoSlug(s) {
40
+ return SLUG_RE.test(s)
41
+ }
42
+
43
+ export function reportRepoGithub(data) {
44
+ const raw = data?.repo?.github
45
+ if (typeof raw !== 'string') return null
46
+ const trimmed = raw.trim().replace(/\/+$/u, '')
47
+ if (!trimmed) return null
48
+ const url = GITHUB_URL_RE.exec(trimmed)
49
+ const slug = (url ? `${url[1]}/${url[2]}` : trimmed).replace(/\.git$/u, '')
50
+ return isRepoSlug(slug) ? slug : null
51
+ }
@@ -0,0 +1,147 @@
1
+ // Codex Security CSV parser. Input is a multi-scan CSV export with
2
+ // rows like:
3
+ // finding_url,repository,repository_url,title,description,severity,
4
+ // status,detected_at,committed_at,author_email,assignee_name,
5
+ // assignee_email,has_patch,configured_scan_id,commit_hash,
6
+ // relevant_paths,resolution_reason
7
+ // One CSV typically merges several scans (each `configured_scan_id`
8
+ // is one scan). We split on that field and emit one report per scan.
9
+ //
10
+ // Display name per scan: `${repository}:${configured_scan_id stripped
11
+ // of its `<prefix>:` head}`. Each scan must contain exactly one
12
+ // repository — asserted, not silently merged.
13
+ //
14
+ // Only the first path in `relevant_paths` becomes `f.file` (some
15
+ // findings list several, as `path1 | path2 | …`); the rest are dropped.
16
+
17
+ const REQUIRED_COLUMNS = [
18
+ 'finding_url', 'repository', 'title', 'description', 'severity',
19
+ 'configured_scan_id', 'relevant_paths',
20
+ ]
21
+
22
+ // RFC 4180-ish CSV parser: handles quoted fields containing commas,
23
+ // embedded newlines, and `""` escaped quotes. Returns rows as arrays
24
+ // of strings (no header / object conversion — caller picks columns by
25
+ // index from the header row).
26
+ function parseCsvRows(text) {
27
+ const rows = []
28
+ let row = []
29
+ let field = ''
30
+ let inQuote = false
31
+ const n = text.length
32
+ for (let i = 0; i < n; i++) {
33
+ const c = text[i]
34
+ if (inQuote) {
35
+ if (c === '"') {
36
+ if (text[i + 1] === '"') { field += '"'; i++ }
37
+ else { inQuote = false }
38
+ } else {
39
+ field += c
40
+ }
41
+ } else if (c === '"') {
42
+ inQuote = true
43
+ } else if (c === ',') {
44
+ row.push(field); field = ''
45
+ } else if (c === '\n' || c === '\r') {
46
+ row.push(field); rows.push(row); row = []; field = ''
47
+ if (c === '\r' && text[i + 1] === '\n') i++
48
+ } else {
49
+ field += c
50
+ }
51
+ }
52
+ // Trailing field / row (no final newline).
53
+ if (field !== '' || row.length > 0) { row.push(field); rows.push(row) }
54
+ // Strip purely-empty trailing rows that come from a trailing newline.
55
+ while (rows.length > 0 && rows.at(-1).length === 1 && rows.at(-1)[0] === '') rows.pop()
56
+ return rows
57
+ }
58
+
59
+ // Rows as `{ <column>: value }` records keyed by the header row, so the
60
+ // mapping below reads columns by name; a column the file lacks reads as
61
+ // undefined, a cell a short row lacks as ''. Null-prototype so a column
62
+ // named `constructor` can't alias an inherited key.
63
+ function csvRecords(header, rows) {
64
+ return rows.map((cells) => {
65
+ const record = Object.create(null)
66
+ header.forEach((name, i) => { record[name] = cells[i] ?? '' })
67
+ return record
68
+ })
69
+ }
70
+
71
+ // Parse + split + convert. Returns one entry per scan:
72
+ // { displayName, data: { type, source, findings: [...] } }
73
+ // where `data` is the same shape ingest.js consumes from JSON.
74
+ export function parseCodexCsvToScans(text) {
75
+ const rows = parseCsvRows(text)
76
+ if (rows.length < 2) throw new Error('Codex CSV: empty or missing header row')
77
+ const [header, ...body] = rows
78
+ for (const required of REQUIRED_COLUMNS) {
79
+ if (!header.includes(required)) throw new Error(`Codex CSV: missing required column "${required}"`)
80
+ }
81
+
82
+ // Group rows by configured_scan_id; a row without one (a blank line
83
+ // included) is skipped.
84
+ const byScan = new Map()
85
+ for (const record of csvRecords(header, body)) {
86
+ const scanId = record.configured_scan_id
87
+ if (!scanId) continue
88
+ if (!byScan.has(scanId)) byScan.set(scanId, [])
89
+ byScan.get(scanId).push(record)
90
+ }
91
+ if (byScan.size === 0) throw new Error('Codex CSV: no rows with a configured_scan_id')
92
+
93
+ const scans = []
94
+ for (const [scanId, records] of byScan) {
95
+ // Each scan must belong to a single repository — surface a real
96
+ // error if upstream ever merges scans across repos rather than
97
+ // silently lumping them under one display name.
98
+ const repos = new Set(records.map((r) => r.repository).filter(Boolean))
99
+ if (repos.size > 1) {
100
+ throw new Error(`Codex CSV: scan ${scanId} contains multiple repositories: ${[...repos].join(', ')}`)
101
+ }
102
+ const repo = [...repos][0] || 'unknown-repo'
103
+ // `${repo}:${suffix}`, the suffix being whatever follows the first
104
+ // `:` in configured_scan_id (`uuid:<github-id>`) — the
105
+ // human-meaningful half.
106
+ const displayName = `${repo}:${scanId.replace(/^[^:]+:/u, '')}`
107
+ scans.push({
108
+ displayName,
109
+ data: { type: 'security', source: 'codex-security', findings: records.map(rowToFinding) },
110
+ })
111
+ }
112
+ return scans
113
+ }
114
+
115
+ function rowToFinding(r) {
116
+ // First non-empty path only; the siblings of a `path1 | path2 | …`
117
+ // list are dropped.
118
+ const file = r.relevant_paths.split(' | ').map((s) => s.trim()).find(Boolean) || 'unknown'
119
+
120
+ // Title + description joined with a blank line so the table view's
121
+ // first-line title shows the headline and the expanded view shows
122
+ // the full body.
123
+ const description = [r.title, r.description].filter(Boolean).join('\n\n')
124
+
125
+ const finding = {
126
+ // finding_url is unique per upstream finding, so triage keys off it
127
+ // and survives a reload. The saver takes any non-numeric id, URLs
128
+ // included (triage.js).
129
+ id: r.finding_url,
130
+ file,
131
+ // Codex CSVs lack line numbers — '?' is the same placeholder
132
+ // markdown findings use when the source has no `#L<n>` anchor.
133
+ line: '?',
134
+ severity: (r.severity || 'medium').toLowerCase(),
135
+ description,
136
+ repo: { github: r.repository },
137
+ // No per-finding `type`: the CSV carries no category column, and a
138
+ // synthetic 'security' would print the same word on every run-meta
139
+ // line with nothing to tell the findings apart. An empty run-meta
140
+ // renders as no line at all. The report-level `type` still defaults
141
+ // to 'security' for document.title.
142
+ }
143
+ if (r.commit_hash) finding.commitHash = r.commit_hash
144
+ if (r.detected_at) finding.detectedAt = r.detected_at
145
+ if (r.committed_at) finding.committedAt = r.committed_at
146
+ return finding
147
+ }
@@ -0,0 +1,197 @@
1
+ // Vercel DeepSec markdown findings parser. Per-finding shape differs
2
+ // from parse-md.js (Claude Security):
3
+ //
4
+ // # Vulnerability Scan Report
5
+ // …project metadata table… / ## Summary …summary table…
6
+ //
7
+ // ## HIGH (2)
8
+ //
9
+ // ### Finding title 1
10
+ //
11
+ // - **File:** `path/file.js`
12
+ // - **Recent committers:** … (ignored)
13
+ // - **Lines:** 26, 28
14
+ // - **Slug:** rule-slug
15
+ // - **Confidence:** high
16
+ // - **Revalidation:** confirmed (only where the pass ran)
17
+ // - **Reasoning:** what it concluded (only where the pass ran)
18
+ //
19
+ // prose body…
20
+ //
21
+ // **Recommendation:** recommendation text
22
+ //
23
+ // ---
24
+ // ### Finding title 2 … ## MEDIUM (5) …
25
+ //
26
+ // The writer is `packages/deepsec/src/commands/report.ts` in
27
+ // vercel-labs/deepsec; the shape above is settled there.
28
+ //
29
+ // Returns `{ type, source: 'deepsec', findings }`, or null when no
30
+ // `## SEVERITY (n)` header appears and the chain moves on.
31
+
32
+ import { normalizeNewlines, splitHeadingLine } from './md-structure.js'
33
+
34
+ // The `## SEVERITY (n)` header that marks a DeepSec document. Splitting
35
+ // on it with the tier captured interleaves tiers and content:
36
+ // [preamble, sevA, contentA, sevB, …].
37
+ const SECTION_RE = /^## ([A-Z][A-Z_]*)\s*\(\d+\)\s*\n/mu
38
+
39
+ // DeepSec's tiers onto the internal ladder. It separates vulnerabilities
40
+ // (CRITICAL … LOW) from non-vuln defects (HIGH_BUG, BUG), and
41
+ // `high_bug` / `bug` keep that apart so the chips count them
42
+ // separately. Anything else falls back to medium, where a renamed or
43
+ // new tier stays visible instead of vanishing.
44
+ function mapSeverity(s) {
45
+ switch (s.toUpperCase()) {
46
+ case 'CRITICAL': return 'critical'
47
+ case 'HIGH': return 'high'
48
+ case 'MEDIUM': return 'medium'
49
+ case 'LOW': return 'low'
50
+ case 'HIGH_BUG': return 'high_bug'
51
+ case 'BUG': return 'bug'
52
+ case 'INFO': case 'INFORMATIONAL': return 'informational'
53
+ default: return 'medium'
54
+ }
55
+ }
56
+
57
+ // A field's value as the word it names, whatever punctuation it arrived
58
+ // in — the writer's `~~false positive~~`, or a hand-edited document's
59
+ // backticks and emphasis.
60
+ const word = (s) => String(s ?? '').toLowerCase().replaceAll(/[^a-z]+/gu, '')
61
+
62
+ // DeepSec's `high` / `medium` / `low` is a closed enum its schema
63
+ // validates, and what the words MEAN is nowhere: the investigate prompt
64
+ // asks for one of the three without saying what separates them, the docs
65
+ // call it "the agent's self-rated confidence", and nothing in DeepSec
66
+ // reads it back. So there is no probability to convert, only three rungs
67
+ // to place on the app's 0—10 scale — where 0 is a claim the revalidation
68
+ // pass withdrew, 10 the no-doubt an unscored import rides at, and a
69
+ // fresh load opens on a floor of 6, 7 or 8 by volume, then walks down
70
+ // through any gap that reveals nothing new (ui filters.js).
71
+ //
72
+ // So `high` clears every floor the tune can pick without claiming the
73
+ // app's no-doubt 10; `medium` is the lowest of those floors, surviving a
74
+ // small report's opening view and dropping out of a big one's; `low`
75
+ // sits under every floor but clear of the 0 that means refuted, since
76
+ // the agent still chose to report it. The even spacing carries as much —
77
+ // the walk settles in the GAPS, one step under the lowest rung it keeps,
78
+ // so a rung packed tighter leaves it nowhere to stop and a rung moved
79
+ // without its gap puts that tier off screen at open.
80
+ const CONFIDENCE = new Map([['high', 8], ['medium', 6], ['low', 4]])
81
+
82
+ // An unknown word reads as the middle rung, for the reason an
83
+ // unrecognized severity falls back to medium: a level DeepSec adds later
84
+ // should neither vanish under the floor nor — as scoring it nothing
85
+ // would — ride the unscored stand-in at 10, above every `high`. A block
86
+ // with no `Confidence:` line rated nothing, and there that stand-in is
87
+ // the honest answer.
88
+ function mapConfidence(s) {
89
+ if (s === undefined) return undefined
90
+ return CONFIDENCE.get(word(s)) ?? CONFIDENCE.get('medium')
91
+ }
92
+
93
+ // The verdict of DeepSec's revalidation pass as its writer spells it —
94
+ // `confirmed`, `~~false positive~~` struck through, `uncertain` for
95
+ // everything else it can answer — onto the app's own outcomes
96
+ // (finding.js REVALIDATE_KINDS).
97
+ //
98
+ // It belongs with the confidence question rather than beside it:
99
+ // `Confidence:` is the INVESTIGATE pass's self-rating, written before
100
+ // the adversarial pass looked at the finding, and `refuted` is the
101
+ // outcome that acts on the number — the range reads a ruled-out row as
102
+ // 0 whatever it claims. A report saying `high` on one line and
103
+ // `~~false positive~~` on the next is not a finding to show at 8/10.
104
+ const REVALIDATION = new Map([
105
+ ['confirmed', 'confirmed'],
106
+ ['falsepositive', 'refuted'],
107
+ ['uncertain', 'unknown'],
108
+ ])
109
+
110
+ export function parseDeepsecFindings(content) {
111
+ const text = normalizeNewlines(content).trim()
112
+ // Format guard — without a single `## SEVERITY (n)` header this isn't
113
+ // a DeepSec doc; bail out so the chain moves on to
114
+ // parseMarkdownFindings.
115
+ const parts = text.split(SECTION_RE)
116
+ if (parts.length === 1) return null
117
+
118
+ const findings = []
119
+ for (let i = 1; i < parts.length; i += 2) {
120
+ const sev = mapSeverity(parts[i])
121
+ // Each finding inside a severity section starts with `### Title`.
122
+ for (const block of parts[i + 1].split(/^### /mu).slice(1)) {
123
+ const f = parseBlock(block, sev)
124
+ if (f) findings.push(f)
125
+ }
126
+ }
127
+ if (findings.length === 0) return null
128
+
129
+ // Report-level 'security' for the document.title fallback. No
130
+ // per-finding `type`: DeepSec categorizes by severity alone, as codex
131
+ // does, and ingest.js's `data.source` gate keeps the report-level one
132
+ // off the findings.
133
+ return { type: 'security', source: 'deepsec', findings }
134
+ }
135
+
136
+ function parseBlock(block, severity) {
137
+ // The `---` separator after each finding in a section is shed.
138
+ const { title, body: rawBody } = splitHeadingLine(block)
139
+ if (!title) return null
140
+ const body = rawBody.replace(/\n---\s*$/u, '').trim()
141
+
142
+ // Bullet metadata: `- **Field:** value`. Field names case-folded.
143
+ const fields = {}
144
+ for (const m of body.matchAll(/^- \*\*([^:*]+):\*\*\s*(.+)$/gmu)) {
145
+ fields[m[1].trim().toLowerCase()] = m[2].trim()
146
+ }
147
+
148
+ // A bold inline label in the body, not a `## Recommended fix` H2 as
149
+ // Claude Security writes — so the split is there.
150
+ const recMatch = /^\*\*Recommendation:\*\*\s*/mu.exec(body)
151
+ let prose = body
152
+ let recommendation = ''
153
+ if (recMatch) {
154
+ prose = body.slice(0, recMatch.index)
155
+ recommendation = body.slice(recMatch.index + recMatch[0].length).trim()
156
+ }
157
+
158
+ // Prose minus the bullet metadata, with `**bold**` stripped — the
159
+ // renderer escapes HTML, so the markers would print literally.
160
+ const description = prose
161
+ .split('\n')
162
+ .filter((line) => !/^\s*- \*\*/u.test(line))
163
+ .join('\n')
164
+ .replaceAll('**', '')
165
+ .trim()
166
+
167
+ // Title first, as parse-md and parse-codex write it, so the table
168
+ // view's first line is the headline.
169
+ const fullDescription = [title, description].filter(Boolean).join('\n\n')
170
+
171
+ // The path arrives backticked (`path/file.js`); the backticks are
172
+ // notation, not part of it.
173
+ const file = (fields.file || 'unknown').replace(/^`(.*)`$/u, '$1')
174
+ // First non-empty line only — the renderer takes a single `f.line`,
175
+ // and lineLink wraps it as a `#L<n>` anchor when a fileUrl is
176
+ // available. The siblings of a `26, 28` list are dropped.
177
+ const line = (fields.lines || '').split(',').map((s) => s.trim()).find(Boolean) || '?'
178
+
179
+ const finding = { file, line, severity, description: fullDescription }
180
+ if (recommendation) finding.recommendation = recommendation.replaceAll('**', '')
181
+ const confidence = mapConfidence(fields.confidence)
182
+ if (confidence !== undefined) finding.confidence = confidence
183
+ // What the pass concluded, where the report has been through it: the
184
+ // verdict as one of the app's outcomes, the reasoning under it as the
185
+ // pass's remark, DeepSec named as whose pass said so. The two are read
186
+ // as a pair because the document writes them as one — a `Reasoning:`
187
+ // line is the pass's, not the finding's. First line only, like every
188
+ // field here; a wrapped remainder stays in the prose where it was.
189
+ const revalidate = REVALIDATION.get(word(fields.revalidation))
190
+ if (revalidate) {
191
+ finding.revalidate = revalidate
192
+ finding.revalidateSource = 'deepsec'
193
+ if (fields.reasoning) finding.revalidateVerdict = fields.reasoning.replaceAll('**', '')
194
+ }
195
+ if (fields.slug) finding.slug = fields.slug
196
+ return finding
197
+ }