@preventive/triage 1.0.0-alpha.13 → 1.0.0-alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/out/brotli-fallback.js +3 -3
  2. package/out/client-admin.js +2 -2
  3. package/out/client-sync.js +14 -14
  4. package/out/graph.js +30 -5
  5. package/out/index.html +47 -7
  6. package/out/prism.js +2 -2
  7. package/out/stasis.svg +45 -0
  8. package/out/terminal.js +255 -43
  9. package/out/view.css +1 -1
  10. package/out/view.js +147 -110
  11. package/package.json +27 -4
  12. package/report/index.js +253 -0
  13. package/report/src/finding-id.js +80 -0
  14. package/report/src/finding.js +300 -0
  15. package/report/src/labels.js +33 -0
  16. package/report/src/md-structure.js +471 -0
  17. package/report/src/md-text.js +167 -0
  18. package/report/src/meta.js +51 -0
  19. package/report/src/parse-codex.js +147 -0
  20. package/report/src/parse-deepsec.js +197 -0
  21. package/report/src/parse-deepview-fields.js +375 -0
  22. package/report/src/parse-deepview-md.js +185 -0
  23. package/report/src/parse-md-id.js +137 -0
  24. package/report/src/parse-md.js +253 -0
  25. package/report/src/parse-piolium-id.js +79 -0
  26. package/report/src/parse-piolium-rows.js +131 -0
  27. package/report/src/parse-piolium-tokens.js +175 -0
  28. package/report/src/parse-piolium.js +400 -0
  29. package/report/src/utf8.js +21 -0
  30. package/report/src/write-md-finding.js +273 -0
  31. package/report/src/write-md.js +291 -0
  32. package/server-e2e/bus-receiver.ts +1 -0
  33. package/server-e2e/hub.ts +37 -6
  34. package/server-e2e/index.ts +3 -2
  35. package/server-e2e/objstore/handlers.ts +6 -5
  36. package/server-e2e/objstore/init.ts +1 -1
  37. package/server-e2e/objstore/rest-mint.ts +2 -2
  38. package/server-e2e/objstore/rest.ts +2 -2
  39. package/server-e2e/pubsub.ts +8 -5
  40. package/server-e2e/sync-handlers.ts +54 -28
@@ -0,0 +1,175 @@
1
+ // Token-level helpers for the Piolium report parser: severity words,
2
+ // finding-id shapes, heading forms, and field-name aliases. Split from
3
+ // parse-piolium.js, which owns the document structure; everything here
4
+ // is a pure string classifier.
5
+
6
+ import { isCommitHash, stripBrackets } from './md-structure.js'
7
+ import { isRepoSlug } from './meta.js'
8
+
9
+ // Piolium grades findings CRITICAL / HIGH / MEDIUM — its assembler
10
+ // rejects Low-severity leakage into `findings/` — but drafts and
11
+ // deferred entries can carry LOW or INFO, so the full ladder is mapped.
12
+ // The call sites fall back to medium, keeping an odd tier visible.
13
+ //
14
+ // Only the first token is read: a bullet value keeps its continuation
15
+ // lines and may carry a parenthetical ("CRITICAL (raised after the PoC
16
+ // ran)"), while the tier is one word. Backticks and asterisks are shed,
17
+ // so `**CRITICAL**` reads.
18
+ export function mapSeverity(s) {
19
+ const first = ((s || '').trim().split(/\s+/u)[0] || '').replaceAll(/[`*]+/gu, '')
20
+ switch (first.toUpperCase()) {
21
+ case 'CRITICAL': return 'critical'
22
+ case 'HIGH': return 'high'
23
+ case 'MEDIUM': return 'medium'
24
+ case 'LOW': return 'low'
25
+ case 'INFO': case 'INFORMATIONAL': return 'informational'
26
+ default: return ''
27
+ }
28
+ }
29
+
30
+ // Final-report ids are severity-prefixed and sequential — `C1`, `H2`,
31
+ // `H-001` in lite consolidation — so the prefix is a second source for
32
+ // the tier. Only that exact scheme counts: a bare leading letter would
33
+ // read `CVE-2024-1234` as critical, and a draft id carries no tier.
34
+ export function severityFromId(id) {
35
+ const m = /^([CHML])-?\d+$/iu.exec((id || '').trim())
36
+ if (!m) return ''
37
+ return { C: 'critical', H: 'high', M: 'medium', L: 'low' }[m[1].toUpperCase()]
38
+ }
39
+
40
+ // A heading that IS a severity — `Critical`, `HIGH (2)`, `Critical
41
+ // Severity`, `Medium-Risk Findings (3)` — marks a GROUP of that tier's
42
+ // findings. Anchored to the whole heading, so "High memory usage in
43
+ // parser" is never mistaken for one.
44
+ export function severityGroupOf(heading) {
45
+ const m = /^(critical|high|medium|low|informational|info)(?:[ -](?:severity|risk))?(?:[ -]findings?)?(?:\s*\(\d+\))?$/iu
46
+ .exec((heading || '').trim())
47
+ return m ? mapSeverity(m[1]) : ''
48
+ }
49
+
50
+ // A leading severity word on a free-form header — `HIGH — 3 findings`,
51
+ // `High: remaining` — for sections recognized by their CONTENT rather
52
+ // than the anchored severityGroupOf shape.
53
+ export function headerSeverity(header) {
54
+ const m = /^(critical|high|medium|low|informational|info)\b/iu.exec((header || '').trim())
55
+ return m ? mapSeverity(m[1]) : ''
56
+ }
57
+
58
+ // A variants heading (`#### Variants`, `### Variants (2)`), not a
59
+ // finding — matched wherever finding headings are read.
60
+ export function isVariantsHeading(heading) {
61
+ return /^variants?\s*(?:\(\d+\))?\s*:?$/iu.test((heading || '').trim())
62
+ }
63
+
64
+ // A token as a piolium finding id, with the directory slug when it
65
+ // carries one. Two schemes: severity-prefixed final ids (`C1`, `H-001`,
66
+ // `C1-command-injection`) and draft-phase ids from the analysis phases
67
+ // (`p10-011`, `q1-001`, `diff-003`). The phase letters are a closed set
68
+ // with a 2+ digit sequence, so prose tokens (`UTF-8`, `SHA-256`) never
69
+ // read as ids. Upper-cased, so `[c1]` meets its `[C1]` index row.
70
+ export function idFromToken(token) {
71
+ const m = /^([CHML]-?\d{1,4}|(?:p|q|b|r|m|l|x|diff)\d{0,4}-\d{2,4})(?:-([A-Za-z0-9][\w-]*))?$/iu
72
+ .exec(token || '')
73
+ return m ? { id: m[1].toUpperCase(), slug: m[2] || '' } : null
74
+ }
75
+
76
+ // `command-injection` → `command injection` — the human-readable title
77
+ // recovered from an <id>-<slug> directory-name reference.
78
+ export function slugTitle(slug) {
79
+ return (slug || '').replaceAll('-', ' ')
80
+ }
81
+
82
+ // An id cell in any of its spellings — `C1`, `[C1]`,
83
+ // `[p12-001](#p12-001)` — as the upper-case id.
84
+ export function idCell(s) {
85
+ const v = (s || '').trim()
86
+ const link = /^\[([^\]]+)\]\([^)]*\)$/u.exec(v)
87
+ return stripBrackets(link ? link[1] : v).toUpperCase()
88
+ }
89
+
90
+ // A heading or item leading with a link —
91
+ // `[C1-command-injection](…/report.md): Title` — as plain text with the
92
+ // url apart: `{ text: 'C1-command-injection Title', link }`.
93
+ export function leadingLink(value) {
94
+ const m = /^\[([^\]]+)\]\(([^)]+)\)\s*[:—–-]*\s*(.*)$/u.exec(value)
95
+ if (!m) return null
96
+ return { text: m[3] ? `${m[1].trim()} ${m[3].trim()}` : m[1].trim(), link: m[2].trim() }
97
+ }
98
+
99
+ // `text` split at its first whitespace when the leading token is an id —
100
+ // `p10-011 — Title`, `C1: Title` — as `{ id, slug, rest }`, trailing
101
+ // punctuation shed from the token and the separator from the rest.
102
+ export function leadingId(text) {
103
+ const space = text.search(/\s/u)
104
+ const first = (space === -1 ? text : text.slice(0, space)).replace(/[:.,—–-]+$/u, '')
105
+ const tok = idFromToken(first)
106
+ if (!tok) return null
107
+ const rest = (space === -1 ? '' : text.slice(space + 1)).replace(/^[:—–-]+\s*/u, '').trim()
108
+ return { id: tok.id, slug: tok.slug, rest }
109
+ }
110
+
111
+ // A finding heading in any of its observed spellings:
112
+ // `[C1] Title` (pentest template)
113
+ // `[C1-command-injection](url)` (mode outline: linked dir name)
114
+ // `C1-command-injection` (bare dir name)
115
+ // `p10-011 — Title` / `C1: Title` (id + separator + title)
116
+ // `Title` (bare title)
117
+ // Returns { id, title, link } — id '' when the heading carries none,
118
+ // link '' unless the heading's leading token is a markdown link.
119
+ export function parseHeading(headingText) {
120
+ const { text, link } = leadingLink(headingText) ?? { text: headingText, link: '' }
121
+ const bracket = /^\[([^\]]+)\] *(.*)$/u.exec(text)
122
+ if (bracket) {
123
+ const tok = idFromToken(bracket[1].trim())
124
+ if (tok) return { id: tok.id, title: bracket[2].trim() || slugTitle(tok.slug) || tok.id, link }
125
+ // Non-id bracket content is still a usable dedupe key for the
126
+ // seen-set, unrecognized scheme and all (`[SEC-001]`).
127
+ return { id: bracket[1].trim().toUpperCase(), title: bracket[2].trim(), link }
128
+ }
129
+ const lead = leadingId(text)
130
+ if (lead) return { id: lead.id, title: lead.rest || slugTitle(lead.slug) || lead.id, link }
131
+ return { id: '', title: text.trim(), link }
132
+ }
133
+
134
+ // The document preamble (before the first `## ` section) carries the
135
+ // audit's run metadata as `**Label** value` lines:
136
+ //
137
+ // # Security Audit Report: owner/repo
138
+ // **Target** `owner/repo` (description)
139
+ // **Commit audited** `<sha>` (prose)
140
+ // **Audit ID** `…` · **Mode** deep (17-phase) · **Report assembled** …
141
+ //
142
+ // `**Target**` declares the audited repository, where the H1 title's
143
+ // bare <project> may be a monorepo path and isn't trusted. The value is
144
+ // the first backtick span or leading token, taken only in strict
145
+ // `owner/repo` shape, the commit only as plain hex. Audit ID and Mode
146
+ // are run bookkeeping with no consumer.
147
+ export function preambleMeta(head) {
148
+ const meta = {}
149
+ const value = (rest) => (/`([^`]+)`/u.exec(rest)?.[1] ?? rest.split(/\s+/u)[0] ?? '').trim()
150
+ const target = /^\s*(?:[-*] +)?\*\*Target:?\*\*\s*(.*)$/imu.exec(head || '')
151
+ if (target) {
152
+ const v = value(target[1])
153
+ if (isRepoSlug(v)) meta.repo = v
154
+ }
155
+ const commit = /^\s*(?:[-*] +)?\*\*Commit[^:*]*:?\*\*\s*(.*)$/imu.exec(head || '')
156
+ if (commit) {
157
+ const v = value(commit[1])
158
+ if (isCommitHash(v)) meta.commitHash = v
159
+ }
160
+ return meta
161
+ }
162
+
163
+ // "Key Code Reference" is the assembler's name for a finding's code
164
+ // location, which real reports shorten and reword, so every observed
165
+ // spelling is accepted, most specific first. Deliberately absent:
166
+ // `files`, which reports use for reproduction attachments, and bare
167
+ // `code`, which matches PoC-code fields.
168
+ export const CODE_REF_FIELDS = [
169
+ 'key code reference', 'key code', 'code reference', 'location',
170
+ 'affected file', 'file', 'path',
171
+ ]
172
+ export function codeRefOf(fields) {
173
+ for (const k of CODE_REF_FIELDS) if (fields[k]) return fields[k]
174
+ return ''
175
+ }
@@ -0,0 +1,400 @@
1
+ // Piolium markdown findings parser. Piolium is Vigolium's agentic
2
+ // repository audit agent (https://github.com/vigolium/piolium), writing
3
+ // its artifacts under `piolium/` in the audited repo. The only file
4
+ // read here is the CONSOLIDATED run report,
5
+ // `piolium/final-audit-report.md`; the per-finding
6
+ // `piolium/findings/<id>-<slug>/report.md` files are deliberately not
7
+ // an input — one file per finding doesn't fit the one-file-per-report
8
+ // model, and the consolidated report already inlines or links each one.
9
+ //
10
+ // The report is COMPOSED BY AN AGENT, so its structure varies by mode
11
+ // and by run. Three observed layouts anchor the parser; everything else
12
+ // is handled by being liberal within them.
13
+ //
14
+ // Layout A — the pentest template: a `## Summary of Findings` index
15
+ // table (`| [C1] | Title | CRITICAL | executed | -- |`) plus
16
+ // `## Technical Findings Detail` with `### [C1] Title` blocks of
17
+ // `- **Severity:** / **Summary:** / **Impact:** / **Root Cause:** /
18
+ // **Key Code Reference:** / **PoC Status:**` bullets and an optional
19
+ // `#### Variants` sub-table.
20
+ //
21
+ // Layout B — the mode task outline (modes/balanced.ts L6c,
22
+ // modes/deep.ts P15): `## Findings by Severity` with severity groups
23
+ // (`### Critical`, counted `### HIGH (2)`, or promoted to
24
+ // `## Critical Findings`) whose findings are `#### ` blocks, an
25
+ // id/title table, or a `- [<id>-<slug>](…/report.md): summary` list.
26
+ //
27
+ // Layout C — real assembler output: anchored draft-phase ids and
28
+ // per-variant entries,
29
+ //
30
+ // <a id="p10-011"></a>
31
+ // ### p10-011 — Title
32
+ //
33
+ // - **Severity:** HIGH
34
+ // - **Key code:** `src/a.js:20` (`fnA`) → `src/b.js:600` → `src/c.js`
35
+ // - **PoC:** executed (…)
36
+ // - **Files:** …reproduction attachments, ignored…
37
+ //
38
+ // #### Variants
39
+ // | ID | Title | Severity | Location | PoC |
40
+ // |----|-------|----------|----------|-----|
41
+ // | [p12-001](#p12-001) | Variant title | MEDIUM | `src/d.js:50-60` | executed |
42
+ //
43
+ // <a id="p12-001"></a>
44
+ // #### p12-001 — Variant title
45
+ // - **Variant of** [p10-011](#p10-011) · **Pattern** `pattern-id`
46
+ //
47
+ // Variants exist BOTH as table rows and as their own full entries; the
48
+ // entry carries the narrative and wins, so rows are deferred and
49
+ // emitted only for ids no entry covered — never twice, and never as a
50
+ // finding titled "Variants". Rows are also registered as index rows, so
51
+ // an entry adopts its row's severity / PoC / parent.
52
+ //
53
+ // Returns `{ type, source: 'piolium', findings }`, or null when the
54
+ // text isn't this format.
55
+ //
56
+ // Deliberately NOT findings: `## Methodology Summary` / `Notes` (a
57
+ // finding's bullet shape, but about the run), `## Attack Surface
58
+ // Summary` and `## Coverage Gaps` (link lists about the audit), and
59
+ // `## Deferred Findings (triage skip)` (drafts triage did not promote).
60
+ // Only findings-labelled sections, severity groups and the index are
61
+ // read — and structural markdown only OUTSIDE fenced code, since
62
+ // piolium inlines PoC snippets and a fenced `## step 2` must not end a
63
+ // section (md-structure.js).
64
+
65
+ import { H2_RE, H3_RE, H4_RE, normalizeNewlines, parseCodeRef, parseLabelledFields, splitByHeading, splitLeading, tableObjects } from './md-structure.js'
66
+ import { frozenIdBasis } from './parse-piolium-id.js'
67
+ import { fromIndexRow, indexRowOf, listFindings, variantFindings } from './parse-piolium-rows.js'
68
+ import {
69
+ CODE_REF_FIELDS, codeRefOf, headerSeverity, idCell, idFromToken,
70
+ isVariantsHeading, mapSeverity, parseHeading, preambleMeta,
71
+ severityFromId, severityGroupOf,
72
+ } from './parse-piolium-tokens.js'
73
+
74
+
75
+ // Section headers whose body holds the findings. Deliberate non-matches:
76
+ // 'summary of findings' (the index, read separately), the excluded
77
+ // appendices, and the prose/link sections.
78
+ const DETAIL_HEADERS = new Set([
79
+ 'technical findings detail', 'technical findings',
80
+ 'detailed findings', 'findings detail', 'findings',
81
+ ])
82
+ function isDetailHeader(header) {
83
+ return DETAIL_HEADERS.has(header) || header.startsWith('findings by severity')
84
+ }
85
+
86
+ // Never mined for findings, even carrying id-shaped headings or tables.
87
+ const EXCLUDED_HEADERS = /^(?:summary of findings|deferred|methodolog|executive|conclusion|attack surface|coverage|discoveries|scope|table of contents|contents|appendix|recommendation|remediation)/u
88
+ function isExcludedHeader(header) {
89
+ return EXCLUDED_HEADERS.test(header)
90
+ }
91
+
92
+ // Section names vary run to run ('## HIGH — 3 findings', '## Confirmed
93
+ // Findings', emoji prefixes), so a non-excluded section whose headings
94
+ // carry id-shaped tokens holds findings whatever it is called.
95
+ function headingHasId(heading) {
96
+ const { id } = parseHeading(heading)
97
+ return Boolean(id && idFromToken(id))
98
+ }
99
+ function hasIdBlocks(body) {
100
+ const blocks = splitByHeading(body, H3_RE)
101
+ const list = blocks.length > 0 ? blocks : splitByHeading(body, H4_RE)
102
+ return list.some(({ heading }) => headingHasId(heading))
103
+ }
104
+
105
+ export function parsePioliumFindings(content) {
106
+ const text = normalizeNewlines(content).trim()
107
+ // Any one signal is enough: a project can retitle the H1, and a
108
+ // hand-trimmed report can drop the prose sections and keep the
109
+ // findings. Requiring none would steal plain `# Title` documents from
110
+ // parse-md.js, which accepts any h1-led markdown.
111
+ if (!/^# +Security Audit Report\b/mu.test(text)
112
+ && !/^## +Technical Findings Detail\s*$/imu.test(text)
113
+ && !/^## +Findings by Severity\b/imu.test(text)) return null
114
+
115
+ const sections = parseSections(text)
116
+ const index = parseIndexTable(sections['summary of findings'] || '')
117
+ const meta = preambleMeta(splitLeading(text, H2_RE).head)
118
+
119
+ const findings = []
120
+ const seen = new Set()
121
+ // Variant rows wait here until the document is read out: a row is
122
+ // emitted only where no entry claimed its id.
123
+ const pending = []
124
+ const push = ({ id, finding }) => {
125
+ findings.push(finding)
126
+ if (id) seen.add(id)
127
+ }
128
+ const emit = (entries) => { for (const entry of entries) push(entry) }
129
+
130
+ for (const [header, body] of Object.entries(sections)) {
131
+ // A findings section: a severity group promoted to section level
132
+ // (`## Critical Findings`), a findings-labelled header, or the
133
+ // content-based fallback for every other spelling a run invents,
134
+ // where a leading severity word still supplies the tier.
135
+ const groupSev = severityGroupOf(header)
136
+ if (groupSev || isDetailHeader(header)) {
137
+ emit(parseFindingsBody(body, groupSev, index, pending))
138
+ } else if (!isExcludedHeader(header) && hasIdBlocks(body)) {
139
+ emit(parseFindingsBody(body, headerSeverity(header), index, pending))
140
+ }
141
+ }
142
+
143
+ for (const entry of pending) {
144
+ if (!entry.id || !seen.has(entry.id)) push(entry)
145
+ }
146
+
147
+ // Anything the index lists but no block described, from the row
148
+ // alone: the index is the authoritative list, so a report with a
149
+ // truncated detail section still triages every finding.
150
+ for (const row of index.values()) {
151
+ if (!seen.has(row.id)) push({ id: row.id, finding: fromIndexRow(row) })
152
+ }
153
+
154
+ if (findings.length === 0) return null
155
+
156
+ // The preamble's `**Target**` names the audited repository, stamped
157
+ // per finding (repo.github is per-finding downstream) with its own
158
+ // object copy. The H1 title alone is not trusted: its <project> holds
159
+ // a monorepo path as easily as a slug, and a wrong `repo.github` is
160
+ // worse than none — format.js's fileUrl prefers it over the editable
161
+ // repo chip, so a bad guess yields dead links the user can't correct.
162
+ // `**Commit audited**` lands as `auditedCommit`, NOT `commitHash`,
163
+ // which the card renders as "introduced in <commit>": the scan commit
164
+ // says where the audit ran, not where the bug landed.
165
+ for (const f of findings) {
166
+ if (meta.repo) f.repo = { github: meta.repo }
167
+ if (meta.commitHash && !f.auditedCommit) f.auditedCommit = meta.commitHash
168
+ }
169
+
170
+ // Report-level 'security' for the document.title fallback. No
171
+ // per-finding `type` — piolium categorizes by severity, so a
172
+ // synthetic one would print the same word on every run-meta line, the
173
+ // call parse-deepsec.js and parse-codex.js also make — and ingest.js's
174
+ // `data.source` gate keeps the report-level one off the findings.
175
+ return { type: 'security', source: 'piolium', findings }
176
+ }
177
+
178
+ // The `### ` blocks of a findings section or section-level severity
179
+ // group, tier in `sev`: each a finding, a severity group of its own, or
180
+ // a `### Variants` block parented to the block before it.
181
+ function parseDetailBlocks(blocks, index, pending, sev) {
182
+ const out = []
183
+ let lastId = ''
184
+ for (const { heading, body } of blocks) {
185
+ const groupSev = severityGroupOf(heading)
186
+ if (groupSev) {
187
+ out.push(...parseFindingsBody(body, groupSev, index, pending))
188
+ lastId = ''
189
+ } else if (isVariantsHeading(heading)) {
190
+ out.push(...parseVariantsBlock(body, index, lastId, pending, sev))
191
+ } else {
192
+ const entries = parseFindingBlock(heading, body, index, pending, sev)
193
+ out.push(...entries)
194
+ lastId = entries[0]?.id || lastId
195
+ }
196
+ }
197
+ return out
198
+ }
199
+
200
+ // A `### ` finding block: the head, before any `#### `, is the finding.
201
+ // A `#### Variants` sub-heading defers its rows to `pending`; any other
202
+ // is a full entry of its own, as Layout C writes each variant.
203
+ function parseFindingBlock(heading, body, index, pending, sev) {
204
+ const { head, subs } = splitLeading(body, H4_RE)
205
+ const parent = parseBlock(heading, head, index, sev)
206
+ // A `### ` heading with no id and no index row, over id-shaped
207
+ // `#### ` entries, is a CATEGORY grouping: the entries are the
208
+ // findings, and emitting the heading would add one title-only result
209
+ // per category.
210
+ const isCategory = parent !== null && !parent.id
211
+ && subs.some((s) => !isVariantsHeading(s.heading) && headingHasId(s.heading))
212
+ const out = parent && !isCategory ? [parent] : []
213
+ out.push(...parseEntries(subs, index, pending, sev, parent?.id || ''))
214
+ return out
215
+ }
216
+
217
+ // `### Variants` as its own block: tables defer to pending, `#### <id>`
218
+ // sub-blocks are the variants' entries. `parentId` is the block before
219
+ // this one, the structural parent for rows naming none.
220
+ function parseVariantsBlock(body, index, parentId, pending, sev) {
221
+ const { head, subs } = splitLeading(body, H4_RE)
222
+ pending.push(...variantFindings(head, index, parentId, sev))
223
+ return parseEntries(subs, index, pending, sev, '')
224
+ }
225
+
226
+ // The `#### ` entries of a finding block, a `### Variants` block or a
227
+ // severity group — each a finding, except a `#### Variants` table, whose
228
+ // rows defer to `pending` under `parentId`, or under the entry before
229
+ // the table when they are siblings at group level.
230
+ function parseEntries(subs, index, pending, sev, parentId) {
231
+ const out = []
232
+ let lastId = ''
233
+ for (const { heading, body } of subs) {
234
+ if (isVariantsHeading(heading)) {
235
+ pending.push(...variantFindings(body, index, parentId ?? lastId, sev))
236
+ continue
237
+ }
238
+ const entry = parseBlock(heading, body, index, sev)
239
+ if (!entry) continue
240
+ out.push(entry)
241
+ lastId = entry.id || lastId
242
+ }
243
+ return out
244
+ }
245
+
246
+ // The body of a findings section or severity group, holding that tier's
247
+ // findings in whichever rendering the assembler chose. A body with
248
+ // `### ` blocks routes through parseDetailBlocks, and `### ` MUST win
249
+ // over `#### ` there, or one `#### Variants` would swallow every `### `
250
+ // sibling as its body. Otherwise the findings are `#### ` sub-blocks, an
251
+ // id/title table, or a list. Prose alone ("None identified.") yields
252
+ // nothing.
253
+ function parseFindingsBody(body, sev, index, pending) {
254
+ const h3 = splitByHeading(body, H3_RE)
255
+ if (h3.length > 0) return parseDetailBlocks(h3, index, pending, sev)
256
+
257
+ const out = parseEntries(splitByHeading(body, H4_RE), index, pending, sev, null)
258
+ if (out.length > 0) return out
259
+
260
+ // An id/title table here is the INDEX in another position — the real
261
+ // reports put the overview under `## Findings by Severity` and the
262
+ // blocks under per-severity sections — so emitting rows eagerly would
263
+ // double-report every finding. They merge into the index instead, and
264
+ // the gated fallback emits only ids no block claimed. A row with no id
265
+ // can't be index-keyed and defers via pending, as list items do.
266
+ const rows = tableObjects(body).map(indexRowOf).filter((r) => r.id || r.title)
267
+ if (rows.length > 0) {
268
+ for (const r of rows) {
269
+ if (!r.severity && sev) r.severity = sev
270
+ if (r.id && !index.has(r.id)) index.set(r.id, r)
271
+ else if (!r.id) pending.push({ id: '', finding: fromIndexRow(r, sev) })
272
+ }
273
+ return []
274
+ }
275
+ pending.push(...listFindings(body, sev, index))
276
+ return []
277
+ }
278
+
279
+ function parseBlock(heading, body, index, groupSeverity = '') {
280
+ const headingText = heading.trim()
281
+ if (!headingText) return null
282
+
283
+ let { id, title, link } = parseHeading(headingText)
284
+ let row = index.get(id)
285
+ // A bare-title heading ADOPTS the index row of the same title, so the
286
+ // block and the row are one finding; without that, the index fallback
287
+ // would emit it a second time.
288
+ if (!id && title) {
289
+ row = [...index.values()].find((r) => r.title.toLowerCase() === title.toLowerCase())
290
+ id = row?.id ?? ''
291
+ }
292
+
293
+ const { fields, labels, prose } = parseLabelledFields(body)
294
+
295
+ // Precedence: the block's own bullet, the index row, the enclosing
296
+ // group, the id's prefix, then medium — where an unrecognized tier
297
+ // stays visible rather than dropping out.
298
+ const severity = mapSeverity(fields.severity)
299
+ || mapSeverity(row?.severity)
300
+ || groupSeverity
301
+ || severityFromId(id)
302
+ || 'medium'
303
+
304
+ const ref = parseCodeRef(codeRefOf(fields))
305
+ // A `**Line:**` / `**Lines:**` bullet supplies the line when the
306
+ // reference itself carries none.
307
+ const lineBullet = /\d+/u.exec(fields.line || fields.lines || '')?.[0] ?? ''
308
+ const line = ref.line === '?' && lineBullet ? lineBullet : ref.line
309
+
310
+ // `- **Variant of** [p10-011](#p10-011) · …` names the parent, with or
311
+ // without the colon that would make it a labelled field; read off the
312
+ // raw body either way, and kept out of the description along with
313
+ // `<a id>` anchor chrome.
314
+ const variantOf = /\*\*Variant of:?\*\*\s*\[?([^\]\s)]+)/iu.exec(body)
315
+ const proseClean = prose.split('\n')
316
+ .filter((l) => !/^\s*<a\s[^>]*>\s*<\/a>\s*$/iu.test(l) && !/\*\*Variant of:?\*\*/iu.test(l))
317
+ .join('\n').trim()
318
+
319
+ const finding = {
320
+ file: ref.file || 'unknown',
321
+ line,
322
+ severity,
323
+ description: buildDescription(title || id, fields, labels, proseClean),
324
+ }
325
+ if (ref.locationLink) finding.location = ref.locationLink
326
+ // Last-resort fingerprint discriminator for an unlocated finding —
327
+ // see fromIndexRow for why.
328
+ else if (finding.file === 'unknown' && id) finding.location = `piolium:${id}`
329
+ // The id fingerprint is parse-piolium-id.js's own reading of the
330
+ // same reference, not the one above: what `parseCodeRef` makes of a
331
+ // reference is presentation and free to improve, the fingerprint is
332
+ // not. finding-id.js prefers `_idBasis` when deriving the uuid; read
333
+ // that module's header before touching either side.
334
+ finding._idBasis = frozenIdBasis({
335
+ severity, description: finding.description, ref: codeRefOf(fields), lineBullet, id,
336
+ })
337
+ // Auxiliary provenance, kept as plain strings so an export can cite
338
+ // the audit's own artifacts — as parse-md.js keeps branch / status.
339
+ const pocStatus = fields['poc status'] || fields.poc || row?.pocStatus
340
+ if (pocStatus) finding.pocStatus = pocStatus
341
+ const reportPath = fields['detailed report'] || (link.endsWith('report.md') ? link : '')
342
+ if (reportPath) finding.reportPath = reportPath
343
+ if (row?.status) finding.status = row.status
344
+ const parent = row?.parent || (variantOf ? idCell(variantOf[1]) : '')
345
+ if (parent) finding.parent = parent
346
+
347
+ return { id, finding }
348
+ }
349
+
350
+ // The `## ` sections, keyed case-folded. A repeated header CONCATENATES
351
+ // rather than overwrites, or concatenated runs (`cat a.md b.md`) and an
352
+ // index split across tables would keep only the last. Null-prototype, so
353
+ // a section named after an Object.prototype member aliases nothing.
354
+ function parseSections(text) {
355
+ const sections = Object.create(null)
356
+ for (const { heading, body } of splitByHeading(text, H2_RE)) {
357
+ const header = heading.trim().toLowerCase()
358
+ if (!header) continue
359
+ sections[header] = header in sections ? `${sections[header]}\n${body}` : body
360
+ }
361
+ return sections
362
+ }
363
+
364
+ // `## Summary of Findings` → id → row: both the gap-filler for a sparse
365
+ // block (PoC status, parent, verdict) and the source of last resort.
366
+ function parseIndexTable(text) {
367
+ const index = new Map()
368
+ for (const obj of tableObjects(text)) {
369
+ const row = indexRowOf(obj)
370
+ if (!row.id) continue
371
+ index.set(row.id, row)
372
+ }
373
+ return index
374
+ }
375
+
376
+ // Mechanical fields, which must not repeat into the description:
377
+ // severity / code-reference / PoC plumbing, attachments, cross-links.
378
+ // Everything ELSE a report labels — `Impact`, `Root cause`, or an
379
+ // invented `Residual risk` — is narrative and belongs in the story.
380
+ const NON_NARRATIVE_FIELDS = new Set([
381
+ ...CODE_REF_FIELDS, 'severity', 'summary', 'files', 'poc',
382
+ 'poc status', 'line', 'lines', 'detailed report', 'proof of concept',
383
+ 'evidence', 'variant of', 'pattern', 'status', 'id', 'title',
384
+ ])
385
+
386
+ // Heading + narrative in document order: the Summary, from a label or
387
+ // from unlabelled prose — both contribute, since a block often carries
388
+ // its labels first and its narrative in the paragraph under them — then
389
+ // every narrative label in its ORIGINAL casing, kept `**bold**`, which
390
+ // the card renders as real <strong> and the export re-emits as markdown.
391
+ function buildDescription(title, fields, labels, prose) {
392
+ const parts = [title]
393
+ if (fields.summary) parts.push(fields.summary)
394
+ if (prose) parts.push(prose)
395
+ for (const [k, v] of Object.entries(fields)) {
396
+ if (!v || NON_NARRATIVE_FIELDS.has(k)) continue
397
+ parts.push(`**${labels[k] || k}:** ${v}`)
398
+ }
399
+ return parts.filter(Boolean).join('\n\n')
400
+ }
@@ -0,0 +1,21 @@
1
+ // UTF-8 encoding for the id hashing — a copy of the app's
2
+ // `common/utf8.js` minus its decoding half, since nothing under
3
+ // `report/` imports from outside it. Both copies' tests assert the same
4
+ // cases byte for byte.
5
+ //
6
+ // Centralised rather than `new TextEncoder().encode(...)` per call site,
7
+ // because the WHATWG encoder silently replaces lone surrogates with
8
+ // U+FFFD: fine for display, a footgun where the bytes feed a hash, where
9
+ // the id comes out stable but hashed from a string nobody meant.
10
+
11
+ const encoder = new TextEncoder()
12
+
13
+ export function encodeUtf8(str) {
14
+ if (typeof str !== 'string') {
15
+ throw new TypeError(`encodeUtf8 expects a string, got ${typeof str}`)
16
+ }
17
+ if (!str.isWellFormed()) {
18
+ throw new TypeError('encodeUtf8: input contains lone surrogates')
19
+ }
20
+ return encoder.encode(str)
21
+ }