@preventive/triage 1.0.0-alpha.13 → 1.0.0-alpha.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/out/brotli-fallback.js +3 -3
- package/out/client-admin.js +2 -2
- package/out/client-sync.js +14 -14
- package/out/graph.js +30 -5
- package/out/index.html +47 -7
- package/out/prism.js +2 -2
- package/out/stasis.svg +45 -0
- package/out/terminal.js +255 -43
- package/out/view.css +1 -1
- package/out/view.js +147 -110
- package/package.json +27 -4
- package/report/index.js +253 -0
- package/report/src/finding-id.js +80 -0
- package/report/src/finding.js +300 -0
- package/report/src/labels.js +33 -0
- package/report/src/md-structure.js +471 -0
- package/report/src/md-text.js +167 -0
- package/report/src/meta.js +51 -0
- package/report/src/parse-codex.js +147 -0
- package/report/src/parse-deepsec.js +197 -0
- package/report/src/parse-deepview-fields.js +375 -0
- package/report/src/parse-deepview-md.js +185 -0
- package/report/src/parse-md-id.js +137 -0
- package/report/src/parse-md.js +253 -0
- package/report/src/parse-piolium-id.js +79 -0
- package/report/src/parse-piolium-rows.js +131 -0
- package/report/src/parse-piolium-tokens.js +175 -0
- package/report/src/parse-piolium.js +400 -0
- package/report/src/utf8.js +21 -0
- package/report/src/write-md-finding.js +273 -0
- package/report/src/write-md.js +291 -0
- package/server-e2e/bus-receiver.ts +1 -0
- package/server-e2e/hub.ts +37 -6
- package/server-e2e/index.ts +3 -2
- package/server-e2e/objstore/handlers.ts +6 -5
- package/server-e2e/objstore/init.ts +1 -1
- package/server-e2e/objstore/rest-mint.ts +2 -2
- package/server-e2e/objstore/rest.ts +2 -2
- package/server-e2e/pubsub.ts +8 -5
- package/server-e2e/sync-handlers.ts +54 -28
|
@@ -0,0 +1,471 @@
|
|
|
1
|
+
// Shared structural-markdown helpers: fence-aware heading splitting,
|
|
2
|
+
// table reading, labelled fields. parse-piolium.js reads through all of
|
|
3
|
+
// them; parse-md.js and parse-deepsec.js share only the heading-line
|
|
4
|
+
// split and keep their own, subtly different, section and label
|
|
5
|
+
// readers — fold those in only with their behavior pinned by tests
|
|
6
|
+
// first, since ids derive from parser output and a drift in parsing
|
|
7
|
+
// silently re-keys stored triage.
|
|
8
|
+
|
|
9
|
+
// Byte ranges of fenced code blocks (``` / ~~~), fences included, read
|
|
10
|
+
// once per text so no structural splitter takes a code line for a `## `
|
|
11
|
+
// heading or a `| ` row. A closing fence must use the opening marker,
|
|
12
|
+
// and a dangling one runs to end of input — the reading markdown gives.
|
|
13
|
+
//
|
|
14
|
+
// A fence may be INDENTED: three spaces at the top level (markdown's
|
|
15
|
+
// own limit, past which a line is indented code), and three past the
|
|
16
|
+
// content column of the innermost open list item, which is how a
|
|
17
|
+
// snippet under a numbered step is written. Tracking that column — a
|
|
18
|
+
// `10.` or a nested bullet pushes it out — is what keeps a block
|
|
19
|
+
// indented FURTHER than its item's text an indented code block, with
|
|
20
|
+
// its ``` lines content.
|
|
21
|
+
const FENCE_RE = /^( *)(```|~~~)/u
|
|
22
|
+
// A list marker and the gap to its text; `m[0].length` is the column
|
|
23
|
+
// the item's continuation lines are indented to.
|
|
24
|
+
const LIST_MARKER_RE = /^( *)(?:[-*+]|\d{1,9}[.)]) +(?=\S)/u
|
|
25
|
+
|
|
26
|
+
export function fenceRanges(text) {
|
|
27
|
+
const ranges = []
|
|
28
|
+
let open = -1
|
|
29
|
+
let marker = ''
|
|
30
|
+
let openIndent = 0
|
|
31
|
+
// Content column of the innermost open list item; 0 outside a list.
|
|
32
|
+
let itemIndent = 0
|
|
33
|
+
let pos = 0
|
|
34
|
+
for (const line of text.split('\n')) {
|
|
35
|
+
const start = pos
|
|
36
|
+
pos += line.length + 1
|
|
37
|
+
const fence = FENCE_RE.exec(line)
|
|
38
|
+
if (open !== -1) {
|
|
39
|
+
// A closing fence carries the item's indentation too, and needn't
|
|
40
|
+
// match the opening one's exactly — but the MARKER still has to,
|
|
41
|
+
// so a ``` inside a ~~~ block stays content.
|
|
42
|
+
if (fence && fence[2] === marker && fence[1].length <= openIndent + 3) {
|
|
43
|
+
ranges.push([open, start + line.length])
|
|
44
|
+
open = -1
|
|
45
|
+
}
|
|
46
|
+
continue
|
|
47
|
+
}
|
|
48
|
+
if (fence && fence[1].length <= itemIndent + 3) {
|
|
49
|
+
open = start
|
|
50
|
+
marker = fence[2]
|
|
51
|
+
openIndent = fence[1].length
|
|
52
|
+
continue
|
|
53
|
+
}
|
|
54
|
+
// List bookkeeping. A blank line doesn't end an item (a loose list
|
|
55
|
+
// is still one list); a marker opens or re-opens one at its own
|
|
56
|
+
// column, and any other line that starts LEFT of the open item's
|
|
57
|
+
// text has left it.
|
|
58
|
+
if (!line.trim()) continue
|
|
59
|
+
const item = LIST_MARKER_RE.exec(line)
|
|
60
|
+
const indent = /^ */u.exec(line)[0].length
|
|
61
|
+
if (item && item[1].length <= itemIndent + 3) itemIndent = item[0].length
|
|
62
|
+
else if (indent < itemIndent) itemIndent = 0
|
|
63
|
+
}
|
|
64
|
+
if (open !== -1) ranges.push([open, text.length])
|
|
65
|
+
return ranges
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function inFence(ranges, index) {
|
|
69
|
+
return ranges.some(([start, end]) => index >= start && index < end)
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// Line endings normalised — what every parser does before reading a
|
|
73
|
+
// line, and the writer before putting prose on the page.
|
|
74
|
+
export function normalizeNewlines(text) {
|
|
75
|
+
return String(text ?? '').replaceAll(/\r\n?/gu, '\n')
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// The headings the parsers split on: global and multiline, heading text
|
|
79
|
+
// in capture 1, as splitByHeading and splitLeading want them. Shared
|
|
80
|
+
// instances — they only ever reach `matchAll`, which doesn't advance a
|
|
81
|
+
// regex.
|
|
82
|
+
export const H2_RE = /^## +(.*)$/gmu
|
|
83
|
+
export const H3_RE = /^### +(.*)$/gmu
|
|
84
|
+
export const H4_RE = /^#### +(.*)$/gmu
|
|
85
|
+
|
|
86
|
+
// `file:line`, the line a number or a `10-20` RANGE kept whole: the
|
|
87
|
+
// displays print it verbatim, link anchors parseInt() it to the start.
|
|
88
|
+
export const FILE_LINE_RE = /^(.+):(\d+(?:-\d+)?)$/u
|
|
89
|
+
|
|
90
|
+
// A git hash as a report writes one: short or full, either case.
|
|
91
|
+
export function isCommitHash(s) {
|
|
92
|
+
return /^[0-9a-f]{7,64}$/iu.test(s)
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// `text` split at every line matching `re` outside a fence. Content
|
|
96
|
+
// before the first heading is dropped.
|
|
97
|
+
export function splitByHeading(text, re) {
|
|
98
|
+
const ranges = fenceRanges(text)
|
|
99
|
+
const marks = [...text.matchAll(re)].filter((m) => !inFence(ranges, m.index))
|
|
100
|
+
return marks.map((m, i) => ({
|
|
101
|
+
heading: m[1],
|
|
102
|
+
body: text.slice(m.index + m[0].length + 1, marks[i + 1]?.index),
|
|
103
|
+
}))
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// splitByHeading, keeping the content before the first heading — the
|
|
107
|
+
// enclosing block's own body — as `head`.
|
|
108
|
+
export function splitLeading(body, re) {
|
|
109
|
+
const ranges = fenceRanges(body)
|
|
110
|
+
const first = [...body.matchAll(re)].find((m) => !inFence(ranges, m.index))
|
|
111
|
+
if (!first) return { head: body, subs: [] }
|
|
112
|
+
return { head: body.slice(0, first.index), subs: splitByHeading(body, re) }
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// A block split off its `# ` / `### ` marker: heading line and body.
|
|
116
|
+
export function splitHeadingLine(block) {
|
|
117
|
+
const nl = block.indexOf('\n')
|
|
118
|
+
if (nl === -1) return { title: block.trim(), body: '' }
|
|
119
|
+
return { title: block.slice(0, nl).trim(), body: block.slice(nl + 1) }
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// Rows of a markdown table, as arrays of trimmed cells; the `|---|---|`
|
|
123
|
+
// delimiter and any non-row line are skipped, so prose around the table
|
|
124
|
+
// is ignored. The delimiter test is one character class: a
|
|
125
|
+
// `[\s:|-]*\|?\s*$` shape carries two overlapping whitespace
|
|
126
|
+
// quantifiers and backtracks quadratically on a padded cell.
|
|
127
|
+
function tableRows(text) {
|
|
128
|
+
const rows = []
|
|
129
|
+
for (const line of text.split('\n')) {
|
|
130
|
+
const trimmed = line.trim()
|
|
131
|
+
if (!trimmed.startsWith('|')) continue
|
|
132
|
+
if (/^[\s:|-]+$/u.test(trimmed) && trimmed.includes('-')) continue
|
|
133
|
+
const cells = trimmed.replace(/^\|/u, '').replace(/\|$/u, '').split('|').map((c) => c.trim())
|
|
134
|
+
rows.push(cells)
|
|
135
|
+
}
|
|
136
|
+
return rows
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// A table as `{ <column>: value }` objects keyed by its own case-folded
|
|
140
|
+
// headers, so callers match columns by name rather than position. A
|
|
141
|
+
// re-stated header row — how a concatenated section arrives — is chrome.
|
|
142
|
+
export function tableObjects(text) {
|
|
143
|
+
const rows = tableRows(text)
|
|
144
|
+
if (rows.length < 2) return []
|
|
145
|
+
const header = rows[0].map((h) => h.toLowerCase())
|
|
146
|
+
const objects = []
|
|
147
|
+
for (const cells of rows.slice(1)) {
|
|
148
|
+
if (cells.length === header.length && cells.every((c, i) => c.toLowerCase() === header[i])) continue
|
|
149
|
+
const obj = {}
|
|
150
|
+
header.forEach((name, i) => { if (name) obj[name] = cells[i] ?? '' })
|
|
151
|
+
objects.push(obj)
|
|
152
|
+
}
|
|
153
|
+
return objects
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// `**Field:** value` labels, with or without a `- ` bullet, keyed
|
|
157
|
+
// case-folded with the original spelling in `labels`; first occurrence
|
|
158
|
+
// wins, and several joined by ` · ` on one line are peeled apart.
|
|
159
|
+
//
|
|
160
|
+
// A value runs to the next label, heading, table row, rule, or BLANK
|
|
161
|
+
// LINE: a wrapped one-liner keeps its continuation lines, while the
|
|
162
|
+
// paragraph under a label block is body prose — a `**Key code:** …` line
|
|
163
|
+
// must not swallow the summary under it. Fenced code in a value is all
|
|
164
|
+
// content. Unlabelled text is collected as `prose`, for reports that
|
|
165
|
+
// narrate without labels. Null-prototype, so "Constructor" aliases
|
|
166
|
+
// nothing.
|
|
167
|
+
export function parseLabelledFields(body) {
|
|
168
|
+
const fields = Object.create(null)
|
|
169
|
+
const labels = Object.create(null)
|
|
170
|
+
const proseLines = []
|
|
171
|
+
let key = null
|
|
172
|
+
let keyLabel = ''
|
|
173
|
+
let buf = []
|
|
174
|
+
let fence = ''
|
|
175
|
+
const setField = (k, label, value) => {
|
|
176
|
+
if (!k || k in fields) return
|
|
177
|
+
fields[k] = value.trim()
|
|
178
|
+
labels[k] = label
|
|
179
|
+
}
|
|
180
|
+
const flush = () => {
|
|
181
|
+
if (key) setField(key, keyLabel, buf.join('\n'))
|
|
182
|
+
key = null
|
|
183
|
+
keyLabel = ''
|
|
184
|
+
buf = []
|
|
185
|
+
}
|
|
186
|
+
// A content line belongs to the open label's value, or to the prose.
|
|
187
|
+
const keep = (line) => { (key ? buf : proseLines).push(line) }
|
|
188
|
+
for (const line of body.split('\n')) {
|
|
189
|
+
// A fence delimiter toggles, and every line up to the closing one
|
|
190
|
+
// (delimiters included) is content, whatever it looks like.
|
|
191
|
+
const fm = /^ {0,3}(```|~~~)/u.exec(line)
|
|
192
|
+
const delimiter = fm !== null && (!fence || fm[1] === fence)
|
|
193
|
+
if (delimiter) fence = fence ? '' : fm[1]
|
|
194
|
+
if (delimiter || fence) {
|
|
195
|
+
keep(line)
|
|
196
|
+
continue
|
|
197
|
+
}
|
|
198
|
+
if (!line.trim()) {
|
|
199
|
+
if (key) flush()
|
|
200
|
+
else proseLines.push(line)
|
|
201
|
+
continue
|
|
202
|
+
}
|
|
203
|
+
const label = /^\s*(?:[-*] +)?\*\*([^:*]+):\*\*\s*(.*)$/u.exec(line)
|
|
204
|
+
if (label) {
|
|
205
|
+
flush()
|
|
206
|
+
let k = label[1].trim()
|
|
207
|
+
let rest = label[2]
|
|
208
|
+
let seg
|
|
209
|
+
while ((seg = /\s+[·•]\s+\*\*([^:*]+):\*\*\s*/u.exec(rest)) !== null) {
|
|
210
|
+
setField(k.toLowerCase(), k, rest.slice(0, seg.index))
|
|
211
|
+
k = seg[1].trim()
|
|
212
|
+
rest = rest.slice(seg.index + seg[0].length)
|
|
213
|
+
}
|
|
214
|
+
key = k.toLowerCase()
|
|
215
|
+
keyLabel = k
|
|
216
|
+
buf = [rest]
|
|
217
|
+
continue
|
|
218
|
+
}
|
|
219
|
+
// Structural line — ends the current value without starting one.
|
|
220
|
+
if (/^\s*(?:#{1,6} |\||[-=*_]{3,}\s*$)/u.test(line)) { flush(); continue }
|
|
221
|
+
keep(line)
|
|
222
|
+
}
|
|
223
|
+
flush()
|
|
224
|
+
return { fields, labels, prose: proseLines.join('\n').trim() }
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
// A code reference is prose-ish: `src/a.js:142 in runHook()`, a
|
|
228
|
+
// backticked path, or a markdown link to the line on GitHub. Out come
|
|
229
|
+
// the path, the line, and the URL when there is one — which
|
|
230
|
+
// finding-id.js keys off when no fileHash is available, so two imports
|
|
231
|
+
// of a report derive the same uuid and share triage. A trailing
|
|
232
|
+
// function qualifier is shed from the path.
|
|
233
|
+
export function parseCodeRef(raw) {
|
|
234
|
+
let text = (raw || '').trim()
|
|
235
|
+
let locationLink = ''
|
|
236
|
+
// Read the link the way findMdLink reads one — brackets, parens and
|
|
237
|
+
// all. Piolium cites paths a Next.js tree is full of
|
|
238
|
+
// (`app/(main)/[id]/page.ts`), and an expression whose label stops at
|
|
239
|
+
// the first `]` finds none of them: the whole `[…](…)` text is left
|
|
240
|
+
// as the file name, or the path comes back off its code span with the
|
|
241
|
+
// url dropped. What the FINGERPRINT reads is that expression, frozen
|
|
242
|
+
// in parse-piolium-id.js, so fixing this moves no ids.
|
|
243
|
+
const link = findMdLink(text)
|
|
244
|
+
if (link) {
|
|
245
|
+
text = link.label.trim()
|
|
246
|
+
locationLink = link.url.trim()
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// A `#L<n>` anchor on the link is the most reliable line source (and
|
|
250
|
+
// reads the start line of a `#L88-L95` range).
|
|
251
|
+
let line = ''
|
|
252
|
+
const anchor = /#L(\d+)/u.exec(locationLink)
|
|
253
|
+
if (anchor) line = anchor[1]
|
|
254
|
+
|
|
255
|
+
// The first PATH-SHAPED backtick span wins when there is one: a value
|
|
256
|
+
// citing a call chain — "see `src/a.js:42` and `src/b.js:9`" — locates
|
|
257
|
+
// the finding at the first quoted path, the rest being prose. Path-
|
|
258
|
+
// shaped means a separator or an extension and no call parens, so a
|
|
259
|
+
// quoted qualifier (`… in \`runHook()\``) never beats a bare path, and
|
|
260
|
+
// a chosen span is the WHOLE path — backticks are what delimit one
|
|
261
|
+
// with spaces in it. The unquoted fallback takes the first whitespace
|
|
262
|
+
// token instead, since the template appends `… in runHook()`. Either
|
|
263
|
+
// way a trailing `#L42` or `:88-95` yields the line, a RANGE keeping
|
|
264
|
+
// its start and shedding the rest from the path.
|
|
265
|
+
const spans = [...text.matchAll(/`([^`]+)`/gu)].map((m) => m[1].trim())
|
|
266
|
+
const pathish = spans.find((s) => !s.includes('(') && (s.includes('/') || /\.\w/u.test(s)))
|
|
267
|
+
let file = pathish ?? (text.replaceAll('`', '').trim().split(/[\s,]+/u).find(Boolean) || '')
|
|
268
|
+
const frag = /^(.*?)#L(\d+)(?:-L?\d+)?$/u.exec(file)
|
|
269
|
+
if (frag) {
|
|
270
|
+
file = frag[1]
|
|
271
|
+
if (!line) line = frag[2]
|
|
272
|
+
}
|
|
273
|
+
const colon = FILE_LINE_RE.exec(file)
|
|
274
|
+
if (colon) {
|
|
275
|
+
if (!line) line = colon[2]
|
|
276
|
+
return { file: colon[1], line, locationLink }
|
|
277
|
+
}
|
|
278
|
+
return { file, line: line || '?', locationLink }
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
export function stripBold(text) { return text.replaceAll('**', '') }
|
|
282
|
+
|
|
283
|
+
// An inline link — `[label](destination)` — the first one in `s`, or
|
|
284
|
+
// null. Scanned rather than matched with one expression: both halves
|
|
285
|
+
// nest, and an expression permissive enough for the nesting can no
|
|
286
|
+
// longer tell where a link STARTS (`[context] see [src/a.ts:7](…)` opens
|
|
287
|
+
// on a bracket pair that is not a link).
|
|
288
|
+
//
|
|
289
|
+
// At each `[` the LABEL is read up to the first `]`, then
|
|
290
|
+
// bracket-balanced. The first reading is all a path with an UNMATCHED
|
|
291
|
+
// bracket has — `[`src/[id.ts:7`](…)` is what this library's own writer
|
|
292
|
+
// emits for one. The second is markdown's rule and what a path carrying
|
|
293
|
+
// brackets needs (`[app/(main)/[id]/page.ts:12](…)`), code spans skipped
|
|
294
|
+
// whole, since backticks make their content literal. Either reading
|
|
295
|
+
// counts only when a `(` follows, so `[context]` is no label.
|
|
296
|
+
//
|
|
297
|
+
// The DESTINATION is `<…>` — what md-text.js `link` writes when a url
|
|
298
|
+
// holds a space, a paren or an angle bracket — else a bare run read the
|
|
299
|
+
// same two ways for the same reasons: balanced parens first, or a url
|
|
300
|
+
// the writer never percent-encoded comes back cut
|
|
301
|
+
// (`…/app/(main)/page.ts` → `…/app/(main`), then up to the first `)`,
|
|
302
|
+
// all an unmatched paren leaves. Whitespace disqualifies a bare
|
|
303
|
+
// destination either way, where markdown would read a title.
|
|
304
|
+
//
|
|
305
|
+
// A backslash hides the character after it from every scan here.
|
|
306
|
+
//
|
|
307
|
+
// `index` comes back too, so a caller can ask that the value START with
|
|
308
|
+
// a link (parse-deepview-fields.js readLink) rather than take the first
|
|
309
|
+
// one in the line (parse-md.js).
|
|
310
|
+
export function findMdLink(s) {
|
|
311
|
+
const text = String(s ?? '')
|
|
312
|
+
// Where each reading would CLOSE, read off the text once rather than
|
|
313
|
+
// rescanned per candidate: every reading here scans to the end when
|
|
314
|
+
// nothing closes it, so 50k of `[` with no `]` would cost each
|
|
315
|
+
// bracket the remainder of the line.
|
|
316
|
+
const labels = balancedLabelEnds(text, codeSpanEnds(text))
|
|
317
|
+
const dests = destinationEnds(text)
|
|
318
|
+
let plain = text.indexOf(']')
|
|
319
|
+
for (let open = text.indexOf('['); open !== -1; open = text.indexOf('[', open + 1)) {
|
|
320
|
+
// The first `]` after this `[`. Carried forward, not looked up
|
|
321
|
+
// again: `open` only advances, so this does too.
|
|
322
|
+
while (plain !== -1 && plain <= open) plain = text.indexOf(']', plain + 1)
|
|
323
|
+
for (const close of [plain, labels.get(open) ?? -1]) {
|
|
324
|
+
// An EMPTY label is no label: `` ahead of a
|
|
325
|
+
// reference is a badge, and the link wanted is the one behind it.
|
|
326
|
+
if (close === -1 || close === open + 1 || text[close + 1] !== '(') continue
|
|
327
|
+
const url = destination(text, close + 1, dests)
|
|
328
|
+
if (url !== null) return { label: text.slice(open + 1, close), url, index: open }
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
return null
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// Every `[` paired with the `]` that closes it once its brackets
|
|
335
|
+
// balance, in one pass with a stack. Escapes and code spans are passed
|
|
336
|
+
// over whole — neither one's brackets are structure, the reading
|
|
337
|
+
// markdown gives them too.
|
|
338
|
+
function balancedLabelEnds(text, spans) {
|
|
339
|
+
const ends = new Map()
|
|
340
|
+
const open = []
|
|
341
|
+
for (let i = 0; i < text.length; i++) {
|
|
342
|
+
const c = text[i]
|
|
343
|
+
if (escapes(text, i)) i++
|
|
344
|
+
else if (c === '`') i = spans.get(i) ?? i
|
|
345
|
+
else if (c === '[') open.push(i)
|
|
346
|
+
else if (c === ']' && open.length > 0) ends.set(open.pop(), i)
|
|
347
|
+
}
|
|
348
|
+
return ends
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
// Every backtick RUN that opens a code span, paired with the last
|
|
352
|
+
// backtick of the run that closes it — the next run of exactly the same
|
|
353
|
+
// length, which is how markdown fences one (md-text.js code writes
|
|
354
|
+
// these). A run nothing matches is absent; its backticks are text.
|
|
355
|
+
//
|
|
356
|
+
// Read backwards, each run remembering the nearest one of its own
|
|
357
|
+
// length ahead of it, so a line of unmatched runs of growing lengths —
|
|
358
|
+
// `` `x``x```x… `` — costs one pass, not a scan per run.
|
|
359
|
+
function codeSpanEnds(text) {
|
|
360
|
+
const runs = []
|
|
361
|
+
for (let i = text.indexOf('`'); i !== -1; i = text.indexOf('`', i)) {
|
|
362
|
+
let n = 1
|
|
363
|
+
while (text[i + n] === '`') n++
|
|
364
|
+
runs.push([i, n])
|
|
365
|
+
i += n
|
|
366
|
+
}
|
|
367
|
+
const ends = new Map()
|
|
368
|
+
const nearest = new Map()
|
|
369
|
+
for (let r = runs.length - 1; r >= 0; r--) {
|
|
370
|
+
const [start, length] = runs[r]
|
|
371
|
+
const close = nearest.get(length)
|
|
372
|
+
if (close !== undefined) ends.set(start, close + length - 1)
|
|
373
|
+
nearest.set(length, start)
|
|
374
|
+
}
|
|
375
|
+
return ends
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
// What a bare destination can close on, at every position: the `)` that
|
|
379
|
+
// balances each `(`, and — for the reading that needs no balance — the
|
|
380
|
+
// next `)` and the next whitespace. Whitespace ends a bare destination
|
|
381
|
+
// either way, so a run of it abandons every open `(`.
|
|
382
|
+
function destinationEnds(text) {
|
|
383
|
+
const n = text.length
|
|
384
|
+
const nextClose = new Int32Array(n + 1).fill(-1)
|
|
385
|
+
const nextSpace = new Int32Array(n + 1).fill(-1)
|
|
386
|
+
// …and what ends an angle-bracket one, for the same reason.
|
|
387
|
+
const nextAngle = new Int32Array(n + 1).fill(-1)
|
|
388
|
+
const nextLine = new Int32Array(n + 1).fill(-1)
|
|
389
|
+
for (let i = n - 1; i >= 0; i--) {
|
|
390
|
+
nextClose[i] = text[i] === ')' ? i : nextClose[i + 1]
|
|
391
|
+
nextSpace[i] = /\s/u.test(text[i]) ? i : nextSpace[i + 1]
|
|
392
|
+
nextAngle[i] = text[i] === '>' ? i : nextAngle[i + 1]
|
|
393
|
+
nextLine[i] = text[i] === '\n' ? i : nextLine[i + 1]
|
|
394
|
+
}
|
|
395
|
+
const balanced = new Map()
|
|
396
|
+
const open = []
|
|
397
|
+
for (let i = 0; i < n; i++) {
|
|
398
|
+
const c = text[i]
|
|
399
|
+
// A backslash hides only a character it can actually escape.
|
|
400
|
+
// `not\ a-url` is a backslash and a SPACE, and that space ends a
|
|
401
|
+
// bare destination — read as an escape, `[badge](not\ a-url)` would
|
|
402
|
+
// be a link, and a reference behind it never reached.
|
|
403
|
+
if (escapes(text, i)) i++
|
|
404
|
+
else if (nextSpace[i] === i) open.length = 0
|
|
405
|
+
else if (c === '(') open.push(i)
|
|
406
|
+
else if (c === ')' && open.length > 0) balanced.set(open.pop(), i)
|
|
407
|
+
}
|
|
408
|
+
return { balanced, nextClose, nextSpace, nextAngle, nextLine }
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
// The destination opened at `open`, as its url, or null: an
|
|
412
|
+
// angle-bracket form, else the bare run its own parens close, else the
|
|
413
|
+
// one the first `)` closes. An EMPTY destination is none of them —
|
|
414
|
+
// `[a]()` is not a link — while an empty `<>` falls through to the bare
|
|
415
|
+
// readings, which take the angle brackets themselves as the url.
|
|
416
|
+
function destination(text, open, dests) {
|
|
417
|
+
const angled = angleDestination(text, open, dests)
|
|
418
|
+
if (angled) return angled
|
|
419
|
+
const balanced = dests.balanced.get(open)
|
|
420
|
+
if (balanced !== undefined) return balanced > open + 1 ? text.slice(open + 1, balanced) : null
|
|
421
|
+
// Failing that, up to the first `)` — all an unmatched paren leaves.
|
|
422
|
+
// Whitespace before it disqualifies the candidate.
|
|
423
|
+
const flat = dests.nextClose[open + 1]
|
|
424
|
+
const space = dests.nextSpace[open + 1]
|
|
425
|
+
if (flat === -1 || flat === open + 1 || (space !== -1 && space < flat)) return null
|
|
426
|
+
return text.slice(open + 1, flat)
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
// The `<…>` form md-text.js `link` writes when a url can't sit bare,
|
|
430
|
+
// or '' when this destination isn't one.
|
|
431
|
+
function angleDestination(text, open, dests) {
|
|
432
|
+
if (text[open + 1] !== '<') return ''
|
|
433
|
+
const close = dests.nextAngle[open + 2]
|
|
434
|
+
const line = dests.nextLine[open + 2]
|
|
435
|
+
if (close === -1 || (line !== -1 && line < close) || text[close + 1] !== ')') return ''
|
|
436
|
+
return text.slice(open + 2, close)
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
// Markdown backslash escapes — `a/b/\_cc\_cc/index.js` is a report
|
|
440
|
+
// escaping underscores that would open emphasis, not a path with
|
|
441
|
+
// backslashes. Undone wherever a value is a NAME rather than prose: a
|
|
442
|
+
// file path, a link's label. Only ASCII punctuation can be escaped
|
|
443
|
+
// (CommonMark), so a `\n` or a Windows `C:\path` keeps its backslash.
|
|
444
|
+
const MD_ESCAPE_RE = /\\([!-/:-@[-`{-~])/gu
|
|
445
|
+
|
|
446
|
+
// The same rule at one position: `\ ` is two characters and `\[` is
|
|
447
|
+
// one, which is what keeps a scanner from reading a space as hidden
|
|
448
|
+
// where markdown reads it as the whitespace ending a destination.
|
|
449
|
+
const MD_ESCAPABLE = /[!-/:-@[-`{-~]/u
|
|
450
|
+
|
|
451
|
+
function escapes(text, i) {
|
|
452
|
+
return text[i] === '\\' && MD_ESCAPABLE.test(text[i + 1] ?? '')
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
export function unescapeMd(s) {
|
|
456
|
+
return typeof s === 'string' ? s.replace(MD_ESCAPE_RE, '$1') : s
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
// `[X]` → `X`, where the brackets are notation. Ids only — a title can
|
|
460
|
+
// legitimately carry square brackets.
|
|
461
|
+
export function stripBrackets(s) {
|
|
462
|
+
const m = /^\[(.+)\]$/u.exec(s.trim())
|
|
463
|
+
return m ? m[1].trim() : s.trim()
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
// Table cells use `--` / `-` — or a typographic `—` / `–` — for
|
|
467
|
+
// "not applicable".
|
|
468
|
+
export function cellValue(s) {
|
|
469
|
+
const v = (s || '').trim()
|
|
470
|
+
return /^[-–—]+$/u.test(v) ? '' : v
|
|
471
|
+
}
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
// Markdown text helpers for the writer (write-md.js): the escaping a
|
|
2
|
+
// value needs for the position it lands in, GitHub-style heading
|
|
3
|
+
// anchors, tables, and the formatters the document header uses. The
|
|
4
|
+
// writing-side sibling of md-structure.js, which reads. Pure string
|
|
5
|
+
// work; nothing here knows what a finding is.
|
|
6
|
+
|
|
7
|
+
import { fenceRanges, inFence, normalizeNewlines } from './md-structure.js'
|
|
8
|
+
|
|
9
|
+
// Parseable http:// / https:// only. What gets linked comes from reports
|
|
10
|
+
// and from the user's own notes, where a fix reference can be "internal
|
|
11
|
+
// ticket #42", and the other schemes are useless (file:) or a footgun
|
|
12
|
+
// (javascript:, data:). The viewer's `<a>` gates read this too.
|
|
13
|
+
export function isHttpUrl(s) {
|
|
14
|
+
if (typeof s !== 'string' || s.length === 0) return false
|
|
15
|
+
try {
|
|
16
|
+
const u = new URL(s)
|
|
17
|
+
return u.protocol === 'http:' || u.protocol === 'https:'
|
|
18
|
+
} catch { return false }
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
// One line of a table cell: newlines collapse to spaces, and the `|`
|
|
22
|
+
// that would end the cell is escaped — inside a code span too, which is
|
|
23
|
+
// where GitHub still reads it as a column break.
|
|
24
|
+
export function cell(text) {
|
|
25
|
+
return String(text ?? '').replaceAll(/\s*\n\s*/gu, ' ').replaceAll('|', '\\|').trim()
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// Inline code — a path, a hash, a package name. Fenced with one more
|
|
29
|
+
// backtick than the longest run inside it, which is how markdown quotes
|
|
30
|
+
// a backtick; padded when the content itself starts or ends on one, so
|
|
31
|
+
// the content can't merge with its fence.
|
|
32
|
+
export function code(text) {
|
|
33
|
+
const s = String(text ?? '')
|
|
34
|
+
if (s === '') return ''
|
|
35
|
+
const longest = Math.max(0, ...[...s.matchAll(/`+/gu)].map((m) => m[0].length))
|
|
36
|
+
const fence = '`'.repeat(longest + 1)
|
|
37
|
+
const pad = s.startsWith('`') || s.endsWith('`') ? ' ' : ''
|
|
38
|
+
return `${fence}${pad}${s}${pad}${fence}`
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// Square brackets in a link's TEXT would open a nested link; escape
|
|
42
|
+
// them. Code spans inside the text need no escaping — they bind tighter
|
|
43
|
+
// than the brackets — so this is for plain-text labels only.
|
|
44
|
+
export function escapeBrackets(text) {
|
|
45
|
+
return String(text ?? '').replaceAll(/[[\]]/gu, '\\$&')
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// `[label](url)`. The URL goes in angle brackets when it carries a
|
|
49
|
+
// character that would end the destination early.
|
|
50
|
+
export function link(label, url) {
|
|
51
|
+
const target = /[\s()<>]/u.test(url) ? `<${url.replaceAll('>', '%3E')}>` : url
|
|
52
|
+
return `[${label}](${target})`
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// A bare URL as an autolink (`<url>`), which every renderer links;
|
|
56
|
+
// anything else — a ticket number, a note — as the text it is.
|
|
57
|
+
export function autolink(s) {
|
|
58
|
+
return isHttpUrl(s) ? `<${s}>` : String(s ?? '')
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// GitHub's heading anchor: lower-cased, punctuation dropped, spaces to
|
|
62
|
+
// hyphens, `-N` on a repeat — what GitHub, GitLab and most editors read.
|
|
63
|
+
// `taken` is the document's registry of anchors handed out.
|
|
64
|
+
export function anchorSlug(text, taken) {
|
|
65
|
+
const base = String(text ?? '').toLowerCase()
|
|
66
|
+
.replaceAll(/[^\p{L}\p{N}\p{M}\s_-]/gu, '')
|
|
67
|
+
.replaceAll(/\s/gu, '-')
|
|
68
|
+
let slug = base
|
|
69
|
+
for (let n = 1; taken.has(slug); n++) slug = `${base}-${n}`
|
|
70
|
+
taken.add(slug)
|
|
71
|
+
return slug
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// A heading at `depth`, clamped to markdown's six levels.
|
|
75
|
+
export function heading(depth, text) {
|
|
76
|
+
return `${'#'.repeat(Math.min(6, Math.max(1, depth)))} ${text}`
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// A table from a header row and rows, every cell already escaped by the
|
|
80
|
+
// caller. `align[i]` is 'right' for a numeric column; the delimiter row
|
|
81
|
+
// carries it.
|
|
82
|
+
export function table(headers, rows, align = []) {
|
|
83
|
+
const delim = headers.map((_, i) => (align[i] === 'right' ? '---:' : '---'))
|
|
84
|
+
return [headers, delim, ...rows].map((r) => `| ${r.join(' | ')} |`).join('\n')
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// `2026-09-05 14:02 UTC` — a moment a reader can compare with the
|
|
88
|
+
// report's own dates without knowing the exporting machine's zone.
|
|
89
|
+
export function formatTimestamp(date) {
|
|
90
|
+
const d = date instanceof Date ? date : new Date(date)
|
|
91
|
+
if (Number.isNaN(d.getTime())) return ''
|
|
92
|
+
const p = (n) => String(n).padStart(2, '0')
|
|
93
|
+
return `${d.getUTCFullYear()}-${p(d.getUTCMonth() + 1)}-${p(d.getUTCDate())} ${p(d.getUTCHours())}:${p(d.getUTCMinutes())} UTC`
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export function plural(n, noun, many = `${noun}s`) {
|
|
97
|
+
return `${n} ${n === 1 ? noun : many}`
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// Blocks joined by one blank line — the paragraph break — with empty
|
|
101
|
+
// blocks dropped and each block's trailing whitespace trimmed, so no
|
|
102
|
+
// block can add a second break of its own.
|
|
103
|
+
export function joinBlocks(blocks) {
|
|
104
|
+
return blocks.filter((b) => typeof b === 'string' && b.trim()).map((b) => b.replace(/\s+$/u, '')).join('\n\n')
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// A run of a report's own markdown as it lands in the document: line
|
|
108
|
+
// endings normalised, edges trimmed, an open fence closed, a line that
|
|
109
|
+
// would read as a heading escaped.
|
|
110
|
+
//
|
|
111
|
+
// A dangling fence runs to the end of the FINDING for every parser — a
|
|
112
|
+
// card's reader sees the snippet, not a problem — but in a document it
|
|
113
|
+
// would swallow every finding after it, so it is closed with the marker
|
|
114
|
+
// that opened it.
|
|
115
|
+
//
|
|
116
|
+
// A `## Internal detail` line in an analyzer's prose is text the card
|
|
117
|
+
// shows, not a section: written bare, a renderer and the document's own
|
|
118
|
+
// reader (parse-deepview-md.js) would both end the finding there. It
|
|
119
|
+
// goes on the page as `\## Internal detail` and is stripped back
|
|
120
|
+
// (unescapeHeadings), with a line already opening on a backslash getting
|
|
121
|
+
// one more, so that strip is exact whatever the prose held. Fenced code
|
|
122
|
+
// is left alone — a `#` there is code.
|
|
123
|
+
const FENCE_OPEN_RE = /^ *(`{3,}|~{3,})/u
|
|
124
|
+
const HEADING_LINE_RE = /^( {0,3})(\\*#)/u
|
|
125
|
+
|
|
126
|
+
export function prose(text) {
|
|
127
|
+
const s = closeFence(normalizeNewlines(text).trim())
|
|
128
|
+
return s ? escapeHeadings(s) : ''
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function closeFence(s) {
|
|
132
|
+
if (!s) return ''
|
|
133
|
+
const last = fenceRanges(s).at(-1)
|
|
134
|
+
if (!last || last[1] < s.length) return s
|
|
135
|
+
const lines = s.slice(last[0]).split('\n')
|
|
136
|
+
const marker = FENCE_OPEN_RE.exec(lines[0])?.[1] ?? '```'
|
|
137
|
+
const closed = lines.length > 1 && FENCE_OPEN_RE.exec(lines.at(-1))?.[1]?.startsWith(marker.slice(0, 3))
|
|
138
|
+
return closed ? s : `${s}\n${marker}`
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// `fn` over every line of `s` outside a fence, in place.
|
|
142
|
+
function mapProseLines(s, fn) {
|
|
143
|
+
const ranges = fenceRanges(s)
|
|
144
|
+
let pos = 0
|
|
145
|
+
return s.split('\n').map((line) => {
|
|
146
|
+
const start = pos
|
|
147
|
+
pos += line.length + 1
|
|
148
|
+
return inFence(ranges, start) ? line : fn(line)
|
|
149
|
+
}).join('\n')
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
function escapeHeadings(s) {
|
|
153
|
+
return mapProseLines(s, (line) => line.replace(HEADING_LINE_RE, '$1\\$2'))
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// The inverse, for the reader: one backslash off a line that opens on
|
|
157
|
+
// backslashes before a `#`.
|
|
158
|
+
export function unescapeHeadings(s) {
|
|
159
|
+
return mapProseLines(String(s ?? ''), (line) => line.replace(/^( {0,3})\\(\\*#)/u, '$1$2'))
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// Continuation lines indented to a list item's content column, so
|
|
163
|
+
// markdown reads them as the item's own. Blank lines stay empty.
|
|
164
|
+
export function indentUnder(marker, text) {
|
|
165
|
+
const pad = ' '.repeat(marker.length)
|
|
166
|
+
return text.split('\n').map((l) => (l ? pad + l : '')).join('\n')
|
|
167
|
+
}
|