@brett_lamy/docstream 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,594 @@
1
+ import type {
2
+ Block,
3
+ ColumnNode,
4
+ DocumentNode,
5
+ HintStyle,
6
+ Inline,
7
+ ListItemNode,
8
+ StepNode,
9
+ TabNode,
10
+ UpdateNode,
11
+ } from "./ast"
12
+ import { parseInline, plainText, refDefinitions } from "./inline"
13
+
14
+ // Minimal HTML-inline → markdown-inline bridge for HTML table cells.
15
+ function htmlToInlineMd(html: string): string {
16
+ return html
17
+ .replace(/<br\s*\/?>/gi, " ")
18
+ .replace(/<strong>(.*?)<\/strong>/gis, "**$1**")
19
+ .replace(/<b>(.*?)<\/b>/gis, "**$1**")
20
+ .replace(/<em>(.*?)<\/em>/gis, "_$1_")
21
+ .replace(/<i>(.*?)<\/i>/gis, "_$1_")
22
+ .replace(/<code>(.*?)<\/code>/gis, "`$1`")
23
+ .replace(/<a[^>]*href="([^"]*)"[^>]*>(.*?)<\/a>/gis, "[$2]($1)")
24
+ .replace(/<[^>]+>/g, "")
25
+ .trim()
26
+ }
27
+
28
+ const TEMPLATE_RE = /^\s*\{%\s*(\S+?)(\s+[^%]*?)?\s*%\}\s*$/
29
+
30
+ function parseAttrs(raw: string | undefined): Record<string, string> {
31
+ const attrs: Record<string, string> = {}
32
+ if (!raw) return attrs
33
+ for (const m of raw.matchAll(/([\w-]+)="([^"]*)"/g)) {
34
+ attrs[m[1]] = m[2]
35
+ }
36
+ return attrs
37
+ }
38
+
39
+ interface TemplateTag {
40
+ name: string
41
+ attrs: Record<string, string>
42
+ }
43
+
44
+ function templateTag(line: string): TemplateTag | null {
45
+ const m = line.match(TEMPLATE_RE)
46
+ if (!m) return null
47
+ return { name: m[1], attrs: parseAttrs(m[2]) }
48
+ }
49
+
50
+ // Collects the lines between an opening {% name %} and its matching
51
+ // {% endname %}, honoring nesting of the same tag.
52
+ function collectUntil(lines: string[], start: number, name: string): { body: string[]; next: number } {
53
+ const body: string[] = []
54
+ let depth = 1
55
+ let i = start
56
+ for (; i < lines.length; i++) {
57
+ const tag = templateTag(lines[i])
58
+ if (tag?.name === name) depth++
59
+ if (tag?.name === `end${name}`) {
60
+ depth--
61
+ if (depth === 0) return { body, next: i + 1 }
62
+ }
63
+ body.push(lines[i])
64
+ }
65
+ return { body, next: i }
66
+ }
67
+
68
+ function collectHtmlUntil(lines: string[], start: number, closeTag: string): { body: string[]; next: number } {
69
+ const body: string[] = []
70
+ let i = start
71
+ for (; i < lines.length; i++) {
72
+ if (lines[i].includes(closeTag)) {
73
+ const before = lines[i].slice(0, lines[i].indexOf(closeTag))
74
+ if (before.trim()) body.push(before)
75
+ return { body, next: i + 1 }
76
+ }
77
+ body.push(lines[i])
78
+ }
79
+ return { body, next: i }
80
+ }
81
+
82
+ const HINT_STYLES: HintStyle[] = ["info", "success", "warning", "danger"]
83
+
84
+ export function parseMarkdown(src: string): DocumentNode {
85
+ const lines = src.split(/\r?\n/)
86
+ refDefinitions.clear()
87
+ const content: string[] = []
88
+ for (const line of lines) {
89
+ const def = line.match(/^\[([^\]]+)\]:\s*(\S+)\s*$/)
90
+ if (def) refDefinitions.set(def[1].toLowerCase(), def[2])
91
+ else content.push(line)
92
+ }
93
+ return { type: "doc", children: parseBlocks(content) }
94
+ }
95
+
96
+ export function parseBlocks(lines: string[]): Block[] {
97
+ const blocks: Block[] = []
98
+ let i = 0
99
+
100
+ const paragraph: string[] = []
101
+ const flushParagraph = () => {
102
+ if (paragraph.length) {
103
+ const text = paragraph.join(" ").trim()
104
+ if (text) blocks.push({ type: "paragraph", children: parseInline(text) })
105
+ paragraph.length = 0
106
+ }
107
+ }
108
+
109
+ while (i < lines.length) {
110
+ const line = lines[i]
111
+ const trimmed = line.trim()
112
+
113
+ if (!trimmed) {
114
+ flushParagraph()
115
+ i++
116
+ continue
117
+ }
118
+
119
+ const tag = templateTag(line)
120
+ if (tag) {
121
+ flushParagraph()
122
+
123
+ if (tag.name === "hint") {
124
+ const { body, next } = collectUntil(lines, i + 1, "hint")
125
+ const style = (HINT_STYLES as string[]).includes(tag.attrs.style)
126
+ ? (tag.attrs.style as HintStyle)
127
+ : "info"
128
+ blocks.push({ type: "hint", style, children: parseBlocks(body) })
129
+ i = next
130
+ continue
131
+ }
132
+
133
+ if (tag.name === "tabs") {
134
+ const { body, next } = collectUntil(lines, i + 1, "tabs")
135
+ blocks.push({ type: "tabs", tabs: parseTabs(body) })
136
+ i = next
137
+ continue
138
+ }
139
+
140
+ if (tag.name === "stepper") {
141
+ const { body, next } = collectUntil(lines, i + 1, "stepper")
142
+ blocks.push({ type: "stepper", steps: parseSteps(body) })
143
+ i = next
144
+ continue
145
+ }
146
+
147
+ if (tag.name === "columns") {
148
+ const { body, next } = collectUntil(lines, i + 1, "columns")
149
+ blocks.push({ type: "columns", columns: parseColumns(body) })
150
+ i = next
151
+ continue
152
+ }
153
+
154
+ if (tag.name === "code") {
155
+ const { body, next } = collectUntil(lines, i + 1, "code")
156
+ const inner = parseBlocks(body)
157
+ const code = inner.find((b) => b.type === "code")
158
+ if (code && code.type === "code") {
159
+ code.title = tag.attrs.title ?? null
160
+ code.lineNumbers = tag.attrs.lineNumbers === "true"
161
+ blocks.push(code)
162
+ }
163
+ i = next
164
+ continue
165
+ }
166
+
167
+ if (tag.name === "embed") {
168
+ blocks.push({ type: "embed", url: tag.attrs.url ?? "" })
169
+ i++
170
+ // tolerate optional {% endembed %}
171
+ if (i < lines.length && templateTag(lines[i])?.name === "endembed") i++
172
+ continue
173
+ }
174
+
175
+ if (tag.name === "content-ref") {
176
+ const { body, next } = collectUntil(lines, i + 1, "content-ref")
177
+ const inner: Inline[] = parseInline(body.join(" ").trim())
178
+ blocks.push({ type: "content-ref", url: tag.attrs.url ?? "", children: inner })
179
+ i = next
180
+ continue
181
+ }
182
+
183
+ if (tag.name === "updates") {
184
+ const { body, next } = collectUntil(lines, i + 1, "updates")
185
+ blocks.push({
186
+ type: "updates",
187
+ format: tag.attrs.format ?? null,
188
+ updates: parseUpdates(body),
189
+ })
190
+ i = next
191
+ continue
192
+ }
193
+
194
+ if (tag.name === "openapi-operation" || tag.name === "openapi") {
195
+ const { body, next } = collectUntil(lines, i + 1, tag.name)
196
+ const link = body.join(" ").match(/\[([^\]]*)\]\(([^)\s]+)\)/)
197
+ blocks.push({
198
+ type: "openapi-operation",
199
+ spec: tag.attrs.spec ?? "",
200
+ path: tag.attrs.path ?? "",
201
+ method: tag.attrs.method ?? "",
202
+ specUrl: link?.[2] ?? "",
203
+ label: link?.[1] ?? "",
204
+ })
205
+ i = next
206
+ continue
207
+ }
208
+
209
+ if (tag.name === "file") {
210
+ // Render file blocks as content-refs for now — same shape, different chrome.
211
+ blocks.push({ type: "content-ref", url: tag.attrs.src ?? "", children: parseInline(tag.attrs.caption ?? tag.attrs.src ?? "") })
212
+ i++
213
+ continue
214
+ }
215
+
216
+ // Unknown template tag: skip the line rather than corrupting output.
217
+ i++
218
+ continue
219
+ }
220
+
221
+ // <details><summary>…</summary> … </details> (GitBook expandable)
222
+ if (/^<details>/i.test(trimmed)) {
223
+ flushParagraph()
224
+ let summary = ""
225
+ let bodyStart = i + 1
226
+ const sameLine = trimmed.match(/<summary>(.*?)<\/summary>/i)
227
+ if (sameLine) {
228
+ summary = sameLine[1]
229
+ } else {
230
+ let j = i + 1
231
+ while (j < lines.length && !lines[j].trim()) j++
232
+ const m = lines[j]?.trim().match(/^<summary>(.*?)<\/summary>$/i)
233
+ if (m) {
234
+ summary = m[1]
235
+ bodyStart = j + 1
236
+ }
237
+ }
238
+ const { body, next } = collectHtmlUntil(lines, bodyStart, "</details>")
239
+ blocks.push({ type: "expandable", summary, children: parseBlocks(body) })
240
+ i = next
241
+ continue
242
+ }
243
+
244
+ // <table …>…</table> — GitBook exports complex/cards tables as HTML
245
+ if (/^<table[\s>]/i.test(trimmed)) {
246
+ flushParagraph()
247
+ const chunkLines = [line]
248
+ let j = i
249
+ while (!chunkLines.join("\n").includes("</table>") && j + 1 < lines.length) {
250
+ j++
251
+ chunkLines.push(lines[j])
252
+ }
253
+ const chunk = chunkLines.join("\n")
254
+ const view = chunk.match(/<table[^>]*data-view="([^"]*)"/i)?.[1]
255
+ const rowsHtml = [...chunk.matchAll(/<tr[^>]*>(.*?)<\/tr>/gis)].map((m) => m[1])
256
+ const parseCells = (rowHtml: string): Inline[][] =>
257
+ [...rowHtml.matchAll(/<t[dh][^>]*>(.*?)<\/t[dh]>/gis)].map((m) =>
258
+ parseInline(htmlToInlineMd(m[1]))
259
+ )
260
+ const allRows = rowsHtml.map(parseCells)
261
+ const [header, ...rest] = allRows.length ? allRows : [[]]
262
+ blocks.push({
263
+ type: "table",
264
+ header: header ?? [],
265
+ rows: rest,
266
+ ...(view ? { view } : {}),
267
+ })
268
+ i = j + 1
269
+ continue
270
+ }
271
+
272
+ // standalone <img …> line (e.g. GitBook drawings)
273
+ const soloHtmlImg = trimmed.match(/^<img[^>]*src="([^"]*)"[^>]*>$/i)
274
+ if (soloHtmlImg) {
275
+ flushParagraph()
276
+ const alt = trimmed.match(/alt="([^"]*)"/i)?.[1] ?? ""
277
+ blocks.push({ type: "figure", src: soloHtmlImg[1], alt, caption: "" })
278
+ i++
279
+ continue
280
+ }
281
+
282
+ // <figure><img …><figcaption>…</figcaption></figure>
283
+ if (/^<figure>/i.test(trimmed)) {
284
+ flushParagraph()
285
+ const chunkLines = [line]
286
+ let j = i
287
+ while (!chunkLines.join("\n").includes("</figure>") && j + 1 < lines.length) {
288
+ j++
289
+ chunkLines.push(lines[j])
290
+ }
291
+ const chunk = chunkLines.join("\n")
292
+ const img = chunk.match(/<img[^>]*src="([^"]*)"[^>]*>/i)
293
+ const alt = chunk.match(/<img[^>]*alt="([^"]*)"[^>]*>/i)
294
+ const cap = chunk.match(/<figcaption>(.*?)<\/figcaption>/is)
295
+ blocks.push({
296
+ type: "figure",
297
+ src: img?.[1] ?? "",
298
+ alt: alt?.[1] ?? "",
299
+ caption: (cap?.[1] ?? "").replace(/<\/?p>/g, "").trim(),
300
+ })
301
+ i = j + 1
302
+ continue
303
+ }
304
+
305
+ // plain markdown image on its own line
306
+ const soloImg = trimmed.match(/^!\[([^\]]*)\]\(([^)\s]+)\)$/)
307
+ if (soloImg) {
308
+ flushParagraph()
309
+ blocks.push({ type: "figure", src: soloImg[2], alt: soloImg[1], caption: "" })
310
+ i++
311
+ continue
312
+ }
313
+
314
+ // $$ math $$
315
+ if (trimmed === "$$") {
316
+ flushParagraph()
317
+ const formula: string[] = []
318
+ i++
319
+ while (i < lines.length && lines[i].trim() !== "$$") {
320
+ formula.push(lines[i])
321
+ i++
322
+ }
323
+ i++
324
+ blocks.push({ type: "math", formula: formula.join("\n") })
325
+ continue
326
+ }
327
+
328
+ // fenced code (``` or ~~~)
329
+ const fence = trimmed.match(/^(`{3,}|~{3,})(\S*)\s*$/)
330
+ if (fence) {
331
+ flushParagraph()
332
+ const marker = fence[1][0]
333
+ const code: string[] = []
334
+ i++
335
+ while (i < lines.length && !lines[i].trim().startsWith(marker.repeat(3))) {
336
+ code.push(lines[i])
337
+ i++
338
+ }
339
+ i++
340
+ blocks.push({
341
+ type: "code",
342
+ language: fence[2] || null,
343
+ title: null,
344
+ lineNumbers: false,
345
+ code: code.join("\n"),
346
+ })
347
+ continue
348
+ }
349
+
350
+ // indented code block (4+ spaces, GFM)
351
+ if (paragraph.length === 0 && /^ {4,}\S/.test(line) && !line.trim().match(/^([-*+]|\d+[.)])\s/)) {
352
+ const code: string[] = []
353
+ while (
354
+ i < lines.length &&
355
+ (/^ {4,}\S/.test(lines[i]) || (!lines[i].trim() && /^ {4,}\S/.test(lines[i + 1] ?? "")))
356
+ ) {
357
+ code.push(lines[i].slice(4))
358
+ i++
359
+ }
360
+ blocks.push({ type: "code", language: null, title: null, lineNumbers: false, code: code.join("\n") })
361
+ continue
362
+ }
363
+
364
+ // setext headings: a paragraph line followed by ==== (h1) or ---- (h2)
365
+ if (paragraph.length && /^=+$/.test(trimmed)) {
366
+ const text = paragraph.join(" ").trim()
367
+ paragraph.length = 0
368
+ blocks.push({ type: "heading", level: 1, children: parseInline(text) })
369
+ i++
370
+ continue
371
+ }
372
+ if (paragraph.length && /^-+$/.test(trimmed)) {
373
+ const text = paragraph.join(" ").trim()
374
+ paragraph.length = 0
375
+ blocks.push({ type: "heading", level: 2, children: parseInline(text) })
376
+ i++
377
+ continue
378
+ }
379
+
380
+ // heading (GitBook exports can contain empty headings like a bare "##")
381
+ const heading = trimmed.match(/^(#{1,6})(?:\s+(.*))?$/)
382
+ if (heading) {
383
+ flushParagraph()
384
+ blocks.push({
385
+ type: "heading",
386
+ level: heading[1].length as 1 | 2 | 3 | 4 | 5 | 6,
387
+ children: parseInline(heading[2] ?? ""),
388
+ })
389
+ i++
390
+ continue
391
+ }
392
+
393
+ // divider
394
+ if (/^(-{3,}|\*{3,}|_{3,})$/.test(trimmed)) {
395
+ flushParagraph()
396
+ blocks.push({ type: "divider" })
397
+ i++
398
+ continue
399
+ }
400
+
401
+ // blockquote
402
+ if (trimmed.startsWith(">")) {
403
+ flushParagraph()
404
+ const quote: string[] = []
405
+ while (i < lines.length && lines[i].trim().startsWith(">")) {
406
+ quote.push(lines[i].trim().replace(/^>\s?/, ""))
407
+ i++
408
+ }
409
+ blocks.push({ type: "blockquote", children: parseBlocks(quote) })
410
+ continue
411
+ }
412
+
413
+ // table
414
+ if (trimmed.startsWith("|") && lines[i + 1]?.trim().match(/^\|[\s:|-]+\|$/)) {
415
+ flushParagraph()
416
+ const parseRow = (row: string): Inline[][] =>
417
+ row
418
+ .trim()
419
+ .replace(/^\||\|$/g, "")
420
+ .split("|")
421
+ .map((cell) => parseInline(cell.trim()))
422
+ const header = parseRow(lines[i])
423
+ i += 2
424
+ const rows: Inline[][][] = []
425
+ while (i < lines.length && lines[i].trim().startsWith("|")) {
426
+ rows.push(parseRow(lines[i]))
427
+ i++
428
+ }
429
+ blocks.push({ type: "table", header, rows })
430
+ continue
431
+ }
432
+
433
+ // list (unordered / ordered / task, with GFM nesting)
434
+ const listMatch = line.match(/^(\s*)([-*+]|\d+[.)])\s+(.*)$/)
435
+ if (listMatch && listMatch[1].length < 4) {
436
+ flushParagraph()
437
+ const { list, next } = parseList(lines, i)
438
+ blocks.push(list)
439
+ i = next
440
+ continue
441
+ }
442
+
443
+ paragraph.push(trimmed)
444
+ i++
445
+ }
446
+
447
+ flushParagraph()
448
+ return blocks
449
+ }
450
+
451
+ function parseTabs(lines: string[]): TabNode[] {
452
+ const tabs: TabNode[] = []
453
+ let i = 0
454
+ while (i < lines.length) {
455
+ const tag = templateTag(lines[i])
456
+ if (tag?.name === "tab") {
457
+ const { body, next } = collectUntil(lines, i + 1, "tab")
458
+ tabs.push({ type: "tab", title: tag.attrs.title ?? "Tab", children: parseBlocks(body) })
459
+ i = next
460
+ } else {
461
+ i++
462
+ }
463
+ }
464
+ return tabs
465
+ }
466
+
467
+ function parseSteps(lines: string[]): StepNode[] {
468
+ const steps: StepNode[] = []
469
+ let i = 0
470
+ while (i < lines.length) {
471
+ const tag = templateTag(lines[i])
472
+ if (tag?.name === "step") {
473
+ const { body, next } = collectUntil(lines, i + 1, "step")
474
+ // GitBook convention: the step's first heading is its title
475
+ const inner = parseBlocks(body)
476
+ let title = ""
477
+ if (inner[0]?.type === "heading") {
478
+ title = plainText(inner[0].children)
479
+ inner.shift()
480
+ }
481
+ steps.push({ type: "step", title, children: inner })
482
+ i = next
483
+ } else {
484
+ i++
485
+ }
486
+ }
487
+ return steps
488
+ }
489
+
490
+ interface ParsedList {
491
+ list: Extract<Block, { type: "list" }>
492
+ next: number
493
+ }
494
+
495
+ // Indent-aware list parser: deeper-indented marker lines become nested lists
496
+ // inside the item (handled recursively via parseBlocks on dedented lines).
497
+ function parseList(lines: string[], start: number): ParsedList {
498
+ const first = lines[start].match(/^(\s*)([-*+]|\d+[.)])\s+/)!
499
+ const baseIndent = first[1].length
500
+ const ordered = /^\d/.test(first[2])
501
+ const items: ListItemNode[] = []
502
+ let task = false
503
+ let i = start
504
+
505
+ while (i < lines.length) {
506
+ const m = lines[i].match(/^(\s*)([-*+]|\d+[.)])\s+(.*)$/)
507
+ if (!m || m[1].length !== baseIndent || /^\d/.test(m[2]) !== ordered) break
508
+
509
+ let content = m[3]
510
+ let checked: boolean | undefined
511
+ const taskMatch = content.match(/^\[([ xX])\]\s+(.*)$/)
512
+ if (taskMatch) {
513
+ task = true
514
+ checked = taskMatch[1] !== " "
515
+ content = taskMatch[2]
516
+ }
517
+
518
+ // Everything indented deeper than the marker belongs to this item.
519
+ const contIndent = baseIndent + m[2].length + 1
520
+ const body: string[] = []
521
+ i++
522
+ while (i < lines.length) {
523
+ const raw = lines[i]
524
+ if (!raw.trim()) {
525
+ // blank line stays in the item only if more indented content follows
526
+ const lookahead = lines[i + 1]
527
+ if (lookahead !== undefined && /^\s+/.test(lookahead) && lookahead.search(/\S/) >= Math.min(contIndent, baseIndent + 2)) {
528
+ body.push("")
529
+ i++
530
+ continue
531
+ }
532
+ break
533
+ }
534
+ const indent = raw.search(/\S/)
535
+ if (indent >= Math.min(contIndent, baseIndent + 2)) {
536
+ body.push(raw.slice(Math.min(indent, contIndent)))
537
+ i++
538
+ continue
539
+ }
540
+ break
541
+ }
542
+
543
+ const para: Block = { type: "paragraph", children: parseInline(content) }
544
+ const children: Block[] = body.length ? [para, ...parseBlocks(body)] : [para]
545
+ items.push(
546
+ checked === undefined ? { type: "listItem", children } : { type: "listItem", children, checked }
547
+ )
548
+
549
+ // skip a single blank line between sibling items
550
+ if (i < lines.length && !lines[i].trim()) {
551
+ const after = lines[i + 1]?.match(/^(\s*)([-*+]|\d+[.)])\s+/)
552
+ if (after && after[1].length === baseIndent) i++
553
+ else break
554
+ }
555
+ }
556
+
557
+ return { list: { type: "list", ordered, task, items }, next: i }
558
+ }
559
+
560
+ function parseUpdates(lines: string[]): UpdateNode[] {
561
+ const updates: UpdateNode[] = []
562
+ let i = 0
563
+ while (i < lines.length) {
564
+ const tag = templateTag(lines[i])
565
+ if (tag?.name === "update") {
566
+ const { body, next } = collectUntil(lines, i + 1, "update")
567
+ updates.push({
568
+ type: "update",
569
+ date: tag.attrs.date ?? "",
570
+ children: parseBlocks(body),
571
+ })
572
+ i = next
573
+ } else {
574
+ i++
575
+ }
576
+ }
577
+ return updates
578
+ }
579
+
580
+ function parseColumns(lines: string[]): ColumnNode[] {
581
+ const cols: ColumnNode[] = []
582
+ let i = 0
583
+ while (i < lines.length) {
584
+ const tag = templateTag(lines[i])
585
+ if (tag?.name === "column") {
586
+ const { body, next } = collectUntil(lines, i + 1, "column")
587
+ cols.push({ type: "column", children: parseBlocks(body) })
588
+ i = next
589
+ } else {
590
+ i++
591
+ }
592
+ }
593
+ return cols
594
+ }