markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,869 @@
1
+ """Defensive, AST-aware Markdown sectioniser.
2
+
3
+ The parser tokenises Markdown with ``markdown-it-py`` and uses the block tokens'
4
+ source maps to cut the *verbatim* source into heading-delimited sections. Only
5
+ headings that are real top-level AST nodes open a section, so ``#`` lines inside
6
+ code fences, block quotes or list items never do.
7
+
8
+ Defences against real-world documentation:
9
+
10
+ * **Skipped levels** (``#`` straight to ``####``) - a heading stack pops every
11
+ entry with level >= L before pushing, so breadcrumbs stay well formed.
12
+ * **Preamble** - badges/summaries before the first heading become
13
+ ``[Overview / Preamble]``.
14
+ * **Front matter** - a leading block that really is YAML is kept out of the AST
15
+ (CommonMark would misread it as a thematic break plus a setext heading) and mined
16
+ for ``title:``. A leading ``---`` rule followed by prose is left alone.
17
+ * **Unclosed code fences** - CommonMark runs such a fence to EOF, swallowing every
18
+ heading after it. ``_resilient_fence`` replaces markdown-it's own fence rule and ends
19
+ the fence at the next plausible heading instead, in the same pass. A fence that
20
+ markdown-it closed normally is left exactly as CommonMark read it, unless it holds an
21
+ opening fence of its own (```` ```bash ```` around ```` ```python ````), which means the
22
+ markers were paired wrongly. Deciding on positive evidence only is what keeps a
23
+ well-formed document - one that simply shows headings inside a fence - untouched; the
24
+ cost is that a stray bare marker, which shifts every pairing after it, leaves the
25
+ headings inside those fences hidden.
26
+ * **Oversized / heading-less text** - split on paragraph boundaries into
27
+ ``Path (Part n)`` parts that reassemble byte-for-byte. A fenced block is only cut
28
+ when it exceeds the limit on its own.
29
+ * **Colliding breadcrumbs** - repeated paths get a ``[n]`` suffix, also against
30
+ generated ``(Part n)`` paths, so every stored path addresses exactly one thing.
31
+
32
+ Every section is additionally broken into *units* (``extract_units``): the plain text of
33
+ each paragraph, list item, table row and code block. They feed passage-level embeddings.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import logging
39
+ import re
40
+ from collections.abc import Sequence
41
+ from dataclasses import dataclass, replace
42
+ from typing import Protocol
43
+
44
+ from markdown_it import MarkdownIt
45
+ from markdown_it.rules_block.fence import make_fence_rule
46
+ from markdown_it.rules_block.state_block import StateBlock
47
+ from markdown_it.token import Token
48
+
49
+ from markdown_memory.exceptions import ASTParseError
50
+ from markdown_memory.models import (
51
+ PATH_SEPARATOR,
52
+ PREAMBLE_TITLE,
53
+ ParsedDocument,
54
+ SectionDraft,
55
+ part_path,
56
+ )
57
+
58
+ DEFAULT_MAX_SECTION_CHARS = 3200 # ~800 tokens
59
+ MAX_UNITS_PER_SECTION = 64
60
+ MAX_UNIT_CHARS = 600
61
+ UNTITLED_HEADING = "(untitled)"
62
+
63
+ _FENCE_OPEN = re.compile(r"^( *)(`{3,}|~{3,})(.*)$")
64
+ _MAX_TOP_LEVEL_INDENT = 3
65
+ logger = logging.getLogger(__name__)
66
+
67
+ _STOCK_FENCE = make_fence_rule()
68
+ _ATX_HEADING = re.compile(r"^ {0,3}(#{1,6})[ \t]+\S")
69
+ _FRONT_MATTER_TITLE = re.compile(r"^title\s*:\s*(.+?)\s*$", re.IGNORECASE)
70
+ # An unquoted key starts with a letter of any script or "_" (`[^\W\d]`), never a digit.
71
+ _YAML_KEY = re.compile(r"""^(?:"[^"]+"|'[^']+'|[^\W\d][^:#]*?)\s*:(\s|$)""")
72
+ _HTML_TAG_NAME = re.compile(r"^</?([A-Za-z][A-Za-z0-9-]*)")
73
+ # Inline HTML that only styles a heading. Any other "tag" is kept as title text: in
74
+ # technical docs `Option<T>` or `<details>` in a heading is almost always literal.
75
+ _FORMATTING_TAGS = frozenset(
76
+ {
77
+ "a", "abbr", "b", "br", "code", "del", "em", "font", "i", "img", "ins", "kbd",
78
+ "mark", "s", "small", "span", "strong", "sub", "sup", "u",
79
+ }
80
+ ) # fmt: skip
81
+ _WHITESPACE = re.compile(r"\s+")
82
+ _HTML_TAG = re.compile(r"<[^>]+>")
83
+ # Removed before tags are: a ">" inside a comment would end the "tag" early and leak the
84
+ # rest of the comment as text. An unterminated comment hides everything after it.
85
+ _HTML_COMMENT = re.compile(r"<!--.*?(?:-->|\Z)", re.DOTALL)
86
+ _BLOCK_CLOSERS = {
87
+ "paragraph_open": "paragraph_close",
88
+ "heading_open": "heading_close",
89
+ "table_open": "table_close",
90
+ "bullet_list_open": "bullet_list_close",
91
+ "ordered_list_open": "ordered_list_close",
92
+ "blockquote_open": "blockquote_close",
93
+ }
94
+
95
+ # Languages where a line starting with "# " cannot be a comment. Inside an unclosed
96
+ # fence of any *other* language a level-1 "# ..." line is assumed to be a comment.
97
+ _HASH_IS_NOT_COMMENT = frozenset(
98
+ {
99
+ "c", "cpp", "c++", "cs", "csharp", "css", "go", "golang", "html", "java",
100
+ "javascript", "js", "json", "jsonc", "jsx", "kotlin", "lua", "rs", "rust",
101
+ "scss", "sql", "swift", "ts", "tsx", "typescript", "xml",
102
+ }
103
+ ) # fmt: skip
104
+
105
+
106
+ @dataclass(slots=True, frozen=True)
107
+ class _Heading:
108
+ line: int # 0-based index of the heading's first source line
109
+ level: int
110
+ title: str
111
+
112
+
113
+ @dataclass(slots=True, frozen=True)
114
+ class _Block:
115
+ """A blank-line-delimited run of text inside one section (character offsets)."""
116
+
117
+ start: int
118
+ end: int
119
+ has_fence: bool
120
+
121
+
122
+ class MarkdownParser:
123
+ """Turn Markdown source into breadcrumbed, size-bounded sections."""
124
+
125
+ def __init__(self, max_section_chars: int = DEFAULT_MAX_SECTION_CHARS) -> None:
126
+ if max_section_chars < 1:
127
+ raise ValueError("max_section_chars must be positive")
128
+ self._max_chars = max_section_chars
129
+ self._md = MarkdownIt("commonmark").enable("table")
130
+ # The chain the stock rule is registered with. Dropping it would stop the fence rule
131
+ # being consulted inside list items and block quotes, so a fence in a list would be
132
+ # read as a paragraph and headings inside it would become real headings.
133
+ self._md.block.ruler.at(
134
+ "fence", _resilient_fence, {"alt": ["paragraph", "reference", "blockquote", "list"]}
135
+ )
136
+ # Passages are extracted with the stock rule. Fence recovery is there to rescue a
137
+ # document's structure; inside one section an unclosed fence is usually a fenced
138
+ # sample that oversized-section splitting cut in half, and reinterpreting its
139
+ # contents as prose would index the template rather than the documentation.
140
+ self._units_md = MarkdownIt("commonmark").enable("table")
141
+
142
+ def parse(self, text: str, *, fallback_title: str = "Untitled") -> ParsedDocument:
143
+ """Parse ``text``; never raises for malformed Markdown, only for tokeniser failure."""
144
+ lines = _split_lines(text)
145
+ body_start, front_matter_title = _scan_front_matter(lines)
146
+ headings = self._find_headings(lines, body_start)
147
+
148
+ used_paths: set[str] = set()
149
+ first_heading_line = headings[0].line if headings else len(lines)
150
+ sections: list[SectionDraft] = self._build_section(
151
+ lines,
152
+ start=0,
153
+ stop=first_heading_line,
154
+ title=PREAMBLE_TITLE,
155
+ level=0,
156
+ path=PREAMBLE_TITLE,
157
+ has_heading_line=False,
158
+ )
159
+ used_paths.update(section.heading_path for section in sections)
160
+ used_paths.add(PREAMBLE_TITLE)
161
+
162
+ stack: list[tuple[int, str]] = []
163
+ next_occurrence: dict[str, int] = {}
164
+ for index, heading in enumerate(headings):
165
+ while stack and stack[-1][0] >= heading.level:
166
+ stack.pop()
167
+ parents = [title for _, title in stack]
168
+ stop = headings[index + 1].line if index + 1 < len(headings) else len(lines)
169
+ # Every stored path must address exactly one thing: a candidate is rejected
170
+ # when it - or any "(Part n)" path generated from it - is already taken. The
171
+ # section is built once; trying another name only renames the drafts.
172
+ base_path = PATH_SEPARATOR.join([*parents, heading.title])
173
+ drafts = self._build_section(
174
+ lines,
175
+ start=heading.line,
176
+ stop=stop,
177
+ title=heading.title,
178
+ level=heading.level,
179
+ path=base_path,
180
+ has_heading_line=True,
181
+ )
182
+ occurrence = next_occurrence.get(base_path, 1)
183
+ while True:
184
+ title = heading.title if occurrence == 1 else f"{heading.title} [{occurrence}]"
185
+ path = PATH_SEPARATOR.join([*parents, title])
186
+ claimed = {path, *(part_path(path, d.part_index) for d in drafts if d.part_index)}
187
+ if not claimed & used_paths:
188
+ break
189
+ occurrence += 1
190
+ next_occurrence[base_path] = occurrence + 1
191
+ if occurrence > 1:
192
+ drafts = [
193
+ replace(
194
+ draft,
195
+ heading_title=title,
196
+ base_path=path,
197
+ heading_path=part_path(path, draft.part_index)
198
+ if draft.part_index
199
+ else path,
200
+ )
201
+ for draft in drafts
202
+ ]
203
+ used_paths |= claimed
204
+ stack.append((heading.level, title))
205
+ sections.extend(drafts)
206
+
207
+ return ParsedDocument(
208
+ title=_pick_title(headings, front_matter_title, fallback_title),
209
+ sections=tuple(sections),
210
+ line_count=len(lines),
211
+ )
212
+
213
+ # ------------------------------------------------------------------ AST walk
214
+
215
+ def _tokenize(self, source: str, *, recover_fences: bool = True) -> list[Token]:
216
+ try:
217
+ return (self._md if recover_fences else self._units_md).parse(source)
218
+ except Exception as exc: # markdown-it has no dedicated error type
219
+ raise ASTParseError(f"markdown-it failed to tokenise the document: {exc}") from exc
220
+
221
+ def _find_headings(self, lines: Sequence[str], body_start: int) -> list[_Heading]:
222
+ """Collect the document's top-level headings in a single tokenisation pass.
223
+
224
+ An unclosed fence no longer hides the rest of the document: ``_resilient_fence``
225
+ ends it at the first plausible heading while the block parser is running, so the
226
+ headings after it are simply there.
227
+ """
228
+ tokens = self._tokenize("\n".join(lines[body_start:]))
229
+ headings: list[_Heading] = []
230
+ for position, token in enumerate(tokens):
231
+ if token.level != 0 or token.map is None or token.type != "heading_open":
232
+ continue
233
+ inline = tokens[position + 1] if position + 1 < len(tokens) else None
234
+ headings.append(
235
+ _Heading(
236
+ line=body_start + token.map[0],
237
+ level=int(token.tag[1:]),
238
+ title=_inline_text(inline),
239
+ )
240
+ )
241
+ return headings
242
+
243
+ # ------------------------------------------------------------------ sections
244
+
245
+ def _build_section(
246
+ self,
247
+ lines: Sequence[str],
248
+ *,
249
+ start: int,
250
+ stop: int,
251
+ title: str,
252
+ level: int,
253
+ path: str,
254
+ has_heading_line: bool,
255
+ ) -> list[SectionDraft]:
256
+ """Build the section for ``lines[start:stop]``, split into parts if oversized."""
257
+ while start < stop and not lines[start].strip():
258
+ start += 1
259
+ while stop > start and not lines[stop - 1].strip():
260
+ stop -= 1
261
+ if start >= stop:
262
+ return []
263
+ content = "\n".join(lines[start:stop])
264
+ first_line = start + 1 # 1-based
265
+ if len(content) <= self._max_chars:
266
+ return [
267
+ SectionDraft(
268
+ heading_title=title,
269
+ heading_level=level,
270
+ heading_path=path,
271
+ base_path=path,
272
+ content=content,
273
+ start_line=first_line,
274
+ end_line=stop,
275
+ units=self.extract_units(content, skip_heading=has_heading_line),
276
+ )
277
+ ]
278
+ # Only trailing newlines are dropped from a part, and they are recoverable from
279
+ # the line numbers, so join_parts() can rebuild the section byte-for-byte.
280
+ spans = split_into_spans(content, self._max_chars, glue_first=has_heading_line)
281
+ table_headers = _table_headers(lines[start:stop])
282
+ drafts: list[SectionDraft] = []
283
+ for number, (begin, end) in enumerate(spans, start=1):
284
+ part = content[begin:end].rstrip("\n")
285
+ first_part_line = content.count("\n", 0, begin)
286
+ part_start = first_line + first_part_line
287
+ # A part that starts in the middle of a table has lost the header row, and
288
+ # without it the rows are just a paragraph. Units are extracted from the part
289
+ # with its header restored; the stored content stays verbatim.
290
+ header = (
291
+ table_headers.get(first_part_line)
292
+ if content[begin - 1 : begin] in {"", "\n"}
293
+ else None
294
+ )
295
+ drafts.append(
296
+ SectionDraft(
297
+ heading_title=title,
298
+ heading_level=level,
299
+ heading_path=part_path(path, number),
300
+ base_path=path,
301
+ content=part,
302
+ start_line=part_start,
303
+ end_line=part_start + part.count("\n"),
304
+ part_index=number,
305
+ units=self.extract_units(
306
+ f"{header}\n{part}" if header else part,
307
+ skip_heading=has_heading_line and number == 1,
308
+ ),
309
+ )
310
+ )
311
+ return drafts
312
+
313
+ # ------------------------------------------------------------------ units
314
+
315
+ def extract_units(self, content: str, *, skip_heading: bool = False) -> tuple[str, ...]:
316
+ """Plain-text passages of ``content``: paragraphs, list items, table rows, code blocks.
317
+
318
+ Table rows are rendered as ``Header: cell; Header: cell`` so that a row keeps its
319
+ meaning without the rest of the table. ``skip_heading`` drops the section's own
320
+ heading (its text already lives in the breadcrumb).
321
+ """
322
+ tokens = self._tokenize(content, recover_fences=False)
323
+ units: list[str] = []
324
+ index = 0
325
+ heading_skipped = not skip_heading
326
+ while index < len(tokens) and len(units) < MAX_UNITS_PER_SECTION:
327
+ token = tokens[index]
328
+ if token.level != 0:
329
+ index += 1
330
+ continue
331
+ end = _block_end(tokens, index)
332
+ block = tokens[index : end + 1]
333
+ if token.type == "heading_open" and not heading_skipped:
334
+ heading_skipped = True
335
+ elif token.type == "table_open":
336
+ units.extend(_table_rows(block))
337
+ elif token.type in {"bullet_list_open", "ordered_list_open"}:
338
+ units.extend(_list_items(block))
339
+ elif token.type == "blockquote_open":
340
+ units.extend(_leaf_texts(block)) # one unit per quoted paragraph / code block
341
+ else:
342
+ units.append(" ".join(_leaf_texts(block)))
343
+ index = end + 1
344
+ passages: list[str] = []
345
+ for unit in units:
346
+ for window in _windows(unit):
347
+ if len(passages) >= MAX_UNITS_PER_SECTION:
348
+ return tuple(passages)
349
+ cleaned = _WHITESPACE.sub(" ", window.replace("|", " ")).strip()
350
+ if cleaned:
351
+ passages.append(cleaned)
352
+ return tuple(passages)
353
+
354
+
355
+ # ---------------------------------------------------------------------- helpers
356
+
357
+
358
+ def _last_space(text: str) -> int:
359
+ """Index just past the last whitespace run, or -1. Any whitespace, not just a space:
360
+ a line of tab-separated columns has no literal space to break on."""
361
+ for index in range(len(text) - 1, -1, -1):
362
+ if text[index].isspace():
363
+ return index
364
+ return -1
365
+
366
+
367
+ def _windows(text: str) -> list[str]:
368
+ """``text`` as consecutive pieces of at most ``MAX_UNIT_CHARS``, nothing discarded.
369
+
370
+ A fenced block arrives here as one unit. Slicing it to the limit - which is what this
371
+ used to do - gave the tail no vector at all: in this project's own CLAUDE.md the
372
+ command list was cut mid-word at ``reindex_docs.py D``, and every command after that
373
+ point was unreachable by passage search while sitting in the index in plain sight.
374
+ Across the vendored corpus that was 73,598 characters with no passage vector.
375
+
376
+ Pieces break at a newline where possible, then at a sentence end, then at a space, so
377
+ a window is a run of whole lines or whole words rather than an arbitrary cut. The
378
+ limit is a window size now, not a truncation point.
379
+ """
380
+ if len(text) <= MAX_UNIT_CHARS:
381
+ return [text]
382
+ pieces: list[str] = []
383
+ remaining = text
384
+ while len(remaining) > MAX_UNIT_CHARS:
385
+ head = remaining[:MAX_UNIT_CHARS]
386
+ cut = max(head.rfind("\n"), head.rfind(". "), _last_space(head))
387
+ # A cut in the first half would make a window mostly empty; a hard cut keeps the
388
+ # windows even, and no character is lost either way.
389
+ if cut < MAX_UNIT_CHARS // 2:
390
+ cut = MAX_UNIT_CHARS
391
+ pieces.append(remaining[:cut])
392
+ remaining = remaining[cut:]
393
+ if remaining:
394
+ pieces.append(remaining)
395
+ return pieces
396
+
397
+
398
+ _TABLE_DELIMITER = re.compile(
399
+ r"^ {0,3}\|?[ \t]*:?-{1,}:?[ \t]*(\|[ \t]*:?-{1,}:?[ \t]*)*\|?[ \t]*$"
400
+ )
401
+
402
+
403
+ def _table_headers(lines: Sequence[str]) -> dict[int, str]:
404
+ """Map each table *body* row (by line index) to its table's header + delimiter rows."""
405
+ headers: dict[int, str] = {}
406
+ index = 0
407
+ while index + 1 < len(lines):
408
+ is_header = "|" in lines[index] and "|" in lines[index + 1]
409
+ if is_header and _TABLE_DELIMITER.match(lines[index + 1]) and lines[index].strip():
410
+ header = f"{lines[index]}\n{lines[index + 1]}"
411
+ index += 2
412
+ while index < len(lines) and lines[index].strip() and "|" in lines[index]:
413
+ headers[index] = header
414
+ index += 1
415
+ else:
416
+ index += 1
417
+ return headers
418
+
419
+
420
+ def _split_lines(text: str) -> list[str]:
421
+ normalized = text.removeprefix("").replace("\r\n", "\n").replace("\r", "\n")
422
+ lines = normalized.split("\n")
423
+ if lines and lines[-1] == "":
424
+ lines.pop()
425
+ return lines
426
+
427
+
428
+ def _scan_front_matter(lines: Sequence[str]) -> tuple[int, str | None]:
429
+ """Return ``(first body line, front-matter title)`` for a leading YAML block."""
430
+ if not lines or lines[0].rstrip() != "---":
431
+ return 0, None
432
+ for index in range(1, len(lines)):
433
+ if lines[index].rstrip() in {"---", "..."}:
434
+ block = lines[1:index]
435
+ if not _looks_like_yaml(block):
436
+ return 0, None
437
+ title: str | None = None
438
+ for line in block:
439
+ match = _FRONT_MATTER_TITLE.match(line)
440
+ if match:
441
+ title = match.group(1).strip("\"'") or None
442
+ break
443
+ return index + 1, title
444
+ return 0, None
445
+
446
+
447
+ # "Note: read this first" between two rules is a setext heading, not metadata. A lone
448
+ # `key: value` line is otherwise taken for front matter, as every site generator does.
449
+ _ADMONITION_KEYS = frozenset(
450
+ {"note", "warning", "tip", "todo", "important", "caution", "hint", "see also", "example"}
451
+ )
452
+
453
+
454
+ def _looks_like_yaml(block: Sequence[str]) -> bool:
455
+ """Distinguish front matter from a document that merely opens with a ``---`` rule.
456
+
457
+ Every non-blank line must be a mapping key, a list item, a comment, a continuation
458
+ or a closing flow bracket, and at least one top-level key must exist. A ``# ...``
459
+ line followed by a blank line is taken for a Markdown heading, not a YAML comment.
460
+ """
461
+ has_key = False
462
+ content_lines = [line for line in block if line.strip()]
463
+ if len(content_lines) == 1:
464
+ key = content_lines[0].split(":", 1)[0].strip().strip("\"'").lower()
465
+ if key in _ADMONITION_KEYS:
466
+ return False
467
+ for index, line in enumerate(block):
468
+ stripped = line.strip()
469
+ if not stripped:
470
+ continue
471
+ if _YAML_KEY.match(line):
472
+ has_key = True
473
+ elif stripped.startswith("#"):
474
+ followed_by_blank = index + 1 < len(block) and not block[index + 1].strip()
475
+ if _ATX_HEADING.match(line) and followed_by_blank:
476
+ return False
477
+ elif not (line[0] in " \t" or stripped.startswith("- ") or stripped in {"-", "]", "}"}):
478
+ return False
479
+ return has_key
480
+
481
+
482
+ def _plain_inline(inline: Token | None) -> str:
483
+ """Visible text of an inline token: markup stripped, code and image alt text kept."""
484
+ if inline is None or inline.type != "inline":
485
+ return ""
486
+ fragments: list[str] = []
487
+ for child in inline.children or []:
488
+ if child.type in {"text", "code_inline", "image"}:
489
+ fragments.append(child.content)
490
+ elif child.type in {"softbreak", "hardbreak"}:
491
+ fragments.append(" ")
492
+ elif child.type == "html_inline":
493
+ tag = _HTML_TAG_NAME.match(child.content)
494
+ # Case-sensitive on purpose: `<u>` is underline, `<U>` is a type parameter.
495
+ if tag is not None and tag.group(1) not in _FORMATTING_TAGS:
496
+ fragments.append(child.content) # `Option<T>`: a type parameter, not markup
497
+ elif tag is not None and tag.group(1) == "br":
498
+ fragments.append(" ") # the only line break a table cell can hold
499
+ return _WHITESPACE.sub(" ", "".join(fragments)).strip()
500
+
501
+
502
+ def _inline_text(inline: Token | None) -> str:
503
+ """Plain text of a heading's inline token, never empty."""
504
+ return _plain_inline(inline) or UNTITLED_HEADING
505
+
506
+
507
+ def _leaf_texts(block: Sequence[Token]) -> list[str]:
508
+ """Text of every leaf in ``block``, at any depth: inline runs, code and raw HTML."""
509
+ texts: list[str] = []
510
+ for token in block:
511
+ if token.type == "inline":
512
+ texts.append(_plain_inline(token))
513
+ elif token.type in {"fence", "code_block"}:
514
+ texts.append(token.content)
515
+ elif token.type == "html_block":
516
+ texts.append(_HTML_TAG.sub(" ", _HTML_COMMENT.sub(" ", token.content)))
517
+ return [text for text in texts if text.strip()]
518
+
519
+
520
+ def _block_end(tokens: Sequence[Token], start: int) -> int:
521
+ """Index of the token closing the top-level block that opens at ``start``."""
522
+ closer = _BLOCK_CLOSERS.get(tokens[start].type)
523
+ if closer is None:
524
+ return start
525
+ for index in range(start + 1, len(tokens)):
526
+ if tokens[index].type == closer and tokens[index].level == tokens[start].level:
527
+ return index
528
+ return len(tokens) - 1
529
+
530
+
531
+ def _table_rows(block: Sequence[Token]) -> list[str]:
532
+ """One unit per body row, each cell labelled with its column header.
533
+
534
+ A table without body rows yields its header cells instead: they are visible text,
535
+ and a section holding nothing else would otherwise pass for a heading-only stub.
536
+ """
537
+ headers: list[str] = []
538
+ rows: list[str] = []
539
+ cells: list[str] = []
540
+ in_head = False
541
+ for position, token in enumerate(block):
542
+ if token.type == "thead_open":
543
+ in_head = True
544
+ elif token.type == "thead_close":
545
+ in_head = False
546
+ elif token.type in {"th_open", "td_open"}:
547
+ text = _plain_inline(block[position + 1] if position + 1 < len(block) else None)
548
+ (headers if in_head else cells).append(text)
549
+ elif token.type == "tr_close" and not in_head:
550
+ labelled = [
551
+ f"{header}: {cell}" if header else cell
552
+ for header, cell in zip(headers + [""] * len(cells), cells, strict=False)
553
+ if cell
554
+ ]
555
+ rows.append("; ".join(labelled))
556
+ cells = []
557
+ if rows:
558
+ return rows
559
+ header_row = "; ".join(header for header in headers if header)
560
+ return [header_row] if header_row else []
561
+
562
+
563
+ def _list_items(block: Sequence[Token]) -> list[str]:
564
+ """One unit per top-level list item, nested content included."""
565
+ items: list[str] = []
566
+ fragments: list[str] = []
567
+ for token in block:
568
+ if token.type == "list_item_open" and token.level == 1:
569
+ fragments = []
570
+ elif token.type in {"inline", "fence", "code_block", "html_block"}:
571
+ fragments.extend(_leaf_texts([token]))
572
+ elif token.type == "list_item_close" and token.level == 1:
573
+ items.append(" ".join(fragments))
574
+ return items
575
+
576
+
577
+ def _pick_title(headings: Sequence[_Heading], front_matter_title: str | None, fallback: str) -> str:
578
+ for heading in headings:
579
+ if heading.level == 1 and heading.title != UNTITLED_HEADING:
580
+ return heading.title
581
+ if front_matter_title:
582
+ return front_matter_title
583
+ for heading in headings:
584
+ if heading.title != UNTITLED_HEADING:
585
+ return heading.title
586
+ return fallback
587
+
588
+
589
+ def _is_closing_fence(line: str, marker: str, max_indent: int) -> bool:
590
+ stripped = line.strip()
591
+ indent = len(line) - len(line.lstrip(" "))
592
+ return (
593
+ indent <= max_indent
594
+ and len(stripped) >= len(marker)
595
+ and stripped == marker[0] * len(stripped)
596
+ )
597
+
598
+
599
+ def _opening_fence(line: str) -> tuple[int, str, str] | None:
600
+ """``(indent, marker, info)`` when ``line`` can open a fenced block."""
601
+ match = _FENCE_OPEN.match(line)
602
+ if match is None:
603
+ return None
604
+ indent, marker, info = len(match.group(1)), match.group(2), match.group(3).strip()
605
+ if marker[0] == "`" and "`" in info: # CommonMark: that is inline code, not a fence
606
+ return None
607
+ return indent, marker, info
608
+
609
+
610
+ def _recovery_line(state: StateBlock, start: int, stop: int, language: str) -> int | None:
611
+ """First line in ``(start, stop)`` that should be read as a heading again.
612
+
613
+ A candidate is an ATX heading preceded by a blank line. A level-1 candidate is only
614
+ trusted when ``#`` cannot start a comment in the fence's language, because a lone
615
+ ``# comment`` is far more common inside shell or Python than a heading is.
616
+ """
617
+ for index in range(start + 2, min(stop, len(state.bMarks) - 1)):
618
+ line = state.src[state.bMarks[index] : state.eMarks[index]]
619
+ previous = state.src[state.bMarks[index - 1] : state.eMarks[index - 1]]
620
+ match = _ATX_HEADING.match(line)
621
+ if match is None or previous.strip():
622
+ continue
623
+ if len(match.group(1)) >= 2 or language in _HASH_IS_NOT_COMMENT:
624
+ return index
625
+ return None
626
+
627
+
628
+ # Fences whose purpose is to *show* Markdown: a fence inside them is sample text, and a
629
+ # heading inside them is part of the sample, never a symptom of a broken document.
630
+ _MARKUP_SAMPLE_LANGUAGES = frozenset(
631
+ {"", "markdown", "md", "mdx", "mdown", "text", "txt", "plain", "plaintext", "rst", "html"}
632
+ )
633
+
634
+
635
+ def _has_nested_opener(state: StateBlock, start: int, stop: int, markup: str) -> bool:
636
+ """True when this fence contains a line that opens a fence of its own.
637
+
638
+ ``` ```bash ... ```python ``` pairs the wrong markers: CommonMark reads the inner
639
+ opener as body text and closes the outer fence somewhere later, hiding whatever lies
640
+ between. An opener with an info string is the giveaway - a bare marker is ambiguous.
641
+ """
642
+ for index in range(start + 1, min(stop, len(state.bMarks) - 1)):
643
+ opener = _opening_fence(state.src[state.bMarks[index] : state.eMarks[index]])
644
+ if opener is None:
645
+ continue
646
+ indent, inner, info = opener
647
+ if (
648
+ indent <= _MAX_TOP_LEVEL_INDENT
649
+ and info
650
+ and inner[0] == markup[0]
651
+ and len(inner) >= len(markup)
652
+ ):
653
+ return True
654
+ return False
655
+
656
+
657
+ def _resilient_fence(state: StateBlock, start_line: int, end_line: int, silent: bool) -> bool:
658
+ """markdown-it's fence rule, but an unclosed fence stops at the next heading.
659
+
660
+ CommonMark runs an unclosed fence to the end of the document, so a single stray
661
+ ``` in a long runbook deletes every heading after it from the outline - the document
662
+ becomes one code block and search can never return those sections. Closing the fence
663
+ at the first plausible heading costs a stray code block at worst; not closing it
664
+ costs the whole tail of the document.
665
+
666
+ A fence that markdown-it closed normally is never touched, so a well-formed document
667
+ parses exactly as CommonMark says it should.
668
+ """
669
+ if not _STOCK_FENCE(state, start_line, end_line, silent):
670
+ return False
671
+ if silent:
672
+ return True
673
+ # Recovery is a top-level concern: resuming inside a list item or block quote would
674
+ # hand the block parser a heading that does not belong to that container. A container
675
+ # rewrites the line start past its own marker ("> ", list indent), so a line whose
676
+ # parsed start is not its physical start is nested. `parentType` is no help here: it
677
+ # is "paragraph" for an ordinary fence that ends a paragraph.
678
+ physical_start = state.src.rfind("\n", 0, state.bMarks[start_line]) + 1
679
+ if state.blkIndent > 0 or physical_start != state.bMarks[start_line]:
680
+ return True
681
+ token = state.tokens[-1]
682
+ info = token.info.strip()
683
+ language = info.split(maxsplit=1)[0].lower() if info else ""
684
+ closing = state.src[state.bMarks[state.line - 1] : state.eMarks[state.line - 1]]
685
+ if closing.strip().startswith(token.markup[0] * len(token.markup)):
686
+ # This fence was closed. Its pairing is still suspect when it holds an opener of
687
+ # its own: ```bash ... ```python pairs the wrong markers, so the closing marker
688
+ # found somewhere later hides everything in between. Never suspect a fence that is
689
+ # showing Markdown, where an inner fence is exactly what the sample is about.
690
+ if language in _MARKUP_SAMPLE_LANGUAGES:
691
+ return True
692
+ if not _has_nested_opener(state, start_line, state.line, token.markup):
693
+ return True
694
+ cut = _recovery_line(state, start_line, state.line, language)
695
+ if cut is None or cut <= start_line + 1:
696
+ return True
697
+ logger.warning(
698
+ "Unclosed %s fence at line %d: closing it before the heading at line %d",
699
+ language or "code",
700
+ start_line + 1,
701
+ cut + 1,
702
+ )
703
+ token.content = state.getLines(start_line + 1, cut, state.sCount[start_line], True)
704
+ token.map = [start_line, cut]
705
+ state.line = cut
706
+ return True
707
+
708
+
709
+ # ---------------------------------------------------------------------- sub-chunker
710
+
711
+
712
+ class SectionPart(Protocol):
713
+ """Anything carrying a part's text and its 1-based inclusive line range."""
714
+
715
+ @property
716
+ def content(self) -> str: ...
717
+ @property
718
+ def start_line(self) -> int: ...
719
+ @property
720
+ def end_line(self) -> int: ...
721
+
722
+
723
+ def join_parts(parts: Sequence[SectionPart]) -> str:
724
+ """Reassemble consecutive parts (in source order) into the original verbatim text."""
725
+ if not parts:
726
+ return ""
727
+ chunks = [parts[0].content]
728
+ for previous, current in zip(parts, parts[1:], strict=False):
729
+ chunks.append("\n" * max(0, current.start_line - previous.end_line))
730
+ chunks.append(current.content)
731
+ return "".join(chunks)
732
+
733
+
734
+ def split_into_spans(
735
+ content: str, max_chars: int, *, glue_first: bool = False
736
+ ) -> list[tuple[int, int]]:
737
+ """Split ``content`` into contiguous ``(start, end)`` character spans of <= ``max_chars``.
738
+
739
+ Spans tile the input exactly (``"".join(content[a:b]) == content``). Boundaries are
740
+ paragraph breaks outside code fences; a single block that is still too large falls
741
+ back to line boundaries, and a single over-long line to whitespace boundaries. Only
742
+ blank lines may take a span past the limit: they stay with the text before them,
743
+ because a span of their own would become a part with no content.
744
+ ``glue_first`` keeps a heading attached to the block that follows it - unless that
745
+ block holds a fence which fits in a part by itself but not together with the heading:
746
+ a lone heading is harmless, a code block cut in two is not.
747
+ """
748
+ blocks = _paragraph_blocks(content)
749
+ if glue_first and len(blocks) > 1:
750
+ heading, following = blocks[0], blocks[1]
751
+ cuts_a_fence = (
752
+ following.has_fence
753
+ and following.end - following.start <= max_chars < following.end - heading.start
754
+ )
755
+ if not cuts_a_fence:
756
+ blocks[0:2] = [_Block(heading.start, following.end, following.has_fence)]
757
+ pieces: list[tuple[int, int]] = []
758
+ for block in blocks:
759
+ if block.end - block.start <= max_chars:
760
+ pieces.append((block.start, block.end))
761
+ else:
762
+ pieces.extend(_split_block(content, block.start, block.end, max_chars))
763
+
764
+ spans: list[tuple[int, int]] = []
765
+ span_start, span_end = pieces[0]
766
+ for begin, end in pieces[1:]:
767
+ if end - span_start > max_chars and content[begin:end].strip():
768
+ spans.append((span_start, span_end))
769
+ span_start = begin
770
+ span_end = end
771
+ spans.append((span_start, span_end))
772
+ return spans
773
+
774
+
775
+ def _paragraph_blocks(content: str) -> list[_Block]:
776
+ """Blank-line-separated blocks; a fenced block (at any indent) stays in one block."""
777
+ blocks: list[_Block] = []
778
+ raw_lines = content.split("\n")
779
+ block_start = 0
780
+ block_has_fence = False
781
+ position = 0
782
+ after_blank = False
783
+ open_fence: tuple[int, str] | None = None # (indent, marker) of the fence we are inside
784
+ for index, line in enumerate(raw_lines):
785
+ if open_fence is None:
786
+ if not line.strip():
787
+ after_blank = True
788
+ else:
789
+ if after_blank and position > block_start:
790
+ blocks.append(_Block(block_start, position, block_has_fence))
791
+ block_start, block_has_fence = position, False
792
+ after_blank = False
793
+ opener = _opening_fence(line)
794
+ if opener is not None:
795
+ # Fences nested in list items are indented past column 3, so any
796
+ # indent opens one here; its closer may sit up to 3 columns deeper.
797
+ open_fence = (opener[0], opener[1])
798
+ block_has_fence = True
799
+ elif _is_closing_fence(line, open_fence[1], open_fence[0] + _MAX_TOP_LEVEL_INDENT):
800
+ open_fence = None
801
+ position += len(line) + (1 if index < len(raw_lines) - 1 else 0)
802
+ blocks.append(_Block(block_start, len(content), block_has_fence))
803
+ return blocks
804
+
805
+
806
+ def _split_block(content: str, begin: int, end: int, max_chars: int) -> list[tuple[int, int]]:
807
+ """Split one oversized block at line boundaries, then at whitespace if needed.
808
+
809
+ A block is oversized as a whole, yet a fence inside it (prose directly above it, a
810
+ list whose items carry code) usually is not: a fenced run that fits in a part stays
811
+ one piece, so it is only ever cut when it exceeds the limit by itself.
812
+ """
813
+ pieces: list[tuple[int, int]] = []
814
+ position = begin
815
+ for fence_start, fence_end in _fenced_runs(content, begin, end):
816
+ pieces.extend(_line_pieces(content, position, fence_start, max_chars))
817
+ if fence_end - fence_start <= max_chars:
818
+ pieces.append((fence_start, fence_end))
819
+ else:
820
+ pieces.extend(_line_pieces(content, fence_start, fence_end, max_chars))
821
+ position = fence_end
822
+ pieces.extend(_line_pieces(content, position, end, max_chars))
823
+ return pieces
824
+
825
+
826
+ def _fenced_runs(content: str, begin: int, end: int) -> list[tuple[int, int]]:
827
+ """Character spans of the fenced runs in ``content[begin:end]``, closing line included.
828
+
829
+ ``begin`` is a block boundary, so it is never inside a fence; a fence left open runs
830
+ to ``end``.
831
+ """
832
+ runs: list[tuple[int, int]] = []
833
+ open_fence: tuple[int, str] | None = None
834
+ run_start = position = begin
835
+ while position < end:
836
+ newline = content.find("\n", position, end)
837
+ line_end = end if newline == -1 else newline + 1
838
+ line = content[position:line_end].rstrip("\n")
839
+ if open_fence is None:
840
+ opener = _opening_fence(line) if line.strip() else None
841
+ if opener is not None:
842
+ open_fence, run_start = (opener[0], opener[1]), position
843
+ elif _is_closing_fence(line, open_fence[1], open_fence[0] + _MAX_TOP_LEVEL_INDENT):
844
+ runs.append((run_start, line_end))
845
+ open_fence = None
846
+ position = line_end
847
+ if open_fence is not None:
848
+ runs.append((run_start, end))
849
+ return runs
850
+
851
+
852
+ def _line_pieces(content: str, begin: int, end: int, max_chars: int) -> list[tuple[int, int]]:
853
+ """One piece per line of ``content[begin:end]``; an over-long line is cut at whitespace."""
854
+ pieces: list[tuple[int, int]] = []
855
+ position = begin
856
+ while position < end:
857
+ newline = content.find("\n", position, end)
858
+ line_end = end if newline == -1 else newline + 1
859
+ # An over-long line is cut into pieces of at most half a part, so that whatever
860
+ # precedes it (typically the heading) can still share a part with its first piece.
861
+ piece = max(1, max_chars // 2) if line_end - position > max_chars else max_chars
862
+ while line_end - position > piece:
863
+ cut = content.rfind(" ", position + 1, position + piece)
864
+ cut = position + piece if cut == -1 else cut + 1
865
+ pieces.append((position, cut))
866
+ position = cut
867
+ pieces.append((position, line_end))
868
+ position = line_end
869
+ return pieces