markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,869 @@
|
|
|
1
|
+
"""Defensive, AST-aware Markdown sectioniser.
|
|
2
|
+
|
|
3
|
+
The parser tokenises Markdown with ``markdown-it-py`` and uses the block tokens'
|
|
4
|
+
source maps to cut the *verbatim* source into heading-delimited sections. Only
|
|
5
|
+
headings that are real top-level AST nodes open a section, so ``#`` lines inside
|
|
6
|
+
code fences, block quotes or list items never do.
|
|
7
|
+
|
|
8
|
+
Defences against real-world documentation:
|
|
9
|
+
|
|
10
|
+
* **Skipped levels** (``#`` straight to ``####``) - a heading stack pops every
|
|
11
|
+
entry with level >= L before pushing, so breadcrumbs stay well formed.
|
|
12
|
+
* **Preamble** - badges/summaries before the first heading become
|
|
13
|
+
``[Overview / Preamble]``.
|
|
14
|
+
* **Front matter** - a leading block that really is YAML is kept out of the AST
|
|
15
|
+
(CommonMark would misread it as a thematic break plus a setext heading) and mined
|
|
16
|
+
for ``title:``. A leading ``---`` rule followed by prose is left alone.
|
|
17
|
+
* **Unclosed code fences** - CommonMark runs such a fence to EOF, swallowing every
|
|
18
|
+
heading after it. ``_resilient_fence`` replaces markdown-it's own fence rule and ends
|
|
19
|
+
the fence at the next plausible heading instead, in the same pass. A fence that
|
|
20
|
+
markdown-it closed normally is left exactly as CommonMark read it, unless it holds an
|
|
21
|
+
opening fence of its own (```` ```bash ```` around ```` ```python ````), which means the
|
|
22
|
+
markers were paired wrongly. Deciding on positive evidence only is what keeps a
|
|
23
|
+
well-formed document - one that simply shows headings inside a fence - untouched; the
|
|
24
|
+
cost is that a stray bare marker, which shifts every pairing after it, leaves the
|
|
25
|
+
headings inside those fences hidden.
|
|
26
|
+
* **Oversized / heading-less text** - split on paragraph boundaries into
|
|
27
|
+
``Path (Part n)`` parts that reassemble byte-for-byte. A fenced block is only cut
|
|
28
|
+
when it exceeds the limit on its own.
|
|
29
|
+
* **Colliding breadcrumbs** - repeated paths get a ``[n]`` suffix, also against
|
|
30
|
+
generated ``(Part n)`` paths, so every stored path addresses exactly one thing.
|
|
31
|
+
|
|
32
|
+
Every section is additionally broken into *units* (``extract_units``): the plain text of
|
|
33
|
+
each paragraph, list item, table row and code block. They feed passage-level embeddings.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import logging
|
|
39
|
+
import re
|
|
40
|
+
from collections.abc import Sequence
|
|
41
|
+
from dataclasses import dataclass, replace
|
|
42
|
+
from typing import Protocol
|
|
43
|
+
|
|
44
|
+
from markdown_it import MarkdownIt
|
|
45
|
+
from markdown_it.rules_block.fence import make_fence_rule
|
|
46
|
+
from markdown_it.rules_block.state_block import StateBlock
|
|
47
|
+
from markdown_it.token import Token
|
|
48
|
+
|
|
49
|
+
from markdown_memory.exceptions import ASTParseError
|
|
50
|
+
from markdown_memory.models import (
|
|
51
|
+
PATH_SEPARATOR,
|
|
52
|
+
PREAMBLE_TITLE,
|
|
53
|
+
ParsedDocument,
|
|
54
|
+
SectionDraft,
|
|
55
|
+
part_path,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
DEFAULT_MAX_SECTION_CHARS = 3200 # ~800 tokens
|
|
59
|
+
MAX_UNITS_PER_SECTION = 64
|
|
60
|
+
MAX_UNIT_CHARS = 600
|
|
61
|
+
UNTITLED_HEADING = "(untitled)"
|
|
62
|
+
|
|
63
|
+
_FENCE_OPEN = re.compile(r"^( *)(`{3,}|~{3,})(.*)$")
|
|
64
|
+
_MAX_TOP_LEVEL_INDENT = 3
|
|
65
|
+
logger = logging.getLogger(__name__)
|
|
66
|
+
|
|
67
|
+
_STOCK_FENCE = make_fence_rule()
|
|
68
|
+
_ATX_HEADING = re.compile(r"^ {0,3}(#{1,6})[ \t]+\S")
|
|
69
|
+
_FRONT_MATTER_TITLE = re.compile(r"^title\s*:\s*(.+?)\s*$", re.IGNORECASE)
|
|
70
|
+
# An unquoted key starts with a letter of any script or "_" (`[^\W\d]`), never a digit.
|
|
71
|
+
_YAML_KEY = re.compile(r"""^(?:"[^"]+"|'[^']+'|[^\W\d][^:#]*?)\s*:(\s|$)""")
|
|
72
|
+
_HTML_TAG_NAME = re.compile(r"^</?([A-Za-z][A-Za-z0-9-]*)")
|
|
73
|
+
# Inline HTML that only styles a heading. Any other "tag" is kept as title text: in
|
|
74
|
+
# technical docs `Option<T>` or `<details>` in a heading is almost always literal.
|
|
75
|
+
_FORMATTING_TAGS = frozenset(
|
|
76
|
+
{
|
|
77
|
+
"a", "abbr", "b", "br", "code", "del", "em", "font", "i", "img", "ins", "kbd",
|
|
78
|
+
"mark", "s", "small", "span", "strong", "sub", "sup", "u",
|
|
79
|
+
}
|
|
80
|
+
) # fmt: skip
|
|
81
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
82
|
+
_HTML_TAG = re.compile(r"<[^>]+>")
|
|
83
|
+
# Removed before tags are: a ">" inside a comment would end the "tag" early and leak the
|
|
84
|
+
# rest of the comment as text. An unterminated comment hides everything after it.
|
|
85
|
+
_HTML_COMMENT = re.compile(r"<!--.*?(?:-->|\Z)", re.DOTALL)
|
|
86
|
+
_BLOCK_CLOSERS = {
|
|
87
|
+
"paragraph_open": "paragraph_close",
|
|
88
|
+
"heading_open": "heading_close",
|
|
89
|
+
"table_open": "table_close",
|
|
90
|
+
"bullet_list_open": "bullet_list_close",
|
|
91
|
+
"ordered_list_open": "ordered_list_close",
|
|
92
|
+
"blockquote_open": "blockquote_close",
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
# Languages where a line starting with "# " cannot be a comment. Inside an unclosed
|
|
96
|
+
# fence of any *other* language a level-1 "# ..." line is assumed to be a comment.
|
|
97
|
+
_HASH_IS_NOT_COMMENT = frozenset(
|
|
98
|
+
{
|
|
99
|
+
"c", "cpp", "c++", "cs", "csharp", "css", "go", "golang", "html", "java",
|
|
100
|
+
"javascript", "js", "json", "jsonc", "jsx", "kotlin", "lua", "rs", "rust",
|
|
101
|
+
"scss", "sql", "swift", "ts", "tsx", "typescript", "xml",
|
|
102
|
+
}
|
|
103
|
+
) # fmt: skip
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass(slots=True, frozen=True)
|
|
107
|
+
class _Heading:
|
|
108
|
+
line: int # 0-based index of the heading's first source line
|
|
109
|
+
level: int
|
|
110
|
+
title: str
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass(slots=True, frozen=True)
|
|
114
|
+
class _Block:
|
|
115
|
+
"""A blank-line-delimited run of text inside one section (character offsets)."""
|
|
116
|
+
|
|
117
|
+
start: int
|
|
118
|
+
end: int
|
|
119
|
+
has_fence: bool
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class MarkdownParser:
|
|
123
|
+
"""Turn Markdown source into breadcrumbed, size-bounded sections."""
|
|
124
|
+
|
|
125
|
+
def __init__(self, max_section_chars: int = DEFAULT_MAX_SECTION_CHARS) -> None:
|
|
126
|
+
if max_section_chars < 1:
|
|
127
|
+
raise ValueError("max_section_chars must be positive")
|
|
128
|
+
self._max_chars = max_section_chars
|
|
129
|
+
self._md = MarkdownIt("commonmark").enable("table")
|
|
130
|
+
# The chain the stock rule is registered with. Dropping it would stop the fence rule
|
|
131
|
+
# being consulted inside list items and block quotes, so a fence in a list would be
|
|
132
|
+
# read as a paragraph and headings inside it would become real headings.
|
|
133
|
+
self._md.block.ruler.at(
|
|
134
|
+
"fence", _resilient_fence, {"alt": ["paragraph", "reference", "blockquote", "list"]}
|
|
135
|
+
)
|
|
136
|
+
# Passages are extracted with the stock rule. Fence recovery is there to rescue a
|
|
137
|
+
# document's structure; inside one section an unclosed fence is usually a fenced
|
|
138
|
+
# sample that oversized-section splitting cut in half, and reinterpreting its
|
|
139
|
+
# contents as prose would index the template rather than the documentation.
|
|
140
|
+
self._units_md = MarkdownIt("commonmark").enable("table")
|
|
141
|
+
|
|
142
|
+
def parse(self, text: str, *, fallback_title: str = "Untitled") -> ParsedDocument:
|
|
143
|
+
"""Parse ``text``; never raises for malformed Markdown, only for tokeniser failure."""
|
|
144
|
+
lines = _split_lines(text)
|
|
145
|
+
body_start, front_matter_title = _scan_front_matter(lines)
|
|
146
|
+
headings = self._find_headings(lines, body_start)
|
|
147
|
+
|
|
148
|
+
used_paths: set[str] = set()
|
|
149
|
+
first_heading_line = headings[0].line if headings else len(lines)
|
|
150
|
+
sections: list[SectionDraft] = self._build_section(
|
|
151
|
+
lines,
|
|
152
|
+
start=0,
|
|
153
|
+
stop=first_heading_line,
|
|
154
|
+
title=PREAMBLE_TITLE,
|
|
155
|
+
level=0,
|
|
156
|
+
path=PREAMBLE_TITLE,
|
|
157
|
+
has_heading_line=False,
|
|
158
|
+
)
|
|
159
|
+
used_paths.update(section.heading_path for section in sections)
|
|
160
|
+
used_paths.add(PREAMBLE_TITLE)
|
|
161
|
+
|
|
162
|
+
stack: list[tuple[int, str]] = []
|
|
163
|
+
next_occurrence: dict[str, int] = {}
|
|
164
|
+
for index, heading in enumerate(headings):
|
|
165
|
+
while stack and stack[-1][0] >= heading.level:
|
|
166
|
+
stack.pop()
|
|
167
|
+
parents = [title for _, title in stack]
|
|
168
|
+
stop = headings[index + 1].line if index + 1 < len(headings) else len(lines)
|
|
169
|
+
# Every stored path must address exactly one thing: a candidate is rejected
|
|
170
|
+
# when it - or any "(Part n)" path generated from it - is already taken. The
|
|
171
|
+
# section is built once; trying another name only renames the drafts.
|
|
172
|
+
base_path = PATH_SEPARATOR.join([*parents, heading.title])
|
|
173
|
+
drafts = self._build_section(
|
|
174
|
+
lines,
|
|
175
|
+
start=heading.line,
|
|
176
|
+
stop=stop,
|
|
177
|
+
title=heading.title,
|
|
178
|
+
level=heading.level,
|
|
179
|
+
path=base_path,
|
|
180
|
+
has_heading_line=True,
|
|
181
|
+
)
|
|
182
|
+
occurrence = next_occurrence.get(base_path, 1)
|
|
183
|
+
while True:
|
|
184
|
+
title = heading.title if occurrence == 1 else f"{heading.title} [{occurrence}]"
|
|
185
|
+
path = PATH_SEPARATOR.join([*parents, title])
|
|
186
|
+
claimed = {path, *(part_path(path, d.part_index) for d in drafts if d.part_index)}
|
|
187
|
+
if not claimed & used_paths:
|
|
188
|
+
break
|
|
189
|
+
occurrence += 1
|
|
190
|
+
next_occurrence[base_path] = occurrence + 1
|
|
191
|
+
if occurrence > 1:
|
|
192
|
+
drafts = [
|
|
193
|
+
replace(
|
|
194
|
+
draft,
|
|
195
|
+
heading_title=title,
|
|
196
|
+
base_path=path,
|
|
197
|
+
heading_path=part_path(path, draft.part_index)
|
|
198
|
+
if draft.part_index
|
|
199
|
+
else path,
|
|
200
|
+
)
|
|
201
|
+
for draft in drafts
|
|
202
|
+
]
|
|
203
|
+
used_paths |= claimed
|
|
204
|
+
stack.append((heading.level, title))
|
|
205
|
+
sections.extend(drafts)
|
|
206
|
+
|
|
207
|
+
return ParsedDocument(
|
|
208
|
+
title=_pick_title(headings, front_matter_title, fallback_title),
|
|
209
|
+
sections=tuple(sections),
|
|
210
|
+
line_count=len(lines),
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
# ------------------------------------------------------------------ AST walk
|
|
214
|
+
|
|
215
|
+
def _tokenize(self, source: str, *, recover_fences: bool = True) -> list[Token]:
|
|
216
|
+
try:
|
|
217
|
+
return (self._md if recover_fences else self._units_md).parse(source)
|
|
218
|
+
except Exception as exc: # markdown-it has no dedicated error type
|
|
219
|
+
raise ASTParseError(f"markdown-it failed to tokenise the document: {exc}") from exc
|
|
220
|
+
|
|
221
|
+
def _find_headings(self, lines: Sequence[str], body_start: int) -> list[_Heading]:
|
|
222
|
+
"""Collect the document's top-level headings in a single tokenisation pass.
|
|
223
|
+
|
|
224
|
+
An unclosed fence no longer hides the rest of the document: ``_resilient_fence``
|
|
225
|
+
ends it at the first plausible heading while the block parser is running, so the
|
|
226
|
+
headings after it are simply there.
|
|
227
|
+
"""
|
|
228
|
+
tokens = self._tokenize("\n".join(lines[body_start:]))
|
|
229
|
+
headings: list[_Heading] = []
|
|
230
|
+
for position, token in enumerate(tokens):
|
|
231
|
+
if token.level != 0 or token.map is None or token.type != "heading_open":
|
|
232
|
+
continue
|
|
233
|
+
inline = tokens[position + 1] if position + 1 < len(tokens) else None
|
|
234
|
+
headings.append(
|
|
235
|
+
_Heading(
|
|
236
|
+
line=body_start + token.map[0],
|
|
237
|
+
level=int(token.tag[1:]),
|
|
238
|
+
title=_inline_text(inline),
|
|
239
|
+
)
|
|
240
|
+
)
|
|
241
|
+
return headings
|
|
242
|
+
|
|
243
|
+
# ------------------------------------------------------------------ sections
|
|
244
|
+
|
|
245
|
+
def _build_section(
|
|
246
|
+
self,
|
|
247
|
+
lines: Sequence[str],
|
|
248
|
+
*,
|
|
249
|
+
start: int,
|
|
250
|
+
stop: int,
|
|
251
|
+
title: str,
|
|
252
|
+
level: int,
|
|
253
|
+
path: str,
|
|
254
|
+
has_heading_line: bool,
|
|
255
|
+
) -> list[SectionDraft]:
|
|
256
|
+
"""Build the section for ``lines[start:stop]``, split into parts if oversized."""
|
|
257
|
+
while start < stop and not lines[start].strip():
|
|
258
|
+
start += 1
|
|
259
|
+
while stop > start and not lines[stop - 1].strip():
|
|
260
|
+
stop -= 1
|
|
261
|
+
if start >= stop:
|
|
262
|
+
return []
|
|
263
|
+
content = "\n".join(lines[start:stop])
|
|
264
|
+
first_line = start + 1 # 1-based
|
|
265
|
+
if len(content) <= self._max_chars:
|
|
266
|
+
return [
|
|
267
|
+
SectionDraft(
|
|
268
|
+
heading_title=title,
|
|
269
|
+
heading_level=level,
|
|
270
|
+
heading_path=path,
|
|
271
|
+
base_path=path,
|
|
272
|
+
content=content,
|
|
273
|
+
start_line=first_line,
|
|
274
|
+
end_line=stop,
|
|
275
|
+
units=self.extract_units(content, skip_heading=has_heading_line),
|
|
276
|
+
)
|
|
277
|
+
]
|
|
278
|
+
# Only trailing newlines are dropped from a part, and they are recoverable from
|
|
279
|
+
# the line numbers, so join_parts() can rebuild the section byte-for-byte.
|
|
280
|
+
spans = split_into_spans(content, self._max_chars, glue_first=has_heading_line)
|
|
281
|
+
table_headers = _table_headers(lines[start:stop])
|
|
282
|
+
drafts: list[SectionDraft] = []
|
|
283
|
+
for number, (begin, end) in enumerate(spans, start=1):
|
|
284
|
+
part = content[begin:end].rstrip("\n")
|
|
285
|
+
first_part_line = content.count("\n", 0, begin)
|
|
286
|
+
part_start = first_line + first_part_line
|
|
287
|
+
# A part that starts in the middle of a table has lost the header row, and
|
|
288
|
+
# without it the rows are just a paragraph. Units are extracted from the part
|
|
289
|
+
# with its header restored; the stored content stays verbatim.
|
|
290
|
+
header = (
|
|
291
|
+
table_headers.get(first_part_line)
|
|
292
|
+
if content[begin - 1 : begin] in {"", "\n"}
|
|
293
|
+
else None
|
|
294
|
+
)
|
|
295
|
+
drafts.append(
|
|
296
|
+
SectionDraft(
|
|
297
|
+
heading_title=title,
|
|
298
|
+
heading_level=level,
|
|
299
|
+
heading_path=part_path(path, number),
|
|
300
|
+
base_path=path,
|
|
301
|
+
content=part,
|
|
302
|
+
start_line=part_start,
|
|
303
|
+
end_line=part_start + part.count("\n"),
|
|
304
|
+
part_index=number,
|
|
305
|
+
units=self.extract_units(
|
|
306
|
+
f"{header}\n{part}" if header else part,
|
|
307
|
+
skip_heading=has_heading_line and number == 1,
|
|
308
|
+
),
|
|
309
|
+
)
|
|
310
|
+
)
|
|
311
|
+
return drafts
|
|
312
|
+
|
|
313
|
+
# ------------------------------------------------------------------ units
|
|
314
|
+
|
|
315
|
+
def extract_units(self, content: str, *, skip_heading: bool = False) -> tuple[str, ...]:
|
|
316
|
+
"""Plain-text passages of ``content``: paragraphs, list items, table rows, code blocks.
|
|
317
|
+
|
|
318
|
+
Table rows are rendered as ``Header: cell; Header: cell`` so that a row keeps its
|
|
319
|
+
meaning without the rest of the table. ``skip_heading`` drops the section's own
|
|
320
|
+
heading (its text already lives in the breadcrumb).
|
|
321
|
+
"""
|
|
322
|
+
tokens = self._tokenize(content, recover_fences=False)
|
|
323
|
+
units: list[str] = []
|
|
324
|
+
index = 0
|
|
325
|
+
heading_skipped = not skip_heading
|
|
326
|
+
while index < len(tokens) and len(units) < MAX_UNITS_PER_SECTION:
|
|
327
|
+
token = tokens[index]
|
|
328
|
+
if token.level != 0:
|
|
329
|
+
index += 1
|
|
330
|
+
continue
|
|
331
|
+
end = _block_end(tokens, index)
|
|
332
|
+
block = tokens[index : end + 1]
|
|
333
|
+
if token.type == "heading_open" and not heading_skipped:
|
|
334
|
+
heading_skipped = True
|
|
335
|
+
elif token.type == "table_open":
|
|
336
|
+
units.extend(_table_rows(block))
|
|
337
|
+
elif token.type in {"bullet_list_open", "ordered_list_open"}:
|
|
338
|
+
units.extend(_list_items(block))
|
|
339
|
+
elif token.type == "blockquote_open":
|
|
340
|
+
units.extend(_leaf_texts(block)) # one unit per quoted paragraph / code block
|
|
341
|
+
else:
|
|
342
|
+
units.append(" ".join(_leaf_texts(block)))
|
|
343
|
+
index = end + 1
|
|
344
|
+
passages: list[str] = []
|
|
345
|
+
for unit in units:
|
|
346
|
+
for window in _windows(unit):
|
|
347
|
+
if len(passages) >= MAX_UNITS_PER_SECTION:
|
|
348
|
+
return tuple(passages)
|
|
349
|
+
cleaned = _WHITESPACE.sub(" ", window.replace("|", " ")).strip()
|
|
350
|
+
if cleaned:
|
|
351
|
+
passages.append(cleaned)
|
|
352
|
+
return tuple(passages)
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
# ---------------------------------------------------------------------- helpers
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _last_space(text: str) -> int:
|
|
359
|
+
"""Index just past the last whitespace run, or -1. Any whitespace, not just a space:
|
|
360
|
+
a line of tab-separated columns has no literal space to break on."""
|
|
361
|
+
for index in range(len(text) - 1, -1, -1):
|
|
362
|
+
if text[index].isspace():
|
|
363
|
+
return index
|
|
364
|
+
return -1
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _windows(text: str) -> list[str]:
|
|
368
|
+
"""``text`` as consecutive pieces of at most ``MAX_UNIT_CHARS``, nothing discarded.
|
|
369
|
+
|
|
370
|
+
A fenced block arrives here as one unit. Slicing it to the limit - which is what this
|
|
371
|
+
used to do - gave the tail no vector at all: in this project's own CLAUDE.md the
|
|
372
|
+
command list was cut mid-word at ``reindex_docs.py D``, and every command after that
|
|
373
|
+
point was unreachable by passage search while sitting in the index in plain sight.
|
|
374
|
+
Across the vendored corpus that was 73,598 characters with no passage vector.
|
|
375
|
+
|
|
376
|
+
Pieces break at a newline where possible, then at a sentence end, then at a space, so
|
|
377
|
+
a window is a run of whole lines or whole words rather than an arbitrary cut. The
|
|
378
|
+
limit is a window size now, not a truncation point.
|
|
379
|
+
"""
|
|
380
|
+
if len(text) <= MAX_UNIT_CHARS:
|
|
381
|
+
return [text]
|
|
382
|
+
pieces: list[str] = []
|
|
383
|
+
remaining = text
|
|
384
|
+
while len(remaining) > MAX_UNIT_CHARS:
|
|
385
|
+
head = remaining[:MAX_UNIT_CHARS]
|
|
386
|
+
cut = max(head.rfind("\n"), head.rfind(". "), _last_space(head))
|
|
387
|
+
# A cut in the first half would make a window mostly empty; a hard cut keeps the
|
|
388
|
+
# windows even, and no character is lost either way.
|
|
389
|
+
if cut < MAX_UNIT_CHARS // 2:
|
|
390
|
+
cut = MAX_UNIT_CHARS
|
|
391
|
+
pieces.append(remaining[:cut])
|
|
392
|
+
remaining = remaining[cut:]
|
|
393
|
+
if remaining:
|
|
394
|
+
pieces.append(remaining)
|
|
395
|
+
return pieces
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
_TABLE_DELIMITER = re.compile(
|
|
399
|
+
r"^ {0,3}\|?[ \t]*:?-{1,}:?[ \t]*(\|[ \t]*:?-{1,}:?[ \t]*)*\|?[ \t]*$"
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _table_headers(lines: Sequence[str]) -> dict[int, str]:
|
|
404
|
+
"""Map each table *body* row (by line index) to its table's header + delimiter rows."""
|
|
405
|
+
headers: dict[int, str] = {}
|
|
406
|
+
index = 0
|
|
407
|
+
while index + 1 < len(lines):
|
|
408
|
+
is_header = "|" in lines[index] and "|" in lines[index + 1]
|
|
409
|
+
if is_header and _TABLE_DELIMITER.match(lines[index + 1]) and lines[index].strip():
|
|
410
|
+
header = f"{lines[index]}\n{lines[index + 1]}"
|
|
411
|
+
index += 2
|
|
412
|
+
while index < len(lines) and lines[index].strip() and "|" in lines[index]:
|
|
413
|
+
headers[index] = header
|
|
414
|
+
index += 1
|
|
415
|
+
else:
|
|
416
|
+
index += 1
|
|
417
|
+
return headers
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _split_lines(text: str) -> list[str]:
|
|
421
|
+
normalized = text.removeprefix("").replace("\r\n", "\n").replace("\r", "\n")
|
|
422
|
+
lines = normalized.split("\n")
|
|
423
|
+
if lines and lines[-1] == "":
|
|
424
|
+
lines.pop()
|
|
425
|
+
return lines
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _scan_front_matter(lines: Sequence[str]) -> tuple[int, str | None]:
|
|
429
|
+
"""Return ``(first body line, front-matter title)`` for a leading YAML block."""
|
|
430
|
+
if not lines or lines[0].rstrip() != "---":
|
|
431
|
+
return 0, None
|
|
432
|
+
for index in range(1, len(lines)):
|
|
433
|
+
if lines[index].rstrip() in {"---", "..."}:
|
|
434
|
+
block = lines[1:index]
|
|
435
|
+
if not _looks_like_yaml(block):
|
|
436
|
+
return 0, None
|
|
437
|
+
title: str | None = None
|
|
438
|
+
for line in block:
|
|
439
|
+
match = _FRONT_MATTER_TITLE.match(line)
|
|
440
|
+
if match:
|
|
441
|
+
title = match.group(1).strip("\"'") or None
|
|
442
|
+
break
|
|
443
|
+
return index + 1, title
|
|
444
|
+
return 0, None
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
# "Note: read this first" between two rules is a setext heading, not metadata. A lone
|
|
448
|
+
# `key: value` line is otherwise taken for front matter, as every site generator does.
|
|
449
|
+
_ADMONITION_KEYS = frozenset(
|
|
450
|
+
{"note", "warning", "tip", "todo", "important", "caution", "hint", "see also", "example"}
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _looks_like_yaml(block: Sequence[str]) -> bool:
|
|
455
|
+
"""Distinguish front matter from a document that merely opens with a ``---`` rule.
|
|
456
|
+
|
|
457
|
+
Every non-blank line must be a mapping key, a list item, a comment, a continuation
|
|
458
|
+
or a closing flow bracket, and at least one top-level key must exist. A ``# ...``
|
|
459
|
+
line followed by a blank line is taken for a Markdown heading, not a YAML comment.
|
|
460
|
+
"""
|
|
461
|
+
has_key = False
|
|
462
|
+
content_lines = [line for line in block if line.strip()]
|
|
463
|
+
if len(content_lines) == 1:
|
|
464
|
+
key = content_lines[0].split(":", 1)[0].strip().strip("\"'").lower()
|
|
465
|
+
if key in _ADMONITION_KEYS:
|
|
466
|
+
return False
|
|
467
|
+
for index, line in enumerate(block):
|
|
468
|
+
stripped = line.strip()
|
|
469
|
+
if not stripped:
|
|
470
|
+
continue
|
|
471
|
+
if _YAML_KEY.match(line):
|
|
472
|
+
has_key = True
|
|
473
|
+
elif stripped.startswith("#"):
|
|
474
|
+
followed_by_blank = index + 1 < len(block) and not block[index + 1].strip()
|
|
475
|
+
if _ATX_HEADING.match(line) and followed_by_blank:
|
|
476
|
+
return False
|
|
477
|
+
elif not (line[0] in " \t" or stripped.startswith("- ") or stripped in {"-", "]", "}"}):
|
|
478
|
+
return False
|
|
479
|
+
return has_key
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _plain_inline(inline: Token | None) -> str:
|
|
483
|
+
"""Visible text of an inline token: markup stripped, code and image alt text kept."""
|
|
484
|
+
if inline is None or inline.type != "inline":
|
|
485
|
+
return ""
|
|
486
|
+
fragments: list[str] = []
|
|
487
|
+
for child in inline.children or []:
|
|
488
|
+
if child.type in {"text", "code_inline", "image"}:
|
|
489
|
+
fragments.append(child.content)
|
|
490
|
+
elif child.type in {"softbreak", "hardbreak"}:
|
|
491
|
+
fragments.append(" ")
|
|
492
|
+
elif child.type == "html_inline":
|
|
493
|
+
tag = _HTML_TAG_NAME.match(child.content)
|
|
494
|
+
# Case-sensitive on purpose: `<u>` is underline, `<U>` is a type parameter.
|
|
495
|
+
if tag is not None and tag.group(1) not in _FORMATTING_TAGS:
|
|
496
|
+
fragments.append(child.content) # `Option<T>`: a type parameter, not markup
|
|
497
|
+
elif tag is not None and tag.group(1) == "br":
|
|
498
|
+
fragments.append(" ") # the only line break a table cell can hold
|
|
499
|
+
return _WHITESPACE.sub(" ", "".join(fragments)).strip()
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _inline_text(inline: Token | None) -> str:
|
|
503
|
+
"""Plain text of a heading's inline token, never empty."""
|
|
504
|
+
return _plain_inline(inline) or UNTITLED_HEADING
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _leaf_texts(block: Sequence[Token]) -> list[str]:
|
|
508
|
+
"""Text of every leaf in ``block``, at any depth: inline runs, code and raw HTML."""
|
|
509
|
+
texts: list[str] = []
|
|
510
|
+
for token in block:
|
|
511
|
+
if token.type == "inline":
|
|
512
|
+
texts.append(_plain_inline(token))
|
|
513
|
+
elif token.type in {"fence", "code_block"}:
|
|
514
|
+
texts.append(token.content)
|
|
515
|
+
elif token.type == "html_block":
|
|
516
|
+
texts.append(_HTML_TAG.sub(" ", _HTML_COMMENT.sub(" ", token.content)))
|
|
517
|
+
return [text for text in texts if text.strip()]
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def _block_end(tokens: Sequence[Token], start: int) -> int:
|
|
521
|
+
"""Index of the token closing the top-level block that opens at ``start``."""
|
|
522
|
+
closer = _BLOCK_CLOSERS.get(tokens[start].type)
|
|
523
|
+
if closer is None:
|
|
524
|
+
return start
|
|
525
|
+
for index in range(start + 1, len(tokens)):
|
|
526
|
+
if tokens[index].type == closer and tokens[index].level == tokens[start].level:
|
|
527
|
+
return index
|
|
528
|
+
return len(tokens) - 1
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _table_rows(block: Sequence[Token]) -> list[str]:
|
|
532
|
+
"""One unit per body row, each cell labelled with its column header.
|
|
533
|
+
|
|
534
|
+
A table without body rows yields its header cells instead: they are visible text,
|
|
535
|
+
and a section holding nothing else would otherwise pass for a heading-only stub.
|
|
536
|
+
"""
|
|
537
|
+
headers: list[str] = []
|
|
538
|
+
rows: list[str] = []
|
|
539
|
+
cells: list[str] = []
|
|
540
|
+
in_head = False
|
|
541
|
+
for position, token in enumerate(block):
|
|
542
|
+
if token.type == "thead_open":
|
|
543
|
+
in_head = True
|
|
544
|
+
elif token.type == "thead_close":
|
|
545
|
+
in_head = False
|
|
546
|
+
elif token.type in {"th_open", "td_open"}:
|
|
547
|
+
text = _plain_inline(block[position + 1] if position + 1 < len(block) else None)
|
|
548
|
+
(headers if in_head else cells).append(text)
|
|
549
|
+
elif token.type == "tr_close" and not in_head:
|
|
550
|
+
labelled = [
|
|
551
|
+
f"{header}: {cell}" if header else cell
|
|
552
|
+
for header, cell in zip(headers + [""] * len(cells), cells, strict=False)
|
|
553
|
+
if cell
|
|
554
|
+
]
|
|
555
|
+
rows.append("; ".join(labelled))
|
|
556
|
+
cells = []
|
|
557
|
+
if rows:
|
|
558
|
+
return rows
|
|
559
|
+
header_row = "; ".join(header for header in headers if header)
|
|
560
|
+
return [header_row] if header_row else []
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _list_items(block: Sequence[Token]) -> list[str]:
|
|
564
|
+
"""One unit per top-level list item, nested content included."""
|
|
565
|
+
items: list[str] = []
|
|
566
|
+
fragments: list[str] = []
|
|
567
|
+
for token in block:
|
|
568
|
+
if token.type == "list_item_open" and token.level == 1:
|
|
569
|
+
fragments = []
|
|
570
|
+
elif token.type in {"inline", "fence", "code_block", "html_block"}:
|
|
571
|
+
fragments.extend(_leaf_texts([token]))
|
|
572
|
+
elif token.type == "list_item_close" and token.level == 1:
|
|
573
|
+
items.append(" ".join(fragments))
|
|
574
|
+
return items
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _pick_title(headings: Sequence[_Heading], front_matter_title: str | None, fallback: str) -> str:
|
|
578
|
+
for heading in headings:
|
|
579
|
+
if heading.level == 1 and heading.title != UNTITLED_HEADING:
|
|
580
|
+
return heading.title
|
|
581
|
+
if front_matter_title:
|
|
582
|
+
return front_matter_title
|
|
583
|
+
for heading in headings:
|
|
584
|
+
if heading.title != UNTITLED_HEADING:
|
|
585
|
+
return heading.title
|
|
586
|
+
return fallback
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def _is_closing_fence(line: str, marker: str, max_indent: int) -> bool:
|
|
590
|
+
stripped = line.strip()
|
|
591
|
+
indent = len(line) - len(line.lstrip(" "))
|
|
592
|
+
return (
|
|
593
|
+
indent <= max_indent
|
|
594
|
+
and len(stripped) >= len(marker)
|
|
595
|
+
and stripped == marker[0] * len(stripped)
|
|
596
|
+
)
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def _opening_fence(line: str) -> tuple[int, str, str] | None:
|
|
600
|
+
"""``(indent, marker, info)`` when ``line`` can open a fenced block."""
|
|
601
|
+
match = _FENCE_OPEN.match(line)
|
|
602
|
+
if match is None:
|
|
603
|
+
return None
|
|
604
|
+
indent, marker, info = len(match.group(1)), match.group(2), match.group(3).strip()
|
|
605
|
+
if marker[0] == "`" and "`" in info: # CommonMark: that is inline code, not a fence
|
|
606
|
+
return None
|
|
607
|
+
return indent, marker, info
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def _recovery_line(state: StateBlock, start: int, stop: int, language: str) -> int | None:
|
|
611
|
+
"""First line in ``(start, stop)`` that should be read as a heading again.
|
|
612
|
+
|
|
613
|
+
A candidate is an ATX heading preceded by a blank line. A level-1 candidate is only
|
|
614
|
+
trusted when ``#`` cannot start a comment in the fence's language, because a lone
|
|
615
|
+
``# comment`` is far more common inside shell or Python than a heading is.
|
|
616
|
+
"""
|
|
617
|
+
for index in range(start + 2, min(stop, len(state.bMarks) - 1)):
|
|
618
|
+
line = state.src[state.bMarks[index] : state.eMarks[index]]
|
|
619
|
+
previous = state.src[state.bMarks[index - 1] : state.eMarks[index - 1]]
|
|
620
|
+
match = _ATX_HEADING.match(line)
|
|
621
|
+
if match is None or previous.strip():
|
|
622
|
+
continue
|
|
623
|
+
if len(match.group(1)) >= 2 or language in _HASH_IS_NOT_COMMENT:
|
|
624
|
+
return index
|
|
625
|
+
return None
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
# Fences whose purpose is to *show* Markdown: a fence inside them is sample text, and a
|
|
629
|
+
# heading inside them is part of the sample, never a symptom of a broken document.
|
|
630
|
+
_MARKUP_SAMPLE_LANGUAGES = frozenset(
|
|
631
|
+
{"", "markdown", "md", "mdx", "mdown", "text", "txt", "plain", "plaintext", "rst", "html"}
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def _has_nested_opener(state: StateBlock, start: int, stop: int, markup: str) -> bool:
|
|
636
|
+
"""True when this fence contains a line that opens a fence of its own.
|
|
637
|
+
|
|
638
|
+
``` ```bash ... ```python ``` pairs the wrong markers: CommonMark reads the inner
|
|
639
|
+
opener as body text and closes the outer fence somewhere later, hiding whatever lies
|
|
640
|
+
between. An opener with an info string is the giveaway - a bare marker is ambiguous.
|
|
641
|
+
"""
|
|
642
|
+
for index in range(start + 1, min(stop, len(state.bMarks) - 1)):
|
|
643
|
+
opener = _opening_fence(state.src[state.bMarks[index] : state.eMarks[index]])
|
|
644
|
+
if opener is None:
|
|
645
|
+
continue
|
|
646
|
+
indent, inner, info = opener
|
|
647
|
+
if (
|
|
648
|
+
indent <= _MAX_TOP_LEVEL_INDENT
|
|
649
|
+
and info
|
|
650
|
+
and inner[0] == markup[0]
|
|
651
|
+
and len(inner) >= len(markup)
|
|
652
|
+
):
|
|
653
|
+
return True
|
|
654
|
+
return False
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def _resilient_fence(state: StateBlock, start_line: int, end_line: int, silent: bool) -> bool:
|
|
658
|
+
"""markdown-it's fence rule, but an unclosed fence stops at the next heading.
|
|
659
|
+
|
|
660
|
+
CommonMark runs an unclosed fence to the end of the document, so a single stray
|
|
661
|
+
``` in a long runbook deletes every heading after it from the outline - the document
|
|
662
|
+
becomes one code block and search can never return those sections. Closing the fence
|
|
663
|
+
at the first plausible heading costs a stray code block at worst; not closing it
|
|
664
|
+
costs the whole tail of the document.
|
|
665
|
+
|
|
666
|
+
A fence that markdown-it closed normally is never touched, so a well-formed document
|
|
667
|
+
parses exactly as CommonMark says it should.
|
|
668
|
+
"""
|
|
669
|
+
if not _STOCK_FENCE(state, start_line, end_line, silent):
|
|
670
|
+
return False
|
|
671
|
+
if silent:
|
|
672
|
+
return True
|
|
673
|
+
# Recovery is a top-level concern: resuming inside a list item or block quote would
|
|
674
|
+
# hand the block parser a heading that does not belong to that container. A container
|
|
675
|
+
# rewrites the line start past its own marker ("> ", list indent), so a line whose
|
|
676
|
+
# parsed start is not its physical start is nested. `parentType` is no help here: it
|
|
677
|
+
# is "paragraph" for an ordinary fence that ends a paragraph.
|
|
678
|
+
physical_start = state.src.rfind("\n", 0, state.bMarks[start_line]) + 1
|
|
679
|
+
if state.blkIndent > 0 or physical_start != state.bMarks[start_line]:
|
|
680
|
+
return True
|
|
681
|
+
token = state.tokens[-1]
|
|
682
|
+
info = token.info.strip()
|
|
683
|
+
language = info.split(maxsplit=1)[0].lower() if info else ""
|
|
684
|
+
closing = state.src[state.bMarks[state.line - 1] : state.eMarks[state.line - 1]]
|
|
685
|
+
if closing.strip().startswith(token.markup[0] * len(token.markup)):
|
|
686
|
+
# This fence was closed. Its pairing is still suspect when it holds an opener of
|
|
687
|
+
# its own: ```bash ... ```python pairs the wrong markers, so the closing marker
|
|
688
|
+
# found somewhere later hides everything in between. Never suspect a fence that is
|
|
689
|
+
# showing Markdown, where an inner fence is exactly what the sample is about.
|
|
690
|
+
if language in _MARKUP_SAMPLE_LANGUAGES:
|
|
691
|
+
return True
|
|
692
|
+
if not _has_nested_opener(state, start_line, state.line, token.markup):
|
|
693
|
+
return True
|
|
694
|
+
cut = _recovery_line(state, start_line, state.line, language)
|
|
695
|
+
if cut is None or cut <= start_line + 1:
|
|
696
|
+
return True
|
|
697
|
+
logger.warning(
|
|
698
|
+
"Unclosed %s fence at line %d: closing it before the heading at line %d",
|
|
699
|
+
language or "code",
|
|
700
|
+
start_line + 1,
|
|
701
|
+
cut + 1,
|
|
702
|
+
)
|
|
703
|
+
token.content = state.getLines(start_line + 1, cut, state.sCount[start_line], True)
|
|
704
|
+
token.map = [start_line, cut]
|
|
705
|
+
state.line = cut
|
|
706
|
+
return True
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
# ---------------------------------------------------------------------- sub-chunker
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
class SectionPart(Protocol):
|
|
713
|
+
"""Anything carrying a part's text and its 1-based inclusive line range."""
|
|
714
|
+
|
|
715
|
+
@property
|
|
716
|
+
def content(self) -> str: ...
|
|
717
|
+
@property
|
|
718
|
+
def start_line(self) -> int: ...
|
|
719
|
+
@property
|
|
720
|
+
def end_line(self) -> int: ...
|
|
721
|
+
|
|
722
|
+
|
|
723
|
+
def join_parts(parts: Sequence[SectionPart]) -> str:
|
|
724
|
+
"""Reassemble consecutive parts (in source order) into the original verbatim text."""
|
|
725
|
+
if not parts:
|
|
726
|
+
return ""
|
|
727
|
+
chunks = [parts[0].content]
|
|
728
|
+
for previous, current in zip(parts, parts[1:], strict=False):
|
|
729
|
+
chunks.append("\n" * max(0, current.start_line - previous.end_line))
|
|
730
|
+
chunks.append(current.content)
|
|
731
|
+
return "".join(chunks)
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def split_into_spans(
|
|
735
|
+
content: str, max_chars: int, *, glue_first: bool = False
|
|
736
|
+
) -> list[tuple[int, int]]:
|
|
737
|
+
"""Split ``content`` into contiguous ``(start, end)`` character spans of <= ``max_chars``.
|
|
738
|
+
|
|
739
|
+
Spans tile the input exactly (``"".join(content[a:b]) == content``). Boundaries are
|
|
740
|
+
paragraph breaks outside code fences; a single block that is still too large falls
|
|
741
|
+
back to line boundaries, and a single over-long line to whitespace boundaries. Only
|
|
742
|
+
blank lines may take a span past the limit: they stay with the text before them,
|
|
743
|
+
because a span of their own would become a part with no content.
|
|
744
|
+
``glue_first`` keeps a heading attached to the block that follows it - unless that
|
|
745
|
+
block holds a fence which fits in a part by itself but not together with the heading:
|
|
746
|
+
a lone heading is harmless, a code block cut in two is not.
|
|
747
|
+
"""
|
|
748
|
+
blocks = _paragraph_blocks(content)
|
|
749
|
+
if glue_first and len(blocks) > 1:
|
|
750
|
+
heading, following = blocks[0], blocks[1]
|
|
751
|
+
cuts_a_fence = (
|
|
752
|
+
following.has_fence
|
|
753
|
+
and following.end - following.start <= max_chars < following.end - heading.start
|
|
754
|
+
)
|
|
755
|
+
if not cuts_a_fence:
|
|
756
|
+
blocks[0:2] = [_Block(heading.start, following.end, following.has_fence)]
|
|
757
|
+
pieces: list[tuple[int, int]] = []
|
|
758
|
+
for block in blocks:
|
|
759
|
+
if block.end - block.start <= max_chars:
|
|
760
|
+
pieces.append((block.start, block.end))
|
|
761
|
+
else:
|
|
762
|
+
pieces.extend(_split_block(content, block.start, block.end, max_chars))
|
|
763
|
+
|
|
764
|
+
spans: list[tuple[int, int]] = []
|
|
765
|
+
span_start, span_end = pieces[0]
|
|
766
|
+
for begin, end in pieces[1:]:
|
|
767
|
+
if end - span_start > max_chars and content[begin:end].strip():
|
|
768
|
+
spans.append((span_start, span_end))
|
|
769
|
+
span_start = begin
|
|
770
|
+
span_end = end
|
|
771
|
+
spans.append((span_start, span_end))
|
|
772
|
+
return spans
|
|
773
|
+
|
|
774
|
+
|
|
775
|
+
def _paragraph_blocks(content: str) -> list[_Block]:
|
|
776
|
+
"""Blank-line-separated blocks; a fenced block (at any indent) stays in one block."""
|
|
777
|
+
blocks: list[_Block] = []
|
|
778
|
+
raw_lines = content.split("\n")
|
|
779
|
+
block_start = 0
|
|
780
|
+
block_has_fence = False
|
|
781
|
+
position = 0
|
|
782
|
+
after_blank = False
|
|
783
|
+
open_fence: tuple[int, str] | None = None # (indent, marker) of the fence we are inside
|
|
784
|
+
for index, line in enumerate(raw_lines):
|
|
785
|
+
if open_fence is None:
|
|
786
|
+
if not line.strip():
|
|
787
|
+
after_blank = True
|
|
788
|
+
else:
|
|
789
|
+
if after_blank and position > block_start:
|
|
790
|
+
blocks.append(_Block(block_start, position, block_has_fence))
|
|
791
|
+
block_start, block_has_fence = position, False
|
|
792
|
+
after_blank = False
|
|
793
|
+
opener = _opening_fence(line)
|
|
794
|
+
if opener is not None:
|
|
795
|
+
# Fences nested in list items are indented past column 3, so any
|
|
796
|
+
# indent opens one here; its closer may sit up to 3 columns deeper.
|
|
797
|
+
open_fence = (opener[0], opener[1])
|
|
798
|
+
block_has_fence = True
|
|
799
|
+
elif _is_closing_fence(line, open_fence[1], open_fence[0] + _MAX_TOP_LEVEL_INDENT):
|
|
800
|
+
open_fence = None
|
|
801
|
+
position += len(line) + (1 if index < len(raw_lines) - 1 else 0)
|
|
802
|
+
blocks.append(_Block(block_start, len(content), block_has_fence))
|
|
803
|
+
return blocks
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _split_block(content: str, begin: int, end: int, max_chars: int) -> list[tuple[int, int]]:
|
|
807
|
+
"""Split one oversized block at line boundaries, then at whitespace if needed.
|
|
808
|
+
|
|
809
|
+
A block is oversized as a whole, yet a fence inside it (prose directly above it, a
|
|
810
|
+
list whose items carry code) usually is not: a fenced run that fits in a part stays
|
|
811
|
+
one piece, so it is only ever cut when it exceeds the limit by itself.
|
|
812
|
+
"""
|
|
813
|
+
pieces: list[tuple[int, int]] = []
|
|
814
|
+
position = begin
|
|
815
|
+
for fence_start, fence_end in _fenced_runs(content, begin, end):
|
|
816
|
+
pieces.extend(_line_pieces(content, position, fence_start, max_chars))
|
|
817
|
+
if fence_end - fence_start <= max_chars:
|
|
818
|
+
pieces.append((fence_start, fence_end))
|
|
819
|
+
else:
|
|
820
|
+
pieces.extend(_line_pieces(content, fence_start, fence_end, max_chars))
|
|
821
|
+
position = fence_end
|
|
822
|
+
pieces.extend(_line_pieces(content, position, end, max_chars))
|
|
823
|
+
return pieces
|
|
824
|
+
|
|
825
|
+
|
|
826
|
+
def _fenced_runs(content: str, begin: int, end: int) -> list[tuple[int, int]]:
|
|
827
|
+
"""Character spans of the fenced runs in ``content[begin:end]``, closing line included.
|
|
828
|
+
|
|
829
|
+
``begin`` is a block boundary, so it is never inside a fence; a fence left open runs
|
|
830
|
+
to ``end``.
|
|
831
|
+
"""
|
|
832
|
+
runs: list[tuple[int, int]] = []
|
|
833
|
+
open_fence: tuple[int, str] | None = None
|
|
834
|
+
run_start = position = begin
|
|
835
|
+
while position < end:
|
|
836
|
+
newline = content.find("\n", position, end)
|
|
837
|
+
line_end = end if newline == -1 else newline + 1
|
|
838
|
+
line = content[position:line_end].rstrip("\n")
|
|
839
|
+
if open_fence is None:
|
|
840
|
+
opener = _opening_fence(line) if line.strip() else None
|
|
841
|
+
if opener is not None:
|
|
842
|
+
open_fence, run_start = (opener[0], opener[1]), position
|
|
843
|
+
elif _is_closing_fence(line, open_fence[1], open_fence[0] + _MAX_TOP_LEVEL_INDENT):
|
|
844
|
+
runs.append((run_start, line_end))
|
|
845
|
+
open_fence = None
|
|
846
|
+
position = line_end
|
|
847
|
+
if open_fence is not None:
|
|
848
|
+
runs.append((run_start, end))
|
|
849
|
+
return runs
|
|
850
|
+
|
|
851
|
+
|
|
852
|
+
def _line_pieces(content: str, begin: int, end: int, max_chars: int) -> list[tuple[int, int]]:
|
|
853
|
+
"""One piece per line of ``content[begin:end]``; an over-long line is cut at whitespace."""
|
|
854
|
+
pieces: list[tuple[int, int]] = []
|
|
855
|
+
position = begin
|
|
856
|
+
while position < end:
|
|
857
|
+
newline = content.find("\n", position, end)
|
|
858
|
+
line_end = end if newline == -1 else newline + 1
|
|
859
|
+
# An over-long line is cut into pieces of at most half a part, so that whatever
|
|
860
|
+
# precedes it (typically the heading) can still share a part with its first piece.
|
|
861
|
+
piece = max(1, max_chars // 2) if line_end - position > max_chars else max_chars
|
|
862
|
+
while line_end - position > piece:
|
|
863
|
+
cut = content.rfind(" ", position + 1, position + piece)
|
|
864
|
+
cut = position + piece if cut == -1 else cut + 1
|
|
865
|
+
pieces.append((position, cut))
|
|
866
|
+
position = cut
|
|
867
|
+
pieces.append((position, line_end))
|
|
868
|
+
position = line_end
|
|
869
|
+
return pieces
|