sphinx-mkdocs-migrate 0.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sphinx_mkdocs_migrate/__init__.py +8 -0
- sphinx_mkdocs_migrate/analyzer/__init__.py +17 -0
- sphinx_mkdocs_migrate/analyzer/ci.py +223 -0
- sphinx_mkdocs_migrate/analyzer/dependencies.py +134 -0
- sphinx_mkdocs_migrate/analyzer/markdown.py +148 -0
- sphinx_mkdocs_migrate/analyzer/mkdocs.py +263 -0
- sphinx_mkdocs_migrate/analyzer/models.py +348 -0
- sphinx_mkdocs_migrate/analyzer/navigation.py +108 -0
- sphinx_mkdocs_migrate/analyzer/project.py +511 -0
- sphinx_mkdocs_migrate/cli.py +507 -0
- sphinx_mkdocs_migrate/parsing/doc_ir.py +533 -0
- sphinx_mkdocs_migrate/parsing/flow_extractor.py +457 -0
- sphinx_mkdocs_migrate/parsing/html_flow_parser.py +349 -0
- sphinx_mkdocs_migrate/parsing/markdown.py +22 -0
- sphinx_mkdocs_migrate/parsing/markdown_ir.py +49 -0
- sphinx_mkdocs_migrate/parsing/markdown_it_adapter.py +496 -0
- sphinx_mkdocs_migrate/parsing/requirements.py +155 -0
- sphinx_mkdocs_migrate/planner/accountability.py +111 -0
- sphinx_mkdocs_migrate/planner/ci.py +142 -0
- sphinx_mkdocs_migrate/planner/conf_builder.py +183 -0
- sphinx_mkdocs_migrate/planner/models.py +379 -0
- sphinx_mkdocs_migrate/planner/planner.py +1867 -0
- sphinx_mkdocs_migrate/planner/policy.py +474 -0
- sphinx_mkdocs_migrate/planner/theme_constants.py +70 -0
- sphinx_mkdocs_migrate/planner/toctree.py +158 -0
- sphinx_mkdocs_migrate/py.typed +1 -0
- sphinx_mkdocs_migrate/rules/catalog.py +154 -0
- sphinx_mkdocs_migrate/rules/engine.py +94 -0
- sphinx_mkdocs_migrate/rules/models.py +176 -0
- sphinx_mkdocs_migrate/transformer/engine.py +897 -0
- sphinx_mkdocs_migrate/transformer/models.py +59 -0
- sphinx_mkdocs_migrate/transformer/myst_transformer.py +393 -0
- sphinx_mkdocs_migrate/validator/models.py +40 -0
- sphinx_mkdocs_migrate/validator/verifier.py +377 -0
- sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/METADATA +199 -0
- sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/RECORD +39 -0
- sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/WHEEL +4 -0
- sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/entry_points.txt +2 -0
- sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
"""Markdown-it-py adapter with controlled MkDocs source extension parsing."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from typing import List, Optional, Tuple
|
|
5
|
+
from markdown_it import MarkdownIt
|
|
6
|
+
from markdown_it.token import Token
|
|
7
|
+
from .markdown_ir import NodeKind, BaseIRNode, DocumentIR
|
|
8
|
+
|
|
9
|
+
# Controlled MkDocs / PyMdown syntax patterns
|
|
10
|
+
RE_ADMONITION_HEADER = re.compile(
|
|
11
|
+
r"^(?P<indent>[ ]{0,3})!{3}[ ]+(?P<type>note|warning|tip|info|important|caution|danger|bug|example|quote|abstract|check|question|fail|success)(?:[ ]+\"(?P<title>[^\"]*)\")?",
|
|
12
|
+
re.IGNORECASE,
|
|
13
|
+
)
|
|
14
|
+
RE_DETAILS_HEADER = re.compile(
|
|
15
|
+
r"^(?P<indent>[ ]{0,3})\?{3}(?P<state>\+|-)?(?:[ ]+(?P<type>note|warning|tip|info|details))?[ ]+\"(?P<title>[^\"]+)\"",
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
RE_TAB_HEADER = re.compile(r"^(?P<indent>[ ]{0,3})={3}[ ]+\"(?P<title>[^\"]+)\"")
|
|
19
|
+
RE_MKDOCSTRINGS = re.compile(r"^:::[ ]+(?P<symbol>[a-zA-Z0-9_\.]+)")
|
|
20
|
+
RE_SNIPPET = re.compile(r"^--8<--[ ]+\"(?P<filepath>[^\"]+)\"")
|
|
21
|
+
RE_CODE_FENCE_START = re.compile(r"^(?P<indent>[ ]{0,3})(?P<char>`|~){3,}")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class MarkdownIRBuilder:
|
|
25
|
+
"""Builds a structured DocumentIR tree combining markdown-it tokens with controlled MkDocs extensions."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, file_path: str = ""):
|
|
28
|
+
self.file_path = file_path
|
|
29
|
+
# Initialize markdown-it for all Standard CommonMark grammar
|
|
30
|
+
self.md_parser = MarkdownIt("commonmark", {"html": True}).enable("table")
|
|
31
|
+
|
|
32
|
+
def build_from_text(self, text: str) -> DocumentIR:
|
|
33
|
+
lines = text.splitlines()
|
|
34
|
+
line_map = list(range(1, len(lines) + 1))
|
|
35
|
+
root_nodes = self._parse_lines(lines, line_map)
|
|
36
|
+
return DocumentIR(file_path=self.file_path, nodes=root_nodes)
|
|
37
|
+
|
|
38
|
+
def _parse_lines(self, lines: List[str], line_map: List[int]) -> List[BaseIRNode]:
|
|
39
|
+
if not lines:
|
|
40
|
+
return []
|
|
41
|
+
|
|
42
|
+
nodes: List[BaseIRNode] = []
|
|
43
|
+
i = 0
|
|
44
|
+
n = len(lines)
|
|
45
|
+
chunk: List[str] = []
|
|
46
|
+
chunk_line_map: List[int] = []
|
|
47
|
+
|
|
48
|
+
def flush_chunk():
|
|
49
|
+
nonlocal chunk, chunk_line_map
|
|
50
|
+
if chunk:
|
|
51
|
+
text_block = "\n".join(chunk)
|
|
52
|
+
sub_nodes = self._parse_standard_markdown_chunk(
|
|
53
|
+
text_block, chunk_line_map
|
|
54
|
+
)
|
|
55
|
+
nodes.extend(sub_nodes)
|
|
56
|
+
chunk = []
|
|
57
|
+
chunk_line_map = []
|
|
58
|
+
|
|
59
|
+
current_tab_set: Optional[BaseIRNode] = None
|
|
60
|
+
|
|
61
|
+
while i < n:
|
|
62
|
+
line = lines[i]
|
|
63
|
+
s_line = line.strip()
|
|
64
|
+
|
|
65
|
+
# 1. Code Fence (Fenced blocks are buffered as standard markdown for markdown-it to parse safely)
|
|
66
|
+
m_fence = RE_CODE_FENCE_START.match(line)
|
|
67
|
+
if m_fence:
|
|
68
|
+
fence_char = m_fence.group("char")
|
|
69
|
+
match_str = m_fence.group(0).strip()
|
|
70
|
+
fence_len = len(match_str)
|
|
71
|
+
|
|
72
|
+
chunk.append(line)
|
|
73
|
+
chunk_line_map.append(line_map[i])
|
|
74
|
+
i += 1
|
|
75
|
+
while i < n:
|
|
76
|
+
cur_line = lines[i]
|
|
77
|
+
chunk.append(cur_line)
|
|
78
|
+
chunk_line_map.append(line_map[i])
|
|
79
|
+
cur_stripped = cur_line.strip()
|
|
80
|
+
if cur_stripped.startswith(fence_char * fence_len):
|
|
81
|
+
i += 1
|
|
82
|
+
break
|
|
83
|
+
i += 1
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
# 2. Content Tabs (=== "Title")
|
|
87
|
+
m_tab = RE_TAB_HEADER.match(line)
|
|
88
|
+
if m_tab:
|
|
89
|
+
flush_chunk()
|
|
90
|
+
title = m_tab.group("title")
|
|
91
|
+
tab_start_l = line_map[i]
|
|
92
|
+
tab_header_line = line
|
|
93
|
+
i += 1
|
|
94
|
+
|
|
95
|
+
tab_body_lines: List[str] = []
|
|
96
|
+
tab_body_line_map: List[int] = []
|
|
97
|
+
|
|
98
|
+
while i < n:
|
|
99
|
+
next_l = lines[i]
|
|
100
|
+
if not next_l.strip():
|
|
101
|
+
tab_body_lines.append("")
|
|
102
|
+
tab_body_line_map.append(line_map[i])
|
|
103
|
+
i += 1
|
|
104
|
+
continue
|
|
105
|
+
if RE_TAB_HEADER.match(next_l):
|
|
106
|
+
break
|
|
107
|
+
if next_l.startswith(" ") or next_l.startswith("\t"):
|
|
108
|
+
stripped = (
|
|
109
|
+
next_l[4:] if next_l.startswith(" ") else next_l[1:]
|
|
110
|
+
)
|
|
111
|
+
tab_body_lines.append(stripped)
|
|
112
|
+
tab_body_line_map.append(line_map[i])
|
|
113
|
+
i += 1
|
|
114
|
+
else:
|
|
115
|
+
break
|
|
116
|
+
|
|
117
|
+
# Strip trailing empty lines to determine precise end line
|
|
118
|
+
while tab_body_lines and not tab_body_lines[-1].strip():
|
|
119
|
+
tab_body_lines.pop()
|
|
120
|
+
tab_body_line_map.pop()
|
|
121
|
+
|
|
122
|
+
child_nodes = self._parse_lines(tab_body_lines, tab_body_line_map)
|
|
123
|
+
tab_end_l = tab_body_line_map[-1] if tab_body_line_map else tab_start_l
|
|
124
|
+
|
|
125
|
+
tab_item = BaseIRNode(
|
|
126
|
+
kind=NodeKind.TAB_ITEM,
|
|
127
|
+
start_line=tab_start_l,
|
|
128
|
+
end_line=tab_end_l,
|
|
129
|
+
raw_text=tab_header_line,
|
|
130
|
+
metadata={"title": title},
|
|
131
|
+
children=child_nodes,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
if current_tab_set is not None:
|
|
135
|
+
current_tab_set.children.append(tab_item)
|
|
136
|
+
current_tab_set.end_line = tab_end_l
|
|
137
|
+
else:
|
|
138
|
+
current_tab_set = BaseIRNode(
|
|
139
|
+
kind=NodeKind.TAB_SET,
|
|
140
|
+
start_line=tab_start_l,
|
|
141
|
+
end_line=tab_end_l,
|
|
142
|
+
raw_text="",
|
|
143
|
+
children=[tab_item],
|
|
144
|
+
)
|
|
145
|
+
nodes.append(current_tab_set)
|
|
146
|
+
continue
|
|
147
|
+
|
|
148
|
+
# Non-tab line breaks consecutive tab sets
|
|
149
|
+
current_tab_set = None
|
|
150
|
+
|
|
151
|
+
# 3. Admonitions (!!! note "Title")
|
|
152
|
+
m_adm = RE_ADMONITION_HEADER.match(line)
|
|
153
|
+
if m_adm:
|
|
154
|
+
flush_chunk()
|
|
155
|
+
adm_type = m_adm.group("type").lower()
|
|
156
|
+
title = m_adm.group("title") or adm_type.capitalize()
|
|
157
|
+
adm_start_l = line_map[i]
|
|
158
|
+
adm_header = line
|
|
159
|
+
i += 1
|
|
160
|
+
|
|
161
|
+
adm_body_lines: List[str] = []
|
|
162
|
+
adm_body_line_map: List[int] = []
|
|
163
|
+
|
|
164
|
+
while i < n:
|
|
165
|
+
next_l = lines[i]
|
|
166
|
+
if not next_l.strip():
|
|
167
|
+
adm_body_lines.append("")
|
|
168
|
+
adm_body_line_map.append(line_map[i])
|
|
169
|
+
i += 1
|
|
170
|
+
continue
|
|
171
|
+
if next_l.startswith(" ") or next_l.startswith("\t"):
|
|
172
|
+
stripped = (
|
|
173
|
+
next_l[4:] if next_l.startswith(" ") else next_l[1:]
|
|
174
|
+
)
|
|
175
|
+
adm_body_lines.append(stripped)
|
|
176
|
+
adm_body_line_map.append(line_map[i])
|
|
177
|
+
i += 1
|
|
178
|
+
else:
|
|
179
|
+
break
|
|
180
|
+
|
|
181
|
+
# Strip trailing empty lines
|
|
182
|
+
while adm_body_lines and not adm_body_lines[-1].strip():
|
|
183
|
+
adm_body_lines.pop()
|
|
184
|
+
adm_body_line_map.pop()
|
|
185
|
+
|
|
186
|
+
child_nodes = self._parse_lines(adm_body_lines, adm_body_line_map)
|
|
187
|
+
adm_end_l = adm_body_line_map[-1] if adm_body_line_map else adm_start_l
|
|
188
|
+
|
|
189
|
+
adm_node = BaseIRNode(
|
|
190
|
+
kind=NodeKind.ADMONITION,
|
|
191
|
+
start_line=adm_start_l,
|
|
192
|
+
end_line=adm_end_l,
|
|
193
|
+
raw_text=adm_header,
|
|
194
|
+
metadata={"admonition_type": adm_type, "title": title},
|
|
195
|
+
children=child_nodes,
|
|
196
|
+
)
|
|
197
|
+
nodes.append(adm_node)
|
|
198
|
+
continue
|
|
199
|
+
|
|
200
|
+
# 4. Details Dropdowns (???+ note "Title")
|
|
201
|
+
m_det = RE_DETAILS_HEADER.match(line)
|
|
202
|
+
if m_det:
|
|
203
|
+
flush_chunk()
|
|
204
|
+
title = m_det.group("title")
|
|
205
|
+
state = m_det.group("state")
|
|
206
|
+
is_open = state == "+"
|
|
207
|
+
det_start_l = line_map[i]
|
|
208
|
+
det_header = line
|
|
209
|
+
i += 1
|
|
210
|
+
|
|
211
|
+
det_body_lines: List[str] = []
|
|
212
|
+
det_body_line_map: List[int] = []
|
|
213
|
+
|
|
214
|
+
while i < n:
|
|
215
|
+
next_l = lines[i]
|
|
216
|
+
if not next_l.strip():
|
|
217
|
+
det_body_lines.append("")
|
|
218
|
+
det_body_line_map.append(line_map[i])
|
|
219
|
+
i += 1
|
|
220
|
+
continue
|
|
221
|
+
if next_l.startswith(" ") or next_l.startswith("\t"):
|
|
222
|
+
stripped = (
|
|
223
|
+
next_l[4:] if next_l.startswith(" ") else next_l[1:]
|
|
224
|
+
)
|
|
225
|
+
det_body_lines.append(stripped)
|
|
226
|
+
det_body_line_map.append(line_map[i])
|
|
227
|
+
i += 1
|
|
228
|
+
else:
|
|
229
|
+
break
|
|
230
|
+
|
|
231
|
+
# Strip trailing empty lines
|
|
232
|
+
while det_body_lines and not det_body_lines[-1].strip():
|
|
233
|
+
det_body_lines.pop()
|
|
234
|
+
det_body_line_map.pop()
|
|
235
|
+
|
|
236
|
+
child_nodes = self._parse_lines(det_body_lines, det_body_line_map)
|
|
237
|
+
det_end_l = det_body_line_map[-1] if det_body_line_map else det_start_l
|
|
238
|
+
|
|
239
|
+
det_node = BaseIRNode(
|
|
240
|
+
kind=NodeKind.DETAILS_DROPDOWN,
|
|
241
|
+
start_line=det_start_l,
|
|
242
|
+
end_line=det_end_l,
|
|
243
|
+
raw_text=det_header,
|
|
244
|
+
metadata={"title": title, "open_state": is_open},
|
|
245
|
+
children=child_nodes,
|
|
246
|
+
)
|
|
247
|
+
nodes.append(det_node)
|
|
248
|
+
continue
|
|
249
|
+
|
|
250
|
+
# 5. API Directives (::: symbol)
|
|
251
|
+
m_api = RE_MKDOCSTRINGS.match(s_line)
|
|
252
|
+
if m_api:
|
|
253
|
+
flush_chunk()
|
|
254
|
+
api_start_l = line_map[i]
|
|
255
|
+
api_raw_lines = [line]
|
|
256
|
+
i += 1
|
|
257
|
+
while i < n:
|
|
258
|
+
next_l = lines[i]
|
|
259
|
+
if (
|
|
260
|
+
next_l.startswith(" ")
|
|
261
|
+
or next_l.startswith("\t")
|
|
262
|
+
or (len(api_raw_lines) > 1 and next_l.startswith(" "))
|
|
263
|
+
):
|
|
264
|
+
api_raw_lines.append(next_l)
|
|
265
|
+
i += 1
|
|
266
|
+
continue
|
|
267
|
+
break
|
|
268
|
+
api_end_l = line_map[i - 1]
|
|
269
|
+
api_node = BaseIRNode(
|
|
270
|
+
kind=NodeKind.API_DIRECTIVE,
|
|
271
|
+
start_line=api_start_l,
|
|
272
|
+
end_line=api_end_l,
|
|
273
|
+
raw_text="\n".join(api_raw_lines),
|
|
274
|
+
metadata={"symbol": m_api.group("symbol")},
|
|
275
|
+
)
|
|
276
|
+
nodes.append(api_node)
|
|
277
|
+
continue
|
|
278
|
+
|
|
279
|
+
# 6. Snippet Includes (--8<-- "...")
|
|
280
|
+
m_snip = RE_SNIPPET.match(s_line)
|
|
281
|
+
if m_snip:
|
|
282
|
+
flush_chunk()
|
|
283
|
+
snip_node = BaseIRNode(
|
|
284
|
+
kind=NodeKind.SNIPPET_INCLUDE,
|
|
285
|
+
start_line=line_map[i],
|
|
286
|
+
end_line=line_map[i],
|
|
287
|
+
raw_text=line,
|
|
288
|
+
metadata={"filepath": m_snip.group("filepath")},
|
|
289
|
+
)
|
|
290
|
+
nodes.append(snip_node)
|
|
291
|
+
i += 1
|
|
292
|
+
continue
|
|
293
|
+
|
|
294
|
+
# Standard Markdown line
|
|
295
|
+
chunk.append(line)
|
|
296
|
+
chunk_line_map.append(line_map[i])
|
|
297
|
+
i += 1
|
|
298
|
+
|
|
299
|
+
flush_chunk()
|
|
300
|
+
return nodes
|
|
301
|
+
|
|
302
|
+
def _parse_standard_markdown_chunk(
|
|
303
|
+
self, text: str, line_map: List[int]
|
|
304
|
+
) -> List[BaseIRNode]:
|
|
305
|
+
tokens = self.md_parser.parse(text)
|
|
306
|
+
nodes: List[BaseIRNode] = []
|
|
307
|
+
stack: List[BaseIRNode] = []
|
|
308
|
+
i = 0
|
|
309
|
+
n = len(tokens)
|
|
310
|
+
|
|
311
|
+
while i < n:
|
|
312
|
+
token = tokens[i]
|
|
313
|
+
|
|
314
|
+
# 1. Code Fence (handled natively by markdown-it)
|
|
315
|
+
if token.type == "fence":
|
|
316
|
+
start_l, end_l = self._get_lines(token, line_map)
|
|
317
|
+
info = token.info.strip()
|
|
318
|
+
kind = (
|
|
319
|
+
NodeKind.MERMAID_DIAGRAM
|
|
320
|
+
if info == "mermaid"
|
|
321
|
+
else NodeKind.CODE_BLOCK
|
|
322
|
+
)
|
|
323
|
+
fence_node = BaseIRNode(
|
|
324
|
+
kind=kind,
|
|
325
|
+
start_line=start_l,
|
|
326
|
+
end_line=end_l,
|
|
327
|
+
raw_text=token.content,
|
|
328
|
+
metadata={"info_string": info},
|
|
329
|
+
)
|
|
330
|
+
if stack:
|
|
331
|
+
stack[-1].children.append(fence_node)
|
|
332
|
+
else:
|
|
333
|
+
nodes.append(fence_node)
|
|
334
|
+
i += 1
|
|
335
|
+
continue
|
|
336
|
+
|
|
337
|
+
# 2. Heading
|
|
338
|
+
if token.type == "heading_open":
|
|
339
|
+
start_l, end_l = self._get_lines(token, line_map)
|
|
340
|
+
level = (
|
|
341
|
+
int(token.tag[1:])
|
|
342
|
+
if len(token.tag) > 1 and token.tag[1:].isdigit()
|
|
343
|
+
else 1
|
|
344
|
+
)
|
|
345
|
+
title = ""
|
|
346
|
+
if i + 1 < n and tokens[i + 1].type == "inline":
|
|
347
|
+
title = tokens[i + 1].content.strip()
|
|
348
|
+
head_node = BaseIRNode(
|
|
349
|
+
kind=NodeKind.HEADING,
|
|
350
|
+
start_line=start_l,
|
|
351
|
+
end_line=end_l,
|
|
352
|
+
raw_text=title,
|
|
353
|
+
metadata={"level": level, "title": title},
|
|
354
|
+
)
|
|
355
|
+
if stack:
|
|
356
|
+
stack[-1].children.append(head_node)
|
|
357
|
+
else:
|
|
358
|
+
nodes.append(head_node)
|
|
359
|
+
while i < n and tokens[i].type != "heading_close":
|
|
360
|
+
i += 1
|
|
361
|
+
i += 1
|
|
362
|
+
continue
|
|
363
|
+
|
|
364
|
+
# 3. Lists & Blockquotes (Containers)
|
|
365
|
+
if token.type in (
|
|
366
|
+
"blockquote_open",
|
|
367
|
+
"bullet_list_open",
|
|
368
|
+
"ordered_list_open",
|
|
369
|
+
"list_item_open",
|
|
370
|
+
):
|
|
371
|
+
start_l, end_l = self._get_lines(token, line_map)
|
|
372
|
+
kind = (
|
|
373
|
+
NodeKind.BLOCK_QUOTE
|
|
374
|
+
if token.type == "blockquote_open"
|
|
375
|
+
else (
|
|
376
|
+
NodeKind.LIST_ITEM
|
|
377
|
+
if token.type == "list_item_open"
|
|
378
|
+
else NodeKind.LIST
|
|
379
|
+
)
|
|
380
|
+
)
|
|
381
|
+
container_node = BaseIRNode(
|
|
382
|
+
kind=kind,
|
|
383
|
+
start_line=start_l,
|
|
384
|
+
end_line=end_l,
|
|
385
|
+
raw_text="",
|
|
386
|
+
metadata={"token_tag": token.tag},
|
|
387
|
+
)
|
|
388
|
+
if stack:
|
|
389
|
+
stack[-1].children.append(container_node)
|
|
390
|
+
else:
|
|
391
|
+
nodes.append(container_node)
|
|
392
|
+
stack.append(container_node)
|
|
393
|
+
i += 1
|
|
394
|
+
continue
|
|
395
|
+
|
|
396
|
+
if token.type in (
|
|
397
|
+
"blockquote_close",
|
|
398
|
+
"bullet_list_close",
|
|
399
|
+
"ordered_list_close",
|
|
400
|
+
"list_item_close",
|
|
401
|
+
):
|
|
402
|
+
if stack:
|
|
403
|
+
stack.pop()
|
|
404
|
+
i += 1
|
|
405
|
+
continue
|
|
406
|
+
|
|
407
|
+
# 4. Paragraphs & Generic Inline Links
|
|
408
|
+
if token.type == "paragraph_open":
|
|
409
|
+
start_l, end_l = self._get_lines(token, line_map)
|
|
410
|
+
content_token = (
|
|
411
|
+
tokens[i + 1]
|
|
412
|
+
if i + 1 < n and tokens[i + 1].type == "inline"
|
|
413
|
+
else None
|
|
414
|
+
)
|
|
415
|
+
content_text = content_token.content.strip() if content_token else ""
|
|
416
|
+
|
|
417
|
+
p_node = BaseIRNode(
|
|
418
|
+
kind=NodeKind.PARAGRAPH,
|
|
419
|
+
start_line=start_l,
|
|
420
|
+
end_line=end_l,
|
|
421
|
+
raw_text=content_text,
|
|
422
|
+
)
|
|
423
|
+
if content_token and content_token.children:
|
|
424
|
+
self._extract_generic_links(content_token.children, p_node, start_l)
|
|
425
|
+
|
|
426
|
+
if stack:
|
|
427
|
+
stack[-1].children.append(p_node)
|
|
428
|
+
else:
|
|
429
|
+
nodes.append(p_node)
|
|
430
|
+
|
|
431
|
+
while i < n and tokens[i].type != "paragraph_close":
|
|
432
|
+
i += 1
|
|
433
|
+
i += 1
|
|
434
|
+
continue
|
|
435
|
+
|
|
436
|
+
# 5. HTML Blocks
|
|
437
|
+
if token.type == "html_block":
|
|
438
|
+
start_l, end_l = self._get_lines(token, line_map)
|
|
439
|
+
html_node = BaseIRNode(
|
|
440
|
+
kind=NodeKind.HTML_BLOCK,
|
|
441
|
+
start_line=start_l,
|
|
442
|
+
end_line=end_l,
|
|
443
|
+
raw_text=token.content,
|
|
444
|
+
)
|
|
445
|
+
if stack:
|
|
446
|
+
stack[-1].children.append(html_node)
|
|
447
|
+
else:
|
|
448
|
+
nodes.append(html_node)
|
|
449
|
+
i += 1
|
|
450
|
+
continue
|
|
451
|
+
|
|
452
|
+
i += 1
|
|
453
|
+
|
|
454
|
+
return nodes
|
|
455
|
+
|
|
456
|
+
def _get_lines(self, token: Token, line_map: List[int]) -> Tuple[int, int]:
|
|
457
|
+
if token.map and line_map:
|
|
458
|
+
start_idx = min(token.map[0], len(line_map) - 1)
|
|
459
|
+
end_idx = min(max(0, token.map[1] - 1), len(line_map) - 1)
|
|
460
|
+
return line_map[start_idx], line_map[end_idx]
|
|
461
|
+
return 1, 1
|
|
462
|
+
|
|
463
|
+
def _extract_generic_links(
|
|
464
|
+
self, inline_children: List[Token], parent: BaseIRNode, line_no: int
|
|
465
|
+
):
|
|
466
|
+
"""Extracts destination-neutral link references from markdown-it inline tokens."""
|
|
467
|
+
j = 0
|
|
468
|
+
m = len(inline_children)
|
|
469
|
+
while j < m:
|
|
470
|
+
child = inline_children[j]
|
|
471
|
+
if child.type == "link_open":
|
|
472
|
+
href = str(child.attrs.get("href", ""))
|
|
473
|
+
link_text = ""
|
|
474
|
+
if j + 1 < m and inline_children[j + 1].type == "text":
|
|
475
|
+
link_text = inline_children[j + 1].content
|
|
476
|
+
|
|
477
|
+
is_ext = (
|
|
478
|
+
href.startswith("http://")
|
|
479
|
+
or href.startswith("https://")
|
|
480
|
+
or href.startswith("mailto:")
|
|
481
|
+
)
|
|
482
|
+
parent.children.append(
|
|
483
|
+
BaseIRNode(
|
|
484
|
+
kind=NodeKind.LINK_REF,
|
|
485
|
+
start_line=line_no,
|
|
486
|
+
end_line=line_no,
|
|
487
|
+
raw_text=link_text,
|
|
488
|
+
metadata={
|
|
489
|
+
"text": link_text,
|
|
490
|
+
"href": href,
|
|
491
|
+
"is_external": is_ext,
|
|
492
|
+
"target_path": href,
|
|
493
|
+
},
|
|
494
|
+
)
|
|
495
|
+
)
|
|
496
|
+
j += 1
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""Structured manifest and dependency parsing."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import List, Optional, Tuple
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
if sys.version_info >= (3, 11):
|
|
10
|
+
import tomllib
|
|
11
|
+
else:
|
|
12
|
+
import tomli as tomllib
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class NormalizedDependency(BaseModel):
|
|
16
|
+
name: str
|
|
17
|
+
raw_spec: str
|
|
18
|
+
source_location: str # e.g., "project.dependencies", "dependency-groups.docs", "requirements.txt"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class NormalizedManifest(BaseModel):
|
|
22
|
+
manifest_type: str # "pyproject.toml", "requirements.txt"
|
|
23
|
+
manifest_path: str
|
|
24
|
+
python_requirement: Optional[str] = None
|
|
25
|
+
dependencies: List[NormalizedDependency] = Field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _normalize_package_name(spec: str) -> Tuple[str, str]:
|
|
29
|
+
clean = spec.strip()
|
|
30
|
+
match = re.match(r"^([a-zA-Z0-9_\-\.]+)(?:\[[^\]]+\])?", clean)
|
|
31
|
+
name = match.group(1).lower().replace("_", "-") if match else clean.lower()
|
|
32
|
+
return name, clean
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def parse_pyproject_toml(pyproject_path: Path) -> Optional[NormalizedManifest]:
|
|
36
|
+
if not pyproject_path.exists():
|
|
37
|
+
return None
|
|
38
|
+
try:
|
|
39
|
+
with open(pyproject_path, "rb") as f:
|
|
40
|
+
data = tomllib.load(f)
|
|
41
|
+
except Exception:
|
|
42
|
+
return None
|
|
43
|
+
|
|
44
|
+
deps: List[NormalizedDependency] = []
|
|
45
|
+
py_req = None
|
|
46
|
+
|
|
47
|
+
# 1. PEP 621 [project]
|
|
48
|
+
project_table = data.get("project", {})
|
|
49
|
+
if isinstance(project_table, dict):
|
|
50
|
+
py_req = project_table.get("requires-python")
|
|
51
|
+
for dep in project_table.get("dependencies", []):
|
|
52
|
+
name, raw = _normalize_package_name(dep)
|
|
53
|
+
deps.append(
|
|
54
|
+
NormalizedDependency(
|
|
55
|
+
name=name, raw_spec=raw, source_location="project.dependencies"
|
|
56
|
+
)
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# Optional dependencies (e.g. [project.optional-dependencies.docs])
|
|
60
|
+
opt_deps = project_table.get("optional-dependencies", {})
|
|
61
|
+
if isinstance(opt_deps, dict):
|
|
62
|
+
for group, group_deps in opt_deps.items():
|
|
63
|
+
if isinstance(group_deps, list):
|
|
64
|
+
for dep in group_deps:
|
|
65
|
+
if isinstance(dep, str):
|
|
66
|
+
name, raw = _normalize_package_name(dep)
|
|
67
|
+
deps.append(
|
|
68
|
+
NormalizedDependency(
|
|
69
|
+
name=name,
|
|
70
|
+
raw_spec=raw,
|
|
71
|
+
source_location=f"project.optional-dependencies.{group}",
|
|
72
|
+
)
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# 2. PEP 735 [dependency-groups]
|
|
76
|
+
dep_groups = data.get("dependency-groups", {})
|
|
77
|
+
if isinstance(dep_groups, dict):
|
|
78
|
+
for group, group_deps in dep_groups.items():
|
|
79
|
+
if isinstance(group_deps, list):
|
|
80
|
+
for dep in group_deps:
|
|
81
|
+
if isinstance(dep, str):
|
|
82
|
+
name, raw = _normalize_package_name(dep)
|
|
83
|
+
deps.append(
|
|
84
|
+
NormalizedDependency(
|
|
85
|
+
name=name,
|
|
86
|
+
raw_spec=raw,
|
|
87
|
+
source_location=f"dependency-groups.{group}",
|
|
88
|
+
)
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
# 3. Poetry [tool.poetry]
|
|
92
|
+
tool_poetry = data.get("tool", {}).get("poetry", {})
|
|
93
|
+
if isinstance(tool_poetry, dict):
|
|
94
|
+
poetry_deps = tool_poetry.get("dependencies", {})
|
|
95
|
+
if isinstance(poetry_deps, dict):
|
|
96
|
+
for pkg_name, spec in poetry_deps.items():
|
|
97
|
+
if pkg_name != "python":
|
|
98
|
+
raw = f"{pkg_name} {spec}" if isinstance(spec, str) else pkg_name
|
|
99
|
+
deps.append(
|
|
100
|
+
NormalizedDependency(
|
|
101
|
+
name=pkg_name.lower().replace("_", "-"),
|
|
102
|
+
raw_spec=raw,
|
|
103
|
+
source_location="tool.poetry.dependencies",
|
|
104
|
+
)
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
# Poetry group dependencies [tool.poetry.group.docs.dependencies]
|
|
108
|
+
poetry_groups = tool_poetry.get("group", {})
|
|
109
|
+
if isinstance(poetry_groups, dict):
|
|
110
|
+
for group, g_data in poetry_groups.items():
|
|
111
|
+
g_deps = g_data.get("dependencies", {})
|
|
112
|
+
if isinstance(g_deps, dict):
|
|
113
|
+
for pkg_name, spec in g_deps.items():
|
|
114
|
+
raw = (
|
|
115
|
+
f"{pkg_name} {spec}" if isinstance(spec, str) else pkg_name
|
|
116
|
+
)
|
|
117
|
+
deps.append(
|
|
118
|
+
NormalizedDependency(
|
|
119
|
+
name=pkg_name.lower().replace("_", "-"),
|
|
120
|
+
raw_spec=raw,
|
|
121
|
+
source_location=f"tool.poetry.group.{group}.dependencies",
|
|
122
|
+
)
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
return NormalizedManifest(
|
|
126
|
+
manifest_type="pyproject.toml",
|
|
127
|
+
manifest_path=str(pyproject_path),
|
|
128
|
+
python_requirement=py_req,
|
|
129
|
+
dependencies=deps,
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def parse_requirements_txt(req_path: Path) -> Optional[NormalizedManifest]:
|
|
134
|
+
if not req_path.exists():
|
|
135
|
+
return None
|
|
136
|
+
try:
|
|
137
|
+
content = req_path.read_text(encoding="utf-8")
|
|
138
|
+
except Exception:
|
|
139
|
+
return None
|
|
140
|
+
|
|
141
|
+
deps: List[NormalizedDependency] = []
|
|
142
|
+
for line in content.splitlines():
|
|
143
|
+
line = line.strip()
|
|
144
|
+
if not line or line.startswith("#") or line.startswith("-r"):
|
|
145
|
+
continue
|
|
146
|
+
name, raw = _normalize_package_name(line)
|
|
147
|
+
deps.append(
|
|
148
|
+
NormalizedDependency(
|
|
149
|
+
name=name, raw_spec=raw, source_location=str(req_path.name)
|
|
150
|
+
)
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
return NormalizedManifest(
|
|
154
|
+
manifest_type="requirements.txt", manifest_path=str(req_path), dependencies=deps
|
|
155
|
+
)
|