sphinx-mkdocs-migrate 0.0.1.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. sphinx_mkdocs_migrate/__init__.py +8 -0
  2. sphinx_mkdocs_migrate/analyzer/__init__.py +17 -0
  3. sphinx_mkdocs_migrate/analyzer/ci.py +223 -0
  4. sphinx_mkdocs_migrate/analyzer/dependencies.py +134 -0
  5. sphinx_mkdocs_migrate/analyzer/markdown.py +148 -0
  6. sphinx_mkdocs_migrate/analyzer/mkdocs.py +263 -0
  7. sphinx_mkdocs_migrate/analyzer/models.py +348 -0
  8. sphinx_mkdocs_migrate/analyzer/navigation.py +108 -0
  9. sphinx_mkdocs_migrate/analyzer/project.py +511 -0
  10. sphinx_mkdocs_migrate/cli.py +507 -0
  11. sphinx_mkdocs_migrate/parsing/doc_ir.py +533 -0
  12. sphinx_mkdocs_migrate/parsing/flow_extractor.py +457 -0
  13. sphinx_mkdocs_migrate/parsing/html_flow_parser.py +349 -0
  14. sphinx_mkdocs_migrate/parsing/markdown.py +22 -0
  15. sphinx_mkdocs_migrate/parsing/markdown_ir.py +49 -0
  16. sphinx_mkdocs_migrate/parsing/markdown_it_adapter.py +496 -0
  17. sphinx_mkdocs_migrate/parsing/requirements.py +155 -0
  18. sphinx_mkdocs_migrate/planner/accountability.py +111 -0
  19. sphinx_mkdocs_migrate/planner/ci.py +142 -0
  20. sphinx_mkdocs_migrate/planner/conf_builder.py +183 -0
  21. sphinx_mkdocs_migrate/planner/models.py +379 -0
  22. sphinx_mkdocs_migrate/planner/planner.py +1867 -0
  23. sphinx_mkdocs_migrate/planner/policy.py +474 -0
  24. sphinx_mkdocs_migrate/planner/theme_constants.py +70 -0
  25. sphinx_mkdocs_migrate/planner/toctree.py +158 -0
  26. sphinx_mkdocs_migrate/py.typed +1 -0
  27. sphinx_mkdocs_migrate/rules/catalog.py +154 -0
  28. sphinx_mkdocs_migrate/rules/engine.py +94 -0
  29. sphinx_mkdocs_migrate/rules/models.py +176 -0
  30. sphinx_mkdocs_migrate/transformer/engine.py +897 -0
  31. sphinx_mkdocs_migrate/transformer/models.py +59 -0
  32. sphinx_mkdocs_migrate/transformer/myst_transformer.py +393 -0
  33. sphinx_mkdocs_migrate/validator/models.py +40 -0
  34. sphinx_mkdocs_migrate/validator/verifier.py +377 -0
  35. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/METADATA +199 -0
  36. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/RECORD +39 -0
  37. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/WHEEL +4 -0
  38. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/entry_points.txt +2 -0
  39. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,349 @@
1
+ """HTML Semantic DOM Flow Parser.
2
+
3
+ Extracts ordered document flow, API object hierarchy, docstrings, and summary tables
4
+ directly from rendered MkDocs HTML (e.g. mkdocs build output) to serve as the empirical
5
+ ground truth for document layout and sequence.
6
+ """
7
+
8
+ from enum import Enum
9
+ from pathlib import Path
10
+ from typing import List, Dict, Any, Optional
11
+ from pydantic import BaseModel, Field
12
+ from bs4 import BeautifulSoup, Tag
13
+
14
+
15
+ class HtmlFlowRole(str, Enum):
16
+ HEADING = "HEADING"
17
+ PROSE = "PROSE"
18
+ AUTOSUMMARY = "AUTOSUMMARY"
19
+ API_MODULE = "API_MODULE"
20
+ API_CLASS = "API_CLASS"
21
+ API_FUNCTION = "API_FUNCTION"
22
+ API_ATTRIBUTE = "API_ATTRIBUTE"
23
+ TABLE = "TABLE"
24
+ CODE_BLOCK = "CODE_BLOCK"
25
+ ADMONITION = "ADMONITION"
26
+ LIST = "LIST"
27
+ UNKNOWN = "UNKNOWN"
28
+
29
+
30
+ def _get_classes(tag: Any) -> List[str]:
31
+ if not isinstance(tag, Tag):
32
+ return []
33
+ cls = tag.get("class")
34
+ if isinstance(cls, list):
35
+ return [str(c) for c in cls]
36
+ elif isinstance(cls, str):
37
+ return cls.split()
38
+ return []
39
+
40
+
41
+ class HtmlFlowElement(BaseModel):
42
+ """An ordered semantic element extracted from the rendered HTML DOM."""
43
+
44
+ order_index: int
45
+ role: HtmlFlowRole
46
+ text: Optional[str] = None
47
+ level: Optional[int] = None
48
+ qname: Optional[str] = None
49
+ directive: Optional[str] = None
50
+ options: Dict[str, Any] = Field(default_factory=dict)
51
+ symbols: List[str] = Field(default_factory=list)
52
+ headers: List[str] = Field(default_factory=list)
53
+ raw_html: Optional[str] = None
54
+
55
+
56
+ class HtmlPageFlow(BaseModel):
57
+ """Normalized ordered flow of a single rendered documentation page."""
58
+
59
+ source_html: str
60
+ logical_route: str
61
+ title: Optional[str] = None
62
+ elements: List[HtmlFlowElement] = Field(default_factory=list)
63
+
64
+
65
+ class HtmlFlowParser:
66
+ """Parses rendered MkDocs HTML into an ordered sequence of semantic elements."""
67
+
68
+ def __init__(self):
69
+ pass
70
+
71
+ def parse_file(
72
+ self, html_path: Path, rel_route: Optional[str] = None
73
+ ) -> HtmlPageFlow:
74
+ """Parse an HTML file on disk."""
75
+ content = html_path.read_text(encoding="utf-8", errors="replace")
76
+ route = rel_route or str(html_path)
77
+ return self.parse_html(content, route)
78
+
79
+ def parse_html(self, html_content: str, rel_route: str = "") -> HtmlPageFlow:
80
+ """Parse HTML string and extract sequential document flow."""
81
+ soup = BeautifulSoup(html_content, "html.parser")
82
+ article = (
83
+ soup.select_one("article.md-content__inner")
84
+ or soup.find("article")
85
+ or soup.select_one("main")
86
+ )
87
+
88
+ elements: List[HtmlFlowElement] = []
89
+ page_title: Optional[str] = None
90
+ order_idx = 0
91
+
92
+ if not article:
93
+ return HtmlPageFlow(
94
+ source_html=rel_route, logical_route=rel_route, title=None, elements=[]
95
+ )
96
+
97
+ # Check if the page is a mkdocstrings API module document
98
+ doc_module = article.select_one("div.doc-module")
99
+ if doc_module:
100
+ # 1. Module Heading
101
+ heading_elem = doc_module.select_one(
102
+ "h1.doc-heading"
103
+ ) or doc_module.select_one("h1")
104
+ mod_qname = ""
105
+ if heading_elem:
106
+ heading_text = (
107
+ heading_elem.get_text(" ", strip=True).replace("🔗", "").strip()
108
+ )
109
+ page_title = heading_text
110
+ mod_qname = heading_text
111
+ elements.append(
112
+ HtmlFlowElement(
113
+ order_index=order_idx,
114
+ role=HtmlFlowRole.HEADING,
115
+ level=1,
116
+ text=heading_text,
117
+ qname=mod_qname,
118
+ )
119
+ )
120
+ order_idx += 1
121
+
122
+ # 2. Walk contents sequentially (Docstring prose, Summary tables, Children)
123
+ contents = doc_module.select_one("div.doc-contents")
124
+ if contents:
125
+ for child in contents.find_all(recursive=False):
126
+ if not isinstance(child, Tag):
127
+ continue
128
+
129
+ classes = _get_classes(child)
130
+
131
+ # Prose paragraphs (module docstring)
132
+ if child.name == "p":
133
+ txt = child.get_text(" ", strip=True)
134
+ if txt:
135
+ elements.append(
136
+ HtmlFlowElement(
137
+ order_index=order_idx,
138
+ role=HtmlFlowRole.PROSE,
139
+ text=txt,
140
+ )
141
+ )
142
+ order_idx += 1
143
+
144
+ # Summary overview tables
145
+ elif child.name == "table":
146
+ headers = [
147
+ th.get_text(strip=True) for th in child.find_all("th")
148
+ ]
149
+ rows = [
150
+ [td.get_text(strip=True) for td in tr.find_all("td")]
151
+ for tr in child.find_all("tr")
152
+ if tr.find_all("td")
153
+ ]
154
+ symbols = [r[0] for r in rows if r]
155
+ has_sigs = any("(" in s for s in symbols)
156
+
157
+ table_role = (
158
+ HtmlFlowRole.AUTOSUMMARY if symbols else HtmlFlowRole.TABLE
159
+ )
160
+ elements.append(
161
+ HtmlFlowElement(
162
+ order_index=order_idx,
163
+ role=table_role,
164
+ directive="autosummary"
165
+ if table_role == HtmlFlowRole.AUTOSUMMARY
166
+ else None,
167
+ options={"nosignatures": not has_sigs}
168
+ if table_role == HtmlFlowRole.AUTOSUMMARY
169
+ else {},
170
+ symbols=symbols,
171
+ headers=headers,
172
+ )
173
+ )
174
+ order_idx += 1
175
+
176
+ # Code blocks
177
+ elif child.name == "pre" or (
178
+ child.name == "div" and "highlight" in classes
179
+ ):
180
+ code_txt = child.get_text().strip()
181
+ elements.append(
182
+ HtmlFlowElement(
183
+ order_index=order_idx,
184
+ role=HtmlFlowRole.CODE_BLOCK,
185
+ text=code_txt,
186
+ )
187
+ )
188
+ order_idx += 1
189
+
190
+ # API Objects / Children container
191
+ elif child.name == "div" and "doc-children" in classes:
192
+ child_objs = child.find_all(
193
+ "div", class_="doc-object", recursive=False
194
+ )
195
+ if child_objs:
196
+ for obj in child_objs:
197
+ if not isinstance(obj, Tag):
198
+ continue
199
+ h = obj.find(["h1", "h2", "h3", "h4", "h5", "h6"])
200
+ hid_val = h.get("id") if isinstance(h, Tag) else None
201
+ hid = str(hid_val) if hid_val is not None else None
202
+ obj_classes = _get_classes(obj)
203
+
204
+ if "doc-class" in obj_classes:
205
+ role = HtmlFlowRole.API_CLASS
206
+ directive = "autoclass"
207
+ options = {
208
+ "members": True,
209
+ "undoc-members": True,
210
+ "show-inheritance": True,
211
+ }
212
+ elif "doc-function" in obj_classes:
213
+ role = HtmlFlowRole.API_FUNCTION
214
+ directive = "autofunction"
215
+ options = {}
216
+ elif "doc-attribute" in obj_classes:
217
+ role = HtmlFlowRole.API_ATTRIBUTE
218
+ directive = "autodata"
219
+ options = {}
220
+ else:
221
+ role = HtmlFlowRole.API_MODULE
222
+ directive = "automodule"
223
+ options = {
224
+ "members": True,
225
+ "undoc-members": True,
226
+ "show-inheritance": True,
227
+ }
228
+
229
+ elements.append(
230
+ HtmlFlowElement(
231
+ order_index=order_idx,
232
+ role=role,
233
+ qname=hid or mod_qname,
234
+ directive=directive,
235
+ options=options,
236
+ )
237
+ )
238
+ order_idx += 1
239
+ else:
240
+ # Fallback for empty doc-children container on module pages
241
+ if not rel_route.endswith("index.md"):
242
+ elements.append(
243
+ HtmlFlowElement(
244
+ order_index=order_idx,
245
+ role=HtmlFlowRole.API_MODULE,
246
+ qname=mod_qname,
247
+ directive="automodule",
248
+ options={
249
+ "members": True,
250
+ "undoc-members": True,
251
+ "show-inheritance": True,
252
+ },
253
+ )
254
+ )
255
+ order_idx += 1
256
+
257
+ return HtmlPageFlow(
258
+ source_html=rel_route,
259
+ logical_route=rel_route,
260
+ title=page_title,
261
+ elements=elements,
262
+ )
263
+
264
+ # Standard documentation page (non-module or generic markdown)
265
+ for child in article.find_all(recursive=False):
266
+ if not isinstance(child, Tag):
267
+ continue
268
+
269
+ tag = child.name
270
+ classes = _get_classes(child)
271
+
272
+ if tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
273
+ lvl = int(tag[1])
274
+ txt = child.get_text(" ", strip=True).replace("🔗", "").strip()
275
+ if lvl == 1 and page_title is None:
276
+ page_title = txt
277
+ elements.append(
278
+ HtmlFlowElement(
279
+ order_index=order_idx,
280
+ role=HtmlFlowRole.HEADING,
281
+ level=lvl,
282
+ text=txt,
283
+ )
284
+ )
285
+ order_idx += 1
286
+
287
+ elif tag == "p":
288
+ txt = child.get_text(" ", strip=True)
289
+ if txt:
290
+ elements.append(
291
+ HtmlFlowElement(
292
+ order_index=order_idx, role=HtmlFlowRole.PROSE, text=txt
293
+ )
294
+ )
295
+ order_idx += 1
296
+
297
+ elif tag == "table":
298
+ headers = [th.get_text(strip=True) for th in child.find_all("th")]
299
+ elements.append(
300
+ HtmlFlowElement(
301
+ order_index=order_idx, role=HtmlFlowRole.TABLE, headers=headers
302
+ )
303
+ )
304
+ order_idx += 1
305
+
306
+ elif tag in ("ul", "ol"):
307
+ items = [
308
+ li.get_text(" ", strip=True)
309
+ for li in child.find_all("li", recursive=False)
310
+ ]
311
+ elements.append(
312
+ HtmlFlowElement(
313
+ order_index=order_idx,
314
+ role=HtmlFlowRole.LIST,
315
+ symbols=items,
316
+ options={"ordered": tag == "ol"},
317
+ )
318
+ )
319
+ order_idx += 1
320
+
321
+ elif tag == "div" and "highlight" in classes:
322
+ code_txt = child.get_text().strip()
323
+ elements.append(
324
+ HtmlFlowElement(
325
+ order_index=order_idx,
326
+ role=HtmlFlowRole.CODE_BLOCK,
327
+ text=code_txt,
328
+ )
329
+ )
330
+ order_idx += 1
331
+
332
+ elif tag in ("div", "details") and "admonition" in classes:
333
+ kind = [c for c in classes if c != "admonition"]
334
+ elements.append(
335
+ HtmlFlowElement(
336
+ order_index=order_idx,
337
+ role=HtmlFlowRole.ADMONITION,
338
+ text=child.get_text(" ", strip=True),
339
+ options={"kind": kind[0] if kind else "note"},
340
+ )
341
+ )
342
+ order_idx += 1
343
+
344
+ return HtmlPageFlow(
345
+ source_html=rel_route,
346
+ logical_route=rel_route,
347
+ title=page_title,
348
+ elements=elements,
349
+ )
@@ -0,0 +1,22 @@
1
+ """Markdown parser entrypoint wrapping the markdown-it token adapter."""
2
+
3
+ from pathlib import Path
4
+ from typing import Optional
5
+ from .markdown_ir import DocumentIR
6
+ from .markdown_it_adapter import MarkdownIRBuilder
7
+
8
+
9
+ class MarkdownParser:
10
+ """Parses markdown documents into destination-neutral DocumentIR trees via markdown-it-py."""
11
+
12
+ def parse_file(self, file_path: Path, rel_path: Optional[str] = None) -> DocumentIR:
13
+ display_path = rel_path if rel_path is not None else str(file_path)
14
+ try:
15
+ content = file_path.read_text(encoding="utf-8")
16
+ except Exception:
17
+ return DocumentIR(file_path=display_path, nodes=[])
18
+ return self.parse_text(content, display_path)
19
+
20
+ def parse_text(self, text: str, file_path: str = "") -> DocumentIR:
21
+ builder = MarkdownIRBuilder(file_path=file_path)
22
+ return builder.build_from_text(text)
@@ -0,0 +1,49 @@
1
+ """Structured Document & Block Intermediate Representation (IR)."""
2
+
3
+ from enum import Enum
4
+ from typing import List, Dict, Any, Iterator
5
+ from pydantic import BaseModel, Field
6
+
7
+
8
+ class NodeKind(str, Enum):
9
+ DOCUMENT = "DOCUMENT"
10
+ HEADING = "HEADING"
11
+ PARAGRAPH = "PARAGRAPH"
12
+ BLOCK_QUOTE = "BLOCK_QUOTE"
13
+ LIST = "LIST"
14
+ LIST_ITEM = "LIST_ITEM"
15
+ CODE_BLOCK = "CODE_BLOCK"
16
+ HTML_BLOCK = "HTML_BLOCK"
17
+ ADMONITION = "ADMONITION"
18
+ DETAILS_DROPDOWN = "DETAILS_DROPDOWN"
19
+ TAB_SET = "TAB_SET"
20
+ TAB_ITEM = "TAB_ITEM"
21
+ API_DIRECTIVE = "API_DIRECTIVE"
22
+ MERMAID_DIAGRAM = "MERMAID_DIAGRAM"
23
+ SNIPPET_INCLUDE = "SNIPPET_INCLUDE"
24
+ LINK_REF = "LINK_REF"
25
+
26
+
27
+ class BaseIRNode(BaseModel):
28
+ kind: NodeKind
29
+ start_line: int
30
+ end_line: int
31
+ raw_text: str = ""
32
+ metadata: Dict[str, Any] = Field(default_factory=dict)
33
+ children: List["BaseIRNode"] = Field(default_factory=list)
34
+
35
+ def walk(self) -> Iterator["BaseIRNode"]:
36
+ """Recursively walk this node and all of its descendants."""
37
+ yield self
38
+ for child in self.children:
39
+ yield from child.walk()
40
+
41
+
42
+ class DocumentIR(BaseModel):
43
+ file_path: str
44
+ nodes: List[BaseIRNode] = Field(default_factory=list)
45
+
46
+ def walk(self) -> Iterator[BaseIRNode]:
47
+ """Recursively walk all nodes in the document."""
48
+ for node in self.nodes:
49
+ yield from node.walk()