sphinx-mkdocs-migrate 0.0.1.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. sphinx_mkdocs_migrate/__init__.py +8 -0
  2. sphinx_mkdocs_migrate/analyzer/__init__.py +17 -0
  3. sphinx_mkdocs_migrate/analyzer/ci.py +223 -0
  4. sphinx_mkdocs_migrate/analyzer/dependencies.py +134 -0
  5. sphinx_mkdocs_migrate/analyzer/markdown.py +148 -0
  6. sphinx_mkdocs_migrate/analyzer/mkdocs.py +263 -0
  7. sphinx_mkdocs_migrate/analyzer/models.py +348 -0
  8. sphinx_mkdocs_migrate/analyzer/navigation.py +108 -0
  9. sphinx_mkdocs_migrate/analyzer/project.py +511 -0
  10. sphinx_mkdocs_migrate/cli.py +507 -0
  11. sphinx_mkdocs_migrate/parsing/doc_ir.py +533 -0
  12. sphinx_mkdocs_migrate/parsing/flow_extractor.py +457 -0
  13. sphinx_mkdocs_migrate/parsing/html_flow_parser.py +349 -0
  14. sphinx_mkdocs_migrate/parsing/markdown.py +22 -0
  15. sphinx_mkdocs_migrate/parsing/markdown_ir.py +49 -0
  16. sphinx_mkdocs_migrate/parsing/markdown_it_adapter.py +496 -0
  17. sphinx_mkdocs_migrate/parsing/requirements.py +155 -0
  18. sphinx_mkdocs_migrate/planner/accountability.py +111 -0
  19. sphinx_mkdocs_migrate/planner/ci.py +142 -0
  20. sphinx_mkdocs_migrate/planner/conf_builder.py +183 -0
  21. sphinx_mkdocs_migrate/planner/models.py +379 -0
  22. sphinx_mkdocs_migrate/planner/planner.py +1867 -0
  23. sphinx_mkdocs_migrate/planner/policy.py +474 -0
  24. sphinx_mkdocs_migrate/planner/theme_constants.py +70 -0
  25. sphinx_mkdocs_migrate/planner/toctree.py +158 -0
  26. sphinx_mkdocs_migrate/py.typed +1 -0
  27. sphinx_mkdocs_migrate/rules/catalog.py +154 -0
  28. sphinx_mkdocs_migrate/rules/engine.py +94 -0
  29. sphinx_mkdocs_migrate/rules/models.py +176 -0
  30. sphinx_mkdocs_migrate/transformer/engine.py +897 -0
  31. sphinx_mkdocs_migrate/transformer/models.py +59 -0
  32. sphinx_mkdocs_migrate/transformer/myst_transformer.py +393 -0
  33. sphinx_mkdocs_migrate/validator/models.py +40 -0
  34. sphinx_mkdocs_migrate/validator/verifier.py +377 -0
  35. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/METADATA +199 -0
  36. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/RECORD +39 -0
  37. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/WHEEL +4 -0
  38. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/entry_points.txt +2 -0
  39. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,457 @@
1
+ """Deterministic Source Document Flow Extractor.
2
+
3
+ Transforms raw Markdown text into a strongly-typed DocumentFlowSpec containing an ordered,
4
+ heterogeneous sequence of DocumentElements with exact SourceSpans, stable construct_ids,
5
+ and parsed ApiDocumentationRequests (including mkdocstrings and mkautodoc options and intent).
6
+ """
7
+
8
+ import re
9
+ from pathlib import Path
10
+ from typing import List, Dict, Any, Optional, Tuple
11
+ from .doc_ir import (
12
+ DocumentFlowSpec,
13
+ DocumentElement,
14
+ DocumentElementType,
15
+ SourceSpan,
16
+ HeadingElement,
17
+ ParagraphElement,
18
+ ListElement,
19
+ ListItemElement,
20
+ CodeBlockElement,
21
+ AdmonitionElement,
22
+ SnippetElement,
23
+ ApiDocumentationRequest,
24
+ ApiObjectKind,
25
+ MemberSelection,
26
+ SummaryMode,
27
+ ResolutionStatus,
28
+ RawHtmlElement,
29
+ UnknownElement,
30
+ DocumentationPage,
31
+ DocumentationLink,
32
+ AssetReference,
33
+ DocumentElementContent,
34
+ )
35
+ from .markdown_ir import BaseIRNode, NodeKind
36
+ from .markdown import MarkdownParser
37
+
38
+ RE_MKAUTODOC_FLAG = re.compile(
39
+ r"^[ ]{4,}:(?P<flag>[a-zA-Z0-9_]+):(?:[ ]+(?P<val>.*))?$"
40
+ )
41
+ RE_KEY_VAL_OPT = re.compile(r"^[ ]{4,}(?P<key>[a-zA-Z0-9_]+):[ ]*(?P<val>.*)$")
42
+
43
+
44
+ class DocumentFlowExtractor:
45
+ """Extracts strongly-typed DocumentFlowSpec and DocumentationPage from Markdown text/files."""
46
+
47
+ def __init__(
48
+ self, file_path: str = "", mkdocs_config: Optional[Dict[str, Any]] = None
49
+ ):
50
+ self.file_path = file_path
51
+ self.mkdocs_config = mkdocs_config or {}
52
+ self.parser = MarkdownParser()
53
+
54
+ def extract_from_file(
55
+ self, file_path: Path, rel_path: Optional[str] = None
56
+ ) -> DocumentationPage:
57
+ display_path = rel_path if rel_path is not None else str(file_path)
58
+ content = file_path.read_text(encoding="utf-8")
59
+ return self.extract_from_text(content, display_path)
60
+
61
+ def extract_from_text(self, text: str, file_path: str = "") -> DocumentationPage:
62
+ display_path = file_path or self.file_path
63
+ doc_ir = self.parser.parse_text(text, file_path=display_path)
64
+
65
+ flow_elements: List[DocumentElement] = []
66
+ outgoing_links: List[DocumentationLink] = []
67
+ referenced_assets: List[AssetReference] = []
68
+ page_title: Optional[str] = None
69
+
70
+ raw_lines = text.splitlines()
71
+ order_index = 0
72
+
73
+ for node in doc_ir.nodes:
74
+ elem, links, assets = self._convert_ir_node_to_flow_element(
75
+ node, order_index, display_path, raw_lines
76
+ )
77
+ if elem:
78
+ flow_elements.append(elem)
79
+ order_index += 1
80
+ if elem.element_type == DocumentElementType.HEADING:
81
+ if (
82
+ page_title is None
83
+ and isinstance(elem.content, HeadingElement)
84
+ and elem.content.level == 1
85
+ ):
86
+ page_title = elem.content.text
87
+ outgoing_links.extend(links)
88
+ referenced_assets.extend(assets)
89
+
90
+ flow_spec = DocumentFlowSpec(source_file=display_path, elements=flow_elements)
91
+
92
+ logical_route = display_path
93
+ if logical_route.endswith(".md"):
94
+ logical_route = logical_route[:-3]
95
+ if logical_route.endswith("/index"):
96
+ logical_route = logical_route[: -len("/index")]
97
+ elif logical_route == "index":
98
+ logical_route = ""
99
+
100
+ return DocumentationPage(
101
+ source_file=display_path,
102
+ logical_route=logical_route,
103
+ title=page_title,
104
+ flow=flow_spec,
105
+ outgoing_links=outgoing_links,
106
+ referenced_assets=referenced_assets,
107
+ )
108
+
109
+ def _convert_ir_node_to_flow_element(
110
+ self, node: BaseIRNode, order_index: int, file_path: str, raw_lines: List[str]
111
+ ) -> Tuple[
112
+ Optional[DocumentElement], List[DocumentationLink], List[AssetReference]
113
+ ]:
114
+ construct_id = f"doc:{file_path}:elem:{order_index:04d}"
115
+ source_span = SourceSpan(
116
+ file=file_path, start_line=node.start_line, end_line=node.end_line
117
+ )
118
+
119
+ links: List[DocumentationLink] = []
120
+ assets: List[AssetReference] = []
121
+
122
+ if node.kind == NodeKind.HEADING:
123
+ level = node.metadata.get("level", 1)
124
+ text = node.metadata.get("text", node.raw_text).strip()
125
+ anchor_id = node.metadata.get("id")
126
+ content: DocumentElementContent = HeadingElement(
127
+ level=level, text=text, anchor_id=anchor_id
128
+ )
129
+ return (
130
+ DocumentElement(
131
+ construct_id=construct_id,
132
+ element_type=DocumentElementType.HEADING,
133
+ source_order_index=order_index,
134
+ source_span=source_span,
135
+ content=content,
136
+ ),
137
+ links,
138
+ assets,
139
+ )
140
+
141
+ elif node.kind == NodeKind.PARAGRAPH:
142
+ text = node.raw_text.strip()
143
+ re_link = re.compile(r"\[(?P<text>[^\]]+)\]\((?P<target>[^\)]+)\)")
144
+ re_img = re.compile(r"!\[(?P<alt>[^\]]*)\]\((?P<src>[^\)]+)\)")
145
+
146
+ for m in re_img.finditer(text):
147
+ src = m.group("src")
148
+ assets.append(
149
+ AssetReference(
150
+ source_construct_id=construct_id,
151
+ source_path=src,
152
+ asset_kind="image",
153
+ )
154
+ )
155
+
156
+ for m in re_link.finditer(text):
157
+ target = m.group("target")
158
+ is_internal = not (
159
+ target.startswith("http://")
160
+ or target.startswith("https://")
161
+ or target.startswith("//")
162
+ )
163
+ link_k = (
164
+ "internal_page"
165
+ if is_internal and target.endswith(".md")
166
+ else (
167
+ "internal_anchor"
168
+ if target.startswith("#")
169
+ else ("external_url" if not is_internal else "internal_page")
170
+ )
171
+ )
172
+ links.append(
173
+ DocumentationLink(
174
+ source_construct_id=construct_id,
175
+ target=target,
176
+ link_kind=link_k,
177
+ resolved=False,
178
+ )
179
+ )
180
+
181
+ content = ParagraphElement(text=text)
182
+ return (
183
+ DocumentElement(
184
+ construct_id=construct_id,
185
+ element_type=DocumentElementType.PARAGRAPH,
186
+ source_order_index=order_index,
187
+ source_span=source_span,
188
+ content=content,
189
+ ),
190
+ links,
191
+ assets,
192
+ )
193
+
194
+ elif node.kind == NodeKind.LIST:
195
+ is_ordered = node.metadata.get("token_tag") == "ol"
196
+ list_items = self._convert_list_items(
197
+ node.children, construct_id, links, assets
198
+ )
199
+ content = ListElement(ordered=is_ordered, items=list_items)
200
+ return (
201
+ DocumentElement(
202
+ construct_id=construct_id,
203
+ element_type=DocumentElementType.LIST,
204
+ source_order_index=order_index,
205
+ source_span=source_span,
206
+ content=content,
207
+ ),
208
+ links,
209
+ assets,
210
+ )
211
+
212
+ elif node.kind == NodeKind.CODE_BLOCK:
213
+ lang = node.metadata.get("language")
214
+ title = node.metadata.get("title")
215
+ content = CodeBlockElement(language=lang, code=node.raw_text, title=title)
216
+ return (
217
+ DocumentElement(
218
+ construct_id=construct_id,
219
+ element_type=DocumentElementType.CODE_BLOCK,
220
+ source_order_index=order_index,
221
+ source_span=source_span,
222
+ content=content,
223
+ ),
224
+ links,
225
+ assets,
226
+ )
227
+
228
+ elif node.kind == NodeKind.ADMONITION:
229
+ kind = node.metadata.get("admonition_type", "note")
230
+ title = node.metadata.get("title")
231
+ body_text = "\n".join(c.raw_text for c in node.children)
232
+ content = AdmonitionElement(kind=kind, title=title, content_text=body_text)
233
+ return (
234
+ DocumentElement(
235
+ construct_id=construct_id,
236
+ element_type=DocumentElementType.ADMONITION,
237
+ source_order_index=order_index,
238
+ source_span=source_span,
239
+ content=content,
240
+ ),
241
+ links,
242
+ assets,
243
+ )
244
+
245
+ elif node.kind == NodeKind.API_DIRECTIVE:
246
+ symbol = node.metadata.get("symbol", "")
247
+ node_text = node.raw_text
248
+ raw_opts: Dict[str, Any] = {}
249
+
250
+ # Parse options from node.raw_text (YAML for mkdocstrings or regex flags for mkautodoc)
251
+ has_mkautodoc_flags = any(
252
+ RE_MKAUTODOC_FLAG.match(line) for line in node_text.splitlines()[1:]
253
+ )
254
+ if not has_mkautodoc_flags:
255
+ remaining_lines = "\n".join(node_text.splitlines()[1:])
256
+ if remaining_lines.strip():
257
+ try:
258
+ import yaml
259
+
260
+ parsed_yaml = yaml.safe_load(remaining_lines)
261
+ if isinstance(parsed_yaml, dict):
262
+ for k, v in parsed_yaml.items():
263
+ raw_opts[k] = v
264
+ if "options" in parsed_yaml and isinstance(
265
+ parsed_yaml["options"], dict
266
+ ):
267
+ for k, v in parsed_yaml["options"].items():
268
+ raw_opts[k] = v
269
+ except Exception:
270
+ pass
271
+
272
+ for line in node_text.splitlines()[1:]:
273
+ m_flag = RE_MKAUTODOC_FLAG.match(line)
274
+ if m_flag:
275
+ f_name = m_flag.group("flag")
276
+ f_val = m_flag.group("val")
277
+ raw_opts[f_name] = f_val.strip() if f_val else True
278
+ continue
279
+ m_opt = RE_KEY_VAL_OPT.match(line)
280
+ if m_opt:
281
+ raw_opts.setdefault(m_opt.group("key"), m_opt.group("val").strip())
282
+
283
+ # Determine handler provenance
284
+ md_exts = self.mkdocs_config.get("markdown_extensions", [])
285
+ has_mkautodoc = any(
286
+ (ext == "mkautodoc" or (isinstance(ext, dict) and "mkautodoc" in ext))
287
+ for ext in md_exts
288
+ )
289
+ handler = "mkautodoc" if has_mkautodoc else "mkdocstrings.python"
290
+
291
+ # Normalize options semantically
292
+ normalized_opts = raw_opts.copy()
293
+ include_docstring = True
294
+ member_selection = MemberSelection.NOT_SPECIFIED
295
+ explicit_members: List[str] = []
296
+
297
+ # Handle :docstring:
298
+ if "docstring" in raw_opts:
299
+ include_docstring = True
300
+
301
+ # Handle :members:
302
+ if "members" in raw_opts:
303
+ m_val = raw_opts["members"]
304
+ if m_val is True or m_val == "":
305
+ member_selection = MemberSelection.ALL_PUBLIC
306
+ else:
307
+ member_selection = MemberSelection.EXPLICIT
308
+ explicit_members = [item for item in m_val.split() if item]
309
+
310
+ # Infer summary mode
311
+ summary_mode = SummaryMode.NOT_REQUESTED
312
+ if "summary" in raw_opts:
313
+ summary_mode = SummaryMode.EXPLICIT
314
+
315
+ # Infer object kind based on symbol naming or leave UNKNOWN
316
+ obj_kind = ApiObjectKind.UNKNOWN
317
+ if symbol:
318
+ last_part = symbol.split(".")[-1]
319
+ if "Error" in last_part or "Exception" in last_part:
320
+ obj_kind = ApiObjectKind.EXCEPTION
321
+ elif last_part[0].isupper():
322
+ obj_kind = ApiObjectKind.CLASS
323
+ elif (
324
+ len(symbol.split(".")) >= 2
325
+ and last_part.islower()
326
+ and "_" not in last_part
327
+ ):
328
+ # Module paths like pythonjsonlogger.core, pythonjsonlogger.defaults
329
+ obj_kind = ApiObjectKind.MODULE
330
+ elif "_" in last_part or last_part.islower():
331
+ obj_kind = ApiObjectKind.FUNCTION
332
+
333
+ api_req = ApiDocumentationRequest(
334
+ construct_id=construct_id,
335
+ source_span=source_span,
336
+ object_path=symbol,
337
+ object_kind=obj_kind,
338
+ handler=handler,
339
+ raw_options=raw_opts,
340
+ normalized_options=normalized_opts,
341
+ include_docstring=include_docstring,
342
+ member_selection=member_selection,
343
+ explicit_members=explicit_members,
344
+ summary_mode=summary_mode,
345
+ resolution_status=ResolutionStatus.UNRESOLVED,
346
+ )
347
+ return (
348
+ DocumentElement(
349
+ construct_id=construct_id,
350
+ element_type=DocumentElementType.API_REQUEST,
351
+ source_order_index=order_index,
352
+ source_span=source_span,
353
+ content=api_req,
354
+ ),
355
+ links,
356
+ assets,
357
+ )
358
+
359
+ elif node.kind == NodeKind.SNIPPET_INCLUDE:
360
+ filepath = node.metadata.get("filepath", "")
361
+ content = SnippetElement(snippet_path=filepath)
362
+ return (
363
+ DocumentElement(
364
+ construct_id=construct_id,
365
+ element_type=DocumentElementType.SNIPPET,
366
+ source_order_index=order_index,
367
+ source_span=source_span,
368
+ content=content,
369
+ ),
370
+ links,
371
+ assets,
372
+ )
373
+
374
+ elif node.kind == NodeKind.HTML_BLOCK:
375
+ content = RawHtmlElement(raw_html=node.raw_text)
376
+ return (
377
+ DocumentElement(
378
+ construct_id=construct_id,
379
+ element_type=DocumentElementType.RAW_HTML,
380
+ source_order_index=order_index,
381
+ source_span=source_span,
382
+ content=content,
383
+ ),
384
+ links,
385
+ assets,
386
+ )
387
+
388
+ else:
389
+ content = UnknownElement(
390
+ tag_or_type=str(node.kind),
391
+ raw_content=node.raw_text,
392
+ rationale=f"Unhandled NodeKind {node.kind}",
393
+ )
394
+ return (
395
+ DocumentElement(
396
+ construct_id=construct_id,
397
+ element_type=DocumentElementType.UNKNOWN,
398
+ source_order_index=order_index,
399
+ source_span=source_span,
400
+ content=content,
401
+ ),
402
+ links,
403
+ assets,
404
+ )
405
+
406
+ def _convert_list_items(
407
+ self,
408
+ nodes: List[BaseIRNode],
409
+ parent_construct_id: str,
410
+ links: List[DocumentationLink],
411
+ assets: List[AssetReference],
412
+ ) -> List[ListItemElement]:
413
+ items: List[ListItemElement] = []
414
+ for node in nodes:
415
+ if node.kind == NodeKind.LIST_ITEM:
416
+ text = node.raw_text.strip()
417
+ child_sub_items: List[ListItemElement] = []
418
+ for ch in node.children:
419
+ if ch.kind == NodeKind.PARAGRAPH:
420
+ p_text = ch.raw_text.strip()
421
+ if not text:
422
+ text = p_text
423
+ elif ch.kind == NodeKind.LIST:
424
+ child_sub_items.extend(
425
+ self._convert_list_items(
426
+ ch.children, parent_construct_id, links, assets
427
+ )
428
+ )
429
+ elif ch.kind == NodeKind.LINK_REF:
430
+ target = ch.metadata.get("target_path", "")
431
+ is_ext = ch.metadata.get("is_external", False)
432
+ link_k = (
433
+ "external_url"
434
+ if is_ext
435
+ else (
436
+ "internal_anchor"
437
+ if target.startswith("#")
438
+ else "internal_page"
439
+ )
440
+ )
441
+ links.append(
442
+ DocumentationLink(
443
+ source_construct_id=parent_construct_id,
444
+ target=target,
445
+ link_kind=link_k,
446
+ resolved=False,
447
+ )
448
+ )
449
+
450
+ items.append(ListItemElement(text=text, children=child_sub_items))
451
+ elif node.kind == NodeKind.LIST:
452
+ items.extend(
453
+ self._convert_list_items(
454
+ node.children, parent_construct_id, links, assets
455
+ )
456
+ )
457
+ return items