sphinx-mkdocs-migrate 0.0.1.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. sphinx_mkdocs_migrate/__init__.py +8 -0
  2. sphinx_mkdocs_migrate/analyzer/__init__.py +17 -0
  3. sphinx_mkdocs_migrate/analyzer/ci.py +223 -0
  4. sphinx_mkdocs_migrate/analyzer/dependencies.py +134 -0
  5. sphinx_mkdocs_migrate/analyzer/markdown.py +148 -0
  6. sphinx_mkdocs_migrate/analyzer/mkdocs.py +263 -0
  7. sphinx_mkdocs_migrate/analyzer/models.py +348 -0
  8. sphinx_mkdocs_migrate/analyzer/navigation.py +108 -0
  9. sphinx_mkdocs_migrate/analyzer/project.py +511 -0
  10. sphinx_mkdocs_migrate/cli.py +507 -0
  11. sphinx_mkdocs_migrate/parsing/doc_ir.py +533 -0
  12. sphinx_mkdocs_migrate/parsing/flow_extractor.py +457 -0
  13. sphinx_mkdocs_migrate/parsing/html_flow_parser.py +349 -0
  14. sphinx_mkdocs_migrate/parsing/markdown.py +22 -0
  15. sphinx_mkdocs_migrate/parsing/markdown_ir.py +49 -0
  16. sphinx_mkdocs_migrate/parsing/markdown_it_adapter.py +496 -0
  17. sphinx_mkdocs_migrate/parsing/requirements.py +155 -0
  18. sphinx_mkdocs_migrate/planner/accountability.py +111 -0
  19. sphinx_mkdocs_migrate/planner/ci.py +142 -0
  20. sphinx_mkdocs_migrate/planner/conf_builder.py +183 -0
  21. sphinx_mkdocs_migrate/planner/models.py +379 -0
  22. sphinx_mkdocs_migrate/planner/planner.py +1867 -0
  23. sphinx_mkdocs_migrate/planner/policy.py +474 -0
  24. sphinx_mkdocs_migrate/planner/theme_constants.py +70 -0
  25. sphinx_mkdocs_migrate/planner/toctree.py +158 -0
  26. sphinx_mkdocs_migrate/py.typed +1 -0
  27. sphinx_mkdocs_migrate/rules/catalog.py +154 -0
  28. sphinx_mkdocs_migrate/rules/engine.py +94 -0
  29. sphinx_mkdocs_migrate/rules/models.py +176 -0
  30. sphinx_mkdocs_migrate/transformer/engine.py +897 -0
  31. sphinx_mkdocs_migrate/transformer/models.py +59 -0
  32. sphinx_mkdocs_migrate/transformer/myst_transformer.py +393 -0
  33. sphinx_mkdocs_migrate/validator/models.py +40 -0
  34. sphinx_mkdocs_migrate/validator/verifier.py +377 -0
  35. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/METADATA +199 -0
  36. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/RECORD +39 -0
  37. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/WHEEL +4 -0
  38. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/entry_points.txt +2 -0
  39. sphinx_mkdocs_migrate-0.0.1.dev0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,897 @@
1
+ """Transformation Engine executing MigrationPlan actions deterministically to produce MyST docs and conf.py."""
2
+
3
+ import difflib
4
+ import hashlib
5
+ import os
6
+ import re
7
+ from pathlib import Path
8
+ from typing import List, Dict, Optional, Tuple
9
+ from ..planner.models import MigrationPlan
10
+ from ..planner.conf_builder import build_conf_py
11
+ from ..planner.toctree import build_semantic_toctree
12
+ from ..planner.ci import (
13
+ build_tox_docs_env,
14
+ build_github_docs_job,
15
+ determine_tox_dependency_line,
16
+ )
17
+ from ..analyzer.models import NavigationItem
18
+ from ..analyzer.dependencies import KNOWN_MKDOCS_PACKAGES
19
+ from ..analyzer.mkdocs import detect_obsolete_generator_scripts
20
+ from ..analyzer.ci import (
21
+ resolve_github_action_ref,
22
+ DEFAULT_CHECKOUT_TAG,
23
+ DEFAULT_CHECKOUT_SHA,
24
+ DEFAULT_SETUP_UV_TAG,
25
+ DEFAULT_SETUP_UV_SHA,
26
+ )
27
+ from ..parsing.markdown import MarkdownParser
28
+ from .models import (
29
+ ProjectTransformationReport,
30
+ DocumentTransformationResult,
31
+ TransformationStatus,
32
+ ConfPyStatus,
33
+ )
34
+ from .myst_transformer import MySTDocumentTransformer
35
+
36
+
37
+ class TransformationEngine:
38
+ """Consumes a MigrationPlan to transform documentation and generate Sphinx artifacts via source-preserving patching."""
39
+
40
+ def __init__(self, plan: MigrationPlan):
41
+ self.plan = plan
42
+ self.parser = MarkdownParser()
43
+ self.project_root = Path(plan.project_root)
44
+
45
+ def execute(
46
+ self, write_to_disk: bool = False, overwrite_conf: bool = False
47
+ ) -> ProjectTransformationReport:
48
+ """Executes document transformations and Sphinx scaffolding derived strictly from the plan."""
49
+ actions_by_file: Dict[str, List] = {}
50
+ for act in self.plan.document_actions:
51
+ actions_by_file.setdefault(act.source_file, []).append(act)
52
+
53
+ # 1. Discover documentation files
54
+ docs_dir_name = (
55
+ self.plan.source_mkdocs_config.docs_dir
56
+ if self.plan.source_mkdocs_config
57
+ else "docs"
58
+ )
59
+ docs_dir = self.project_root / docs_dir_name
60
+ all_md_files: List[Path] = (
61
+ sorted(list(docs_dir.rglob("*.md"))) if docs_dir.exists() else []
62
+ )
63
+ documents_examined = len(all_md_files)
64
+
65
+ nav_entries = (
66
+ self.plan.navigation_analysis.tree
67
+ if (self.plan.navigation_analysis and self.plan.navigation_analysis.has_nav)
68
+ else []
69
+ )
70
+ autorefs_targets = self._autorefs_target_documents()
71
+
72
+ # 2. Transform existing markdown files
73
+ (
74
+ doc_results,
75
+ documents_changed,
76
+ files_written,
77
+ ) = self._transform_all_documents(
78
+ all_md_files=all_md_files,
79
+ docs_dir=docs_dir,
80
+ actions_by_file=actions_by_file,
81
+ nav_entries=nav_entries,
82
+ autorefs_targets=autorefs_targets,
83
+ write_to_disk=write_to_disk,
84
+ )
85
+
86
+ generated_files: Dict[str, str] = {}
87
+
88
+ # 3. Ensure root index.md exists
89
+ (
90
+ root_idx_files,
91
+ root_idx_results,
92
+ idx_written,
93
+ idx_changed,
94
+ ) = self._ensure_root_index(
95
+ docs_dir=docs_dir,
96
+ docs_dir_name=docs_dir_name,
97
+ nav_entries=nav_entries,
98
+ all_md_files=all_md_files,
99
+ doc_results=doc_results,
100
+ write_to_disk=write_to_disk,
101
+ )
102
+ generated_files.update(root_idx_files)
103
+ doc_results.extend(root_idx_results)
104
+ files_written += idx_written
105
+ documents_changed += idx_changed
106
+
107
+ # 4. Materialize Planned Generated Documents (e.g. API reference stubs from gen-files/mkdocstrings)
108
+ (
109
+ gen_files,
110
+ gen_results,
111
+ gen_written,
112
+ gen_changed,
113
+ ) = self._materialize_generated_documents(write_to_disk=write_to_disk)
114
+ generated_files.update(gen_files)
115
+ doc_results.extend(gen_results)
116
+ files_written += gen_written
117
+ documents_changed += gen_changed
118
+
119
+ # 5. Generate Sphinx conf.py scaffolding
120
+ (
121
+ conf_content,
122
+ conf_target_key,
123
+ conf_status,
124
+ conf_diff,
125
+ conf_written,
126
+ ) = self._manage_conf_py(
127
+ docs_dir_name=docs_dir_name,
128
+ overwrite_conf=overwrite_conf,
129
+ write_to_disk=write_to_disk,
130
+ )
131
+ generated_files[conf_target_key] = conf_content
132
+ files_written += conf_written
133
+
134
+ # 6. Update dependencies, CI workflows, and clean up obsolete scripts
135
+ dep_changes = self._migrate_dependencies(write_to_disk=write_to_disk)
136
+ if write_to_disk and dep_changes:
137
+ files_written += 1
138
+
139
+ ci_changes = self._migrate_ci_workflows(write_to_disk=write_to_disk)
140
+ if write_to_disk and ci_changes:
141
+ files_written += len(ci_changes)
142
+
143
+ cleaned_files = self._clean_obsolete_mkdocs_files(write_to_disk=write_to_disk)
144
+
145
+ return ProjectTransformationReport(
146
+ project_root=str(self.project_root),
147
+ plan_hash=self.plan.canonical_hash(),
148
+ transformed_documents=doc_results,
149
+ generated_sphinx_files=generated_files,
150
+ conf_py_status=conf_status,
151
+ conf_py_conflict_diff=conf_diff,
152
+ documents_examined=documents_examined,
153
+ documents_changed=documents_changed,
154
+ files_written_to_disk=files_written,
155
+ total_transforms_executed=sum(d.transforms_applied for d in doc_results),
156
+ total_stale_actions=sum(d.stale_actions_count for d in doc_results),
157
+ cleaned_files=cleaned_files,
158
+ dry_run=not write_to_disk,
159
+ )
160
+
161
+ def _transform_markdown_file(
162
+ self,
163
+ md_file: Path,
164
+ actions: List,
165
+ docs_dir: Path,
166
+ nav_entries: List[NavigationItem],
167
+ all_md_files: List[Path],
168
+ autorefs_targets: Dict[str, Path],
169
+ ) -> Tuple[DocumentTransformationResult, bool]:
170
+ """Transforms a single Markdown file applying MyST actions, autorefs, and frontmatter sanitization."""
171
+ source_file_rel = md_file.relative_to(self.project_root).as_posix()
172
+ orig_content = md_file.read_text(encoding="utf-8")
173
+ source_fp = hashlib.sha256(orig_content.encode("utf-8")).hexdigest()[:16]
174
+
175
+ doc_ir = self.parser.parse_text(orig_content, file_path=source_file_rel)
176
+ transformer = MySTDocumentTransformer(actions)
177
+ (
178
+ transformed_content,
179
+ applied_cnt,
180
+ preserved_cnt,
181
+ manual_cnt,
182
+ unsupported_cnt,
183
+ stale_cnt,
184
+ stale_details,
185
+ ) = transformer.transform_document(doc_ir, orig_content)
186
+ # MkDocs autorefs accepts compact reference links such as
187
+ # ``[orjson][pythonjsonlogger.orjson]``. They have no CommonMark
188
+ # reference definition, so MyST would render them literally. Only
189
+ # rewrite targets we can resolve to a generated Sphinx document.
190
+ if autorefs_targets:
191
+ transformed_content, autorefs_count = self._transform_autorefs_links(
192
+ transformed_content, md_file, docs_dir, autorefs_targets
193
+ )
194
+ applied_cnt += autorefs_count
195
+
196
+ # Universal YAML frontmatter date object sanitizer
197
+ if transformed_content.startswith("---"):
198
+ parts = transformed_content.split("---", 2)
199
+ if len(parts) >= 3:
200
+ sanitized_header = re.sub(
201
+ r"(:\s*)(\d{4}-\d{2}-\d{2})\b", r'\1"\2"', parts[1]
202
+ )
203
+ transformed_content = "---" + sanitized_header + "---" + parts[2]
204
+ # Only the documentation root owns the site-level toctree. Nested
205
+ # ``index.md`` files are section/package landing pages and must keep
206
+ # their own child toctrees (for example generated API modules).
207
+ if md_file == docs_dir / "index.md":
208
+ toctree_block = self._generate_semantic_toctree(
209
+ nav_entries, all_md_files, docs_dir
210
+ )
211
+ if toctree_block:
212
+ if "```{toctree}" in transformed_content:
213
+ transformed_content = re.sub(
214
+ r"```\{toctree\}[\s\S]*?```",
215
+ toctree_block,
216
+ transformed_content,
217
+ )
218
+ else:
219
+ transformed_content = (
220
+ transformed_content.rstrip() + "\n\n" + toctree_block + "\n"
221
+ )
222
+
223
+ is_modified = orig_content != transformed_content
224
+ if manual_cnt > 0 and applied_cnt == 0:
225
+ doc_status = TransformationStatus.MANUAL_REQUIRED
226
+ elif is_modified:
227
+ doc_status = TransformationStatus.APPLIED
228
+ else:
229
+ doc_status = TransformationStatus.UNCHANGED
230
+
231
+ diff_str = None
232
+ if is_modified:
233
+ diff_lines = list(
234
+ difflib.unified_diff(
235
+ orig_content.splitlines(keepends=True),
236
+ transformed_content.splitlines(keepends=True),
237
+ fromfile=f"a/{source_file_rel}",
238
+ tofile=f"b/{source_file_rel}",
239
+ )
240
+ )
241
+ diff_str = "".join(diff_lines)
242
+
243
+ result = DocumentTransformationResult(
244
+ source_file=source_file_rel,
245
+ target_file=source_file_rel,
246
+ original_content=orig_content,
247
+ transformed_content=transformed_content,
248
+ source_fingerprint=source_fp,
249
+ transforms_applied=applied_cnt,
250
+ constructs_preserved=preserved_cnt,
251
+ manual_items_reported=manual_cnt,
252
+ unsupported_items_reported=unsupported_cnt,
253
+ stale_actions_count=stale_cnt,
254
+ stale_action_details=stale_details,
255
+ status=doc_status,
256
+ diff=diff_str,
257
+ is_modified=is_modified,
258
+ )
259
+ return result, is_modified
260
+
261
+ def _transform_all_documents(
262
+ self,
263
+ all_md_files: List[Path],
264
+ docs_dir: Path,
265
+ actions_by_file: Dict[str, List],
266
+ nav_entries: List[NavigationItem],
267
+ autorefs_targets: Dict[str, Path],
268
+ write_to_disk: bool,
269
+ ) -> Tuple[List[DocumentTransformationResult], int, int]:
270
+ """Processes all discovered Markdown documents."""
271
+ doc_results: List[DocumentTransformationResult] = []
272
+ documents_changed = 0
273
+ files_written = 0
274
+
275
+ for md_file in all_md_files:
276
+ source_file_rel = md_file.relative_to(self.project_root).as_posix()
277
+ actions = actions_by_file.get(source_file_rel, [])
278
+ result, is_mod = self._transform_markdown_file(
279
+ md_file=md_file,
280
+ actions=actions,
281
+ docs_dir=docs_dir,
282
+ nav_entries=nav_entries,
283
+ all_md_files=all_md_files,
284
+ autorefs_targets=autorefs_targets,
285
+ )
286
+ doc_results.append(result)
287
+ if is_mod:
288
+ documents_changed += 1
289
+ if write_to_disk:
290
+ md_file.write_text(result.transformed_content, encoding="utf-8")
291
+ files_written += 1
292
+
293
+ return doc_results, documents_changed, files_written
294
+
295
+ def _ensure_root_index(
296
+ self,
297
+ docs_dir: Path,
298
+ docs_dir_name: str,
299
+ nav_entries: List[NavigationItem],
300
+ all_md_files: List[Path],
301
+ doc_results: List[DocumentTransformationResult],
302
+ write_to_disk: bool,
303
+ ) -> Tuple[Dict[str, str], List[DocumentTransformationResult], int, int]:
304
+ """Ensures a root index.md exists with site-level semantic toctree if needed."""
305
+ generated: Dict[str, str] = {}
306
+ added_results: List[DocumentTransformationResult] = []
307
+ files_written = 0
308
+ documents_changed = 0
309
+
310
+ root_index_path = docs_dir / "index.md"
311
+ root_index_rel = f"{docs_dir_name}/index.md" if docs_dir_name else "index.md"
312
+ has_root_index = any(
313
+ Path(doc.target_file).as_posix()
314
+ in (
315
+ Path(root_index_rel).as_posix(),
316
+ "index.md",
317
+ f"{docs_dir_name}/index.md",
318
+ )
319
+ for doc in doc_results
320
+ )
321
+
322
+ if not has_root_index:
323
+ toctree_block = self._generate_semantic_toctree(
324
+ nav_entries, all_md_files, docs_dir
325
+ )
326
+ index_title = (
327
+ self.plan.proposed_sphinx_config.project_name
328
+ if self.plan.proposed_sphinx_config
329
+ else "Documentation"
330
+ )
331
+ generated_index_content = f"# {index_title}\n\n"
332
+ if toctree_block:
333
+ generated_index_content += toctree_block + "\n"
334
+
335
+ generated[root_index_rel] = generated_index_content
336
+ if write_to_disk:
337
+ root_index_path.parent.mkdir(parents=True, exist_ok=True)
338
+ root_index_path.write_text(generated_index_content, encoding="utf-8")
339
+ files_written += 1
340
+
341
+ added_results.append(
342
+ DocumentTransformationResult(
343
+ source_file=root_index_rel,
344
+ target_file=root_index_rel,
345
+ original_content="",
346
+ transformed_content=generated_index_content,
347
+ source_fingerprint="generated_root_index",
348
+ transforms_applied=1,
349
+ status=TransformationStatus.APPLIED,
350
+ is_modified=True,
351
+ )
352
+ )
353
+ documents_changed += 1
354
+
355
+ return generated, added_results, files_written, documents_changed
356
+
357
+ def _materialize_generated_documents(
358
+ self, write_to_disk: bool
359
+ ) -> Tuple[Dict[str, str], List[DocumentTransformationResult], int, int]:
360
+ """Materializes planned generated documents (e.g. API reference stubs from gen-files/mkdocstrings)."""
361
+ generated: Dict[str, str] = {}
362
+ doc_results: List[DocumentTransformationResult] = []
363
+ files_written = 0
364
+ documents_changed = 0
365
+
366
+ for gen_doc in self.plan.generated_documents:
367
+ target_disk_path = self.project_root / gen_doc.target_path
368
+ generated[gen_doc.target_path] = gen_doc.content
369
+
370
+ orig_doc_content = (
371
+ target_disk_path.read_text(encoding="utf-8")
372
+ if target_disk_path.exists()
373
+ else ""
374
+ )
375
+ is_mod = orig_doc_content != gen_doc.content
376
+
377
+ if write_to_disk:
378
+ target_disk_path.parent.mkdir(parents=True, exist_ok=True)
379
+ target_disk_path.write_text(gen_doc.content, encoding="utf-8")
380
+ files_written += 1
381
+
382
+ if is_mod:
383
+ documents_changed += 1
384
+
385
+ doc_results.append(
386
+ DocumentTransformationResult(
387
+ source_file=gen_doc.target_path,
388
+ target_file=gen_doc.target_path,
389
+ original_content=orig_doc_content,
390
+ transformed_content=gen_doc.content,
391
+ source_fingerprint=f"generated_{gen_doc.generator_plugin}",
392
+ transforms_applied=1,
393
+ status=TransformationStatus.APPLIED
394
+ if is_mod
395
+ else TransformationStatus.UNCHANGED,
396
+ is_modified=is_mod,
397
+ )
398
+ )
399
+
400
+ return generated, doc_results, files_written, documents_changed
401
+
402
+ def _manage_conf_py(
403
+ self, docs_dir_name: str, overwrite_conf: bool, write_to_disk: bool
404
+ ) -> Tuple[str, str, ConfPyStatus, Optional[str], int]:
405
+ """Generates and writes Sphinx conf.py scaffolding with conflict detection."""
406
+ conf_py_content = self._generate_conf_py()
407
+ conf_target_key = f"{docs_dir_name}/conf.py"
408
+ conf_disk_path = self.project_root / docs_dir_name / "conf.py"
409
+ conf_status = ConfPyStatus.CREATED
410
+ conf_diff = None
411
+ files_written = 0
412
+
413
+ if conf_disk_path.exists():
414
+ existing_conf = conf_disk_path.read_text(encoding="utf-8")
415
+ if existing_conf == conf_py_content:
416
+ conf_status = ConfPyStatus.UNCHANGED
417
+ else:
418
+ conf_status = ConfPyStatus.CONFLICT
419
+ diff_lines = list(
420
+ difflib.unified_diff(
421
+ existing_conf.splitlines(keepends=True),
422
+ conf_py_content.splitlines(keepends=True),
423
+ fromfile=f"a/{docs_dir_name}/conf.py (existing)",
424
+ tofile=f"b/{docs_dir_name}/conf.py (planned)",
425
+ )
426
+ )
427
+ conf_diff = "".join(diff_lines)
428
+
429
+ if write_to_disk:
430
+ if (
431
+ not conf_disk_path.exists()
432
+ or conf_status == ConfPyStatus.UNCHANGED
433
+ or overwrite_conf
434
+ ):
435
+ conf_disk_path.parent.mkdir(parents=True, exist_ok=True)
436
+ conf_disk_path.write_text(conf_py_content, encoding="utf-8")
437
+ files_written = 1
438
+
439
+ return conf_py_content, conf_target_key, conf_status, conf_diff, files_written
440
+
441
+ def _autorefs_target_documents(self) -> Dict[str, Path]:
442
+ """Map known Python module identities to generated documentation paths."""
443
+ if (
444
+ not self.plan.source_mkdocs_config
445
+ or "autorefs" not in self.plan.source_mkdocs_config.plugins
446
+ ):
447
+ return {}
448
+
449
+ targets: Dict[str, Path] = {}
450
+ # 1. Prefer pre-planned cross reference mappings from documentation plan
451
+ if (
452
+ self.plan.documentation_plan
453
+ and self.plan.documentation_plan.cross_reference_mappings
454
+ ):
455
+ for (
456
+ name,
457
+ rel_path,
458
+ ) in self.plan.documentation_plan.cross_reference_mappings.items():
459
+ if rel_path.endswith(".md"):
460
+ targets[name] = self.project_root / rel_path
461
+
462
+ # 2. Fallback to scanning generated documents if not mapped
463
+ if not targets:
464
+ currentmodule_re = re.compile(
465
+ r"^\.\. currentmodule::\s+(?P<name>[\w.]+)\s*$", re.MULTILINE
466
+ )
467
+ for proposal in self.plan.generated_documents:
468
+ match = currentmodule_re.search(proposal.content)
469
+ if match:
470
+ targets[match.group("name")] = (
471
+ self.project_root / proposal.target_path
472
+ )
473
+ return targets
474
+
475
+ @staticmethod
476
+ def _transform_autorefs_links(
477
+ content: str,
478
+ source_path: Path,
479
+ docs_dir: Path,
480
+ targets: Dict[str, Path],
481
+ ) -> Tuple[str, int]:
482
+ """Rewrite resolvable MkDocs autorefs compact links to relative MyST links."""
483
+ reference_re = re.compile(
484
+ r"(?<!!)\[(?P<label>[^\]\n]+)\]\[(?P<target>[A-Za-z_]\w*(?:\.\w+)*)?\]"
485
+ )
486
+ source_dir = source_path.parent
487
+ changes = 0
488
+
489
+ def replace(match: re.Match) -> str:
490
+ nonlocal changes
491
+ label = match.group("label")
492
+ target = match.group("target")
493
+ effective_target = target if target else label
494
+ target_doc = targets.get(effective_target)
495
+ if not target_doc and "." in effective_target:
496
+ target_doc = targets.get(effective_target.rsplit(".", 1)[0])
497
+ if target_doc is None:
498
+ return match.group(0)
499
+ display_label = label.split(".")[-1] if not target else label
500
+ relative = Path(os.path.relpath(target_doc, source_dir)).as_posix()
501
+ changes += 1
502
+ return f"[{display_label}]({relative})"
503
+
504
+ return reference_re.sub(replace, content), changes
505
+
506
+ def _generate_semantic_toctree(
507
+ self, nav_entries: List[NavigationItem], all_files: List[Path], docs_dir: Path
508
+ ) -> Optional[str]:
509
+ """Generate the root toctree without flattening section landing pages."""
510
+ docs_prefix = docs_dir.relative_to(self.project_root)
511
+ generated_targets = set()
512
+ for proposal in self.plan.generated_documents:
513
+ try:
514
+ generated_targets.add(
515
+ Path(proposal.target_path)
516
+ .relative_to(docs_prefix)
517
+ .with_suffix("")
518
+ .as_posix()
519
+ )
520
+ except ValueError:
521
+ continue
522
+ return build_semantic_toctree(
523
+ nav_entries=nav_entries,
524
+ all_files=all_files,
525
+ docs_dir=docs_dir,
526
+ project_root=self.project_root,
527
+ generated_targets=generated_targets,
528
+ )
529
+
530
+ def _generate_conf_py(self) -> str:
531
+ """Generates a clean Sphinx conf.py based strictly on MigrationPlan requirements."""
532
+ if (
533
+ self.plan.proposed_sphinx_config
534
+ and self.plan.proposed_sphinx_config.rendered_content
535
+ ):
536
+ return self.plan.proposed_sphinx_config.rendered_content
537
+ return build_conf_py(self.plan)
538
+
539
+ def _migrate_dependencies(self, write_to_disk: bool) -> List[str]:
540
+ """Update pyproject.toml or requirements.txt removing MkDocs packages and adding Sphinx packages."""
541
+ changes: List[str] = []
542
+ if not self.plan.dependency_analysis:
543
+ return changes
544
+
545
+ to_remove = set(
546
+ p.lower().replace("_", "-") for p in (self.plan.packages_to_remove or [])
547
+ )
548
+ if self.plan.dependency_analysis:
549
+ to_remove.update(
550
+ p.lower().replace("_", "-")
551
+ for p in self.plan.dependency_analysis.detected_packages_to_remove
552
+ )
553
+ to_remove.update({p.lower().replace("_", "-") for p in KNOWN_MKDOCS_PACKAGES})
554
+
555
+ # Deduplicate packages to add by normalized name, keeping version constraints
556
+ raw_candidates = list(self.plan.dependency_analysis.suggested_packages_to_add)
557
+ if self.plan.documentation_plan:
558
+ for req_pkg in self.plan.documentation_plan.required_packages or []:
559
+ raw_candidates.append(req_pkg)
560
+ if "sphinx_copybutton" in (
561
+ self.plan.documentation_plan.required_extensions or []
562
+ ):
563
+ raw_candidates.append("sphinx-copybutton")
564
+
565
+ seen_names = set()
566
+ to_add: List[str] = []
567
+ for candidate in sorted(raw_candidates, key=lambda s: len(s), reverse=True):
568
+ m_pkg = re.match(r"^([a-zA-Z0-9_\-\.]+)", candidate)
569
+ norm = (
570
+ m_pkg.group(1).lower().replace("_", "-") if m_pkg else candidate.lower()
571
+ )
572
+ if norm not in seen_names:
573
+ seen_names.add(norm)
574
+ to_add.append(candidate)
575
+ to_add.sort()
576
+
577
+ # 1. Update pyproject.toml if present
578
+ pyproject_path = self.project_root / "pyproject.toml"
579
+ if pyproject_path.exists():
580
+ content = pyproject_path.read_text(encoding="utf-8")
581
+ lines = content.splitlines()
582
+ new_lines: List[str] = []
583
+ inserted = False
584
+ i = 0
585
+ while i < len(lines):
586
+ line = lines[i]
587
+ m = re.search(
588
+ r'["\']([a-zA-Z0-9_\-\.]+)(?:\[[^\]]+\])?(?:[<>=!~;].*)?["\']', line
589
+ )
590
+ if m:
591
+ pkg_name = m.group(1).lower().replace("_", "-")
592
+ if pkg_name in to_remove:
593
+ if not inserted:
594
+ m_indent = re.match(r"^\s*", line)
595
+ indent = m_indent.group(0) if m_indent else ""
596
+ for add_pkg in to_add:
597
+ new_lines.append(f'{indent}"{add_pkg}",')
598
+ inserted = True
599
+ changes.append(f"Removed {pkg_name} from pyproject.toml")
600
+ i += 1
601
+ continue
602
+ new_lines.append(line)
603
+ i += 1
604
+
605
+ updated_content = "\n".join(new_lines) + (
606
+ "\n" if content.endswith("\n") else ""
607
+ )
608
+ if updated_content != content:
609
+ if write_to_disk:
610
+ pyproject_path.write_text(updated_content, encoding="utf-8")
611
+ changes.append("Updated pyproject.toml with Sphinx dependencies")
612
+
613
+ # 2. Update requirements.txt / docs-requirements.txt if present
614
+ for req_name in [
615
+ "requirements.txt",
616
+ "docs-requirements.txt",
617
+ "docs/requirements.txt",
618
+ ]:
619
+ req_path = self.project_root / req_name
620
+ if req_path.exists():
621
+ content = req_path.read_text(encoding="utf-8")
622
+ lines = content.splitlines()
623
+ new_lines = []
624
+ inserted = False
625
+ for line in lines:
626
+ stripped = line.strip()
627
+ if stripped and not stripped.startswith("#"):
628
+ m = re.match(r"^([a-zA-Z0-9_\-\.]+)", stripped)
629
+ if m:
630
+ pkg_name = m.group(1).lower().replace("_", "-")
631
+ if pkg_name in to_remove:
632
+ if not inserted:
633
+ for add_pkg in to_add:
634
+ new_lines.append(add_pkg)
635
+ inserted = True
636
+ changes.append(f"Removed {pkg_name} from {req_name}")
637
+ continue
638
+ new_lines.append(line)
639
+ if not inserted and to_add:
640
+ new_lines.extend(to_add)
641
+ updated_content = "\n".join(new_lines) + (
642
+ "\n" if content.endswith("\n") else ""
643
+ )
644
+ if updated_content != content:
645
+ if write_to_disk:
646
+ req_path.write_text(updated_content, encoding="utf-8")
647
+ changes.append(f"Updated {req_name} with Sphinx dependencies")
648
+
649
+ return changes
650
+
651
+ def _resolve_github_action_ref(
652
+ self, action_repo: str, fallback_tag: str, fallback_sha: str
653
+ ) -> Tuple[str, str]:
654
+ """Resolves the latest release tag and commit SHA for a GitHub Action repository via git ls-remote."""
655
+ return resolve_github_action_ref(action_repo, fallback_tag, fallback_sha)
656
+
657
+ def _migrate_ci_workflows(self, write_to_disk: bool) -> List[str]:
658
+ """Update CI workflows and configuration to build Sphinx documentation using the project's native tooling."""
659
+ changes: List[str] = []
660
+ github_workflows = self.project_root / ".github" / "workflows"
661
+ tox_ini = self.project_root / "tox.ini"
662
+
663
+ # 1. Update existing workflow files if they invoke mkdocs
664
+ if github_workflows.exists():
665
+ for yml_file in sorted(github_workflows.glob("*.y*ml")):
666
+ try:
667
+ content = yml_file.read_text(encoding="utf-8")
668
+ if (
669
+ "mkdocs build" in content
670
+ or "mkdocs gh-deploy" in content
671
+ or "mkdocs" in content
672
+ ):
673
+ new_content = content.replace(
674
+ "mkdocs gh-deploy",
675
+ "sphinx-build -b html docs site/_build/html",
676
+ )
677
+ new_content = new_content.replace(
678
+ "mkdocs build", "sphinx-build -b html docs site/_build/html"
679
+ )
680
+ if new_content != content:
681
+ if write_to_disk:
682
+ yml_file.write_text(new_content, encoding="utf-8")
683
+ changes.append(
684
+ f"Updated {yml_file.name} to use sphinx-build"
685
+ )
686
+ except Exception:
687
+ pass
688
+
689
+ # 2. If the project uses tox (tox.ini exists), add or update [testenv:docs]
690
+ if tox_ini.exists():
691
+ try:
692
+ tox_content = tox_ini.read_text(encoding="utf-8")
693
+
694
+ # Use planned tox env or generate using planner helper
695
+ if self.plan.ci_plan and self.plan.ci_plan.tox_docs_env:
696
+ docs_env = self.plan.ci_plan.tox_docs_env
697
+ else:
698
+ dep_config_line = determine_tox_dependency_line(
699
+ self.plan.dependency_analysis, self.project_root
700
+ )
701
+ docs_env = build_tox_docs_env(dep_config_line)
702
+
703
+ if "[testenv:docs]" not in tox_content:
704
+ updated_tox = tox_content.rstrip() + docs_env
705
+ if write_to_disk:
706
+ tox_ini.write_text(updated_tox, encoding="utf-8")
707
+ changes.append("Added [testenv:docs] to tox.ini")
708
+ elif "mkdocs build" in tox_content:
709
+ new_tox = tox_content.replace(
710
+ "mkdocs build", "sphinx-build -b html docs site/_build/html"
711
+ )
712
+ if new_tox != tox_content:
713
+ if write_to_disk:
714
+ tox_ini.write_text(new_tox, encoding="utf-8")
715
+ changes.append(
716
+ "Updated [testenv:docs] in tox.ini to use sphinx-build"
717
+ )
718
+ except Exception:
719
+ pass
720
+
721
+ # Also integrate docs job into existing GitHub workflow if it uses tox
722
+ if github_workflows.exists():
723
+ try:
724
+ for yml_file in sorted(github_workflows.glob("*.y*ml")):
725
+ wf_content = yml_file.read_text(encoding="utf-8")
726
+ if "tox" in wf_content:
727
+ checkout_ref = (
728
+ self.plan.ci_plan.checkout_pinned_ref
729
+ if (
730
+ self.plan.ci_plan
731
+ and self.plan.ci_plan.checkout_pinned_ref
732
+ )
733
+ else (
734
+ self.plan.ci_analysis.checkout_pinned_ref
735
+ if (
736
+ self.plan.ci_analysis
737
+ and self.plan.ci_analysis.checkout_pinned_ref
738
+ )
739
+ else None
740
+ )
741
+ )
742
+ uv_ref = (
743
+ self.plan.ci_plan.setup_uv_pinned_ref
744
+ if (
745
+ self.plan.ci_plan
746
+ and self.plan.ci_plan.setup_uv_pinned_ref
747
+ )
748
+ else (
749
+ self.plan.ci_analysis.setup_uv_pinned_ref
750
+ if (
751
+ self.plan.ci_analysis
752
+ and self.plan.ci_analysis.setup_uv_pinned_ref
753
+ )
754
+ else None
755
+ )
756
+ )
757
+ if not checkout_ref or not uv_ref:
758
+ ch_tag, ch_sha = self._resolve_github_action_ref(
759
+ "actions/checkout",
760
+ DEFAULT_CHECKOUT_TAG,
761
+ DEFAULT_CHECKOUT_SHA,
762
+ )
763
+ u_tag, u_sha = self._resolve_github_action_ref(
764
+ "astral-sh/setup-uv",
765
+ DEFAULT_SETUP_UV_TAG,
766
+ DEFAULT_SETUP_UV_SHA,
767
+ )
768
+ checkout_ref = checkout_ref or (
769
+ f"{ch_sha} # {ch_tag}" if ch_sha else ch_tag
770
+ )
771
+ uv_ref = uv_ref or (
772
+ f"{u_sha} # {u_tag}" if u_sha else u_tag
773
+ )
774
+
775
+ docs_job = (
776
+ self.plan.ci_plan.github_docs_job
777
+ if (self.plan.ci_plan and self.plan.ci_plan.github_docs_job)
778
+ else build_github_docs_job(
779
+ checkout_ref=checkout_ref, uv_ref=uv_ref
780
+ )
781
+ )
782
+
783
+ if "tox -e docs" not in wf_content and "jobs:" in wf_content:
784
+ updated_wf = wf_content.rstrip() + "\n" + docs_job
785
+ updated_wf = re.sub(
786
+ r"uses:\s*actions/checkout@[^\s\n]+(?:\s*#[^\n]*)?",
787
+ f"uses: actions/checkout@{checkout_ref}",
788
+ updated_wf,
789
+ )
790
+ updated_wf = re.sub(
791
+ r"uses:\s*astral-sh/setup-uv@[^\s\n]+(?:\s*#[^\n]*)?",
792
+ f"uses: astral-sh/setup-uv@{uv_ref}",
793
+ updated_wf,
794
+ )
795
+ if write_to_disk:
796
+ yml_file.write_text(updated_wf, encoding="utf-8")
797
+ changes.append(
798
+ f"Added docs job with git-pinned action hashes to {yml_file.name}"
799
+ )
800
+ break
801
+ elif "tox -e docs" in wf_content:
802
+ updated_wf = re.sub(
803
+ r"uses:\s*actions/checkout@[^\s\n]+(?:\s*#[^\n]*)?",
804
+ f"uses: actions/checkout@{checkout_ref}",
805
+ wf_content,
806
+ )
807
+ updated_wf = re.sub(
808
+ r"uses:\s*astral-sh/setup-uv@[^\s\n]+(?:\s*#[^\n]*)?",
809
+ f"uses: astral-sh/setup-uv@{uv_ref}",
810
+ updated_wf,
811
+ )
812
+ if updated_wf != wf_content:
813
+ if write_to_disk:
814
+ yml_file.write_text(updated_wf, encoding="utf-8")
815
+ changes.append(
816
+ f"Pinned action versions and commit hashes in {yml_file.name}"
817
+ )
818
+ break
819
+ except Exception:
820
+ pass
821
+
822
+ # 3. ReadTheDocs configuration (.readthedocs.yaml)
823
+ for rtd_name in [".readthedocs.yaml", ".readthedocs.yml"]:
824
+ rtd_path = self.project_root / rtd_name
825
+ if rtd_path.exists():
826
+ try:
827
+ content = rtd_path.read_text(encoding="utf-8")
828
+ new_content = re.sub(
829
+ r"mkdocs:\s*\n(\s+configuration:.*)?",
830
+ "sphinx:\n configuration: docs/conf.py\n",
831
+ content,
832
+ )
833
+ if new_content != content:
834
+ if write_to_disk:
835
+ rtd_path.write_text(new_content, encoding="utf-8")
836
+ changes.append(f"Updated {rtd_name} for Sphinx")
837
+ except Exception:
838
+ pass
839
+
840
+ return changes
841
+
842
+ def _clean_obsolete_mkdocs_files(self, write_to_disk: bool) -> List[str]:
843
+ """Remove obsolete MkDocs-specific generator scripts and hooks (e.g. scripts/gen_ref_nav.py).
844
+
845
+ Why this cleanup is necessary:
846
+ 1. In MkDocs, dynamic generator plugins like `mkdocs-gen-files` execute scripts at build
847
+ time to generate in-memory virtual markdown stubs (`with mkdocs_gen_files.open(...)`).
848
+ 2. During migration to Sphinx, `sphinx-mkdocs-migrate` statically materializes permanent,
849
+ checked-in MyST markdown documentation stubs (e.g., in `docs/reference/`) using native
850
+ Sphinx autodoc/autosummary directives.
851
+ 3. Once MkDocs dependencies are removed from the project's dependency manifest (`pyproject.toml`),
852
+ any remaining generator script importing `mkdocs_gen_files` becomes broken and unrunnable
853
+ (`ModuleNotFoundError: No module named 'mkdocs_gen_files'`).
854
+ 4. Leaving these scripts behind also triggers false positive failures in repo linters and formatters.
855
+ 5. Therefore, after all permanent documentation artifacts are synthesized, this cleanup phase
856
+ safely unlinks obsolete generator scripts and removes their enclosing directory if it becomes empty.
857
+ """
858
+ removed: List[str] = []
859
+ if not self.plan.source_mkdocs_config:
860
+ return removed
861
+
862
+ candidate_scripts = (
863
+ self.plan.obsolete_files
864
+ if self.plan.obsolete_files
865
+ else detect_obsolete_generator_scripts(
866
+ self.project_root,
867
+ self.plan.source_mkdocs_config,
868
+ additional_scripts=[
869
+ p.generator_script
870
+ for p in (
871
+ self.plan.documentation_plan.generated_pipelines
872
+ if self.plan.documentation_plan
873
+ else []
874
+ )
875
+ if p.generator_script
876
+ ],
877
+ )
878
+ )
879
+
880
+ for script_rel in candidate_scripts:
881
+ script_path = self.project_root / script_rel
882
+ if script_path.is_file():
883
+ if write_to_disk:
884
+ try:
885
+ script_path.unlink()
886
+ removed.append(script_rel)
887
+ # Remove parent directory if empty (e.g. scripts/)
888
+ parent_dir = script_path.parent
889
+ if parent_dir != self.project_root and parent_dir.is_dir():
890
+ if not any(parent_dir.iterdir()):
891
+ parent_dir.rmdir()
892
+ except Exception:
893
+ pass
894
+ else:
895
+ removed.append(script_rel)
896
+
897
+ return sorted(removed)