mdfetch 0.5.1__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (207) hide show
  1. {mdfetch-0.5.1 → mdfetch-0.5.2}/PKG-INFO +1 -1
  2. {mdfetch-0.5.1 → mdfetch-0.5.2}/pyproject.toml +1 -1
  3. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/base.py +33 -10
  4. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/cli.py +23 -3
  5. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/devto.py +9 -9
  6. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/medium.py +11 -3
  7. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/substack.py +10 -10
  8. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/router.py +7 -6
  9. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_cli.py +25 -0
  10. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_medium_extractor.py +23 -6
  11. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_router.py +20 -0
  12. {mdfetch-0.5.1 → mdfetch-0.5.2}/uv.lock +1 -1
  13. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-analyze/SKILL.md +0 -0
  14. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
  15. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-checklist/SKILL.md +0 -0
  16. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-clarify/SKILL.md +0 -0
  17. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-constitution/SKILL.md +0 -0
  18. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
  19. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
  20. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
  21. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
  22. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
  23. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-implement/SKILL.md +0 -0
  24. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-plan/SKILL.md +0 -0
  25. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
  26. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-specify/SKILL.md +0 -0
  27. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-tasks/SKILL.md +0 -0
  28. {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
  29. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.analyze.toml +0 -0
  30. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.archive.run.toml +0 -0
  31. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.checklist.toml +0 -0
  32. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.clarify.toml +0 -0
  33. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.constitution.toml +0 -0
  34. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.implement.toml +0 -0
  35. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.plan.toml +0 -0
  36. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.reconcile.run.toml +0 -0
  37. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.specify.toml +0 -0
  38. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.tasks.toml +0 -0
  39. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.taskstoissues.toml +0 -0
  40. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gitattributes +0 -0
  41. {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/copilot-instructions.md +0 -0
  42. {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/ci.yml +0 -0
  43. {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/integration.yml +0 -0
  44. {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/publish.yml +0 -0
  45. {mdfetch-0.5.1 → mdfetch-0.5.2}/.gitignore +0 -0
  46. {mdfetch-0.5.1 → mdfetch-0.5.2}/.python-version +0 -0
  47. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/.registry +0 -0
  48. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/LICENSE +0 -0
  49. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/README.md +0 -0
  50. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/commands/archive.md +0 -0
  51. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/extension.yml +0 -0
  52. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/README.md +0 -0
  53. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
  54. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
  55. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
  56. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
  57. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
  58. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/config-template.yml +0 -0
  59. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/extension.yml +0 -0
  60. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/git-config.yml +0 -0
  61. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
  62. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
  63. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
  64. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
  65. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
  66. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
  67. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
  68. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
  69. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/LICENSE +0 -0
  70. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/README.md +0 -0
  71. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
  72. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/extension.yml +0 -0
  73. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions.yml +0 -0
  74. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/feature.json +0 -0
  75. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/init-options.json +0 -0
  76. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integration.json +0 -0
  77. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/claude.manifest.json +0 -0
  78. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/gemini.manifest.json +0 -0
  79. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/speckit.manifest.json +0 -0
  80. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/changelog.md +0 -0
  81. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/constitution.md +0 -0
  82. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/plan.md +0 -0
  83. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/spec.md +0 -0
  84. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/check-prerequisites.sh +0 -0
  85. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/common.sh +0 -0
  86. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/create-new-feature.sh +0 -0
  87. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/setup-plan.sh +0 -0
  88. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/setup-tasks.sh +0 -0
  89. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/checklist-template.md +0 -0
  90. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/constitution-template.md +0 -0
  91. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/plan-template.md +0 -0
  92. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/spec-template.md +0 -0
  93. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/tasks-template.md +0 -0
  94. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/workflows/speckit/workflow.yml +0 -0
  95. {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/workflows/workflow-registry.json +0 -0
  96. {mdfetch-0.5.1 → mdfetch-0.5.2}/.vscode/settings.json +0 -0
  97. {mdfetch-0.5.1 → mdfetch-0.5.2}/CLAUDE.md +0 -0
  98. {mdfetch-0.5.1 → mdfetch-0.5.2}/GEMINI.md +0 -0
  99. {mdfetch-0.5.1 → mdfetch-0.5.2}/LICENSE +0 -0
  100. {mdfetch-0.5.1 → mdfetch-0.5.2}/Makefile +0 -0
  101. {mdfetch-0.5.1 → mdfetch-0.5.2}/README.md +0 -0
  102. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
  103. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
  104. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
  105. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
  106. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
  107. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/research.md +0 -0
  108. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
  109. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
  110. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/checklists/requirements.md +0 -0
  111. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/contracts/public-api.md +0 -0
  112. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/data-model.md +0 -0
  113. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/plan.md +0 -0
  114. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/quickstart.md +0 -0
  115. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/research.md +0 -0
  116. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/spec.md +0 -0
  117. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/tasks.md +0 -0
  118. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/checklists/requirements.md +0 -0
  119. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/contracts/extract-api.md +0 -0
  120. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/plan.md +0 -0
  121. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/research.md +0 -0
  122. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/spec.md +0 -0
  123. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/tasks.md +0 -0
  124. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/checklists/requirements.md +0 -0
  125. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/plan.md +0 -0
  126. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/research.md +0 -0
  127. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/spec.md +0 -0
  128. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/tasks.md +0 -0
  129. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/checklists/requirements.md +0 -0
  130. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/contracts/extractor-api.md +0 -0
  131. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/data-model.md +0 -0
  132. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/plan.md +0 -0
  133. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/quickstart.md +0 -0
  134. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/research.md +0 -0
  135. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/spec.md +0 -0
  136. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/tasks.md +0 -0
  137. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/checklists/requirements.md +0 -0
  138. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/contracts/public-api.md +0 -0
  139. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/data-model.md +0 -0
  140. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/plan.md +0 -0
  141. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/quickstart.md +0 -0
  142. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/research.md +0 -0
  143. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/spec.md +0 -0
  144. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/tasks.md +0 -0
  145. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/checklists/requirements.md +0 -0
  146. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/contracts/public-api.md +0 -0
  147. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/data-model.md +0 -0
  148. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/plan.md +0 -0
  149. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/quickstart.md +0 -0
  150. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/research.md +0 -0
  151. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/spec.md +0 -0
  152. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/tasks.md +0 -0
  153. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/checklists/requirements.md +0 -0
  154. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/contracts/cli.md +0 -0
  155. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/data-model.md +0 -0
  156. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/plan.md +0 -0
  157. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/quickstart.md +0 -0
  158. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/research.md +0 -0
  159. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/spec.md +0 -0
  160. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/tasks.md +0 -0
  161. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/checklists/requirements.md +0 -0
  162. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/contracts/formula.md +0 -0
  163. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/contracts/tap-update-job.md +0 -0
  164. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/data-model.md +0 -0
  165. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/plan.md +0 -0
  166. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/quickstart.md +0 -0
  167. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/research.md +0 -0
  168. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/spec.md +0 -0
  169. {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/tasks.md +0 -0
  170. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/__init__.py +0 -0
  171. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/exceptions.py +0 -0
  172. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/__init__.py +0 -0
  173. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/dzone.py +0 -0
  174. {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/thenewstack.py +0 -0
  175. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/__init__.py +0 -0
  176. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/conftest.py +0 -0
  177. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/__init__.py +0 -0
  178. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/conftest.py +0 -0
  179. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
  180. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-december-2025.md +0 -0
  181. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-july-2025.md +0 -0
  182. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-march-2026.md +0 -0
  183. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-image-classification-pipeline-camel-djl.md +0 -0
  184. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-integration-patterns-fail-production.md +0 -0
  185. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-kiro-feature-to-requirements-design-tasks.md +0 -0
  186. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
  187. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
  188. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/substack-api-trends-2025.md +0 -0
  189. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/substack-kafka-topic-types.md +0 -0
  190. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-api-mcp-agent.md +0 -0
  191. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-async-apis.md +0 -0
  192. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-developer-portal-api.md +0 -0
  193. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-json-schema-ai.md +0 -0
  194. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-mcp-api-governance.md +0 -0
  195. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_cli_integration.py +0 -0
  196. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_devto_integration.py +0 -0
  197. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_dzone_integration.py +0 -0
  198. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_medium_integration.py +0 -0
  199. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_substack_integration.py +0 -0
  200. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_thenewstack_integration.py +0 -0
  201. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/__init__.py +0 -0
  202. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_devto_extractor.py +0 -0
  203. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_dzone_extractor.py +0 -0
  204. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_fetch_errors.py +0 -0
  205. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_silent.py +0 -0
  206. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_substack_extractor.py +0 -0
  207. {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_thenewstack_extractor.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mdfetch
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: Extract article content from web platforms and return it as clean Markdown.
5
5
  Project-URL: Homepage, https://github.com/stn1slv/md-fetch
6
6
  Project-URL: Source, https://github.com/stn1slv/md-fetch
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "mdfetch"
7
- version = "0.5.1"
7
+ version = "0.5.2"
8
8
  description = "Extract article content from web platforms and return it as clean Markdown."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
  import re
6
6
  import time
7
7
  from abc import ABC, abstractmethod
8
+ from collections.abc import Iterable
8
9
  from typing import Any
9
10
 
10
11
  import httpx
@@ -28,6 +29,10 @@ class BaseExtractor(ABC):
28
29
  """Contract all platform-specific extractors must fulfil."""
29
30
 
30
31
  DOMAINS: frozenset[str] = frozenset()
32
+ # When True, the router also matches hostnames that are subdomains of any
33
+ # entry in DOMAINS (e.g. ``foo.medium.com`` → MediumExtractor). Leave as
34
+ # False for single-tenant sites where subdomains are not article URLs.
35
+ MATCH_SUBDOMAINS: bool = False
31
36
  _no_retry_status_codes: frozenset[int] = frozenset()
32
37
 
33
38
  # FR-014: use a browser-like UA (no mdfetch-specific branding) so servers serve readable HTML
@@ -139,17 +144,35 @@ class BaseExtractor(ABC):
139
144
 
140
145
  return md
141
146
 
142
- @staticmethod
143
- def _replace_iframes_with_links(container: Tag, soup: BeautifulSoup) -> None:
144
- """Replace ``<iframe>`` elements inside *container* with plain anchor links."""
145
- for iframe in container.find_all("iframe"):
146
- src = str(iframe.get("src") or iframe.get("data-src") or "")
147
- if src:
148
- link = soup.new_tag("a", href=src)
149
- link.string = src
150
- iframe.replace_with(link)
147
+ _DEFAULT_EMBED_URL_ATTRS: tuple[str, ...] = ("src", "data-src", "data-url", "href")
148
+
149
+ @classmethod
150
+ def _replace_embeds_with_links(
151
+ cls,
152
+ embeds: Iterable[Tag],
153
+ soup: BeautifulSoup,
154
+ *,
155
+ attrs: tuple[str, ...] = _DEFAULT_EMBED_URL_ATTRS,
156
+ ) -> None:
157
+ """Replace each tag in *embeds* with an anchor pointing at its URL, or decompose."""
158
+ for embed in embeds:
159
+ url = ""
160
+ for attr in attrs:
161
+ val = embed.get(attr)
162
+ if val:
163
+ url = str(val)
164
+ break
165
+ if url:
166
+ link = soup.new_tag("a", href=url)
167
+ link.string = url
168
+ embed.replace_with(link)
151
169
  else:
152
- iframe.decompose()
170
+ embed.decompose()
171
+
172
+ @classmethod
173
+ def _replace_iframes_with_links(cls, container: Tag, soup: BeautifulSoup) -> None:
174
+ """Replace ``<iframe>`` elements inside *container* with plain anchor links."""
175
+ cls._replace_embeds_with_links(container.find_all("iframe"), soup)
153
176
 
154
177
  def extract(self, url: str, *, retries: int = 3, retry_delay: float = 2.0) -> str:
155
178
  """Orchestrate fetch → clean → convert and return Markdown."""
@@ -4,7 +4,9 @@ Command-line interface for mdfetch.
4
4
 
5
5
  from __future__ import annotations
6
6
 
7
+ import os
7
8
  import sys
9
+ from urllib.parse import urlparse
8
10
 
9
11
  import click
10
12
 
@@ -36,8 +38,28 @@ from mdfetch.router import supported_domains
36
38
  show_default=True,
37
39
  help="Seconds to wait between retry attempts",
38
40
  )
39
- def main(url: str, output: str | None, retries: int, retry_delay: float) -> None:
41
+ @click.option(
42
+ "-f",
43
+ "--force",
44
+ is_flag=True,
45
+ default=False,
46
+ help="Overwrite the output file if it already exists",
47
+ )
48
+ def main(
49
+ url: str,
50
+ output: str | None,
51
+ retries: int,
52
+ retry_delay: float,
53
+ force: bool,
54
+ ) -> None:
40
55
  """Fetch and extract Markdown from the given URL."""
56
+ if output and os.path.exists(output) and not force:
57
+ click.secho(
58
+ f"Error: '{output}' already exists. Use --force to overwrite.",
59
+ err=True,
60
+ fg="red",
61
+ )
62
+ sys.exit(1)
41
63
  try:
42
64
  content = extract(url, retries=retries, retry_delay=retry_delay)
43
65
 
@@ -47,8 +69,6 @@ def main(url: str, output: str | None, retries: int, retry_delay: float) -> None
47
69
  else:
48
70
  click.echo(content)
49
71
  except UnsupportedPlatformError as e:
50
- from urllib.parse import urlparse
51
-
52
72
  domain = (urlparse(e.url).hostname if e.url else None) or str(e)
53
73
  domains = ", ".join(sorted(supported_domains()))
54
74
  click.secho(
@@ -30,15 +30,15 @@ class DevToExtractor(BaseExtractor):
30
30
  # Replace iframes with plain anchor links (FR-008)
31
31
  self._replace_iframes_with_links(body, soup)
32
32
 
33
- # Replace dev.to liquid-tag embeds with plain anchor links (FR-008)
34
- for embed in body.find_all(class_=re.compile(r"ltag", re.IGNORECASE)):
35
- src = str(embed.get("data-url") or embed.get("data-src") or embed.get("src") or "")
36
- if src:
37
- link = soup.new_tag("a", href=src)
38
- link.string = src
39
- embed.replace_with(link)
40
- else:
41
- embed.decompose()
33
+ # Replace dev.to liquid-tag embeds with plain anchor links (FR-008).
34
+ # Anchor at the start of the class name so unrelated classes that merely
35
+ # contain "ltag" (e.g. "ultraltagrelated") are not matched — bs4 matches
36
+ # each class independently via re.search.
37
+ self._replace_embeds_with_links(
38
+ body.find_all(class_=re.compile(r"^ltag(?:[-_]|$)", re.IGNORECASE)),
39
+ soup,
40
+ attrs=("data-url", "data-src", "src"),
41
+ )
42
42
 
43
43
  # Strip empty anchor-name links inserted before headings
44
44
  for anchor in body.find_all("a", attrs={"name": True}):
@@ -22,6 +22,7 @@ class MediumExtractor(BaseExtractor):
22
22
  """Extracts article content from medium.com and its subdomains."""
23
23
 
24
24
  DOMAINS: frozenset[str] = frozenset({"medium.com"})
25
+ MATCH_SUBDOMAINS = True
25
26
  _FREEDIUM_BASE = "https://freedium-mirror.cfd/"
26
27
  _no_retry_status_codes: frozenset[int] = frozenset({403, 429})
27
28
  # Smart-quote characters that Medium serves but Freedium replaces with ASCII;
@@ -91,12 +92,19 @@ class MediumExtractor(BaseExtractor):
91
92
  return md
92
93
 
93
94
  def _parse_freedium(self, soup: BeautifulSoup) -> str:
94
- """Parse Freedium mirror HTML, which uses div.main-content instead of <article>."""
95
- content = soup.find("div", class_="main-content")
95
+ """Parse Freedium mirror HTML, whose Svelte rebuild holds the body in div.prose."""
96
+ # Scope to the article so a stray .prose block (e.g. a bio/summary) can't
97
+ # match first; fall back to a bare .prose if the <article> wrapper is absent.
98
+ content = soup.select_one("article .prose") or soup.find("div", class_="prose")
96
99
  if not isinstance(content, Tag):
97
100
  raise UnsupportedContentTypeError(
98
- "Fallback page missing main-content element",
101
+ "Fallback page missing prose element",
99
102
  )
103
+ # Freedium's Shiki highlighter emits each code block twice — a light-theme
104
+ # and a dark-theme variant — so the visible text is duplicated. Drop the
105
+ # dark variant before conversion to avoid repeated fenced-code output.
106
+ for pre in content.select("pre.github-dark"):
107
+ pre.decompose()
100
108
  # Freedium renders section headings one level deeper than medium.com (h4 vs h3).
101
109
  # Remap so the output heading levels match the medium.com direct path.
102
110
  for level in (4, 5, 6):
@@ -18,6 +18,7 @@ class SubstackExtractor(BaseExtractor):
18
18
  """Extracts article content from substack.com and its subdomains."""
19
19
 
20
20
  DOMAINS: frozenset[str] = frozenset({"substack.com"})
21
+ MATCH_SUBDOMAINS = True
21
22
 
22
23
  def clean_html(self, soup: BeautifulSoup) -> Tag:
23
24
  """Isolate the article body and strip all non-content elements."""
@@ -36,16 +37,15 @@ class SubstackExtractor(BaseExtractor):
36
37
 
37
38
  # Convert other Substack embed containers to plain anchor links (FR-011)
38
39
  _safe_components = {"SubscribeWidget", "Image2ToDOM"}
39
- for embed in body.find_all(attrs={"data-component-name": True}):
40
- if embed.get("data-component-name") in _safe_components:
41
- continue
42
- url = str(embed.get("href") or embed.get("data-url") or embed.get("src") or "")
43
- if url:
44
- link = soup.new_tag("a", href=url)
45
- link.string = url
46
- embed.replace_with(link)
47
- else:
48
- embed.decompose()
40
+ self._replace_embeds_with_links(
41
+ [
42
+ e
43
+ for e in body.find_all(attrs={"data-component-name": True})
44
+ if e.get("data-component-name") not in _safe_components
45
+ ],
46
+ soup,
47
+ attrs=("href", "data-url", "src"),
48
+ )
49
49
 
50
50
  # Prepend subtitle from post-header (FR-005)
51
51
  header = soup.find("div", class_="post-header")
@@ -37,15 +37,16 @@ def route(url: str) -> BaseExtractor:
37
37
  if parsed.scheme not in ("http", "https") or not hostname:
38
38
  raise InvalidURLError(f"Invalid URL: {url!r}", url=url)
39
39
 
40
- # Exact match first; fall back to subdomain suffix check so any provider whose
41
- # DOMAINS entry is a parent domain automatically handles its subdomains.
42
- # Sort candidates by length descending so the most-specific suffix wins when
43
- # multiple registered domains are suffixes of the same hostname.
40
+ # Exact match first; fall back to a subdomain suffix check only for providers
41
+ # that opt in via MATCH_SUBDOMAINS=True (multi-tenant sites like Medium and
42
+ # Substack). Sort candidates by length descending so the most-specific
43
+ # suffix wins when multiple registered domains are suffixes of the same host.
44
44
  provider_cls = _REGISTRY.get(hostname)
45
45
  if provider_cls is None:
46
46
  for domain in sorted(_REGISTRY, key=len, reverse=True):
47
- if hostname.endswith(f".{domain}"):
48
- provider_cls = _REGISTRY[domain]
47
+ candidate = _REGISTRY[domain]
48
+ if candidate.MATCH_SUBDOMAINS and hostname.endswith(f".{domain}"):
49
+ provider_cls = candidate
49
50
  break
50
51
 
51
52
  if provider_cls is None:
@@ -2,6 +2,8 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import pathlib
6
+
5
7
  import pytest
6
8
  import pytest_mock
7
9
  from click.testing import CliRunner
@@ -41,3 +43,26 @@ def test_unsupported_domain_error_message(runner: CliRunner) -> None:
41
43
  assert result.exit_code == 1
42
44
  assert "'google.com' is not a supported platform." in result.output
43
45
  assert "Supported domains:" in result.output
46
+
47
+
48
+ def test_output_refuses_to_clobber_existing_file(
49
+ mocker: pytest_mock.MockerFixture, runner: CliRunner, tmp_path: pathlib.Path
50
+ ) -> None:
51
+ mocker.patch("mdfetch.cli.extract", return_value="# Test")
52
+ out = tmp_path / "out.md"
53
+ out.write_text("existing content")
54
+ result = runner.invoke(main, ["https://dev.to/test", "-o", str(out)])
55
+ assert result.exit_code == 1
56
+ assert "already exists" in result.output
57
+ assert out.read_text() == "existing content"
58
+
59
+
60
+ def test_output_force_overwrites_existing_file(
61
+ mocker: pytest_mock.MockerFixture, runner: CliRunner, tmp_path: pathlib.Path
62
+ ) -> None:
63
+ mocker.patch("mdfetch.cli.extract", return_value="# Test")
64
+ out = tmp_path / "out.md"
65
+ out.write_text("existing content")
66
+ result = runner.invoke(main, ["https://dev.to/test", "-o", str(out), "--force"])
67
+ assert result.exit_code == 0
68
+ assert out.read_text() == "# Test"
@@ -174,27 +174,38 @@ class TestConvertToMarkdown:
174
174
  assert "line one \nline two" in md
175
175
 
176
176
 
177
+ # Mirrors the Svelte/Tailwind Freedium rebuild: body in div.prose, code blocks
178
+ # rendered twice via Shiki (light + dark theme variants).
177
179
  _FREEDIUM_ARTICLE_HTML = """
178
180
  <html><body>
179
- <div class="main-content">
181
+ <main>
182
+ <article>
183
+ <div class="prose max-w-none prose-external-links">
180
184
  <h4>Section One</h4>
181
185
  <p>First paragraph of the article.</p>
182
186
  <h4>Section Two</h4>
183
187
  <p>Second paragraph with more content.</p>
184
- <pre><code>print("hello")</code></pre>
188
+ <div class="dark:hidden">
189
+ <pre class="shiki github-light"><code>print("hello")</code></pre>
190
+ </div>
191
+ <div class="hidden dark:block">
192
+ <pre class="shiki github-dark"><code>print("hello")</code></pre>
193
+ </div>
185
194
  </div>
195
+ </article>
196
+ </main>
186
197
  </body></html>
187
198
  """
188
199
 
189
200
  _FREEDIUM_NO_CONTENT_HTML = """
190
201
  <html><body>
191
- <div class="header">No main-content div here</div>
202
+ <div class="header">No prose div here</div>
192
203
  </body></html>
193
204
  """
194
205
 
195
206
 
196
207
  class TestParseFreedium:
197
- def test_returns_markdown_from_main_content(self, extractor: MediumExtractor) -> None:
208
+ def test_returns_markdown_from_prose(self, extractor: MediumExtractor) -> None:
198
209
  soup = BeautifulSoup(_FREEDIUM_ARTICLE_HTML, "lxml")
199
210
  md = extractor._parse_freedium(soup)
200
211
  assert "Section One" in md
@@ -206,9 +217,15 @@ class TestParseFreedium:
206
217
  assert "###" in md
207
218
  assert "####" not in md
208
219
 
209
- def test_raises_when_main_content_missing(self, extractor: MediumExtractor) -> None:
220
+ def test_dark_theme_code_block_deduplicated(self, extractor: MediumExtractor) -> None:
221
+ """Shiki renders each code block twice (light + dark); only one must survive."""
222
+ soup = BeautifulSoup(_FREEDIUM_ARTICLE_HTML, "lxml")
223
+ md = extractor._parse_freedium(soup)
224
+ assert md.count('print("hello")') == 1
225
+
226
+ def test_raises_when_prose_missing(self, extractor: MediumExtractor) -> None:
210
227
  soup = BeautifulSoup(_FREEDIUM_NO_CONTENT_HTML, "lxml")
211
- with pytest.raises(UnsupportedContentTypeError, match="main-content"):
228
+ with pytest.raises(UnsupportedContentTypeError, match="prose"):
212
229
  extractor._parse_freedium(soup)
213
230
 
214
231
 
@@ -81,6 +81,26 @@ class TestUrlValidation:
81
81
  provider = route("https://dzone.com/articles/some-article")
82
82
  assert isinstance(provider, DZoneExtractor)
83
83
 
84
+ def test_rejects_subdomain_of_single_tenant_domain(self) -> None:
85
+ # dev.to does not opt in to MATCH_SUBDOMAINS, so foo.dev.to must not
86
+ # be routed to DevToExtractor — it should raise UnsupportedPlatformError.
87
+ with pytest.raises(UnsupportedPlatformError):
88
+ route("https://foo.dev.to/some-article")
89
+
90
+ def test_rejects_subdomain_of_dzone(self) -> None:
91
+ with pytest.raises(UnsupportedPlatformError):
92
+ route("https://blog.dzone.com/articles/some-article")
93
+
94
+ def test_rejects_subdomain_of_thenewstack(self) -> None:
95
+ with pytest.raises(UnsupportedPlatformError):
96
+ route("https://blog.thenewstack.io/some-article")
97
+
98
+ def test_routes_substack_subdomain(self) -> None:
99
+ from mdfetch.providers.substack import SubstackExtractor
100
+
101
+ provider = route("https://newsletter.substack.com/p/post")
102
+ assert isinstance(provider, SubstackExtractor)
103
+
84
104
 
85
105
  class TestSupportedDomains:
86
106
  def test_returns_frozenset(self) -> None:
@@ -310,7 +310,7 @@ wheels = [
310
310
 
311
311
  [[package]]
312
312
  name = "mdfetch"
313
- version = "0.5.1"
313
+ version = "0.5.2"
314
314
  source = { editable = "." }
315
315
  dependencies = [
316
316
  { name = "beautifulsoup4" },
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes