mdfetch 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. mdfetch-0.2.0/.specify/feature.json +3 -0
  2. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/changelog.md +27 -0
  3. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/plan.md +27 -8
  4. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/spec.md +58 -5
  5. {mdfetch-0.1.0 → mdfetch-0.2.0}/CLAUDE.md +1 -0
  6. {mdfetch-0.1.0 → mdfetch-0.2.0}/PKG-INFO +5 -2
  7. {mdfetch-0.1.0 → mdfetch-0.2.0}/README.md +3 -0
  8. {mdfetch-0.1.0 → mdfetch-0.2.0}/pyproject.toml +2 -2
  9. mdfetch-0.2.0/specs/002-devto-provider/checklists/requirements.md +34 -0
  10. mdfetch-0.2.0/specs/002-devto-provider/contracts/public-api.md +78 -0
  11. mdfetch-0.2.0/specs/002-devto-provider/data-model.md +55 -0
  12. mdfetch-0.2.0/specs/002-devto-provider/plan.md +165 -0
  13. mdfetch-0.2.0/specs/002-devto-provider/quickstart.md +57 -0
  14. mdfetch-0.2.0/specs/002-devto-provider/research.md +77 -0
  15. mdfetch-0.2.0/specs/002-devto-provider/spec.md +117 -0
  16. mdfetch-0.2.0/specs/002-devto-provider/tasks.md +205 -0
  17. mdfetch-0.2.0/src/mdfetch/providers/devto.py +80 -0
  18. mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-december-2025.md +135 -0
  19. mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-july-2025.md +111 -0
  20. mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-march-2026.md +237 -0
  21. mdfetch-0.2.0/tests/integration/test_devto_integration.py +50 -0
  22. mdfetch-0.2.0/tests/unit/test_devto_extractor.py +255 -0
  23. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_router.py +9 -3
  24. {mdfetch-0.1.0 → mdfetch-0.2.0}/uv.lock +1 -1
  25. mdfetch-0.1.0/.specify/feature.json +0 -3
  26. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-analyze/SKILL.md +0 -0
  27. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
  28. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-checklist/SKILL.md +0 -0
  29. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-clarify/SKILL.md +0 -0
  30. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-constitution/SKILL.md +0 -0
  31. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
  32. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
  33. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
  34. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
  35. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
  36. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-implement/SKILL.md +0 -0
  37. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-plan/SKILL.md +0 -0
  38. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
  39. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-specify/SKILL.md +0 -0
  40. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-tasks/SKILL.md +0 -0
  41. {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
  42. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.analyze.toml +0 -0
  43. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.archive.run.toml +0 -0
  44. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.checklist.toml +0 -0
  45. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.clarify.toml +0 -0
  46. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.constitution.toml +0 -0
  47. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.implement.toml +0 -0
  48. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.plan.toml +0 -0
  49. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.reconcile.run.toml +0 -0
  50. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.specify.toml +0 -0
  51. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.tasks.toml +0 -0
  52. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.taskstoissues.toml +0 -0
  53. {mdfetch-0.1.0 → mdfetch-0.2.0}/.github/workflows/ci.yml +0 -0
  54. {mdfetch-0.1.0 → mdfetch-0.2.0}/.github/workflows/publish.yml +0 -0
  55. {mdfetch-0.1.0 → mdfetch-0.2.0}/.gitignore +0 -0
  56. {mdfetch-0.1.0 → mdfetch-0.2.0}/.python-version +0 -0
  57. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/.registry +0 -0
  58. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/LICENSE +0 -0
  59. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/README.md +0 -0
  60. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/commands/archive.md +0 -0
  61. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/extension.yml +0 -0
  62. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/README.md +0 -0
  63. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
  64. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
  65. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
  66. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
  67. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
  68. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/config-template.yml +0 -0
  69. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/extension.yml +0 -0
  70. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/git-config.yml +0 -0
  71. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
  72. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
  73. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
  74. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
  75. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
  76. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
  77. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
  78. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
  79. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/LICENSE +0 -0
  80. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/README.md +0 -0
  81. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
  82. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/extension.yml +0 -0
  83. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions.yml +0 -0
  84. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/init-options.json +0 -0
  85. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integration.json +0 -0
  86. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/claude.manifest.json +0 -0
  87. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/gemini.manifest.json +0 -0
  88. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/speckit.manifest.json +0 -0
  89. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/constitution.md +0 -0
  90. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/check-prerequisites.sh +0 -0
  91. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/common.sh +0 -0
  92. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/create-new-feature.sh +0 -0
  93. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/setup-plan.sh +0 -0
  94. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/setup-tasks.sh +0 -0
  95. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/checklist-template.md +0 -0
  96. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/constitution-template.md +0 -0
  97. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/plan-template.md +0 -0
  98. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/spec-template.md +0 -0
  99. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/tasks-template.md +0 -0
  100. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/workflows/speckit/workflow.yml +0 -0
  101. {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/workflows/workflow-registry.json +0 -0
  102. {mdfetch-0.1.0 → mdfetch-0.2.0}/GEMINI.md +0 -0
  103. {mdfetch-0.1.0 → mdfetch-0.2.0}/LICENSE +0 -0
  104. {mdfetch-0.1.0 → mdfetch-0.2.0}/Makefile +0 -0
  105. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
  106. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
  107. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
  108. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
  109. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
  110. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/research.md +0 -0
  111. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
  112. {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
  113. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/__init__.py +0 -0
  114. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/base.py +0 -0
  115. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/exceptions.py +0 -0
  116. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/providers/__init__.py +0 -0
  117. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/providers/medium.py +0 -0
  118. {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/router.py +0 -0
  119. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/__init__.py +0 -0
  120. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/conftest.py +0 -0
  121. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/__init__.py +0 -0
  122. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
  123. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
  124. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
  125. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/test_medium_integration.py +0 -0
  126. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/__init__.py +0 -0
  127. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_fetch_errors.py +0 -0
  128. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_medium_extractor.py +0 -0
  129. {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_silent.py +0 -0
@@ -0,0 +1,3 @@
1
+ {
2
+ "feature_directory": "specs/002-devto-provider"
3
+ }
@@ -2,6 +2,33 @@
2
2
 
3
3
  ---
4
4
 
5
+ ### mdfetch — dev.to Extractor — 2026-05-14
6
+
7
+ **Branch**: `002-devto-provider`
8
+ **Spec**: specs/002-devto-provider
9
+
10
+ **What was added**:
11
+ - `DevToExtractor` provider for `dev.to` articles, auto-discovered via `@register` decorator
12
+ - Article body isolation from `<div id="article-body">` with cover image extracted from `<header class="crayons-article__header">` and prepended to output
13
+ - `<iframe>` and liquid-tag embed (`ltag__*`) replacement with plain Markdown links (FR-019)
14
+ - `UnsupportedContentTypeError` raised for non-article dev.to pages (profiles, tag listings)
15
+ - 17 new unit tests in `tests/unit/test_devto_extractor.py`
16
+ - 3 dev.to integration tests in `tests/integration/test_devto_integration.py` with snapshot golden files
17
+ - Library version bumped from `0.1.0` to `0.2.0`
18
+ - `"dev.to"` added to `pyproject.toml` keywords (T013)
19
+
20
+ **New Components**:
21
+ - `src/mdfetch/providers/devto.py` — DevToExtractor
22
+ - `tests/unit/test_devto_extractor.py` — 17 unit tests
23
+ - `tests/integration/test_devto_integration.py` — 3 integration tests
24
+ - `tests/integration/snapshots/devto-integration-digest-december-2025.md`
25
+ - `tests/integration/snapshots/devto-integration-digest-july-2025.md`
26
+ - `tests/integration/snapshots/devto-integration-digest-march-2026.md`
27
+
28
+ **Tasks Completed**: 13/13
29
+
30
+ ---
31
+
5
32
  ### mdfetch — Medium Extractor (Initial Release) — 2026-05-14
6
33
 
7
34
  **Branch**: `feature/first-draft`
@@ -1,7 +1,7 @@
1
1
  # mdfetch — Main Implementation Plan
2
2
 
3
3
  **Last Updated**: 2026-05-14
4
- **Sources**: [specs/001-mdfetch-medium-extractor/plan.md]
4
+ **Sources**: [specs/001-mdfetch-medium-extractor/plan.md], [specs/002-devto-provider/plan.md]
5
5
 
6
6
  ---
7
7
 
@@ -47,6 +47,13 @@ MediumExtractor(BaseExtractor) — src/mdfetch/providers/medium.py
47
47
  ├── DOMAINS = frozenset({"medium.com"})
48
48
  ├── clean_html() → removes nav, clap buttons, sidebars, share elements, post-footer, author bio
49
49
  └── convert_to_markdown() → markdownify with ATX headings, fenced code blocks
50
+
51
+ DevToExtractor(BaseExtractor) — src/mdfetch/providers/devto.py
52
+ ├── DOMAINS = frozenset({"dev.to"})
53
+ ├── clean_html() → locates div#article-body; raises UnsupportedContentTypeError if absent;
54
+ │ replaces iframes and ltag embeds with anchor links; strips empty anchor-name
55
+ │ elements; prepends h1 + cover image from crayons-article__header
56
+ └── convert_to_markdown() → markdownify with ATX headings; collapses 3+ newlines to 2
50
57
  ```
51
58
 
52
59
  ### Router / Auto-Discovery
@@ -82,20 +89,26 @@ src/
82
89
  ├── base.py # BaseExtractor ABC + fetch_html() + extract() template
83
90
  └── providers/
84
91
  ├── __init__.py # Empty — auto-discovery handles registration
85
- └── medium.py # MediumExtractor
92
+ ├── medium.py # MediumExtractor
93
+ └── devto.py # DevToExtractor [002-devto-provider]
86
94
 
87
95
  tests/
88
96
  ├── unit/
89
97
  │ ├── test_router.py
90
98
  │ ├── test_medium_extractor.py
91
99
  │ ├── test_fetch_errors.py
92
- │ └── test_silent.py
100
+ │ ├── test_silent.py
101
+ │ └── test_devto_extractor.py # [002-devto-provider]
93
102
  └── integration/
94
103
  ├── snapshots/ # Golden Markdown files (article body snapshots)
95
104
  │ ├── from-drift-to-parity.md
96
105
  │ ├── architecting-the-asynchronous-agent.md
97
- │ └── integration-digest-december-2025.md
98
- └── test_medium_integration.py
106
+ │ ├── integration-digest-december-2025.md
107
+ │ ├── devto-integration-digest-december-2025.md # [002-devto-provider]
108
+ │ ├── devto-integration-digest-july-2025.md # [002-devto-provider]
109
+ │ └── devto-integration-digest-march-2026.md # [002-devto-provider]
110
+ ├── test_medium_integration.py
111
+ └── test_devto_integration.py # [002-devto-provider]
99
112
 
100
113
  specs/ # Speckit feature specifications
101
114
  pyproject.toml # hatchling build backend, uv package manager
@@ -118,14 +131,16 @@ Makefile # setup / test / integration / lint / typecheck / f
118
131
 
119
132
  ## Testing Strategy
120
133
 
121
- **Unit tests** (30 tests, offline):
134
+ **Unit tests** (47 tests, offline):
122
135
  - Router: domain routing, subdomain suffix matching, duplicate registration, invalid URLs, unsupported platforms
123
136
  - MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages
137
+ - DevToExtractor: clean_html (title/cover/heading/image preservation, iframe/ltag embed→link, anchor stripping, non-article error), convert_to_markdown (headings/code/lists/images, no raw HTML, empty content error) [002-devto-provider]
124
138
  - Fetch errors: HTTP 404, 503, timeout, connection error, size limit exceeded
125
139
  - Silent: no stdout/stderr output, no logging during extraction
126
140
 
127
- **Integration tests** (3 tests, network required):
141
+ **Integration tests** (6 tests, network required):
128
142
  - Parametrized over 3 real stn1slv.medium.com articles
143
+ - Parametrized over 3 real dev.to/stn1slv articles [002-devto-provider]
129
144
  - Snapshot-based containment check: `expected_body in extracted_result`
130
145
  - 3 retries with 2-second delay on `FetchError` (covers transient 403s/timeouts)
131
146
  - Run with: `make integration` or `uv run pytest tests/integration/ --override-ini=addopts=`
@@ -155,10 +170,14 @@ Makefile # setup / test / integration / lint / typecheck / f
155
170
  |----------|--------|-----------|
156
171
  | HTTP client | `httpx` (sync) | Supports async in v2 without swapping dependency |
157
172
  | HTML parser | `lxml` | Fast, tolerant; BeautifulSoup backend |
158
- | Article targeting | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
173
+ | Article targeting (Medium) | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
174
+ | Article targeting (dev.to) | `div#article-body` | dev.to does not use `<article>`; the id-scoped div already excludes all chrome | [002-devto-provider]
175
+ | dev.to cover image | Extracted from `header.crayons-article__header`, prepended to body | Cover image is outside the article body div — must be fetched separately | [002-devto-provider]
176
+ | dev.to embed handling | Replace `<iframe>` and `ltag__*` divs with anchor links | Embeds must not be silently dropped (FR-019); plain links are durable | [002-devto-provider]
159
177
  | Markdown converter | `markdownify` with ATX + `code_language=""` | Fenced code blocks, standard headings |
160
178
  | Routing | `pkgutil.iter_modules` auto-discovery + `@register` | SC-006: one new file = one new platform |
161
179
  | Integration tests | Snapshot containment + retry | Durable against minor HTML changes; resilient to transient 403s |
180
+ | test_router.py domain example | Changed from `dev.to` to `substack.com` for "unsupported domain" test | Once DevToExtractor registers `dev.to`, those tests would no longer raise UnsupportedPlatformError | [002-devto-provider]
162
181
 
163
182
  ---
164
183
 
@@ -1,13 +1,13 @@
1
1
  # mdfetch — Main Specification
2
2
 
3
3
  **Last Updated**: 2026-05-14
4
- **Sources**: [specs/001-mdfetch-medium-extractor/spec.md]
4
+ **Sources**: [specs/001-mdfetch-medium-extractor/spec.md], [specs/002-devto-provider/spec.md]
5
5
 
6
6
  ---
7
7
 
8
8
  ## Overview
9
9
 
10
- `mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider. The initial release supports `medium.com` and its subdomains only.
10
+ `mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider. Supported platforms: `medium.com` (and subdomains), `dev.to`.
11
11
 
12
12
  ---
13
13
 
@@ -48,6 +48,41 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
48
48
 
49
49
  ---
50
50
 
51
+ ### US-004 — Extract a dev.to Article to Markdown (P1)
52
+ [Source: specs/002-devto-provider]
53
+
54
+ A developer retrieves the readable content of a dev.to article as clean Markdown by calling the same single library function used for Medium — passing only the article URL, with no additional configuration. The library routes the request to the dev.to provider, fetches the page, isolates the article body, strips all non-content elements, and returns formatted Markdown.
55
+
56
+ **Acceptance Scenarios**:
57
+ 1. Given a valid, publicly accessible dev.to article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
58
+ 2. Given a valid dev.to article URL, the returned Markdown contains no navigation menus, reaction buttons, comment sections, or author sidebar widgets.
59
+ 3. Given a dev.to article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
60
+
61
+ ---
62
+
63
+ ### US-005 — Receive Meaningful Errors for Non-Article dev.to Pages (P2)
64
+ [Source: specs/002-devto-provider]
65
+
66
+ A developer passing a dev.to URL that points to a profile page, a tag listing, or a podcast page receives a clear, typed error indicating the dev.to domain is recognised but the page type is not extractable.
67
+
68
+ **Acceptance Scenarios**:
69
+ 1. Given a dev.to URL pointing to an author profile page, `extract()` raises `UnsupportedContentTypeError` (domain recognised, content type not extractable).
70
+ 2. Given a dev.to URL pointing to a tag listing page (e.g., `dev.to/t/kafka`), `extract()` raises `UnsupportedContentTypeError`.
71
+ 3. Given a URL from a domain other than dev.to, `extract()` raises `UnsupportedPlatformError` (unchanged from existing behaviour).
72
+
73
+ ---
74
+
75
+ ### US-006 — Integration Tests Pass Against Real dev.to Article URLs (P3)
76
+ [Source: specs/002-devto-provider]
77
+
78
+ A developer runs the integration test suite and all dev.to integration tests pass against three reference articles. The tests confirm title, structural elements, and absence of HTML tags on live content.
79
+
80
+ **Acceptance Scenarios**:
81
+ 1. When integration tests for the three reference dev.to URLs execute with a stable internet connection, all three pass without errors or assertion failures.
82
+ 2. For each reference dev.to article, the extraction function returns non-empty Markdown free of HTML tags and containing recognisable content from the article.
83
+
84
+ ---
85
+
51
86
  ## Functional Requirements
52
87
 
53
88
  ### Extraction
@@ -70,7 +105,14 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
70
105
 
71
106
  ### Packaging & Testing
72
107
  - **FR-009**: The library MUST be packaged and distributed via PyPI using modern Python packaging best practices, enabling installation through the standard package manager without additional steps.
73
- - **FR-010**: The test suite MUST include integration tests that supply real Medium article URLs to the extraction function and assert that the output matches expected Markdown structure and content.
108
+ - **FR-010**: The test suite MUST include integration tests that supply real article URLs (Medium and dev.to) to the extraction function and assert that the output matches expected Markdown structure and content.
109
+
110
+ ### dev.to Platform
111
+ - **FR-015**: The library MUST add `dev.to` to the provider router so that any URL with the `dev.to` domain is dispatched to the dev.to provider without any change to the caller's code. [Source: specs/002-devto-provider]
112
+ - **FR-016**: The library MUST include a dev.to provider that fetches the article page, isolates the main article body from `<div id="article-body">`, removes all non-content elements (navigation, social reaction widgets, comments, author sidebar, tag links), and returns the body as Markdown. [Source: specs/002-devto-provider]
113
+ - **FR-017**: The dev.to provider MUST produce Markdown that preserves both the cover image (from the article header) and all inline body images as Markdown image syntax (`![alt](url)`). [Source: specs/002-devto-provider]
114
+ - **FR-018**: The dev.to provider MUST raise `UnsupportedContentTypeError` when a `dev.to` URL is provided but the page is not an article (e.g., an author profile, a tag listing, or an organisation page). [Source: specs/002-devto-provider]
115
+ - **FR-019**: The dev.to provider MUST replace embedded third-party content (GitHub Gists, CodePen demos, YouTube videos, liquid-tag embeds) with a plain Markdown link to the embedded resource URL. Embeds must not be silently dropped. [Source: specs/002-devto-provider]
74
116
 
75
117
  ---
76
118
 
@@ -89,9 +131,9 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
89
131
  ### Provider (Internal)
90
132
  | Attribute | Type | Description |
91
133
  |-----------|------|-------------|
92
- | `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`) |
134
+ | `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`, `{"dev.to"}`) |
93
135
 
94
- **Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent.
136
+ **Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent. Registered providers: `MediumExtractor` (medium.com), `DevToExtractor` (dev.to).
95
137
 
96
138
  ### ExtractionResult (Output)
97
139
  | Attribute | Type | Description |
@@ -150,6 +192,11 @@ caller provides URL string
150
192
  - **Oversized responses**: Responses exceeding 10 MB are rejected with `FetchError` to prevent OOM.
151
193
  - **Profile/tag pages**: When a `medium.com` URL points to a non-article page, `UnsupportedContentTypeError` is raised (distinct from `UnsupportedPlatformError`).
152
194
  - **HTTP 403 / transient failures**: Integration tests use a 3-retry helper with 2-second delay to handle transient rate limits.
195
+ - **dev.to profile pages**: When a `dev.to` URL points to an author profile (no `div#article-body`), `UnsupportedContentTypeError` is raised.
196
+ - **dev.to tag listing pages**: When a `dev.to` URL points to a tag page (e.g., `dev.to/t/kafka`), `UnsupportedContentTypeError` is raised.
197
+ - **dev.to liquid-tag embeds**: Embedded third-party widgets (GitHub Gists, CodePen, YouTube) serialised as `<div class="ltag__*" data-url="...">` are replaced with plain Markdown links; they are never silently dropped.
198
+ - **dev.to cover image**: The cover image lives in `<header class="crayons-article__header">`, not in `div#article-body` — the provider explicitly extracts and prepends it.
199
+ - **dev.to HTML structure changes**: If `div#article-body` is absent, `UnsupportedContentTypeError` is raised.
153
200
 
154
201
  ---
155
202
 
@@ -161,6 +208,12 @@ caller provides URL string
161
208
  - **SC-004**: The returned Markdown for an article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax.
162
209
  - **SC-005**: 100% of integration tests pass against a curated set of real Medium article URLs at the time of initial release.
163
210
  - **SC-006**: Adding support for a second platform requires creating exactly one new file (one new provider class decorated with `@register`). The router auto-discovers all provider modules at import time; no changes to shared library code are required.
211
+ - **SC-007**: A developer already using the library for Medium can extract a dev.to article without any code change — only the URL changes. [Source: specs/002-devto-provider]
212
+ - **SC-008**: The extraction function returns a result for a standard dev.to article in under 10 seconds on a stable internet connection. [Source: specs/002-devto-provider]
213
+ - **SC-009**: The returned Markdown for any of the three reference dev.to articles contains no raw HTML tags (verified by automated assertion). [Source: specs/002-devto-provider]
214
+ - **SC-010**: The returned Markdown for a dev.to article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax. [Source: specs/002-devto-provider]
215
+ - **SC-011**: 100% of integration tests pass against the three provided reference dev.to article URLs at the time of release. [Source: specs/002-devto-provider]
216
+ - **SC-012**: The dev.to provider is delivered as exactly one new file; no existing source files are modified (except `test_router.py` for expected domain-example maintenance when the provider registers `dev.to`). [Source: specs/002-devto-provider]
164
217
 
165
218
  ---
166
219
 
@@ -45,6 +45,7 @@ uv run mypy src/ # type check
45
45
 
46
46
  ## Recent Changes
47
47
 
48
+ - 002-devto-provider: dev.to article extraction provider — `DevToExtractor`, cover image + embed→link handling, 17 new unit tests, 3 integration tests, version 0.2.0
48
49
  - 001-mdfetch-medium-extractor: Initial library release — `extract()` API, Medium provider, typed exceptions, auto-discovery routing, PyPI packaging, snapshot integration tests
49
50
 
50
51
  <!-- SPECKIT START -->
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mdfetch
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Extract article content from web platforms and return it as clean Markdown.
5
5
  Project-URL: Homepage, https://github.com/stn1slv/md-fetch
6
6
  Project-URL: Source, https://github.com/stn1slv/md-fetch
@@ -8,7 +8,7 @@ Project-URL: Issues, https://github.com/stn1slv/md-fetch/issues
8
8
  Author-email: Stanislav Deviatov <devyatov@gmail.com>
9
9
  License: MIT
10
10
  License-File: LICENSE
11
- Keywords: article,extraction,markdown,medium,scraping
11
+ Keywords: article,dev.to,extraction,markdown,medium,scraping
12
12
  Classifier: Development Status :: 3 - Alpha
13
13
  Classifier: Intended Audience :: Developers
14
14
  Classifier: License :: OSI Approved :: MIT License
@@ -44,7 +44,9 @@ pip install mdfetch
44
44
  ```python
45
45
  from mdfetch import extract
46
46
 
47
+ # Works with any supported platform — just pass the URL
47
48
  markdown = extract("https://medium.com/some-publication/article-slug-abc123")
49
+ markdown = extract("https://dev.to/username/article-slug")
48
50
  print(markdown)
49
51
  ```
50
52
 
@@ -84,6 +86,7 @@ except EmptyContentError as e:
84
86
  | Platform | Domains |
85
87
  |----------|---------|
86
88
  | Medium | `medium.com`, `*.medium.com` |
89
+ | dev.to | `dev.to` |
87
90
 
88
91
  ## Development
89
92
 
@@ -13,7 +13,9 @@ pip install mdfetch
13
13
  ```python
14
14
  from mdfetch import extract
15
15
 
16
+ # Works with any supported platform — just pass the URL
16
17
  markdown = extract("https://medium.com/some-publication/article-slug-abc123")
18
+ markdown = extract("https://dev.to/username/article-slug")
17
19
  print(markdown)
18
20
  ```
19
21
 
@@ -53,6 +55,7 @@ except EmptyContentError as e:
53
55
  | Platform | Domains |
54
56
  |----------|---------|
55
57
  | Medium | `medium.com`, `*.medium.com` |
58
+ | dev.to | `dev.to` |
56
59
 
57
60
  ## Development
58
61
 
@@ -4,12 +4,12 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "mdfetch"
7
- version = "0.1.0"
7
+ version = "0.2.0"
8
8
  description = "Extract article content from web platforms and return it as clean Markdown."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
11
11
  authors = [{ name = "Stanislav Deviatov", email = "devyatov@gmail.com" }]
12
- keywords = ["markdown", "scraping", "medium", "article", "extraction"]
12
+ keywords = ["markdown", "scraping", "medium", "dev.to", "article", "extraction"]
13
13
  classifiers = [
14
14
  "Development Status :: 3 - Alpha",
15
15
  "Intended Audience :: Developers",
@@ -0,0 +1,34 @@
1
+ # Specification Quality Checklist: mdfetch — dev.to Extractor
2
+
3
+ **Purpose**: Validate specification completeness and quality before proceeding to planning
4
+ **Created**: 2026-05-14
5
+ **Feature**: [spec.md](../spec.md)
6
+
7
+ ## Content Quality
8
+
9
+ - [x] No implementation details (languages, frameworks, APIs)
10
+ - [x] Focused on user value and business needs
11
+ - [x] Written for non-technical stakeholders
12
+ - [x] All mandatory sections completed
13
+
14
+ ## Requirement Completeness
15
+
16
+ - [x] No [NEEDS CLARIFICATION] markers remain
17
+ - [x] Requirements are testable and unambiguous
18
+ - [x] Success criteria are measurable
19
+ - [x] Success criteria are technology-agnostic (no implementation details)
20
+ - [x] All acceptance scenarios are defined
21
+ - [x] Edge cases are identified
22
+ - [x] Scope is clearly bounded
23
+ - [x] Dependencies and assumptions identified
24
+
25
+ ## Feature Readiness
26
+
27
+ - [x] All functional requirements have clear acceptance criteria
28
+ - [x] User scenarios cover primary flows
29
+ - [x] Feature meets measurable outcomes defined in Success Criteria
30
+ - [x] No implementation details leak into specification
31
+
32
+ ## Notes
33
+
34
+ - All items pass. Specification is ready for `/speckit-plan`.
@@ -0,0 +1,78 @@
1
+ # Public API Contract: mdfetch — dev.to Extractor
2
+
3
+ **Branch**: `002-devto-provider` | **Date**: 2026-05-14
4
+
5
+ ## Overview
6
+
7
+ The dev.to extractor introduces no new public API surface. The existing `mdfetch.extract()` function is the sole entry point — callers use it identically for Medium and dev.to URLs.
8
+
9
+ ## `mdfetch.extract(url, *, retries=3, retry_delay=2.0) -> str`
10
+
11
+ ```python
12
+ from mdfetch import extract
13
+
14
+ markdown = extract("https://dev.to/stn1slv/integration-digest-for-december-2025-5dlp")
15
+ ```
16
+
17
+ ### Parameters
18
+
19
+ | Parameter | Type | Default | Description |
20
+ |-----------|------|---------|-------------|
21
+ | `url` | `str` | required | Full URL of a dev.to article |
22
+ | `retries` | `int` | `3` | Number of fetch attempts before raising |
23
+ | `retry_delay` | `float` | `2.0` | Seconds between retry attempts |
24
+
25
+ ### Return value
26
+
27
+ A `str` containing the article's content as Markdown. Structure:
28
+
29
+ ```markdown
30
+ # Article Title
31
+
32
+ ![Cover image for Article Title](https://...)
33
+
34
+ ## Section Heading
35
+
36
+ Paragraph text with [links](https://...).
37
+
38
+ - List item
39
+ - List item
40
+
41
+ ```python
42
+ code block
43
+ ```
44
+
45
+ [https://gist.github.com/...](https://gist.github.com/...)
46
+ ```
47
+
48
+ ### Exceptions
49
+
50
+ All exceptions inherit from `mdfetch.MdfetchError` and carry `.url` and `.message` attributes.
51
+
52
+ | Exception | When raised |
53
+ |-----------|------------|
54
+ | `InvalidURLError` | `url` is syntactically invalid |
55
+ | `UnsupportedPlatformError` | Domain is not `dev.to` (or other registered domain) |
56
+ | `UnsupportedContentTypeError` | `dev.to` URL is not an article (profile, tag, etc.) |
57
+ | `FetchError` | Network error or timeout |
58
+ | `HTTPStatusError` | Non-2xx HTTP response (adds `.status_code`) |
59
+ | `EmptyContentError` | Article body found but contains no text |
60
+
61
+ ### Routing
62
+
63
+ The router auto-discovers `DevToExtractor` at import time. No configuration required. Adding the provider file to `src/mdfetch/providers/devto.py` is sufficient for routing to work.
64
+
65
+ ## `DevToExtractor` (internal)
66
+
67
+ Not part of the public API. Exposed only for testing via direct instantiation.
68
+
69
+ ```python
70
+ from mdfetch.providers.devto import DevToExtractor
71
+
72
+ extractor = DevToExtractor()
73
+ html = extractor.fetch_html(url)
74
+ from bs4 import BeautifulSoup
75
+ soup = BeautifulSoup(html, "lxml")
76
+ cleaned = extractor.clean_html(soup)
77
+ markdown = extractor.convert_to_markdown(cleaned)
78
+ ```
@@ -0,0 +1,55 @@
1
+ # Data Model: mdfetch — dev.to Extractor
2
+
3
+ **Branch**: `002-devto-provider` | **Date**: 2026-05-14
4
+
5
+ ## Entities
6
+
7
+ This feature adds no new data structures. All entities are shared with the existing library model.
8
+
9
+ ### URL
10
+ - **Representation**: Plain Python `str`
11
+ - **Routing key**: `hostname` extracted via `urllib.parse.urlparse`; dev.to provider registers `"dev.to"` — the router's existing subdomain suffix matching handles `www.dev.to` automatically
12
+ - **Validation**: Existing `InvalidURLError` raised by the router before the provider is invoked
13
+ - **Accepted forms**:
14
+ - `https://dev.to/<username>/<article-slug>` → article
15
+ - `https://dev.to/<username>` → profile → `UnsupportedContentTypeError`
16
+ - `https://dev.to/t/<tag>` → tag listing → `UnsupportedContentTypeError`
17
+
18
+ ### Provider
19
+ - **New provider**: `DevToExtractor` implementing `BaseExtractor`
20
+ - **DOMAINS**: `frozenset({"dev.to"})`
21
+ - **Registered via**: `@register` decorator (auto-discovered by `_autodiscover_providers()`)
22
+ - **File**: `src/mdfetch/providers/devto.py` (one new file; no existing files modified)
23
+
24
+ ### Extracted Article
25
+ - **Output**: `str` — clean Markdown with ATX-style headings
26
+ - **Content included**:
27
+ - Title (`# Title` from article `<h1>`)
28
+ - Cover image as `![alt](url)` (if present)
29
+ - Article body: paragraphs, headings, code blocks, inline code, lists, blockquotes
30
+ - Body images as `![alt](url)`
31
+ - Embedded resources (Gists, CodePens, YouTube) as plain Markdown links `[url](url)`
32
+ - **Content excluded**:
33
+ - Author name, avatar, publication date
34
+ - Series navigation panels
35
+ - Article tags (`#kafka`, `#api`, etc.)
36
+ - Comments section
37
+ - Reaction buttons
38
+ - Advertisements
39
+
40
+ ## State Transitions
41
+
42
+ None — extraction is a stateless, single-call operation with no persistent state.
43
+
44
+ ## Error Cases
45
+
46
+ | Condition | Exception |
47
+ |-----------|-----------|
48
+ | URL has no `dev.to` domain | `UnsupportedPlatformError` (raised by router) |
49
+ | URL is syntactically invalid | `InvalidURLError` (raised by router) |
50
+ | HTTP request fails | `FetchError` |
51
+ | Non-2xx HTTP response | `HTTPStatusError` |
52
+ | `div#article-body` not found | `UnsupportedContentTypeError` |
53
+ | Article body has no extractable text | `EmptyContentError` |
54
+
55
+ All exception classes are defined in `mdfetch.exceptions` — no new classes needed.
@@ -0,0 +1,165 @@
1
+ # Implementation Plan: mdfetch — dev.to Extractor
2
+
3
+ **Branch**: `002-devto-provider` | **Date**: 2026-05-14 | **Spec**: [spec.md](spec.md)
4
+
5
+ **Input**: Feature specification from `specs/002-devto-provider/spec.md`
6
+
7
+ ## Summary
8
+
9
+ Add a `DevToExtractor` provider that fetches dev.to article pages, extracts the article body from `<div id="article-body">`, prepends the title and cover image from the page header, converts embedded resources to plain Markdown links, and returns clean Markdown. The provider is one new file; no shared library code is modified.
10
+
11
+ ## Technical Context
12
+
13
+ **Language/Version**: Python 3.12+
14
+
15
+ **Primary Dependencies**: `httpx` (HTTP fetch), `BeautifulSoup` / `lxml` (HTML parsing), `markdownify` (Markdown conversion), `pytest` (testing)
16
+
17
+ **Storage**: N/A — stateless extraction library
18
+
19
+ **Testing**: `pytest` — unit tests (no network) + integration tests (`-m integration`, real URLs)
20
+
21
+ **Target Platform**: PyPI library (cross-platform)
22
+
23
+ **Project Type**: library
24
+
25
+ **Performance Goals**: Extraction in under 10 seconds per article on stable internet (SC-002)
26
+
27
+ **Constraints**: One new provider file; no changes to existing shared code
28
+
29
+ **Scale/Scope**: Single-article extraction per call; no volume concerns
30
+
31
+ ## Constitution Check
32
+
33
+ *GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.*
34
+
35
+ - [x] Validates Provider Pattern Architecture — `DevToExtractor` inherits `BaseExtractor`; uses `@register`; one new file only; no shared code changes
36
+ - [x] Confirms Technology Stack — `httpx`, `BeautifulSoup`, `markdownify`, `pytest` (all already in use; no new deps needed)
37
+ - [x] Adheres to Coding Standards — strict type hints, PEP 8, clear English naming
38
+ - [x] Incorporates Integration Testing — three reference dev.to URLs verified in `tests/integration/test_devto_integration.py`
39
+ - [x] Respects Packaging and Distribution standards — `pyproject.toml` + `src/` layout unchanged; `uv` for all dev commands
40
+
41
+ All constitution gates pass. No violations to justify.
42
+
43
+ ## Project Structure
44
+
45
+ ### Documentation (this feature)
46
+
47
+ ```text
48
+ specs/002-devto-provider/
49
+ ├── plan.md # This file
50
+ ├── research.md # Phase 0 output — dev.to HTML structure analysis
51
+ ├── data-model.md # Phase 1 output — entities and error cases
52
+ ├── quickstart.md # Phase 1 output — usage examples
53
+ ├── contracts/
54
+ │ └── public-api.md # Phase 1 output — API contract (unchanged from Medium)
55
+ └── tasks.md # Phase 2 output (/speckit-tasks — not created here)
56
+ ```
57
+
58
+ ### Source Code Changes
59
+
60
+ ```text
61
+ src/mdfetch/providers/
62
+ ├── __init__.py # UNCHANGED — auto-discovery handles new file
63
+ └── devto.py # NEW — DevToExtractor (single new file)
64
+
65
+ tests/unit/
66
+ └── test_devto_extractor.py # NEW — unit tests (no network)
67
+
68
+ tests/integration/
69
+ ├── snapshots/
70
+ │ ├── devto-integration-digest-december-2025.md # NEW — snapshot
71
+ │ ├── devto-integration-digest-july-2025.md # NEW — snapshot
72
+ │ └── devto-integration-digest-march-2026.md # NEW — snapshot
73
+ └── test_devto_integration.py # NEW — integration tests (real URLs)
74
+ ```
75
+
76
+ No existing library source files are modified (`src/mdfetch/` unchanged).
77
+
78
+ **Side effect documented**: `tests/unit/test_router.py` required a one-line update — its "unsupported domain" example used `dev.to`, which became a supported domain once this provider was registered. The two test cases were updated to use `substack.com` instead. This is an expected consequence of any new provider registration.
79
+
80
+ **Version**: Library version bumped from `0.1.0` to `0.2.0` in `pyproject.toml` at time of release.
81
+
82
+ ## Implementation Specification
83
+
84
+ ### `src/mdfetch/providers/devto.py`
85
+
86
+ ```python
87
+ @register
88
+ class DevToExtractor(BaseExtractor):
89
+ DOMAINS: frozenset[str] = frozenset({"dev.to"})
90
+
91
+ def clean_html(self, soup: BeautifulSoup) -> Tag:
92
+ # 1. Locate article body — absence means non-article page
93
+ body = soup.find("div", id="article-body")
94
+ if not isinstance(body, Tag):
95
+ raise UnsupportedContentTypeError(...)
96
+
97
+ # 2. Handle embedded content: replace iframes with anchor links
98
+ for iframe in body.find_all("iframe"):
99
+ src = iframe.get("src") or iframe.get("data-src") or ""
100
+ link = soup.new_tag("a", href=src)
101
+ link.string = src
102
+ iframe.replace_with(link)
103
+
104
+ # 3. Handle liquid tag embeds (class containing "ltag")
105
+ for embed in body.find_all(class_=re.compile(r"ltag", re.I)):
106
+ src = embed.get("data-src") or embed.get("src") or str(embed.get("data-url", ""))
107
+ if src:
108
+ link = soup.new_tag("a", href=src)
109
+ link.string = src
110
+ embed.replace_with(link)
111
+ else:
112
+ embed.decompose()
113
+
114
+ # 4. Strip empty anchor-name links inserted before headings
115
+ for anchor in body.find_all("a", attrs={"name": True}):
116
+ if not anchor.get_text(strip=True):
117
+ anchor.decompose()
118
+
119
+ # 5. Extract title and cover image from header; prepend to body
120
+ header = soup.find("header", class_="crayons-article__header")
121
+ if header:
122
+ h1 = header.find("h1")
123
+ cover_img = header.find("img", class_="crayons-article__cover__image")
124
+ if h1:
125
+ body.insert(0, h1) # prepend <h1> as first child of body
126
+ if cover_img:
127
+ body.insert(1, cover_img) # prepend cover image after title
128
+
129
+ return body
130
+
131
+ def convert_to_markdown(self, tag: Tag) -> str:
132
+ md = markdownify(str(tag), heading_style="ATX", code_language="", strip=["script", "style"])
133
+ md = md.strip()
134
+ md = re.sub(r"\n{3,}", "\n\n", md)
135
+ if not md:
136
+ raise EmptyContentError(...)
137
+ return md
138
+ ```
139
+
140
+ ### `tests/unit/test_devto_extractor.py`
141
+
142
+ Unit tests cover (no network):
143
+ - `clean_html` raises `UnsupportedContentTypeError` when `div#article-body` absent
144
+ - `clean_html` raises `UnsupportedContentTypeError` for profile-page HTML (no article-body)
145
+ - `clean_html` preserves `<h1>` title and cover image
146
+ - `clean_html` replaces `<iframe>` with anchor link
147
+ - `clean_html` strips empty anchor-name links
148
+ - `convert_to_markdown` produces ATX headings, code blocks, lists
149
+ - `convert_to_markdown` raises `EmptyContentError` on blank body
150
+ - No raw HTML tags in converted Markdown
151
+
152
+ ### `tests/integration/test_devto_integration.py`
153
+
154
+ Integration tests mirror the Medium pattern:
155
+ - Parametrized over 3 reference URLs + snapshot filenames
156
+ - `@pytest.mark.integration` marker
157
+ - Snapshot containment check: `expected in result`
158
+ - Snapshots generated by running extraction once and saving output
159
+
160
+ ## Complexity Tracking
161
+
162
+ *No constitution violations — section left intentionally blank.*
163
+
164
+ ### Revision: Implementation Sync 2026-05-14
165
+ - Reason: Documented two implementation side-effects not captured in the original plan: (1) `tests/unit/test_router.py` required domain-example update when dev.to became a registered provider; (2) library version bumped from 0.1.0 to 0.2.0 at release. `pyproject.toml` keywords gap (missing "dev.to") tracked as T013.