mdfetch 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mdfetch-0.2.0/.specify/feature.json +3 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/changelog.md +27 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/plan.md +27 -8
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/spec.md +58 -5
- {mdfetch-0.1.0 → mdfetch-0.2.0}/CLAUDE.md +1 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/PKG-INFO +5 -2
- {mdfetch-0.1.0 → mdfetch-0.2.0}/README.md +3 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/pyproject.toml +2 -2
- mdfetch-0.2.0/specs/002-devto-provider/checklists/requirements.md +34 -0
- mdfetch-0.2.0/specs/002-devto-provider/contracts/public-api.md +78 -0
- mdfetch-0.2.0/specs/002-devto-provider/data-model.md +55 -0
- mdfetch-0.2.0/specs/002-devto-provider/plan.md +165 -0
- mdfetch-0.2.0/specs/002-devto-provider/quickstart.md +57 -0
- mdfetch-0.2.0/specs/002-devto-provider/research.md +77 -0
- mdfetch-0.2.0/specs/002-devto-provider/spec.md +117 -0
- mdfetch-0.2.0/specs/002-devto-provider/tasks.md +205 -0
- mdfetch-0.2.0/src/mdfetch/providers/devto.py +80 -0
- mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-december-2025.md +135 -0
- mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-july-2025.md +111 -0
- mdfetch-0.2.0/tests/integration/snapshots/devto-integration-digest-march-2026.md +237 -0
- mdfetch-0.2.0/tests/integration/test_devto_integration.py +50 -0
- mdfetch-0.2.0/tests/unit/test_devto_extractor.py +255 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_router.py +9 -3
- {mdfetch-0.1.0 → mdfetch-0.2.0}/uv.lock +1 -1
- mdfetch-0.1.0/.specify/feature.json +0 -3
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-analyze/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-checklist/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-clarify/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-constitution/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-implement/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-plan/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-specify/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-tasks/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.analyze.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.archive.run.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.checklist.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.clarify.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.constitution.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.implement.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.plan.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.reconcile.run.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.specify.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.tasks.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gemini/commands/speckit.taskstoissues.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.github/workflows/ci.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.github/workflows/publish.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.gitignore +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.python-version +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/.registry +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/commands/archive.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/archive/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/config-template.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/git-config.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions/reconcile/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/extensions.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/init-options.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integration.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/claude.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/gemini.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/integrations/speckit.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/memory/constitution.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/check-prerequisites.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/common.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/setup-plan.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/scripts/bash/setup-tasks.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/checklist-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/constitution-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/plan-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/spec-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/templates/tasks-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/workflows/speckit/workflow.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/.specify/workflows/workflow-registry.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/GEMINI.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/Makefile +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/research.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/base.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/exceptions.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/providers/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/providers/medium.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/src/mdfetch/router.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/conftest.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/integration/test_medium_integration.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_fetch_errors.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_medium_extractor.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.0}/tests/unit/test_silent.py +0 -0
|
@@ -2,6 +2,33 @@
|
|
|
2
2
|
|
|
3
3
|
---
|
|
4
4
|
|
|
5
|
+
### mdfetch — dev.to Extractor — 2026-05-14
|
|
6
|
+
|
|
7
|
+
**Branch**: `002-devto-provider`
|
|
8
|
+
**Spec**: specs/002-devto-provider
|
|
9
|
+
|
|
10
|
+
**What was added**:
|
|
11
|
+
- `DevToExtractor` provider for `dev.to` articles, auto-discovered via `@register` decorator
|
|
12
|
+
- Article body isolation from `<div id="article-body">` with cover image extracted from `<header class="crayons-article__header">` and prepended to output
|
|
13
|
+
- `<iframe>` and liquid-tag embed (`ltag__*`) replacement with plain Markdown links (FR-019)
|
|
14
|
+
- `UnsupportedContentTypeError` raised for non-article dev.to pages (profiles, tag listings)
|
|
15
|
+
- 17 new unit tests in `tests/unit/test_devto_extractor.py`
|
|
16
|
+
- 3 dev.to integration tests in `tests/integration/test_devto_integration.py` with snapshot golden files
|
|
17
|
+
- Library version bumped from `0.1.0` to `0.2.0`
|
|
18
|
+
- `"dev.to"` added to `pyproject.toml` keywords (T013)
|
|
19
|
+
|
|
20
|
+
**New Components**:
|
|
21
|
+
- `src/mdfetch/providers/devto.py` — DevToExtractor
|
|
22
|
+
- `tests/unit/test_devto_extractor.py` — 17 unit tests
|
|
23
|
+
- `tests/integration/test_devto_integration.py` — 3 integration tests
|
|
24
|
+
- `tests/integration/snapshots/devto-integration-digest-december-2025.md`
|
|
25
|
+
- `tests/integration/snapshots/devto-integration-digest-july-2025.md`
|
|
26
|
+
- `tests/integration/snapshots/devto-integration-digest-march-2026.md`
|
|
27
|
+
|
|
28
|
+
**Tasks Completed**: 13/13
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
5
32
|
### mdfetch — Medium Extractor (Initial Release) — 2026-05-14
|
|
6
33
|
|
|
7
34
|
**Branch**: `feature/first-draft`
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# mdfetch — Main Implementation Plan
|
|
2
2
|
|
|
3
3
|
**Last Updated**: 2026-05-14
|
|
4
|
-
**Sources**: [specs/001-mdfetch-medium-extractor/plan.md]
|
|
4
|
+
**Sources**: [specs/001-mdfetch-medium-extractor/plan.md], [specs/002-devto-provider/plan.md]
|
|
5
5
|
|
|
6
6
|
---
|
|
7
7
|
|
|
@@ -47,6 +47,13 @@ MediumExtractor(BaseExtractor) — src/mdfetch/providers/medium.py
|
|
|
47
47
|
├── DOMAINS = frozenset({"medium.com"})
|
|
48
48
|
├── clean_html() → removes nav, clap buttons, sidebars, share elements, post-footer, author bio
|
|
49
49
|
└── convert_to_markdown() → markdownify with ATX headings, fenced code blocks
|
|
50
|
+
|
|
51
|
+
DevToExtractor(BaseExtractor) — src/mdfetch/providers/devto.py
|
|
52
|
+
├── DOMAINS = frozenset({"dev.to"})
|
|
53
|
+
├── clean_html() → locates div#article-body; raises UnsupportedContentTypeError if absent;
|
|
54
|
+
│ replaces iframes and ltag embeds with anchor links; strips empty anchor-name
|
|
55
|
+
│ elements; prepends h1 + cover image from crayons-article__header
|
|
56
|
+
└── convert_to_markdown() → markdownify with ATX headings; collapses 3+ newlines to 2
|
|
50
57
|
```
|
|
51
58
|
|
|
52
59
|
### Router / Auto-Discovery
|
|
@@ -82,20 +89,26 @@ src/
|
|
|
82
89
|
├── base.py # BaseExtractor ABC + fetch_html() + extract() template
|
|
83
90
|
└── providers/
|
|
84
91
|
├── __init__.py # Empty — auto-discovery handles registration
|
|
85
|
-
|
|
92
|
+
├── medium.py # MediumExtractor
|
|
93
|
+
└── devto.py # DevToExtractor [002-devto-provider]
|
|
86
94
|
|
|
87
95
|
tests/
|
|
88
96
|
├── unit/
|
|
89
97
|
│ ├── test_router.py
|
|
90
98
|
│ ├── test_medium_extractor.py
|
|
91
99
|
│ ├── test_fetch_errors.py
|
|
92
|
-
│
|
|
100
|
+
│ ├── test_silent.py
|
|
101
|
+
│ └── test_devto_extractor.py # [002-devto-provider]
|
|
93
102
|
└── integration/
|
|
94
103
|
├── snapshots/ # Golden Markdown files (article body snapshots)
|
|
95
104
|
│ ├── from-drift-to-parity.md
|
|
96
105
|
│ ├── architecting-the-asynchronous-agent.md
|
|
97
|
-
│
|
|
98
|
-
|
|
106
|
+
│ ├── integration-digest-december-2025.md
|
|
107
|
+
│ ├── devto-integration-digest-december-2025.md # [002-devto-provider]
|
|
108
|
+
│ ├── devto-integration-digest-july-2025.md # [002-devto-provider]
|
|
109
|
+
│ └── devto-integration-digest-march-2026.md # [002-devto-provider]
|
|
110
|
+
├── test_medium_integration.py
|
|
111
|
+
└── test_devto_integration.py # [002-devto-provider]
|
|
99
112
|
|
|
100
113
|
specs/ # Speckit feature specifications
|
|
101
114
|
pyproject.toml # hatchling build backend, uv package manager
|
|
@@ -118,14 +131,16 @@ Makefile # setup / test / integration / lint / typecheck / f
|
|
|
118
131
|
|
|
119
132
|
## Testing Strategy
|
|
120
133
|
|
|
121
|
-
**Unit tests** (
|
|
134
|
+
**Unit tests** (47 tests, offline):
|
|
122
135
|
- Router: domain routing, subdomain suffix matching, duplicate registration, invalid URLs, unsupported platforms
|
|
123
136
|
- MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages
|
|
137
|
+
- DevToExtractor: clean_html (title/cover/heading/image preservation, iframe/ltag embed→link, anchor stripping, non-article error), convert_to_markdown (headings/code/lists/images, no raw HTML, empty content error) [002-devto-provider]
|
|
124
138
|
- Fetch errors: HTTP 404, 503, timeout, connection error, size limit exceeded
|
|
125
139
|
- Silent: no stdout/stderr output, no logging during extraction
|
|
126
140
|
|
|
127
|
-
**Integration tests** (
|
|
141
|
+
**Integration tests** (6 tests, network required):
|
|
128
142
|
- Parametrized over 3 real stn1slv.medium.com articles
|
|
143
|
+
- Parametrized over 3 real dev.to/stn1slv articles [002-devto-provider]
|
|
129
144
|
- Snapshot-based containment check: `expected_body in extracted_result`
|
|
130
145
|
- 3 retries with 2-second delay on `FetchError` (covers transient 403s/timeouts)
|
|
131
146
|
- Run with: `make integration` or `uv run pytest tests/integration/ --override-ini=addopts=`
|
|
@@ -155,10 +170,14 @@ Makefile # setup / test / integration / lint / typecheck / f
|
|
|
155
170
|
|----------|--------|-----------|
|
|
156
171
|
| HTTP client | `httpx` (sync) | Supports async in v2 without swapping dependency |
|
|
157
172
|
| HTML parser | `lxml` | Fast, tolerant; BeautifulSoup backend |
|
|
158
|
-
| Article targeting | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
|
|
173
|
+
| Article targeting (Medium) | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
|
|
174
|
+
| Article targeting (dev.to) | `div#article-body` | dev.to does not use `<article>`; the id-scoped div already excludes all chrome | [002-devto-provider]
|
|
175
|
+
| dev.to cover image | Extracted from `header.crayons-article__header`, prepended to body | Cover image is outside the article body div — must be fetched separately | [002-devto-provider]
|
|
176
|
+
| dev.to embed handling | Replace `<iframe>` and `ltag__*` divs with anchor links | Embeds must not be silently dropped (FR-019); plain links are durable | [002-devto-provider]
|
|
159
177
|
| Markdown converter | `markdownify` with ATX + `code_language=""` | Fenced code blocks, standard headings |
|
|
160
178
|
| Routing | `pkgutil.iter_modules` auto-discovery + `@register` | SC-006: one new file = one new platform |
|
|
161
179
|
| Integration tests | Snapshot containment + retry | Durable against minor HTML changes; resilient to transient 403s |
|
|
180
|
+
| test_router.py domain example | Changed from `dev.to` to `substack.com` for "unsupported domain" test | Once DevToExtractor registers `dev.to`, those tests would no longer raise UnsupportedPlatformError | [002-devto-provider]
|
|
162
181
|
|
|
163
182
|
---
|
|
164
183
|
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
# mdfetch — Main Specification
|
|
2
2
|
|
|
3
3
|
**Last Updated**: 2026-05-14
|
|
4
|
-
**Sources**: [specs/001-mdfetch-medium-extractor/spec.md]
|
|
4
|
+
**Sources**: [specs/001-mdfetch-medium-extractor/spec.md], [specs/002-devto-provider/spec.md]
|
|
5
5
|
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
## Overview
|
|
9
9
|
|
|
10
|
-
`mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider.
|
|
10
|
+
`mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider. Supported platforms: `medium.com` (and subdomains), `dev.to`.
|
|
11
11
|
|
|
12
12
|
---
|
|
13
13
|
|
|
@@ -48,6 +48,41 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
|
|
|
48
48
|
|
|
49
49
|
---
|
|
50
50
|
|
|
51
|
+
### US-004 — Extract a dev.to Article to Markdown (P1)
|
|
52
|
+
[Source: specs/002-devto-provider]
|
|
53
|
+
|
|
54
|
+
A developer retrieves the readable content of a dev.to article as clean Markdown by calling the same single library function used for Medium — passing only the article URL, with no additional configuration. The library routes the request to the dev.to provider, fetches the page, isolates the article body, strips all non-content elements, and returns formatted Markdown.
|
|
55
|
+
|
|
56
|
+
**Acceptance Scenarios**:
|
|
57
|
+
1. Given a valid, publicly accessible dev.to article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
|
|
58
|
+
2. Given a valid dev.to article URL, the returned Markdown contains no navigation menus, reaction buttons, comment sections, or author sidebar widgets.
|
|
59
|
+
3. Given a dev.to article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
|
|
60
|
+
|
|
61
|
+
---
|
|
62
|
+
|
|
63
|
+
### US-005 — Receive Meaningful Errors for Non-Article dev.to Pages (P2)
|
|
64
|
+
[Source: specs/002-devto-provider]
|
|
65
|
+
|
|
66
|
+
A developer passing a dev.to URL that points to a profile page, a tag listing, or a podcast page receives a clear, typed error indicating the dev.to domain is recognised but the page type is not extractable.
|
|
67
|
+
|
|
68
|
+
**Acceptance Scenarios**:
|
|
69
|
+
1. Given a dev.to URL pointing to an author profile page, `extract()` raises `UnsupportedContentTypeError` (domain recognised, content type not extractable).
|
|
70
|
+
2. Given a dev.to URL pointing to a tag listing page (e.g., `dev.to/t/kafka`), `extract()` raises `UnsupportedContentTypeError`.
|
|
71
|
+
3. Given a URL from a domain other than dev.to, `extract()` raises `UnsupportedPlatformError` (unchanged from existing behaviour).
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
### US-006 — Integration Tests Pass Against Real dev.to Article URLs (P3)
|
|
76
|
+
[Source: specs/002-devto-provider]
|
|
77
|
+
|
|
78
|
+
A developer runs the integration test suite and all dev.to integration tests pass against three reference articles. The tests confirm title, structural elements, and absence of HTML tags on live content.
|
|
79
|
+
|
|
80
|
+
**Acceptance Scenarios**:
|
|
81
|
+
1. When integration tests for the three reference dev.to URLs execute with a stable internet connection, all three pass without errors or assertion failures.
|
|
82
|
+
2. For each reference dev.to article, the extraction function returns non-empty Markdown free of HTML tags and containing recognisable content from the article.
|
|
83
|
+
|
|
84
|
+
---
|
|
85
|
+
|
|
51
86
|
## Functional Requirements
|
|
52
87
|
|
|
53
88
|
### Extraction
|
|
@@ -70,7 +105,14 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
|
|
|
70
105
|
|
|
71
106
|
### Packaging & Testing
|
|
72
107
|
- **FR-009**: The library MUST be packaged and distributed via PyPI using modern Python packaging best practices, enabling installation through the standard package manager without additional steps.
|
|
73
|
-
- **FR-010**: The test suite MUST include integration tests that supply real
|
|
108
|
+
- **FR-010**: The test suite MUST include integration tests that supply real article URLs (Medium and dev.to) to the extraction function and assert that the output matches expected Markdown structure and content.
|
|
109
|
+
|
|
110
|
+
### dev.to Platform
|
|
111
|
+
- **FR-015**: The library MUST add `dev.to` to the provider router so that any URL with the `dev.to` domain is dispatched to the dev.to provider without any change to the caller's code. [Source: specs/002-devto-provider]
|
|
112
|
+
- **FR-016**: The library MUST include a dev.to provider that fetches the article page, isolates the main article body from `<div id="article-body">`, removes all non-content elements (navigation, social reaction widgets, comments, author sidebar, tag links), and returns the body as Markdown. [Source: specs/002-devto-provider]
|
|
113
|
+
- **FR-017**: The dev.to provider MUST produce Markdown that preserves both the cover image (from the article header) and all inline body images as Markdown image syntax (``). [Source: specs/002-devto-provider]
|
|
114
|
+
- **FR-018**: The dev.to provider MUST raise `UnsupportedContentTypeError` when a `dev.to` URL is provided but the page is not an article (e.g., an author profile, a tag listing, or an organisation page). [Source: specs/002-devto-provider]
|
|
115
|
+
- **FR-019**: The dev.to provider MUST replace embedded third-party content (GitHub Gists, CodePen demos, YouTube videos, liquid-tag embeds) with a plain Markdown link to the embedded resource URL. Embeds must not be silently dropped. [Source: specs/002-devto-provider]
|
|
74
116
|
|
|
75
117
|
---
|
|
76
118
|
|
|
@@ -89,9 +131,9 @@ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `ex
|
|
|
89
131
|
### Provider (Internal)
|
|
90
132
|
| Attribute | Type | Description |
|
|
91
133
|
|-----------|------|-------------|
|
|
92
|
-
| `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`) |
|
|
134
|
+
| `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`, `{"dev.to"}`) |
|
|
93
135
|
|
|
94
|
-
**Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent.
|
|
136
|
+
**Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent. Registered providers: `MediumExtractor` (medium.com), `DevToExtractor` (dev.to).
|
|
95
137
|
|
|
96
138
|
### ExtractionResult (Output)
|
|
97
139
|
| Attribute | Type | Description |
|
|
@@ -150,6 +192,11 @@ caller provides URL string
|
|
|
150
192
|
- **Oversized responses**: Responses exceeding 10 MB are rejected with `FetchError` to prevent OOM.
|
|
151
193
|
- **Profile/tag pages**: When a `medium.com` URL points to a non-article page, `UnsupportedContentTypeError` is raised (distinct from `UnsupportedPlatformError`).
|
|
152
194
|
- **HTTP 403 / transient failures**: Integration tests use a 3-retry helper with 2-second delay to handle transient rate limits.
|
|
195
|
+
- **dev.to profile pages**: When a `dev.to` URL points to an author profile (no `div#article-body`), `UnsupportedContentTypeError` is raised.
|
|
196
|
+
- **dev.to tag listing pages**: When a `dev.to` URL points to a tag page (e.g., `dev.to/t/kafka`), `UnsupportedContentTypeError` is raised.
|
|
197
|
+
- **dev.to liquid-tag embeds**: Embedded third-party widgets (GitHub Gists, CodePen, YouTube) serialised as `<div class="ltag__*" data-url="...">` are replaced with plain Markdown links; they are never silently dropped.
|
|
198
|
+
- **dev.to cover image**: The cover image lives in `<header class="crayons-article__header">`, not in `div#article-body` — the provider explicitly extracts and prepends it.
|
|
199
|
+
- **dev.to HTML structure changes**: If `div#article-body` is absent, `UnsupportedContentTypeError` is raised.
|
|
153
200
|
|
|
154
201
|
---
|
|
155
202
|
|
|
@@ -161,6 +208,12 @@ caller provides URL string
|
|
|
161
208
|
- **SC-004**: The returned Markdown for an article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax.
|
|
162
209
|
- **SC-005**: 100% of integration tests pass against a curated set of real Medium article URLs at the time of initial release.
|
|
163
210
|
- **SC-006**: Adding support for a second platform requires creating exactly one new file (one new provider class decorated with `@register`). The router auto-discovers all provider modules at import time; no changes to shared library code are required.
|
|
211
|
+
- **SC-007**: A developer already using the library for Medium can extract a dev.to article without any code change — only the URL changes. [Source: specs/002-devto-provider]
|
|
212
|
+
- **SC-008**: The extraction function returns a result for a standard dev.to article in under 10 seconds on a stable internet connection. [Source: specs/002-devto-provider]
|
|
213
|
+
- **SC-009**: The returned Markdown for any of the three reference dev.to articles contains no raw HTML tags (verified by automated assertion). [Source: specs/002-devto-provider]
|
|
214
|
+
- **SC-010**: The returned Markdown for a dev.to article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax. [Source: specs/002-devto-provider]
|
|
215
|
+
- **SC-011**: 100% of integration tests pass against the three provided reference dev.to article URLs at the time of release. [Source: specs/002-devto-provider]
|
|
216
|
+
- **SC-012**: The dev.to provider is delivered as exactly one new file; no existing source files are modified (except `test_router.py` for expected domain-example maintenance when the provider registers `dev.to`). [Source: specs/002-devto-provider]
|
|
164
217
|
|
|
165
218
|
---
|
|
166
219
|
|
|
@@ -45,6 +45,7 @@ uv run mypy src/ # type check
|
|
|
45
45
|
|
|
46
46
|
## Recent Changes
|
|
47
47
|
|
|
48
|
+
- 002-devto-provider: dev.to article extraction provider — `DevToExtractor`, cover image + embed→link handling, 17 new unit tests, 3 integration tests, version 0.2.0
|
|
48
49
|
- 001-mdfetch-medium-extractor: Initial library release — `extract()` API, Medium provider, typed exceptions, auto-discovery routing, PyPI packaging, snapshot integration tests
|
|
49
50
|
|
|
50
51
|
<!-- SPECKIT START -->
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mdfetch
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Extract article content from web platforms and return it as clean Markdown.
|
|
5
5
|
Project-URL: Homepage, https://github.com/stn1slv/md-fetch
|
|
6
6
|
Project-URL: Source, https://github.com/stn1slv/md-fetch
|
|
@@ -8,7 +8,7 @@ Project-URL: Issues, https://github.com/stn1slv/md-fetch/issues
|
|
|
8
8
|
Author-email: Stanislav Deviatov <devyatov@gmail.com>
|
|
9
9
|
License: MIT
|
|
10
10
|
License-File: LICENSE
|
|
11
|
-
Keywords: article,extraction,markdown,medium,scraping
|
|
11
|
+
Keywords: article,dev.to,extraction,markdown,medium,scraping
|
|
12
12
|
Classifier: Development Status :: 3 - Alpha
|
|
13
13
|
Classifier: Intended Audience :: Developers
|
|
14
14
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -44,7 +44,9 @@ pip install mdfetch
|
|
|
44
44
|
```python
|
|
45
45
|
from mdfetch import extract
|
|
46
46
|
|
|
47
|
+
# Works with any supported platform — just pass the URL
|
|
47
48
|
markdown = extract("https://medium.com/some-publication/article-slug-abc123")
|
|
49
|
+
markdown = extract("https://dev.to/username/article-slug")
|
|
48
50
|
print(markdown)
|
|
49
51
|
```
|
|
50
52
|
|
|
@@ -84,6 +86,7 @@ except EmptyContentError as e:
|
|
|
84
86
|
| Platform | Domains |
|
|
85
87
|
|----------|---------|
|
|
86
88
|
| Medium | `medium.com`, `*.medium.com` |
|
|
89
|
+
| dev.to | `dev.to` |
|
|
87
90
|
|
|
88
91
|
## Development
|
|
89
92
|
|
|
@@ -13,7 +13,9 @@ pip install mdfetch
|
|
|
13
13
|
```python
|
|
14
14
|
from mdfetch import extract
|
|
15
15
|
|
|
16
|
+
# Works with any supported platform — just pass the URL
|
|
16
17
|
markdown = extract("https://medium.com/some-publication/article-slug-abc123")
|
|
18
|
+
markdown = extract("https://dev.to/username/article-slug")
|
|
17
19
|
print(markdown)
|
|
18
20
|
```
|
|
19
21
|
|
|
@@ -53,6 +55,7 @@ except EmptyContentError as e:
|
|
|
53
55
|
| Platform | Domains |
|
|
54
56
|
|----------|---------|
|
|
55
57
|
| Medium | `medium.com`, `*.medium.com` |
|
|
58
|
+
| dev.to | `dev.to` |
|
|
56
59
|
|
|
57
60
|
## Development
|
|
58
61
|
|
|
@@ -4,12 +4,12 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "mdfetch"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Extract article content from web platforms and return it as clean Markdown."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
11
11
|
authors = [{ name = "Stanislav Deviatov", email = "devyatov@gmail.com" }]
|
|
12
|
-
keywords = ["markdown", "scraping", "medium", "article", "extraction"]
|
|
12
|
+
keywords = ["markdown", "scraping", "medium", "dev.to", "article", "extraction"]
|
|
13
13
|
classifiers = [
|
|
14
14
|
"Development Status :: 3 - Alpha",
|
|
15
15
|
"Intended Audience :: Developers",
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Specification Quality Checklist: mdfetch — dev.to Extractor
|
|
2
|
+
|
|
3
|
+
**Purpose**: Validate specification completeness and quality before proceeding to planning
|
|
4
|
+
**Created**: 2026-05-14
|
|
5
|
+
**Feature**: [spec.md](../spec.md)
|
|
6
|
+
|
|
7
|
+
## Content Quality
|
|
8
|
+
|
|
9
|
+
- [x] No implementation details (languages, frameworks, APIs)
|
|
10
|
+
- [x] Focused on user value and business needs
|
|
11
|
+
- [x] Written for non-technical stakeholders
|
|
12
|
+
- [x] All mandatory sections completed
|
|
13
|
+
|
|
14
|
+
## Requirement Completeness
|
|
15
|
+
|
|
16
|
+
- [x] No [NEEDS CLARIFICATION] markers remain
|
|
17
|
+
- [x] Requirements are testable and unambiguous
|
|
18
|
+
- [x] Success criteria are measurable
|
|
19
|
+
- [x] Success criteria are technology-agnostic (no implementation details)
|
|
20
|
+
- [x] All acceptance scenarios are defined
|
|
21
|
+
- [x] Edge cases are identified
|
|
22
|
+
- [x] Scope is clearly bounded
|
|
23
|
+
- [x] Dependencies and assumptions identified
|
|
24
|
+
|
|
25
|
+
## Feature Readiness
|
|
26
|
+
|
|
27
|
+
- [x] All functional requirements have clear acceptance criteria
|
|
28
|
+
- [x] User scenarios cover primary flows
|
|
29
|
+
- [x] Feature meets measurable outcomes defined in Success Criteria
|
|
30
|
+
- [x] No implementation details leak into specification
|
|
31
|
+
|
|
32
|
+
## Notes
|
|
33
|
+
|
|
34
|
+
- All items pass. Specification is ready for `/speckit-plan`.
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# Public API Contract: mdfetch — dev.to Extractor
|
|
2
|
+
|
|
3
|
+
**Branch**: `002-devto-provider` | **Date**: 2026-05-14
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
The dev.to extractor introduces no new public API surface. The existing `mdfetch.extract()` function is the sole entry point — callers use it identically for Medium and dev.to URLs.
|
|
8
|
+
|
|
9
|
+
## `mdfetch.extract(url, *, retries=3, retry_delay=2.0) -> str`
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from mdfetch import extract
|
|
13
|
+
|
|
14
|
+
markdown = extract("https://dev.to/stn1slv/integration-digest-for-december-2025-5dlp")
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
### Parameters
|
|
18
|
+
|
|
19
|
+
| Parameter | Type | Default | Description |
|
|
20
|
+
|-----------|------|---------|-------------|
|
|
21
|
+
| `url` | `str` | required | Full URL of a dev.to article |
|
|
22
|
+
| `retries` | `int` | `3` | Number of fetch attempts before raising |
|
|
23
|
+
| `retry_delay` | `float` | `2.0` | Seconds between retry attempts |
|
|
24
|
+
|
|
25
|
+
### Return value
|
|
26
|
+
|
|
27
|
+
A `str` containing the article's content as Markdown. Structure:
|
|
28
|
+
|
|
29
|
+
```markdown
|
|
30
|
+
# Article Title
|
|
31
|
+
|
|
32
|
+

|
|
33
|
+
|
|
34
|
+
## Section Heading
|
|
35
|
+
|
|
36
|
+
Paragraph text with [links](https://...).
|
|
37
|
+
|
|
38
|
+
- List item
|
|
39
|
+
- List item
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
code block
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
[https://gist.github.com/...](https://gist.github.com/...)
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Exceptions
|
|
49
|
+
|
|
50
|
+
All exceptions inherit from `mdfetch.MdfetchError` and carry `.url` and `.message` attributes.
|
|
51
|
+
|
|
52
|
+
| Exception | When raised |
|
|
53
|
+
|-----------|------------|
|
|
54
|
+
| `InvalidURLError` | `url` is syntactically invalid |
|
|
55
|
+
| `UnsupportedPlatformError` | Domain is not `dev.to` (or other registered domain) |
|
|
56
|
+
| `UnsupportedContentTypeError` | `dev.to` URL is not an article (profile, tag, etc.) |
|
|
57
|
+
| `FetchError` | Network error or timeout |
|
|
58
|
+
| `HTTPStatusError` | Non-2xx HTTP response (adds `.status_code`) |
|
|
59
|
+
| `EmptyContentError` | Article body found but contains no text |
|
|
60
|
+
|
|
61
|
+
### Routing
|
|
62
|
+
|
|
63
|
+
The router auto-discovers `DevToExtractor` at import time. No configuration required. Adding the provider file to `src/mdfetch/providers/devto.py` is sufficient for routing to work.
|
|
64
|
+
|
|
65
|
+
## `DevToExtractor` (internal)
|
|
66
|
+
|
|
67
|
+
Not part of the public API. Exposed only for testing via direct instantiation.
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
from mdfetch.providers.devto import DevToExtractor
|
|
71
|
+
|
|
72
|
+
extractor = DevToExtractor()
|
|
73
|
+
html = extractor.fetch_html(url)
|
|
74
|
+
from bs4 import BeautifulSoup
|
|
75
|
+
soup = BeautifulSoup(html, "lxml")
|
|
76
|
+
cleaned = extractor.clean_html(soup)
|
|
77
|
+
markdown = extractor.convert_to_markdown(cleaned)
|
|
78
|
+
```
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Data Model: mdfetch — dev.to Extractor
|
|
2
|
+
|
|
3
|
+
**Branch**: `002-devto-provider` | **Date**: 2026-05-14
|
|
4
|
+
|
|
5
|
+
## Entities
|
|
6
|
+
|
|
7
|
+
This feature adds no new data structures. All entities are shared with the existing library model.
|
|
8
|
+
|
|
9
|
+
### URL
|
|
10
|
+
- **Representation**: Plain Python `str`
|
|
11
|
+
- **Routing key**: `hostname` extracted via `urllib.parse.urlparse`; dev.to provider registers `"dev.to"` — the router's existing subdomain suffix matching handles `www.dev.to` automatically
|
|
12
|
+
- **Validation**: Existing `InvalidURLError` raised by the router before the provider is invoked
|
|
13
|
+
- **Accepted forms**:
|
|
14
|
+
- `https://dev.to/<username>/<article-slug>` → article
|
|
15
|
+
- `https://dev.to/<username>` → profile → `UnsupportedContentTypeError`
|
|
16
|
+
- `https://dev.to/t/<tag>` → tag listing → `UnsupportedContentTypeError`
|
|
17
|
+
|
|
18
|
+
### Provider
|
|
19
|
+
- **New provider**: `DevToExtractor` implementing `BaseExtractor`
|
|
20
|
+
- **DOMAINS**: `frozenset({"dev.to"})`
|
|
21
|
+
- **Registered via**: `@register` decorator (auto-discovered by `_autodiscover_providers()`)
|
|
22
|
+
- **File**: `src/mdfetch/providers/devto.py` (one new file; no existing files modified)
|
|
23
|
+
|
|
24
|
+
### Extracted Article
|
|
25
|
+
- **Output**: `str` — clean Markdown with ATX-style headings
|
|
26
|
+
- **Content included**:
|
|
27
|
+
- Title (`# Title` from article `<h1>`)
|
|
28
|
+
- Cover image as `` (if present)
|
|
29
|
+
- Article body: paragraphs, headings, code blocks, inline code, lists, blockquotes
|
|
30
|
+
- Body images as ``
|
|
31
|
+
- Embedded resources (Gists, CodePens, YouTube) as plain Markdown links `[url](url)`
|
|
32
|
+
- **Content excluded**:
|
|
33
|
+
- Author name, avatar, publication date
|
|
34
|
+
- Series navigation panels
|
|
35
|
+
- Article tags (`#kafka`, `#api`, etc.)
|
|
36
|
+
- Comments section
|
|
37
|
+
- Reaction buttons
|
|
38
|
+
- Advertisements
|
|
39
|
+
|
|
40
|
+
## State Transitions
|
|
41
|
+
|
|
42
|
+
None — extraction is a stateless, single-call operation with no persistent state.
|
|
43
|
+
|
|
44
|
+
## Error Cases
|
|
45
|
+
|
|
46
|
+
| Condition | Exception |
|
|
47
|
+
|-----------|-----------|
|
|
48
|
+
| URL has no `dev.to` domain | `UnsupportedPlatformError` (raised by router) |
|
|
49
|
+
| URL is syntactically invalid | `InvalidURLError` (raised by router) |
|
|
50
|
+
| HTTP request fails | `FetchError` |
|
|
51
|
+
| Non-2xx HTTP response | `HTTPStatusError` |
|
|
52
|
+
| `div#article-body` not found | `UnsupportedContentTypeError` |
|
|
53
|
+
| Article body has no extractable text | `EmptyContentError` |
|
|
54
|
+
|
|
55
|
+
All exception classes are defined in `mdfetch.exceptions` — no new classes needed.
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# Implementation Plan: mdfetch — dev.to Extractor
|
|
2
|
+
|
|
3
|
+
**Branch**: `002-devto-provider` | **Date**: 2026-05-14 | **Spec**: [spec.md](spec.md)
|
|
4
|
+
|
|
5
|
+
**Input**: Feature specification from `specs/002-devto-provider/spec.md`
|
|
6
|
+
|
|
7
|
+
## Summary
|
|
8
|
+
|
|
9
|
+
Add a `DevToExtractor` provider that fetches dev.to article pages, extracts the article body from `<div id="article-body">`, prepends the title and cover image from the page header, converts embedded resources to plain Markdown links, and returns clean Markdown. The provider is one new file; no shared library code is modified.
|
|
10
|
+
|
|
11
|
+
## Technical Context
|
|
12
|
+
|
|
13
|
+
**Language/Version**: Python 3.12+
|
|
14
|
+
|
|
15
|
+
**Primary Dependencies**: `httpx` (HTTP fetch), `BeautifulSoup` / `lxml` (HTML parsing), `markdownify` (Markdown conversion), `pytest` (testing)
|
|
16
|
+
|
|
17
|
+
**Storage**: N/A — stateless extraction library
|
|
18
|
+
|
|
19
|
+
**Testing**: `pytest` — unit tests (no network) + integration tests (`-m integration`, real URLs)
|
|
20
|
+
|
|
21
|
+
**Target Platform**: PyPI library (cross-platform)
|
|
22
|
+
|
|
23
|
+
**Project Type**: library
|
|
24
|
+
|
|
25
|
+
**Performance Goals**: Extraction in under 10 seconds per article on stable internet (SC-002)
|
|
26
|
+
|
|
27
|
+
**Constraints**: One new provider file; no changes to existing shared code
|
|
28
|
+
|
|
29
|
+
**Scale/Scope**: Single-article extraction per call; no volume concerns
|
|
30
|
+
|
|
31
|
+
## Constitution Check
|
|
32
|
+
|
|
33
|
+
*GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.*
|
|
34
|
+
|
|
35
|
+
- [x] Validates Provider Pattern Architecture — `DevToExtractor` inherits `BaseExtractor`; uses `@register`; one new file only; no shared code changes
|
|
36
|
+
- [x] Confirms Technology Stack — `httpx`, `BeautifulSoup`, `markdownify`, `pytest` (all already in use; no new deps needed)
|
|
37
|
+
- [x] Adheres to Coding Standards — strict type hints, PEP 8, clear English naming
|
|
38
|
+
- [x] Incorporates Integration Testing — three reference dev.to URLs verified in `tests/integration/test_devto_integration.py`
|
|
39
|
+
- [x] Respects Packaging and Distribution standards — `pyproject.toml` + `src/` layout unchanged; `uv` for all dev commands
|
|
40
|
+
|
|
41
|
+
All constitution gates pass. No violations to justify.
|
|
42
|
+
|
|
43
|
+
## Project Structure
|
|
44
|
+
|
|
45
|
+
### Documentation (this feature)
|
|
46
|
+
|
|
47
|
+
```text
|
|
48
|
+
specs/002-devto-provider/
|
|
49
|
+
├── plan.md # This file
|
|
50
|
+
├── research.md # Phase 0 output — dev.to HTML structure analysis
|
|
51
|
+
├── data-model.md # Phase 1 output — entities and error cases
|
|
52
|
+
├── quickstart.md # Phase 1 output — usage examples
|
|
53
|
+
├── contracts/
|
|
54
|
+
│ └── public-api.md # Phase 1 output — API contract (unchanged from Medium)
|
|
55
|
+
└── tasks.md # Phase 2 output (/speckit-tasks — not created here)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Source Code Changes
|
|
59
|
+
|
|
60
|
+
```text
|
|
61
|
+
src/mdfetch/providers/
|
|
62
|
+
├── __init__.py # UNCHANGED — auto-discovery handles new file
|
|
63
|
+
└── devto.py # NEW — DevToExtractor (single new file)
|
|
64
|
+
|
|
65
|
+
tests/unit/
|
|
66
|
+
└── test_devto_extractor.py # NEW — unit tests (no network)
|
|
67
|
+
|
|
68
|
+
tests/integration/
|
|
69
|
+
├── snapshots/
|
|
70
|
+
│ ├── devto-integration-digest-december-2025.md # NEW — snapshot
|
|
71
|
+
│ ├── devto-integration-digest-july-2025.md # NEW — snapshot
|
|
72
|
+
│ └── devto-integration-digest-march-2026.md # NEW — snapshot
|
|
73
|
+
└── test_devto_integration.py # NEW — integration tests (real URLs)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
No existing library source files are modified (`src/mdfetch/` unchanged).
|
|
77
|
+
|
|
78
|
+
**Side effect documented**: `tests/unit/test_router.py` required a one-line update — its "unsupported domain" example used `dev.to`, which became a supported domain once this provider was registered. The two test cases were updated to use `substack.com` instead. This is an expected consequence of any new provider registration.
|
|
79
|
+
|
|
80
|
+
**Version**: Library version bumped from `0.1.0` to `0.2.0` in `pyproject.toml` at time of release.
|
|
81
|
+
|
|
82
|
+
## Implementation Specification
|
|
83
|
+
|
|
84
|
+
### `src/mdfetch/providers/devto.py`
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
@register
|
|
88
|
+
class DevToExtractor(BaseExtractor):
|
|
89
|
+
DOMAINS: frozenset[str] = frozenset({"dev.to"})
|
|
90
|
+
|
|
91
|
+
def clean_html(self, soup: BeautifulSoup) -> Tag:
|
|
92
|
+
# 1. Locate article body — absence means non-article page
|
|
93
|
+
body = soup.find("div", id="article-body")
|
|
94
|
+
if not isinstance(body, Tag):
|
|
95
|
+
raise UnsupportedContentTypeError(...)
|
|
96
|
+
|
|
97
|
+
# 2. Handle embedded content: replace iframes with anchor links
|
|
98
|
+
for iframe in body.find_all("iframe"):
|
|
99
|
+
src = iframe.get("src") or iframe.get("data-src") or ""
|
|
100
|
+
link = soup.new_tag("a", href=src)
|
|
101
|
+
link.string = src
|
|
102
|
+
iframe.replace_with(link)
|
|
103
|
+
|
|
104
|
+
# 3. Handle liquid tag embeds (class containing "ltag")
|
|
105
|
+
for embed in body.find_all(class_=re.compile(r"ltag", re.I)):
|
|
106
|
+
src = embed.get("data-src") or embed.get("src") or str(embed.get("data-url", ""))
|
|
107
|
+
if src:
|
|
108
|
+
link = soup.new_tag("a", href=src)
|
|
109
|
+
link.string = src
|
|
110
|
+
embed.replace_with(link)
|
|
111
|
+
else:
|
|
112
|
+
embed.decompose()
|
|
113
|
+
|
|
114
|
+
# 4. Strip empty anchor-name links inserted before headings
|
|
115
|
+
for anchor in body.find_all("a", attrs={"name": True}):
|
|
116
|
+
if not anchor.get_text(strip=True):
|
|
117
|
+
anchor.decompose()
|
|
118
|
+
|
|
119
|
+
# 5. Extract title and cover image from header; prepend to body
|
|
120
|
+
header = soup.find("header", class_="crayons-article__header")
|
|
121
|
+
if header:
|
|
122
|
+
h1 = header.find("h1")
|
|
123
|
+
cover_img = header.find("img", class_="crayons-article__cover__image")
|
|
124
|
+
if h1:
|
|
125
|
+
body.insert(0, h1) # prepend <h1> as first child of body
|
|
126
|
+
if cover_img:
|
|
127
|
+
body.insert(1, cover_img) # prepend cover image after title
|
|
128
|
+
|
|
129
|
+
return body
|
|
130
|
+
|
|
131
|
+
def convert_to_markdown(self, tag: Tag) -> str:
|
|
132
|
+
md = markdownify(str(tag), heading_style="ATX", code_language="", strip=["script", "style"])
|
|
133
|
+
md = md.strip()
|
|
134
|
+
md = re.sub(r"\n{3,}", "\n\n", md)
|
|
135
|
+
if not md:
|
|
136
|
+
raise EmptyContentError(...)
|
|
137
|
+
return md
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
### `tests/unit/test_devto_extractor.py`
|
|
141
|
+
|
|
142
|
+
Unit tests cover (no network):
|
|
143
|
+
- `clean_html` raises `UnsupportedContentTypeError` when `div#article-body` absent
|
|
144
|
+
- `clean_html` raises `UnsupportedContentTypeError` for profile-page HTML (no article-body)
|
|
145
|
+
- `clean_html` preserves `<h1>` title and cover image
|
|
146
|
+
- `clean_html` replaces `<iframe>` with anchor link
|
|
147
|
+
- `clean_html` strips empty anchor-name links
|
|
148
|
+
- `convert_to_markdown` produces ATX headings, code blocks, lists
|
|
149
|
+
- `convert_to_markdown` raises `EmptyContentError` on blank body
|
|
150
|
+
- No raw HTML tags in converted Markdown
|
|
151
|
+
|
|
152
|
+
### `tests/integration/test_devto_integration.py`
|
|
153
|
+
|
|
154
|
+
Integration tests mirror the Medium pattern:
|
|
155
|
+
- Parametrized over 3 reference URLs + snapshot filenames
|
|
156
|
+
- `@pytest.mark.integration` marker
|
|
157
|
+
- Snapshot containment check: `expected in result`
|
|
158
|
+
- Snapshots generated by running extraction once and saving output
|
|
159
|
+
|
|
160
|
+
## Complexity Tracking
|
|
161
|
+
|
|
162
|
+
*No constitution violations — section left intentionally blank.*
|
|
163
|
+
|
|
164
|
+
### Revision: Implementation Sync 2026-05-14
|
|
165
|
+
- Reason: Documented two implementation side-effects not captured in the original plan: (1) `tests/unit/test_router.py` required domain-example update when dev.to became a registered provider; (2) library version bumped from 0.1.0 to 0.2.0 at release. `pyproject.toml` keywords gap (missing "dev.to") tracked as T013.
|