mdfetch 0.1.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. mdfetch-0.2.1/.github/workflows/integration.yml +86 -0
  2. mdfetch-0.2.1/.specify/feature.json +3 -0
  3. mdfetch-0.2.1/.specify/memory/changelog.md +79 -0
  4. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/memory/plan.md +48 -14
  5. mdfetch-0.2.1/.specify/memory/spec.md +284 -0
  6. {mdfetch-0.1.0 → mdfetch-0.2.1}/CLAUDE.md +13 -10
  7. {mdfetch-0.1.0 → mdfetch-0.2.1}/PKG-INFO +5 -2
  8. {mdfetch-0.1.0 → mdfetch-0.2.1}/README.md +3 -0
  9. {mdfetch-0.1.0 → mdfetch-0.2.1}/pyproject.toml +2 -2
  10. mdfetch-0.2.1/specs/002-devto-provider/checklists/requirements.md +34 -0
  11. mdfetch-0.2.1/specs/002-devto-provider/contracts/public-api.md +78 -0
  12. mdfetch-0.2.1/specs/002-devto-provider/data-model.md +55 -0
  13. mdfetch-0.2.1/specs/002-devto-provider/plan.md +165 -0
  14. mdfetch-0.2.1/specs/002-devto-provider/quickstart.md +57 -0
  15. mdfetch-0.2.1/specs/002-devto-provider/research.md +77 -0
  16. mdfetch-0.2.1/specs/002-devto-provider/spec.md +117 -0
  17. mdfetch-0.2.1/specs/002-devto-provider/tasks.md +205 -0
  18. mdfetch-0.2.1/specs/003-medium-freedium-fallback/checklists/requirements.md +34 -0
  19. mdfetch-0.2.1/specs/003-medium-freedium-fallback/contracts/extract-api.md +41 -0
  20. mdfetch-0.2.1/specs/003-medium-freedium-fallback/plan.md +187 -0
  21. mdfetch-0.2.1/specs/003-medium-freedium-fallback/research.md +73 -0
  22. mdfetch-0.2.1/specs/003-medium-freedium-fallback/spec.md +108 -0
  23. mdfetch-0.2.1/specs/003-medium-freedium-fallback/tasks.md +174 -0
  24. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/__init__.py +4 -3
  25. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/base.py +20 -4
  26. mdfetch-0.2.1/src/mdfetch/providers/devto.py +80 -0
  27. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/providers/medium.py +42 -1
  28. mdfetch-0.2.1/tests/integration/conftest.py +17 -0
  29. mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-december-2025.md +135 -0
  30. mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-july-2025.md +111 -0
  31. mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-march-2026.md +237 -0
  32. mdfetch-0.2.1/tests/integration/test_devto_integration.py +52 -0
  33. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/test_medium_integration.py +6 -4
  34. mdfetch-0.2.1/tests/unit/test_devto_extractor.py +255 -0
  35. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_fetch_errors.py +60 -1
  36. mdfetch-0.2.1/tests/unit/test_medium_extractor.py +329 -0
  37. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_router.py +9 -3
  38. {mdfetch-0.1.0 → mdfetch-0.2.1}/uv.lock +1 -1
  39. mdfetch-0.1.0/.specify/feature.json +0 -3
  40. mdfetch-0.1.0/.specify/memory/changelog.md +0 -30
  41. mdfetch-0.1.0/.specify/memory/spec.md +0 -175
  42. mdfetch-0.1.0/tests/unit/test_medium_extractor.py +0 -144
  43. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-analyze/SKILL.md +0 -0
  44. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
  45. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-checklist/SKILL.md +0 -0
  46. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-clarify/SKILL.md +0 -0
  47. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-constitution/SKILL.md +0 -0
  48. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
  49. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
  50. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
  51. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
  52. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
  53. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-implement/SKILL.md +0 -0
  54. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-plan/SKILL.md +0 -0
  55. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
  56. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-specify/SKILL.md +0 -0
  57. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-tasks/SKILL.md +0 -0
  58. {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
  59. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.analyze.toml +0 -0
  60. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.archive.run.toml +0 -0
  61. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.checklist.toml +0 -0
  62. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.clarify.toml +0 -0
  63. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.constitution.toml +0 -0
  64. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.implement.toml +0 -0
  65. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.plan.toml +0 -0
  66. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.reconcile.run.toml +0 -0
  67. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.specify.toml +0 -0
  68. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.tasks.toml +0 -0
  69. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.taskstoissues.toml +0 -0
  70. {mdfetch-0.1.0 → mdfetch-0.2.1}/.github/workflows/ci.yml +0 -0
  71. {mdfetch-0.1.0 → mdfetch-0.2.1}/.github/workflows/publish.yml +0 -0
  72. {mdfetch-0.1.0 → mdfetch-0.2.1}/.gitignore +0 -0
  73. {mdfetch-0.1.0 → mdfetch-0.2.1}/.python-version +0 -0
  74. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/.registry +0 -0
  75. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/LICENSE +0 -0
  76. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/README.md +0 -0
  77. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/commands/archive.md +0 -0
  78. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/extension.yml +0 -0
  79. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/README.md +0 -0
  80. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
  81. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
  82. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
  83. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
  84. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
  85. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/config-template.yml +0 -0
  86. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/extension.yml +0 -0
  87. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/git-config.yml +0 -0
  88. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
  89. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
  90. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
  91. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
  92. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
  93. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
  94. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
  95. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
  96. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/LICENSE +0 -0
  97. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/README.md +0 -0
  98. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
  99. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/extension.yml +0 -0
  100. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions.yml +0 -0
  101. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/init-options.json +0 -0
  102. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integration.json +0 -0
  103. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/claude.manifest.json +0 -0
  104. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/gemini.manifest.json +0 -0
  105. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/speckit.manifest.json +0 -0
  106. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/memory/constitution.md +0 -0
  107. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/check-prerequisites.sh +0 -0
  108. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/common.sh +0 -0
  109. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/create-new-feature.sh +0 -0
  110. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/setup-plan.sh +0 -0
  111. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/setup-tasks.sh +0 -0
  112. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/checklist-template.md +0 -0
  113. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/constitution-template.md +0 -0
  114. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/plan-template.md +0 -0
  115. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/spec-template.md +0 -0
  116. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/tasks-template.md +0 -0
  117. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/workflows/speckit/workflow.yml +0 -0
  118. {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/workflows/workflow-registry.json +0 -0
  119. {mdfetch-0.1.0 → mdfetch-0.2.1}/GEMINI.md +0 -0
  120. {mdfetch-0.1.0 → mdfetch-0.2.1}/LICENSE +0 -0
  121. {mdfetch-0.1.0 → mdfetch-0.2.1}/Makefile +0 -0
  122. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
  123. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
  124. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
  125. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
  126. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
  127. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/research.md +0 -0
  128. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
  129. {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
  130. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/exceptions.py +0 -0
  131. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/providers/__init__.py +0 -0
  132. {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/router.py +0 -0
  133. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/__init__.py +0 -0
  134. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/conftest.py +0 -0
  135. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/__init__.py +0 -0
  136. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
  137. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
  138. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
  139. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/__init__.py +0 -0
  140. {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_silent.py +0 -0
@@ -0,0 +1,86 @@
1
+ name: Integration Tests
2
+
3
+ on:
4
+ schedule:
5
+ - cron: "30 23 * * 5"
6
+ workflow_dispatch:
7
+
8
+ permissions:
9
+ contents: read
10
+ issues: write
11
+
12
+ jobs:
13
+ integration:
14
+ runs-on: [self-hosted, Linux, ARM64]
15
+ timeout-minutes: 30
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+
19
+ - name: Install uv
20
+ uses: astral-sh/setup-uv@v4
21
+ with:
22
+ enable-cache: true
23
+ cache-dependency-glob: "uv.lock"
24
+
25
+ - name: Set up Python
26
+ run: uv python pin 3.12
27
+
28
+ - name: Install dependencies
29
+ run: uv sync --frozen --all-extras
30
+
31
+ - name: Run integration tests
32
+ id: integration
33
+ env:
34
+ MDFETCH_RETRIES: "6"
35
+ MDFETCH_RETRY_DELAY: "2.0"
36
+ run: make integration
37
+
38
+ - name: Create issue on failure
39
+ if: failure() && steps.integration.outcome == 'failure'
40
+ uses: actions/github-script@v7
41
+ with:
42
+ script: |
43
+ const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`;
44
+ const date = new Date().toISOString().slice(0, 10);
45
+
46
+ const search = await github.rest.search.issuesAndPullRequests({
47
+ q: `repo:${context.repo.owner}/${context.repo.repo} is:issue is:open label:integration-failure`,
48
+ per_page: 1,
49
+ });
50
+
51
+ if (search.data.total_count > 0) {
52
+ await github.rest.issues.createComment({
53
+ owner: context.repo.owner,
54
+ repo: context.repo.repo,
55
+ issue_number: search.data.items[0].number,
56
+ body: `Integration tests failed again on ${date}.\n\n**Failed run:** ${runUrl}`,
57
+ });
58
+ return;
59
+ }
60
+
61
+ const issueParams = {
62
+ owner: context.repo.owner,
63
+ repo: context.repo.repo,
64
+ title: `Integration tests failed on ${date} — possible HTML structure change`,
65
+ body: [
66
+ "## Integration test failure",
67
+ "",
68
+ "The scheduled integration tests failed. This usually means a provider's",
69
+ "upstream HTML structure has changed and the extractor needs updating.",
70
+ "",
71
+ `**Failed run:** ${runUrl}`,
72
+ "",
73
+ "### Suggested steps",
74
+ "1. Open the failed run above and inspect the test output.",
75
+ "2. Identify which provider is broken (Medium, dev.to, …).",
76
+ "3. Update the relevant extractor in `src/mdfetch/providers/`.",
77
+ "4. Add or update snapshot fixtures in `tests/integration/` if needed.",
78
+ ].join("\n"),
79
+ };
80
+ try {
81
+ issueParams.labels = ["bug", "integration-failure"];
82
+ await github.rest.issues.create(issueParams);
83
+ } catch {
84
+ delete issueParams.labels;
85
+ await github.rest.issues.create(issueParams);
86
+ }
@@ -0,0 +1,3 @@
1
+ {
2
+ "feature_directory": "specs/003-medium-freedium-fallback"
3
+ }
@@ -0,0 +1,79 @@
1
+ # Merged Features Log
2
+
3
+ ---
4
+
5
+ ### mdfetch — Medium Freedium Fallback — 2026-05-15
6
+
7
+ **Branch**: `003-medium-freedium-fallback`
8
+ **Spec**: specs/003-medium-freedium-fallback
9
+
10
+ **What was added**:
11
+ - Transparent fallback to `https://freedium-mirror.cfd/` when medium.com returns HTTP 403 (paywall) or HTTP 429 (rate limit) — caller sees no difference in the `extract()` interface
12
+ - `_no_retry_status_codes: frozenset[int]` class attribute on `BaseExtractor`; codes in this set skip retry/backoff and raise immediately (defaults to `frozenset()` — safe for all existing providers)
13
+ - `_no_retry_codes: frozenset[int] | None = None` keyword-only parameter on `fetch_html()` for per-call override without instance mutation (thread-safe)
14
+ - `_parse_freedium(soup)` method on `MediumExtractor`: locates `div.main-content`, remaps h4→h3/h5→h4/h6→h5, converts to Markdown; heading remap ensures output is structurally identical to the direct medium.com path
15
+ - `extract()` override on `MediumExtractor`: on 403/429, fetches `freedium_url` with `_no_retry_codes=frozenset()`, routes to `_parse_freedium()`; always sets `exc.url` to the original Medium URL on failure
16
+ - 18 new unit tests across `test_medium_extractor.py` (TestParseFreedium, TestFreediumFallback, TestRateLimitFallback, TestNoFallbackOnSuccess) and `test_fetch_errors.py`
17
+ - Integration test suite now resilient to medium.com 403 responses — paywalled URL included in snapshot tests
18
+
19
+ **New Components**:
20
+ - Changes to `src/mdfetch/base.py` — `_no_retry_status_codes` attribute + `_no_retry_codes` param on `fetch_html()`
21
+ - Changes to `src/mdfetch/providers/medium.py` — `_parse_freedium()` + `extract()` override + Freedium constants
22
+
23
+ **Tasks Completed**: 12/12
24
+
25
+ ---
26
+
27
+ ### mdfetch — dev.to Extractor — 2026-05-14
28
+
29
+ **Branch**: `002-devto-provider`
30
+ **Spec**: specs/002-devto-provider
31
+
32
+ **What was added**:
33
+ - `DevToExtractor` provider for `dev.to` articles, auto-discovered via `@register` decorator
34
+ - Article body isolation from `<div id="article-body">` with cover image extracted from `<header class="crayons-article__header">` and prepended to output
35
+ - `<iframe>` and liquid-tag embed (`ltag__*`) replacement with plain Markdown links (FR-019)
36
+ - `UnsupportedContentTypeError` raised for non-article dev.to pages (profiles, tag listings)
37
+ - 17 new unit tests in `tests/unit/test_devto_extractor.py`
38
+ - 3 dev.to integration tests in `tests/integration/test_devto_integration.py` with snapshot golden files
39
+ - Library version bumped from `0.1.0` to `0.2.0`
40
+ - `"dev.to"` added to `pyproject.toml` keywords (T013)
41
+
42
+ **New Components**:
43
+ - `src/mdfetch/providers/devto.py` — DevToExtractor
44
+ - `tests/unit/test_devto_extractor.py` — 17 unit tests
45
+ - `tests/integration/test_devto_integration.py` — 3 integration tests
46
+ - `tests/integration/snapshots/devto-integration-digest-december-2025.md`
47
+ - `tests/integration/snapshots/devto-integration-digest-july-2025.md`
48
+ - `tests/integration/snapshots/devto-integration-digest-march-2026.md`
49
+
50
+ **Tasks Completed**: 13/13
51
+
52
+ ---
53
+
54
+ ### mdfetch — Medium Extractor (Initial Release) — 2026-05-14
55
+
56
+ **Branch**: `feature/first-draft`
57
+ **Spec**: specs/001-mdfetch-medium-extractor
58
+
59
+ **What was added**:
60
+ - `extract(url: str) -> str` public API for converting Medium articles to Markdown
61
+ - Provider pattern with `BaseExtractor` ABC and `MediumExtractor` implementation
62
+ - Auto-discovery routing via `pkgutil.iter_modules` + `@register` decorator (SC-006 compliant)
63
+ - Full typed exception hierarchy: `MdfetchError` → `InvalidURLError`, `UnsupportedPlatformError`, `UnsupportedContentTypeError`, `FetchError` → `HTTPStatusError`, `EmptyContentError`
64
+ - Streaming HTTP fetch with 10 MB response size cap
65
+ - Browser-like User-Agent (FR-014 compliant, no library branding)
66
+ - Snapshot-based integration tests with retry logic (3 retries, 2-second delay)
67
+ - PyPI-ready package: `pyproject.toml` + `src/` layout + `hatchling` build backend
68
+ - `Makefile` with full dev workflow (setup, test, integration, lint, typecheck, format, build, upgrade-deps, clean)
69
+
70
+ **New Components**:
71
+ - `src/mdfetch/__init__.py` — public surface
72
+ - `src/mdfetch/exceptions.py` — typed exception hierarchy
73
+ - `src/mdfetch/base.py` — BaseExtractor ABC + fetch_html + extract template method
74
+ - `src/mdfetch/router.py` — @register, auto-discovery, route()
75
+ - `src/mdfetch/providers/medium.py` — MediumExtractor
76
+ - `tests/unit/` — 30 unit tests (offline)
77
+ - `tests/integration/` — 3 integration tests with 3 snapshot golden files
78
+
79
+ **Tasks Completed**: 32/32
@@ -1,7 +1,7 @@
1
1
  # mdfetch — Main Implementation Plan
2
2
 
3
3
  **Last Updated**: 2026-05-14
4
- **Sources**: [specs/001-mdfetch-medium-extractor/plan.md]
4
+ **Sources**: [specs/001-mdfetch-medium-extractor/plan.md], [specs/002-devto-provider/plan.md]
5
5
 
6
6
  ---
7
7
 
@@ -38,15 +38,30 @@
38
38
 
39
39
  ```
40
40
  BaseExtractor (ABC) — src/mdfetch/base.py
41
- ├── fetch_html(url) → str — concrete: streaming HTTP with 30s timeout, 10 MB cap
41
+ ├── _no_retry_status_codes: frozenset[int] = frozenset() — codes that skip retry; overridden by providers
42
+ ├── fetch_html(url, *, retries, retry_delay, _no_retry_codes=None) → str
43
+ │ — streaming HTTP, 30s timeout, 10 MB cap;
44
+ │ codes in _no_retry_codes (or class attribute) raise immediately
42
45
  ├── clean_html(soup) → Tag — abstract: platform-specific HTML isolation
43
46
  ├── convert_to_markdown(tag) → str— abstract: platform-specific Markdown conversion
44
47
  └── extract(url) → str — concrete template method (orchestrates the above)
45
48
 
46
49
  MediumExtractor(BaseExtractor) — src/mdfetch/providers/medium.py
47
50
  ├── DOMAINS = frozenset({"medium.com"})
51
+ ├── _FREEDIUM_BASE = "https://freedium-mirror.cfd/"
52
+ ├── _no_retry_status_codes = frozenset({403, 429}) — immediate fallback, no medium.com retry
48
53
  ├── clean_html() → removes nav, clap buttons, sidebars, share elements, post-footer, author bio
49
- └── convert_to_markdown() → markdownify with ATX headings, fenced code blocks
54
+ ├── convert_to_markdown() → markdownify with ATX headings, fenced code blocks
55
+ ├── _parse_freedium(soup) → remaps h4→h3/h5→h4/h6→h5; finds div.main-content; convert_to_markdown
56
+ └── extract() → override: on 403/429 calls fetch_html(freedium_url, _no_retry_codes=frozenset());
57
+ exc.url always set to original Medium URL on any Freedium failure
58
+
59
+ DevToExtractor(BaseExtractor) — src/mdfetch/providers/devto.py
60
+ ├── DOMAINS = frozenset({"dev.to"})
61
+ ├── clean_html() → locates div#article-body; raises UnsupportedContentTypeError if absent;
62
+ │ replaces iframes and ltag embeds with anchor links; strips empty anchor-name
63
+ │ elements; prepends h1 + cover image from crayons-article__header
64
+ └── convert_to_markdown() → markdownify with ATX headings; collapses 3+ newlines to 2
50
65
  ```
51
66
 
52
67
  ### Router / Auto-Discovery
@@ -82,20 +97,26 @@ src/
82
97
  ├── base.py # BaseExtractor ABC + fetch_html() + extract() template
83
98
  └── providers/
84
99
  ├── __init__.py # Empty — auto-discovery handles registration
85
- └── medium.py # MediumExtractor
100
+ ├── medium.py # MediumExtractor
101
+ └── devto.py # DevToExtractor [002-devto-provider]
86
102
 
87
103
  tests/
88
104
  ├── unit/
89
105
  │ ├── test_router.py
90
106
  │ ├── test_medium_extractor.py
91
107
  │ ├── test_fetch_errors.py
92
- │ └── test_silent.py
108
+ │ ├── test_silent.py
109
+ │ └── test_devto_extractor.py # [002-devto-provider]
93
110
  └── integration/
94
111
  ├── snapshots/ # Golden Markdown files (article body snapshots)
95
112
  │ ├── from-drift-to-parity.md
96
113
  │ ├── architecting-the-asynchronous-agent.md
97
- │ └── integration-digest-december-2025.md
98
- └── test_medium_integration.py
114
+ │ ├── integration-digest-december-2025.md
115
+ │ ├── devto-integration-digest-december-2025.md # [002-devto-provider]
116
+ │ ├── devto-integration-digest-july-2025.md # [002-devto-provider]
117
+ │ └── devto-integration-digest-march-2026.md # [002-devto-provider]
118
+ ├── test_medium_integration.py
119
+ └── test_devto_integration.py # [002-devto-provider]
99
120
 
100
121
  specs/ # Speckit feature specifications
101
122
  pyproject.toml # hatchling build backend, uv package manager
@@ -118,15 +139,17 @@ Makefile # setup / test / integration / lint / typecheck / f
118
139
 
119
140
  ## Testing Strategy
120
141
 
121
- **Unit tests** (30 tests, offline):
142
+ **Unit tests** (65 tests, offline):
122
143
  - Router: domain routing, subdomain suffix matching, duplicate registration, invalid URLs, unsupported platforms
123
- - MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages
124
- - Fetch errors: HTTP 404, 503, timeout, connection error, size limit exceeded
144
+ - MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages, _parse_freedium (heading remap, missing main-content), fallback on 403/429 (URL construction, exc.url contract, no-sleep on 429), no-fallback on 200, UnsupportedContentTypeError.url on Freedium path [003-medium-freedium-fallback]
145
+ - DevToExtractor: clean_html (title/cover/heading/image preservation, iframe/ltag embed→link, anchor stripping, non-article error), convert_to_markdown (headings/code/lists/images, no raw HTML, empty content error) [002-devto-provider]
146
+ - Fetch errors: HTTP 404, 503, timeout, connection error, size limit exceeded; `_no_retry_status_codes` immediate-raise + `_no_retry_codes` override [003-medium-freedium-fallback]
125
147
  - Silent: no stdout/stderr output, no logging during extraction
126
148
 
127
- **Integration tests** (3 tests, network required):
128
- - Parametrized over 3 real stn1slv.medium.com articles
129
- - Snapshot-based containment check: `expected_body in extracted_result`
149
+ **Integration tests** (6 tests, network required):
150
+ - Parametrized over 3 real stn1slv.medium.com articles (including a known paywalled URL that exercises the Freedium fallback when medium.com returns 403) [003-medium-freedium-fallback]
151
+ - Parametrized over 3 real dev.to/stn1slv articles [002-devto-provider]
152
+ - Snapshot-based containment check: `expected_body in extracted_result` — tests pass regardless of whether medium.com or Freedium served the content (heading normalisation ensures identical output)
130
153
  - 3 retries with 2-second delay on `FetchError` (covers transient 403s/timeouts)
131
154
  - Run with: `make integration` or `uv run pytest tests/integration/ --override-ini=addopts=`
132
155
  - Excluded from default `pytest` run via `addopts = "-m 'not integration'"` in pyproject.toml
@@ -155,10 +178,17 @@ Makefile # setup / test / integration / lint / typecheck / f
155
178
  |----------|--------|-----------|
156
179
  | HTTP client | `httpx` (sync) | Supports async in v2 without swapping dependency |
157
180
  | HTML parser | `lxml` | Fast, tolerant; BeautifulSoup backend |
158
- | Article targeting | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
181
+ | Article targeting (Medium) | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
182
+ | Article targeting (dev.to) | `div#article-body` | dev.to does not use `<article>`; the id-scoped div already excludes all chrome | [002-devto-provider]
183
+ | dev.to cover image | Extracted from `header.crayons-article__header`, prepended to body | Cover image is outside the article body div — must be fetched separately | [002-devto-provider]
184
+ | dev.to embed handling | Replace `<iframe>` and `ltag__*` divs with anchor links | Embeds must not be silently dropped (FR-019); plain links are durable | [002-devto-provider]
159
185
  | Markdown converter | `markdownify` with ATX + `code_language=""` | Fenced code blocks, standard headings |
160
186
  | Routing | `pkgutil.iter_modules` auto-discovery + `@register` | SC-006: one new file = one new platform |
161
187
  | Integration tests | Snapshot containment + retry | Durable against minor HTML changes; resilient to transient 403s |
188
+ | test_router.py domain example | Changed from `dev.to` to `substack.com` for "unsupported domain" test | Once DevToExtractor registers `dev.to`, those tests would no longer raise UnsupportedPlatformError | [002-devto-provider]
189
+ | Medium 403/429 fallback | Override `extract()` in `MediumExtractor`; `_no_retry_status_codes=frozenset({403,429})` on class | Immediate fallback with no medium.com retries; `BaseExtractor` extended with `_no_retry_codes` param for thread safety | [003-medium-freedium-fallback]
190
+ | Freedium HTML parsing | Dedicated `_parse_freedium()` method; `div.main-content`; h4→h3 remap | Freedium HTML is structurally incompatible with `clean_html()` (no `<article>`); heading remap ensures snapshot tests pass for both paths | [003-medium-freedium-fallback]
191
+ | Freedium exc.url contract | `inner_exc.url = url` unconditionally; error message is source-agnostic ("Fallback page…") | Preserves transparent-fallback contract (FR-028); `exc.url` is the authoritative field; message content is internal | [003-medium-freedium-fallback]
162
192
 
163
193
  ---
164
194
 
@@ -169,3 +199,7 @@ Makefile # setup / test / integration / lint / typecheck / f
169
199
  - [x] Coding Standards — PEP 8, strict type hints, `mypy --strict` passes
170
200
  - [x] Integration Testing — real Medium URLs, snapshot-based containment assertions
171
201
  - [x] Packaging and Distribution — `pyproject.toml` + `src/` layout + `hatchling`; all Makefile targets use `uv run`
202
+
203
+ ---
204
+
205
+ *Last Updated: 2026-05-15 | Sources appended: [specs/003-medium-freedium-fallback/plan.md]*
@@ -0,0 +1,284 @@
1
+ # mdfetch — Main Specification
2
+
3
+ **Last Updated**: 2026-05-14
4
+ **Sources**: [specs/001-mdfetch-medium-extractor/spec.md], [specs/002-devto-provider/spec.md]
5
+
6
+ ---
7
+
8
+ ## Overview
9
+
10
+ `mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider. Supported platforms: `medium.com` (and subdomains), `dev.to`.
11
+
12
+ ---
13
+
14
+ ## User Stories
15
+
16
+ ### US-001 — Extract a Medium Article to Markdown (P1)
17
+ [Source: specs/001-mdfetch-medium-extractor]
18
+
19
+ A developer retrieves the readable content of a Medium article as clean Markdown by calling a single library function with the article URL. The library handles fetching, isolating the article body, stripping non-content elements, and returning formatted Markdown — with no configuration required.
20
+
21
+ **Acceptance Scenarios**:
22
+ 1. Given a valid, publicly accessible Medium article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
23
+ 2. Given a valid Medium article URL, the returned Markdown contains no navigation menus, "clap" interaction elements, social sharing buttons, or author biography sections.
24
+ 3. Given a Medium article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
25
+
26
+ ---
27
+
28
+ ### US-002 — Receive Meaningful Errors for Unsupported or Invalid URLs (P2)
29
+ [Source: specs/001-mdfetch-medium-extractor]
30
+
31
+ A developer passing an unsupported domain, malformed URL, or unreachable page receives a clear, descriptive typed exception.
32
+
33
+ **Acceptance Scenarios**:
34
+ 1. Given a URL from a domain other than Medium, `extract()` raises `UnsupportedPlatformError`.
35
+ 2. Given a syntactically invalid URL string, `extract()` raises `InvalidURLError`.
36
+ 3. Given a well-formed URL that returns an HTTP error (e.g., 404, 503), `extract()` raises `HTTPStatusError` including the status code.
37
+
38
+ ---
39
+
40
+ ### US-003 — Install and Use via Standard Package Manager (P3)
41
+ [Source: specs/001-mdfetch-medium-extractor]
42
+
43
+ A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `extract`, and immediately converts Medium URLs to Markdown with zero configuration.
44
+
45
+ **Acceptance Scenarios**:
46
+ 1. `pip install mdfetch` completes without errors and resolves all dependencies automatically.
47
+ 2. `from mdfetch import extract` succeeds and the function is callable.
48
+
49
+ ---
50
+
51
+ ### US-004 — Extract a dev.to Article to Markdown (P1)
52
+ [Source: specs/002-devto-provider]
53
+
54
+ A developer retrieves the readable content of a dev.to article as clean Markdown by calling the same single library function used for Medium — passing only the article URL, with no additional configuration. The library routes the request to the dev.to provider, fetches the page, isolates the article body, strips all non-content elements, and returns formatted Markdown.
55
+
56
+ **Acceptance Scenarios**:
57
+ 1. Given a valid, publicly accessible dev.to article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
58
+ 2. Given a valid dev.to article URL, the returned Markdown contains no navigation menus, reaction buttons, comment sections, or author sidebar widgets.
59
+ 3. Given a dev.to article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
60
+
61
+ ---
62
+
63
+ ### US-005 — Receive Meaningful Errors for Non-Article dev.to Pages (P2)
64
+ [Source: specs/002-devto-provider]
65
+
66
+ A developer passing a dev.to URL that points to a profile page, a tag listing, or a podcast page receives a clear, typed error indicating the dev.to domain is recognised but the page type is not extractable.
67
+
68
+ **Acceptance Scenarios**:
69
+ 1. Given a dev.to URL pointing to an author profile page, `extract()` raises `UnsupportedContentTypeError` (domain recognised, content type not extractable).
70
+ 2. Given a dev.to URL pointing to a tag listing page (e.g., `dev.to/t/kafka`), `extract()` raises `UnsupportedContentTypeError`.
71
+ 3. Given a URL from a domain other than dev.to, `extract()` raises `UnsupportedPlatformError` (unchanged from existing behaviour).
72
+
73
+ ---
74
+
75
+ ### US-007 — Transparent Fallback on Blocked Medium Article (P1)
76
+ [Source: specs/003-medium-freedium-fallback]
77
+
78
+ A developer calls the library's extract function with a Medium URL. The article is behind a paywall or the user is geo-blocked, causing Medium to return a 403 error. Without any code changes, the library automatically retrieves the same article via the Freedium mirror and returns clean Markdown content.
79
+
80
+ **Acceptance Scenarios**:
81
+ 1. Given a valid Medium article URL that returns 403 from medium.com, when `extract()` is called, then the library returns clean Markdown content retrieved via the Freedium mirror.
82
+ 2. Given a valid Medium article URL that returns 403 from medium.com, when the Freedium mirror also fails, then the library raises an appropriate extraction error with `exc.url` set to the original Medium URL.
83
+ 3. Given a valid Medium article URL that returns 403 from medium.com, when the library falls back to Freedium, then the caller receives the result without any knowledge of which source was used.
84
+
85
+ ---
86
+
87
+ ### US-008 — Automatic Fallback on Rate Limiting (P2)
88
+ [Source: specs/003-medium-freedium-fallback]
89
+
90
+ A developer calls the library's extract function for a Medium URL. Medium responds with 429 Too Many Requests. The library automatically uses the Freedium mirror as a fallback and returns clean Markdown without requiring the caller to retry.
91
+
92
+ **Acceptance Scenarios**:
93
+ 1. Given a valid Medium article URL that returns 429 from medium.com, when `extract()` is called, then the library returns clean Markdown content retrieved via the Freedium mirror.
94
+ 2. Given a valid Medium article URL that returns 429 from medium.com, when the Freedium mirror also fails, then the library raises an appropriate extraction error.
95
+
96
+ ---
97
+
98
+ ### US-009 — No Fallback When Primary Succeeds (P3)
99
+ [Source: specs/003-medium-freedium-fallback]
100
+
101
+ A developer calls the library's extract function for a publicly accessible Medium article. Medium responds successfully. The library returns the content directly without involving the Freedium mirror, preserving the existing happy-path behavior.
102
+
103
+ **Acceptance Scenarios**:
104
+ 1. Given a valid Medium article URL that returns a successful response, when `extract()` is called, then the library returns clean Markdown content without making any request to the Freedium mirror.
105
+
106
+ ---
107
+
108
+ ### US-006 — Integration Tests Pass Against Real dev.to Article URLs (P3)
109
+ [Source: specs/002-devto-provider]
110
+
111
+ A developer runs the integration test suite and all dev.to integration tests pass against three reference articles. The tests confirm title, structural elements, and absence of HTML tags on live content.
112
+
113
+ **Acceptance Scenarios**:
114
+ 1. When integration tests for the three reference dev.to URLs execute with a stable internet connection, all three pass without errors or assertion failures.
115
+ 2. For each reference dev.to article, the extraction function returns non-empty Markdown free of HTML tags and containing recognisable content from the article.
116
+
117
+ ---
118
+
119
+ ## Functional Requirements
120
+
121
+ ### Extraction
122
+ - **FR-001**: The library MUST expose a single primary function `extract(url: str) -> str` that accepts a URL string and returns the extracted article content as a Markdown string.
123
+ - **FR-002**: The library MUST enforce a provider pattern: a shared abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider.
124
+ - **FR-003**: The library MUST include a Medium provider that fetches the article page, isolates the main article body, removes all non-content elements (navigation, social sharing widgets, reader interaction components, author biographies), and returns the body as Markdown.
125
+ - **FR-008**: The library MUST produce Markdown that preserves the structural hierarchy of the source article, including headings, code blocks, inline code, lists, and blockquotes.
126
+ - **FR-011**: The library MUST raise a descriptive error when the article body is located but yields no extractable text content.
127
+ - **FR-012**: The library MUST raise a distinct error — separate from the "unsupported platform" error — when a `medium.com` URL is provided but the page is not an article (e.g., an author profile or tag page).
128
+ - **FR-013**: The library MUST NOT emit any logging, print output, or diagnostic messages. All failure information is communicated exclusively through raised exceptions.
129
+
130
+ ### Routing
131
+ - **FR-004**: The library MUST route extraction requests to the correct provider based on the URL's domain without requiring the caller to specify the provider explicitly. For v1, routing recognises only `medium.com` and its subdomains (e.g., `username.medium.com`).
132
+ - **FR-005**: The library MUST raise a descriptive error when given a URL whose domain has no registered provider.
133
+ - **FR-007**: The library MUST raise a descriptive error when given a URL that is syntactically invalid.
134
+
135
+ ### Network
136
+ - **FR-006**: The library MUST raise a descriptive error when a network request fails (connection error, timeout, non-2xx HTTP status).
137
+ - **FR-014**: HTTP requests MUST use a standard browser-like User-Agent string so that web servers return readable HTML. The User-Agent MUST NOT identify the library by name or version. The library does not check or respect `robots.txt` in v1.
138
+
139
+ ### Packaging & Testing
140
+ - **FR-009**: The library MUST be packaged and distributed via PyPI using modern Python packaging best practices, enabling installation through the standard package manager without additional steps.
141
+ - **FR-010**: The test suite MUST include integration tests that supply real article URLs (Medium and dev.to) to the extraction function and assert that the output matches expected Markdown structure and content.
142
+
143
+ ### Freedium Fallback (Medium)
144
+ - **FR-020**: When a Medium article extraction results in HTTP 403, the system MUST immediately attempt extraction via the Freedium mirror (`https://freedium-mirror.cfd/{url}`) — no retries against medium.com are made first. [Source: specs/003-medium-freedium-fallback]
145
+ - **FR-021**: When a Medium article extraction results in HTTP 429, the system MUST immediately attempt extraction via the Freedium mirror — no retries against medium.com are made first. [Source: specs/003-medium-freedium-fallback]
146
+ - **FR-022**: When the Freedium mirror is used as a fallback, the system MUST return content in the same clean Markdown format as direct extraction; Freedium's `<h4>` headings are remapped to `<h3>` so both paths produce identical heading-level output. [Source: specs/003-medium-freedium-fallback]
147
+ - **FR-023**: When the primary Medium request succeeds (HTTP 200), the system MUST NOT make any request to the Freedium mirror. [Source: specs/003-medium-freedium-fallback]
148
+ - **FR-024**: When both the primary Medium request and the Freedium fallback fail, the system MUST raise an error consistent with the existing exception hierarchy; `exc.url` MUST be set to the original Medium URL (never the Freedium URL). [Source: specs/003-medium-freedium-fallback]
149
+ - **FR-025**: The Freedium fallback mechanism MUST require no changes to the caller's code — the public `extract()` interface remains unchanged. [Source: specs/003-medium-freedium-fallback]
150
+ - **FR-026**: The Freedium fallback MUST only apply to Medium provider URLs; other providers are unaffected. [Source: specs/003-medium-freedium-fallback]
151
+ - **FR-027**: The Freedium fallback MUST be unconditionally active for all Medium URL extractions — no caller configuration, opt-in flag, or extractor parameter is required or supported. [Source: specs/003-medium-freedium-fallback]
152
+ - **FR-028**: The Freedium fallback MUST be fully transparent to the caller — no warning, signal, metadata, or result field shall indicate which source (medium.com or Freedium) provided the content. [Source: specs/003-medium-freedium-fallback]
153
+
154
+ ### dev.to Platform
155
+ - **FR-015**: The library MUST add `dev.to` to the provider router so that any URL with the `dev.to` domain is dispatched to the dev.to provider without any change to the caller's code. [Source: specs/002-devto-provider]
156
+ - **FR-016**: The library MUST include a dev.to provider that fetches the article page, isolates the main article body from `<div id="article-body">`, removes all non-content elements (navigation, social reaction widgets, comments, author sidebar, tag links), and returns the body as Markdown. [Source: specs/002-devto-provider]
157
+ - **FR-017**: The dev.to provider MUST produce Markdown that preserves both the cover image (from the article header) and all inline body images as Markdown image syntax (`![alt](url)`). [Source: specs/002-devto-provider]
158
+ - **FR-018**: The dev.to provider MUST raise `UnsupportedContentTypeError` when a `dev.to` URL is provided but the page is not an article (e.g., an author profile, a tag listing, or an organisation page). [Source: specs/002-devto-provider]
159
+ - **FR-019**: The dev.to provider MUST replace embedded third-party content (GitHub Gists, CodePen demos, YouTube videos, liquid-tag embeds) with a plain Markdown link to the embedded resource URL. Embeds must not be silently dropped. [Source: specs/002-devto-provider]
160
+
161
+ ---
162
+
163
+ ## Key Entities
164
+
165
+ ### URL (Input)
166
+ | Attribute | Type | Description |
167
+ |-----------|------|-------------|
168
+ | `raw` | `str` | The original string as supplied by the caller |
169
+ | `scheme` | `str` | Must be `http` or `https`; any other value triggers `InvalidURLError` |
170
+ | `hostname` | `str` | Lowercased hostname used for provider routing (e.g., `medium.com`) |
171
+ | `path` | `str` | Used to distinguish article pages from profile/tag pages |
172
+
173
+ **Validation**: syntactically parseable; scheme must be http/https; hostname must match a registered provider.
174
+
175
+ ### Provider (Internal)
176
+ | Attribute | Type | Description |
177
+ |-----------|------|-------------|
178
+ | `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`, `{"dev.to"}`) |
179
+
180
+ **Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent. Registered providers: `MediumExtractor` (medium.com), `DevToExtractor` (dev.to).
181
+
182
+ ### ExtractionResult (Output)
183
+ | Attribute | Type | Description |
184
+ |-----------|------|-------------|
185
+ | `content` | `str` | The extracted article body as Markdown text |
186
+
187
+ **Validation**: Non-empty string; contains no raw HTML tags; preserves at least one heading or paragraph.
188
+
189
+ ### Exception Hierarchy
190
+ ```
191
+ MdfetchError (base)
192
+ ├── InvalidURLError — syntactically invalid URL
193
+ ├── UnsupportedPlatformError — domain not recognised by any provider
194
+ ├── UnsupportedContentTypeError — recognised domain, but page is not an article
195
+ ├── FetchError — network/HTTP failure
196
+ │ └── HTTPStatusError — non-2xx response; carries .status_code (int)
197
+ └── EmptyContentError — article body found but no extractable text
198
+ ```
199
+
200
+ All exceptions carry `message: str` and `url: str | None`. `HTTPStatusError` additionally carries `status_code: int`.
201
+
202
+ ---
203
+
204
+ ## Call Lifecycle / State Transitions
205
+
206
+ ```
207
+ caller provides URL string
208
+ │
209
+ ▼
210
+ [VALIDATE URL] ──── invalid syntax ────► InvalidURLError
211
+ │
212
+ ▼
213
+ [ROUTE by domain] ── no provider ──────► UnsupportedPlatformError
214
+ │
215
+ ▼
216
+ [FETCH HTML] ─────── network failure ──► FetchError / HTTPStatusError
217
+ │
218
+ ▼
219
+ [LOCATE ARTICLE] ─── not an article ───► UnsupportedContentTypeError
220
+ │
221
+ ▼
222
+ [CLEAN & CONVERT] ── empty result ─────► EmptyContentError
223
+ │
224
+ ▼
225
+ return str (Markdown)
226
+ ```
227
+
228
+ ---
229
+
230
+ ## Edge Cases and Error Handling
231
+
232
+ - **Paywalled content (HTTP 403)**: When Medium returns HTTP 403 (paywall or geo-block), the library automatically retries via the Freedium mirror (`https://freedium-mirror.cfd/`). If Freedium also fails, the error raised carries `exc.url` set to the original Medium URL. [Source: specs/003-medium-freedium-fallback]
233
+ - **Freedium mirror unreachable**: If the Freedium mirror returns a network error, timeout, or non-2xx status, the fallback fails and the exception is propagated with `exc.url` set to the original Medium URL — Freedium's URL never appears in `exc.url`.
234
+ - **Freedium HTML structure**: Freedium uses `<div class="main-content">` (no `<article>`); if this element is absent, `UnsupportedContentTypeError` is raised with a source-agnostic message ("Fallback page missing main-content element").
235
+ - **Freedium heading levels**: Freedium renders section headings as `<h4>` vs. medium.com's `<h2>`/`<h3>`; `_parse_freedium()` remaps h4→h3, h5→h4, h6→h5 before conversion so snapshot tests pass regardless of which source served the content.
236
+ - **HTML structure changes**: If Medium changes its HTML structure and `<article>` is absent, `UnsupportedContentTypeError` is raised.
237
+ - **Empty article body**: If `<article>` is found but contains no extractable text, `EmptyContentError` is raised.
238
+ - **Network timeouts**: Covered by `FetchError` (30-second fixed timeout).
239
+ - **Oversized responses**: Responses exceeding 10 MB are rejected with `FetchError` to prevent OOM.
240
+ - **Profile/tag pages**: When a `medium.com` URL points to a non-article page, `UnsupportedContentTypeError` is raised (distinct from `UnsupportedPlatformError`).
241
+ - **HTTP 403 / transient failures**: Integration tests use a 3-retry helper with 2-second delay to handle transient rate limits.
242
+ - **dev.to profile pages**: When a `dev.to` URL points to an author profile (no `div#article-body`), `UnsupportedContentTypeError` is raised.
243
+ - **dev.to tag listing pages**: When a `dev.to` URL points to a tag page (e.g., `dev.to/t/kafka`), `UnsupportedContentTypeError` is raised.
244
+ - **dev.to liquid-tag embeds**: Embedded third-party widgets (GitHub Gists, CodePen, YouTube) serialised as `<div class="ltag__*" data-url="...">` are replaced with plain Markdown links; they are never silently dropped.
245
+ - **dev.to cover image**: The cover image lives in `<header class="crayons-article__header">`, not in `div#article-body` — the provider explicitly extracts and prepends it.
246
+ - **dev.to HTML structure changes**: If `div#article-body` is absent, `UnsupportedContentTypeError` is raised.
247
+
248
+ ---
249
+
250
+ ## Success Criteria
251
+
252
+ - **SC-001**: A developer can go from installing the library to extracting a real Medium article in fewer than 5 minutes with zero configuration.
253
+ - **SC-002**: The extraction function returns a result for a standard Medium article in under 10 seconds on a stable internet connection.
254
+ - **SC-003**: The returned Markdown for a standard Medium article contains no raw HTML tags.
255
+ - **SC-004**: The returned Markdown for an article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax.
256
+ - **SC-005**: 100% of integration tests pass against a curated set of real Medium article URLs at the time of initial release.
257
+ - **SC-006**: Adding support for a second platform requires creating exactly one new file (one new provider class decorated with `@register`). The router auto-discovers all provider modules at import time; no changes to shared library code are required.
258
+ - **SC-007**: A developer already using the library for Medium can extract a dev.to article without any code change — only the URL changes. [Source: specs/002-devto-provider]
259
+ - **SC-008**: The extraction function returns a result for a standard dev.to article in under 10 seconds on a stable internet connection. [Source: specs/002-devto-provider]
260
+ - **SC-009**: The returned Markdown for any of the three reference dev.to articles contains no raw HTML tags (verified by automated assertion). [Source: specs/002-devto-provider]
261
+ - **SC-010**: The returned Markdown for a dev.to article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax. [Source: specs/002-devto-provider]
262
+ - **SC-011**: 100% of integration tests pass against the three provided reference dev.to article URLs at the time of release. [Source: specs/002-devto-provider]
263
+ - **SC-012**: The dev.to provider is delivered as exactly one new file; no existing source files are modified (except `test_router.py` for expected domain-example maintenance when the provider registers `dev.to`). [Source: specs/002-devto-provider]
264
+
265
+ ---
266
+
267
+ ## Assumptions
268
+
269
+ - The library targets developers as its primary users; no graphical interface or configuration file is required.
270
+ - Medium articles used in integration testing are publicly accessible (not behind a paywall).
271
+ - No response caching; each call performs a fresh network request.
272
+ - Rate limiting or authentication with Medium's servers is out of scope for v1.
273
+ - The library supports Python 3.12 and later.
274
+ - The library operates on publicly accessible HTML; it does not execute JavaScript or render dynamic content.
275
+ - Network timeouts use a fixed default of 30 seconds (not user-configurable in v1).
276
+ - **SC-013**: Articles that previously failed with a 403 paywall error are successfully extracted in at least 90% of cases where the Freedium mirror has the content available. [Source: specs/003-medium-freedium-fallback]
277
+ - **SC-014**: Articles that previously failed with a 429 rate-limit error are successfully extracted via fallback without requiring the caller to retry. [Source: specs/003-medium-freedium-fallback]
278
+ - **SC-015**: Zero changes are required in existing caller code to benefit from the Freedium fallback — existing integrations continue to work as-is. [Source: specs/003-medium-freedium-fallback]
279
+ - **SC-016**: When the primary Medium request succeeds, there is no additional latency attributable to the fallback mechanism. [Source: specs/003-medium-freedium-fallback]
280
+ - **SC-017**: All existing unit and integration tests for the Medium extractor continue to pass without modification after the fallback is introduced. [Source: specs/003-medium-freedium-fallback]
281
+
282
+ ---
283
+
284
+ *Last Updated: 2026-05-15 | Sources appended: [specs/003-medium-freedium-fallback/spec.md]*
@@ -11,11 +11,18 @@ src/mdfetch/
11
11
  ├── base.py # BaseExtractor ABC
12
12
  ├── router.py # Domain-to-provider routing
13
13
  └── providers/
14
- └── medium.py # MediumExtractor (medium.com + *.medium.com)
14
+ ├── __init__.py
15
+ ├── medium.py # MediumExtractor (medium.com + *.medium.com)
16
+ └── devto.py # DevToExtractor (dev.to)
15
17
 
16
18
  tests/
17
19
  ├── unit/ # pytest unit tests (no network)
18
- └── integration/ # pytest -m integration (real Medium URLs)
20
+ └── integration/ # real network tests (Medium + dev.to URLs + snapshots)
21
+
22
+ .github/workflows/
23
+ ├── ci.yml # lint + unit tests on push/PR (Python 3.12–3.14)
24
+ ├── integration.yml # scheduled integration tests every Friday 23:30 UTC
25
+ └── publish.yml # PyPI publish on release
19
26
 
20
27
  specs/ # Speckit feature specifications
21
28
  pyproject.toml # hatchling build, uv package manager
@@ -38,15 +45,11 @@ make test # unit tests only
38
45
  make lint # ruff check
39
46
  make format # ruff format
40
47
  make build # uv build (wheel + sdist)
41
- make upgrade-deps # uv sync --upgrade
42
- uv run pytest -m integration # integration tests (network required)
43
- uv run mypy src/ # type check
48
+ make upgrade-deps # uv sync --all-extras --upgrade
49
+ make integration # integration tests (network required)
50
+ uv run mypy src/ # type check
44
51
  ```
45
52
 
46
- ## Recent Changes
47
-
48
- - 001-mdfetch-medium-extractor: Initial library release — `extract()` API, Medium provider, typed exceptions, auto-discovery routing, PyPI packaging, snapshot integration tests
49
-
50
53
  <!-- SPECKIT START -->
51
- **Active feature plan**: none
54
+ **Recent changes**: `003-medium-freedium-fallback` — Transparent Freedium mirror fallback for Medium 403/429 responses; `_no_retry_status_codes` hook on `BaseExtractor`; `_parse_freedium()` with h4→h3 heading remap on `MediumExtractor`
52
55
  <!-- SPECKIT END -->