mdfetch 0.1.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mdfetch-0.2.1/.github/workflows/integration.yml +86 -0
- mdfetch-0.2.1/.specify/feature.json +3 -0
- mdfetch-0.2.1/.specify/memory/changelog.md +79 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/memory/plan.md +48 -14
- mdfetch-0.2.1/.specify/memory/spec.md +284 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/CLAUDE.md +13 -10
- {mdfetch-0.1.0 → mdfetch-0.2.1}/PKG-INFO +5 -2
- {mdfetch-0.1.0 → mdfetch-0.2.1}/README.md +3 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/pyproject.toml +2 -2
- mdfetch-0.2.1/specs/002-devto-provider/checklists/requirements.md +34 -0
- mdfetch-0.2.1/specs/002-devto-provider/contracts/public-api.md +78 -0
- mdfetch-0.2.1/specs/002-devto-provider/data-model.md +55 -0
- mdfetch-0.2.1/specs/002-devto-provider/plan.md +165 -0
- mdfetch-0.2.1/specs/002-devto-provider/quickstart.md +57 -0
- mdfetch-0.2.1/specs/002-devto-provider/research.md +77 -0
- mdfetch-0.2.1/specs/002-devto-provider/spec.md +117 -0
- mdfetch-0.2.1/specs/002-devto-provider/tasks.md +205 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/checklists/requirements.md +34 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/contracts/extract-api.md +41 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/plan.md +187 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/research.md +73 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/spec.md +108 -0
- mdfetch-0.2.1/specs/003-medium-freedium-fallback/tasks.md +174 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/__init__.py +4 -3
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/base.py +20 -4
- mdfetch-0.2.1/src/mdfetch/providers/devto.py +80 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/providers/medium.py +42 -1
- mdfetch-0.2.1/tests/integration/conftest.py +17 -0
- mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-december-2025.md +135 -0
- mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-july-2025.md +111 -0
- mdfetch-0.2.1/tests/integration/snapshots/devto-integration-digest-march-2026.md +237 -0
- mdfetch-0.2.1/tests/integration/test_devto_integration.py +52 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/test_medium_integration.py +6 -4
- mdfetch-0.2.1/tests/unit/test_devto_extractor.py +255 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_fetch_errors.py +60 -1
- mdfetch-0.2.1/tests/unit/test_medium_extractor.py +329 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_router.py +9 -3
- {mdfetch-0.1.0 → mdfetch-0.2.1}/uv.lock +1 -1
- mdfetch-0.1.0/.specify/feature.json +0 -3
- mdfetch-0.1.0/.specify/memory/changelog.md +0 -30
- mdfetch-0.1.0/.specify/memory/spec.md +0 -175
- mdfetch-0.1.0/tests/unit/test_medium_extractor.py +0 -144
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-analyze/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-checklist/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-clarify/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-constitution/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-implement/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-plan/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-specify/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-tasks/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.analyze.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.archive.run.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.checklist.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.clarify.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.constitution.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.implement.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.plan.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.reconcile.run.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.specify.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.tasks.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gemini/commands/speckit.taskstoissues.toml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.github/workflows/ci.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.github/workflows/publish.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.gitignore +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.python-version +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/.registry +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/commands/archive.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/archive/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/config-template.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/git-config.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/README.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions/reconcile/extension.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/extensions.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/init-options.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integration.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/claude.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/gemini.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/integrations/speckit.manifest.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/memory/constitution.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/check-prerequisites.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/common.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/setup-plan.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/scripts/bash/setup-tasks.sh +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/checklist-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/constitution-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/plan-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/spec-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/templates/tasks-template.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/workflows/speckit/workflow.yml +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/.specify/workflows/workflow-registry.json +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/GEMINI.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/LICENSE +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/Makefile +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/research.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/exceptions.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/providers/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/src/mdfetch/router.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/conftest.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/__init__.py +0 -0
- {mdfetch-0.1.0 → mdfetch-0.2.1}/tests/unit/test_silent.py +0 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
name: Integration Tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
schedule:
|
|
5
|
+
- cron: "30 23 * * 5"
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
issues: write
|
|
11
|
+
|
|
12
|
+
jobs:
|
|
13
|
+
integration:
|
|
14
|
+
runs-on: [self-hosted, Linux, ARM64]
|
|
15
|
+
timeout-minutes: 30
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
|
|
19
|
+
- name: Install uv
|
|
20
|
+
uses: astral-sh/setup-uv@v4
|
|
21
|
+
with:
|
|
22
|
+
enable-cache: true
|
|
23
|
+
cache-dependency-glob: "uv.lock"
|
|
24
|
+
|
|
25
|
+
- name: Set up Python
|
|
26
|
+
run: uv python pin 3.12
|
|
27
|
+
|
|
28
|
+
- name: Install dependencies
|
|
29
|
+
run: uv sync --frozen --all-extras
|
|
30
|
+
|
|
31
|
+
- name: Run integration tests
|
|
32
|
+
id: integration
|
|
33
|
+
env:
|
|
34
|
+
MDFETCH_RETRIES: "6"
|
|
35
|
+
MDFETCH_RETRY_DELAY: "2.0"
|
|
36
|
+
run: make integration
|
|
37
|
+
|
|
38
|
+
- name: Create issue on failure
|
|
39
|
+
if: failure() && steps.integration.outcome == 'failure'
|
|
40
|
+
uses: actions/github-script@v7
|
|
41
|
+
with:
|
|
42
|
+
script: |
|
|
43
|
+
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`;
|
|
44
|
+
const date = new Date().toISOString().slice(0, 10);
|
|
45
|
+
|
|
46
|
+
const search = await github.rest.search.issuesAndPullRequests({
|
|
47
|
+
q: `repo:${context.repo.owner}/${context.repo.repo} is:issue is:open label:integration-failure`,
|
|
48
|
+
per_page: 1,
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
if (search.data.total_count > 0) {
|
|
52
|
+
await github.rest.issues.createComment({
|
|
53
|
+
owner: context.repo.owner,
|
|
54
|
+
repo: context.repo.repo,
|
|
55
|
+
issue_number: search.data.items[0].number,
|
|
56
|
+
body: `Integration tests failed again on ${date}.\n\n**Failed run:** ${runUrl}`,
|
|
57
|
+
});
|
|
58
|
+
return;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const issueParams = {
|
|
62
|
+
owner: context.repo.owner,
|
|
63
|
+
repo: context.repo.repo,
|
|
64
|
+
title: `Integration tests failed on ${date} — possible HTML structure change`,
|
|
65
|
+
body: [
|
|
66
|
+
"## Integration test failure",
|
|
67
|
+
"",
|
|
68
|
+
"The scheduled integration tests failed. This usually means a provider's",
|
|
69
|
+
"upstream HTML structure has changed and the extractor needs updating.",
|
|
70
|
+
"",
|
|
71
|
+
`**Failed run:** ${runUrl}`,
|
|
72
|
+
"",
|
|
73
|
+
"### Suggested steps",
|
|
74
|
+
"1. Open the failed run above and inspect the test output.",
|
|
75
|
+
"2. Identify which provider is broken (Medium, dev.to, …).",
|
|
76
|
+
"3. Update the relevant extractor in `src/mdfetch/providers/`.",
|
|
77
|
+
"4. Add or update snapshot fixtures in `tests/integration/` if needed.",
|
|
78
|
+
].join("\n"),
|
|
79
|
+
};
|
|
80
|
+
try {
|
|
81
|
+
issueParams.labels = ["bug", "integration-failure"];
|
|
82
|
+
await github.rest.issues.create(issueParams);
|
|
83
|
+
} catch {
|
|
84
|
+
delete issueParams.labels;
|
|
85
|
+
await github.rest.issues.create(issueParams);
|
|
86
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Merged Features Log
|
|
2
|
+
|
|
3
|
+
---
|
|
4
|
+
|
|
5
|
+
### mdfetch — Medium Freedium Fallback — 2026-05-15
|
|
6
|
+
|
|
7
|
+
**Branch**: `003-medium-freedium-fallback`
|
|
8
|
+
**Spec**: specs/003-medium-freedium-fallback
|
|
9
|
+
|
|
10
|
+
**What was added**:
|
|
11
|
+
- Transparent fallback to `https://freedium-mirror.cfd/` when medium.com returns HTTP 403 (paywall) or HTTP 429 (rate limit) — caller sees no difference in the `extract()` interface
|
|
12
|
+
- `_no_retry_status_codes: frozenset[int]` class attribute on `BaseExtractor`; codes in this set skip retry/backoff and raise immediately (defaults to `frozenset()` — safe for all existing providers)
|
|
13
|
+
- `_no_retry_codes: frozenset[int] | None = None` keyword-only parameter on `fetch_html()` for per-call override without instance mutation (thread-safe)
|
|
14
|
+
- `_parse_freedium(soup)` method on `MediumExtractor`: locates `div.main-content`, remaps h4→h3/h5→h4/h6→h5, converts to Markdown; heading remap ensures output is structurally identical to the direct medium.com path
|
|
15
|
+
- `extract()` override on `MediumExtractor`: on 403/429, fetches `freedium_url` with `_no_retry_codes=frozenset()`, routes to `_parse_freedium()`; always sets `exc.url` to the original Medium URL on failure
|
|
16
|
+
- 18 new unit tests across `test_medium_extractor.py` (TestParseFreedium, TestFreediumFallback, TestRateLimitFallback, TestNoFallbackOnSuccess) and `test_fetch_errors.py`
|
|
17
|
+
- Integration test suite now resilient to medium.com 403 responses — paywalled URL included in snapshot tests
|
|
18
|
+
|
|
19
|
+
**New Components**:
|
|
20
|
+
- Changes to `src/mdfetch/base.py` — `_no_retry_status_codes` attribute + `_no_retry_codes` param on `fetch_html()`
|
|
21
|
+
- Changes to `src/mdfetch/providers/medium.py` — `_parse_freedium()` + `extract()` override + Freedium constants
|
|
22
|
+
|
|
23
|
+
**Tasks Completed**: 12/12
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
### mdfetch — dev.to Extractor — 2026-05-14
|
|
28
|
+
|
|
29
|
+
**Branch**: `002-devto-provider`
|
|
30
|
+
**Spec**: specs/002-devto-provider
|
|
31
|
+
|
|
32
|
+
**What was added**:
|
|
33
|
+
- `DevToExtractor` provider for `dev.to` articles, auto-discovered via `@register` decorator
|
|
34
|
+
- Article body isolation from `<div id="article-body">` with cover image extracted from `<header class="crayons-article__header">` and prepended to output
|
|
35
|
+
- `<iframe>` and liquid-tag embed (`ltag__*`) replacement with plain Markdown links (FR-019)
|
|
36
|
+
- `UnsupportedContentTypeError` raised for non-article dev.to pages (profiles, tag listings)
|
|
37
|
+
- 17 new unit tests in `tests/unit/test_devto_extractor.py`
|
|
38
|
+
- 3 dev.to integration tests in `tests/integration/test_devto_integration.py` with snapshot golden files
|
|
39
|
+
- Library version bumped from `0.1.0` to `0.2.0`
|
|
40
|
+
- `"dev.to"` added to `pyproject.toml` keywords (T013)
|
|
41
|
+
|
|
42
|
+
**New Components**:
|
|
43
|
+
- `src/mdfetch/providers/devto.py` — DevToExtractor
|
|
44
|
+
- `tests/unit/test_devto_extractor.py` — 17 unit tests
|
|
45
|
+
- `tests/integration/test_devto_integration.py` — 3 integration tests
|
|
46
|
+
- `tests/integration/snapshots/devto-integration-digest-december-2025.md`
|
|
47
|
+
- `tests/integration/snapshots/devto-integration-digest-july-2025.md`
|
|
48
|
+
- `tests/integration/snapshots/devto-integration-digest-march-2026.md`
|
|
49
|
+
|
|
50
|
+
**Tasks Completed**: 13/13
|
|
51
|
+
|
|
52
|
+
---
|
|
53
|
+
|
|
54
|
+
### mdfetch — Medium Extractor (Initial Release) — 2026-05-14
|
|
55
|
+
|
|
56
|
+
**Branch**: `feature/first-draft`
|
|
57
|
+
**Spec**: specs/001-mdfetch-medium-extractor
|
|
58
|
+
|
|
59
|
+
**What was added**:
|
|
60
|
+
- `extract(url: str) -> str` public API for converting Medium articles to Markdown
|
|
61
|
+
- Provider pattern with `BaseExtractor` ABC and `MediumExtractor` implementation
|
|
62
|
+
- Auto-discovery routing via `pkgutil.iter_modules` + `@register` decorator (SC-006 compliant)
|
|
63
|
+
- Full typed exception hierarchy: `MdfetchError` → `InvalidURLError`, `UnsupportedPlatformError`, `UnsupportedContentTypeError`, `FetchError` → `HTTPStatusError`, `EmptyContentError`
|
|
64
|
+
- Streaming HTTP fetch with 10 MB response size cap
|
|
65
|
+
- Browser-like User-Agent (FR-014 compliant, no library branding)
|
|
66
|
+
- Snapshot-based integration tests with retry logic (3 retries, 2-second delay)
|
|
67
|
+
- PyPI-ready package: `pyproject.toml` + `src/` layout + `hatchling` build backend
|
|
68
|
+
- `Makefile` with full dev workflow (setup, test, integration, lint, typecheck, format, build, upgrade-deps, clean)
|
|
69
|
+
|
|
70
|
+
**New Components**:
|
|
71
|
+
- `src/mdfetch/__init__.py` — public surface
|
|
72
|
+
- `src/mdfetch/exceptions.py` — typed exception hierarchy
|
|
73
|
+
- `src/mdfetch/base.py` — BaseExtractor ABC + fetch_html + extract template method
|
|
74
|
+
- `src/mdfetch/router.py` — @register, auto-discovery, route()
|
|
75
|
+
- `src/mdfetch/providers/medium.py` — MediumExtractor
|
|
76
|
+
- `tests/unit/` — 30 unit tests (offline)
|
|
77
|
+
- `tests/integration/` — 3 integration tests with 3 snapshot golden files
|
|
78
|
+
|
|
79
|
+
**Tasks Completed**: 32/32
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# mdfetch — Main Implementation Plan
|
|
2
2
|
|
|
3
3
|
**Last Updated**: 2026-05-14
|
|
4
|
-
**Sources**: [specs/001-mdfetch-medium-extractor/plan.md]
|
|
4
|
+
**Sources**: [specs/001-mdfetch-medium-extractor/plan.md], [specs/002-devto-provider/plan.md]
|
|
5
5
|
|
|
6
6
|
---
|
|
7
7
|
|
|
@@ -38,15 +38,30 @@
|
|
|
38
38
|
|
|
39
39
|
```
|
|
40
40
|
BaseExtractor (ABC) — src/mdfetch/base.py
|
|
41
|
-
├──
|
|
41
|
+
├── _no_retry_status_codes: frozenset[int] = frozenset() — codes that skip retry; overridden by providers
|
|
42
|
+
├── fetch_html(url, *, retries, retry_delay, _no_retry_codes=None) → str
|
|
43
|
+
│ — streaming HTTP, 30s timeout, 10 MB cap;
|
|
44
|
+
│ codes in _no_retry_codes (or class attribute) raise immediately
|
|
42
45
|
├── clean_html(soup) → Tag — abstract: platform-specific HTML isolation
|
|
43
46
|
├── convert_to_markdown(tag) → str— abstract: platform-specific Markdown conversion
|
|
44
47
|
└── extract(url) → str — concrete template method (orchestrates the above)
|
|
45
48
|
|
|
46
49
|
MediumExtractor(BaseExtractor) — src/mdfetch/providers/medium.py
|
|
47
50
|
├── DOMAINS = frozenset({"medium.com"})
|
|
51
|
+
├── _FREEDIUM_BASE = "https://freedium-mirror.cfd/"
|
|
52
|
+
├── _no_retry_status_codes = frozenset({403, 429}) — immediate fallback, no medium.com retry
|
|
48
53
|
├── clean_html() → removes nav, clap buttons, sidebars, share elements, post-footer, author bio
|
|
49
|
-
|
|
54
|
+
├── convert_to_markdown() → markdownify with ATX headings, fenced code blocks
|
|
55
|
+
├── _parse_freedium(soup) → remaps h4→h3/h5→h4/h6→h5; finds div.main-content; convert_to_markdown
|
|
56
|
+
└── extract() → override: on 403/429 calls fetch_html(freedium_url, _no_retry_codes=frozenset());
|
|
57
|
+
exc.url always set to original Medium URL on any Freedium failure
|
|
58
|
+
|
|
59
|
+
DevToExtractor(BaseExtractor) — src/mdfetch/providers/devto.py
|
|
60
|
+
├── DOMAINS = frozenset({"dev.to"})
|
|
61
|
+
├── clean_html() → locates div#article-body; raises UnsupportedContentTypeError if absent;
|
|
62
|
+
│ replaces iframes and ltag embeds with anchor links; strips empty anchor-name
|
|
63
|
+
│ elements; prepends h1 + cover image from crayons-article__header
|
|
64
|
+
└── convert_to_markdown() → markdownify with ATX headings; collapses 3+ newlines to 2
|
|
50
65
|
```
|
|
51
66
|
|
|
52
67
|
### Router / Auto-Discovery
|
|
@@ -82,20 +97,26 @@ src/
|
|
|
82
97
|
├── base.py # BaseExtractor ABC + fetch_html() + extract() template
|
|
83
98
|
└── providers/
|
|
84
99
|
├── __init__.py # Empty — auto-discovery handles registration
|
|
85
|
-
|
|
100
|
+
├── medium.py # MediumExtractor
|
|
101
|
+
└── devto.py # DevToExtractor [002-devto-provider]
|
|
86
102
|
|
|
87
103
|
tests/
|
|
88
104
|
├── unit/
|
|
89
105
|
│ ├── test_router.py
|
|
90
106
|
│ ├── test_medium_extractor.py
|
|
91
107
|
│ ├── test_fetch_errors.py
|
|
92
|
-
│
|
|
108
|
+
│ ├── test_silent.py
|
|
109
|
+
│ └── test_devto_extractor.py # [002-devto-provider]
|
|
93
110
|
└── integration/
|
|
94
111
|
├── snapshots/ # Golden Markdown files (article body snapshots)
|
|
95
112
|
│ ├── from-drift-to-parity.md
|
|
96
113
|
│ ├── architecting-the-asynchronous-agent.md
|
|
97
|
-
│
|
|
98
|
-
|
|
114
|
+
│ ├── integration-digest-december-2025.md
|
|
115
|
+
│ ├── devto-integration-digest-december-2025.md # [002-devto-provider]
|
|
116
|
+
│ ├── devto-integration-digest-july-2025.md # [002-devto-provider]
|
|
117
|
+
│ └── devto-integration-digest-march-2026.md # [002-devto-provider]
|
|
118
|
+
├── test_medium_integration.py
|
|
119
|
+
└── test_devto_integration.py # [002-devto-provider]
|
|
99
120
|
|
|
100
121
|
specs/ # Speckit feature specifications
|
|
101
122
|
pyproject.toml # hatchling build backend, uv package manager
|
|
@@ -118,15 +139,17 @@ Makefile # setup / test / integration / lint / typecheck / f
|
|
|
118
139
|
|
|
119
140
|
## Testing Strategy
|
|
120
141
|
|
|
121
|
-
**Unit tests** (
|
|
142
|
+
**Unit tests** (65 tests, offline):
|
|
122
143
|
- Router: domain routing, subdomain suffix matching, duplicate registration, invalid URLs, unsupported platforms
|
|
123
|
-
- MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages
|
|
124
|
-
-
|
|
144
|
+
- MediumExtractor: clean_html, convert_to_markdown, empty content, non-article pages, _parse_freedium (heading remap, missing main-content), fallback on 403/429 (URL construction, exc.url contract, no-sleep on 429), no-fallback on 200, UnsupportedContentTypeError.url on Freedium path [003-medium-freedium-fallback]
|
|
145
|
+
- DevToExtractor: clean_html (title/cover/heading/image preservation, iframe/ltag embed→link, anchor stripping, non-article error), convert_to_markdown (headings/code/lists/images, no raw HTML, empty content error) [002-devto-provider]
|
|
146
|
+
- Fetch errors: HTTP 404, 503, timeout, connection error, size limit exceeded; `_no_retry_status_codes` immediate-raise + `_no_retry_codes` override [003-medium-freedium-fallback]
|
|
125
147
|
- Silent: no stdout/stderr output, no logging during extraction
|
|
126
148
|
|
|
127
|
-
**Integration tests** (
|
|
128
|
-
- Parametrized over 3 real stn1slv.medium.com articles
|
|
129
|
-
-
|
|
149
|
+
**Integration tests** (6 tests, network required):
|
|
150
|
+
- Parametrized over 3 real stn1slv.medium.com articles (including a known paywalled URL that exercises the Freedium fallback when medium.com returns 403) [003-medium-freedium-fallback]
|
|
151
|
+
- Parametrized over 3 real dev.to/stn1slv articles [002-devto-provider]
|
|
152
|
+
- Snapshot-based containment check: `expected_body in extracted_result` — tests pass regardless of whether medium.com or Freedium served the content (heading normalisation ensures identical output)
|
|
130
153
|
- 3 retries with 2-second delay on `FetchError` (covers transient 403s/timeouts)
|
|
131
154
|
- Run with: `make integration` or `uv run pytest tests/integration/ --override-ini=addopts=`
|
|
132
155
|
- Excluded from default `pytest` run via `addopts = "-m 'not integration'"` in pyproject.toml
|
|
@@ -155,10 +178,17 @@ Makefile # setup / test / integration / lint / typecheck / f
|
|
|
155
178
|
|----------|--------|-----------|
|
|
156
179
|
| HTTP client | `httpx` (sync) | Supports async in v2 without swapping dependency |
|
|
157
180
|
| HTML parser | `lxml` | Fast, tolerant; BeautifulSoup backend |
|
|
158
|
-
| Article targeting | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
|
|
181
|
+
| Article targeting (Medium) | `<article>` element | Stable semantic HTML5; consistent across Medium versions |
|
|
182
|
+
| Article targeting (dev.to) | `div#article-body` | dev.to does not use `<article>`; the id-scoped div already excludes all chrome | [002-devto-provider]
|
|
183
|
+
| dev.to cover image | Extracted from `header.crayons-article__header`, prepended to body | Cover image is outside the article body div — must be fetched separately | [002-devto-provider]
|
|
184
|
+
| dev.to embed handling | Replace `<iframe>` and `ltag__*` divs with anchor links | Embeds must not be silently dropped (FR-019); plain links are durable | [002-devto-provider]
|
|
159
185
|
| Markdown converter | `markdownify` with ATX + `code_language=""` | Fenced code blocks, standard headings |
|
|
160
186
|
| Routing | `pkgutil.iter_modules` auto-discovery + `@register` | SC-006: one new file = one new platform |
|
|
161
187
|
| Integration tests | Snapshot containment + retry | Durable against minor HTML changes; resilient to transient 403s |
|
|
188
|
+
| test_router.py domain example | Changed from `dev.to` to `substack.com` for "unsupported domain" test | Once DevToExtractor registers `dev.to`, those tests would no longer raise UnsupportedPlatformError | [002-devto-provider]
|
|
189
|
+
| Medium 403/429 fallback | Override `extract()` in `MediumExtractor`; `_no_retry_status_codes=frozenset({403,429})` on class | Immediate fallback with no medium.com retries; `BaseExtractor` extended with `_no_retry_codes` param for thread safety | [003-medium-freedium-fallback]
|
|
190
|
+
| Freedium HTML parsing | Dedicated `_parse_freedium()` method; `div.main-content`; h4→h3 remap | Freedium HTML is structurally incompatible with `clean_html()` (no `<article>`); heading remap ensures snapshot tests pass for both paths | [003-medium-freedium-fallback]
|
|
191
|
+
| Freedium exc.url contract | `inner_exc.url = url` unconditionally; error message is source-agnostic ("Fallback page…") | Preserves transparent-fallback contract (FR-028); `exc.url` is the authoritative field; message content is internal | [003-medium-freedium-fallback]
|
|
162
192
|
|
|
163
193
|
---
|
|
164
194
|
|
|
@@ -169,3 +199,7 @@ Makefile # setup / test / integration / lint / typecheck / f
|
|
|
169
199
|
- [x] Coding Standards — PEP 8, strict type hints, `mypy --strict` passes
|
|
170
200
|
- [x] Integration Testing — real Medium URLs, snapshot-based containment assertions
|
|
171
201
|
- [x] Packaging and Distribution — `pyproject.toml` + `src/` layout + `hatchling`; all Makefile targets use `uv run`
|
|
202
|
+
|
|
203
|
+
---
|
|
204
|
+
|
|
205
|
+
*Last Updated: 2026-05-15 | Sources appended: [specs/003-medium-freedium-fallback/plan.md]*
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
# mdfetch — Main Specification
|
|
2
|
+
|
|
3
|
+
**Last Updated**: 2026-05-14
|
|
4
|
+
**Sources**: [specs/001-mdfetch-medium-extractor/spec.md], [specs/002-devto-provider/spec.md]
|
|
5
|
+
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
## Overview
|
|
9
|
+
|
|
10
|
+
`mdfetch` is a Python library that extracts article content from web platforms and returns it as clean, well-structured Markdown. The library enforces a provider pattern — an abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider. Supported platforms: `medium.com` (and subdomains), `dev.to`.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## User Stories
|
|
15
|
+
|
|
16
|
+
### US-001 — Extract a Medium Article to Markdown (P1)
|
|
17
|
+
[Source: specs/001-mdfetch-medium-extractor]
|
|
18
|
+
|
|
19
|
+
A developer retrieves the readable content of a Medium article as clean Markdown by calling a single library function with the article URL. The library handles fetching, isolating the article body, stripping non-content elements, and returning formatted Markdown — with no configuration required.
|
|
20
|
+
|
|
21
|
+
**Acceptance Scenarios**:
|
|
22
|
+
1. Given a valid, publicly accessible Medium article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
|
|
23
|
+
2. Given a valid Medium article URL, the returned Markdown contains no navigation menus, "clap" interaction elements, social sharing buttons, or author biography sections.
|
|
24
|
+
3. Given a Medium article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
### US-002 — Receive Meaningful Errors for Unsupported or Invalid URLs (P2)
|
|
29
|
+
[Source: specs/001-mdfetch-medium-extractor]
|
|
30
|
+
|
|
31
|
+
A developer passing an unsupported domain, malformed URL, or unreachable page receives a clear, descriptive typed exception.
|
|
32
|
+
|
|
33
|
+
**Acceptance Scenarios**:
|
|
34
|
+
1. Given a URL from a domain other than Medium, `extract()` raises `UnsupportedPlatformError`.
|
|
35
|
+
2. Given a syntactically invalid URL string, `extract()` raises `InvalidURLError`.
|
|
36
|
+
3. Given a well-formed URL that returns an HTTP error (e.g., 404, 503), `extract()` raises `HTTPStatusError` including the status code.
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
### US-003 — Install and Use via Standard Package Manager (P3)
|
|
41
|
+
[Source: specs/001-mdfetch-medium-extractor]
|
|
42
|
+
|
|
43
|
+
A developer installs `mdfetch` from PyPI with `pip install mdfetch`, imports `extract`, and immediately converts Medium URLs to Markdown with zero configuration.
|
|
44
|
+
|
|
45
|
+
**Acceptance Scenarios**:
|
|
46
|
+
1. `pip install mdfetch` completes without errors and resolves all dependencies automatically.
|
|
47
|
+
2. `from mdfetch import extract` succeeds and the function is callable.
|
|
48
|
+
|
|
49
|
+
---
|
|
50
|
+
|
|
51
|
+
### US-004 — Extract a dev.to Article to Markdown (P1)
|
|
52
|
+
[Source: specs/002-devto-provider]
|
|
53
|
+
|
|
54
|
+
A developer retrieves the readable content of a dev.to article as clean Markdown by calling the same single library function used for Medium — passing only the article URL, with no additional configuration. The library routes the request to the dev.to provider, fetches the page, isolates the article body, strips all non-content elements, and returns formatted Markdown.
|
|
55
|
+
|
|
56
|
+
**Acceptance Scenarios**:
|
|
57
|
+
1. Given a valid, publicly accessible dev.to article URL, when `extract(url)` is called, then it returns a non-empty Markdown string containing the article's title and at least one paragraph, with no raw HTML tags.
|
|
58
|
+
2. Given a valid dev.to article URL, the returned Markdown contains no navigation menus, reaction buttons, comment sections, or author sidebar widgets.
|
|
59
|
+
3. Given a dev.to article with headings, code blocks, and lists, the returned Markdown preserves those structural elements as proper Markdown equivalents (`#`, ` ``` `, `-`).
|
|
60
|
+
|
|
61
|
+
---
|
|
62
|
+
|
|
63
|
+
### US-005 — Receive Meaningful Errors for Non-Article dev.to Pages (P2)
|
|
64
|
+
[Source: specs/002-devto-provider]
|
|
65
|
+
|
|
66
|
+
A developer passing a dev.to URL that points to a profile page, a tag listing, or a podcast page receives a clear, typed error indicating the dev.to domain is recognised but the page type is not extractable.
|
|
67
|
+
|
|
68
|
+
**Acceptance Scenarios**:
|
|
69
|
+
1. Given a dev.to URL pointing to an author profile page, `extract()` raises `UnsupportedContentTypeError` (domain recognised, content type not extractable).
|
|
70
|
+
2. Given a dev.to URL pointing to a tag listing page (e.g., `dev.to/t/kafka`), `extract()` raises `UnsupportedContentTypeError`.
|
|
71
|
+
3. Given a URL from a domain other than dev.to, `extract()` raises `UnsupportedPlatformError` (unchanged from existing behaviour).
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
### US-007 — Transparent Fallback on Blocked Medium Article (P1)
|
|
76
|
+
[Source: specs/003-medium-freedium-fallback]
|
|
77
|
+
|
|
78
|
+
A developer calls the library's extract function with a Medium URL. The article is behind a paywall or the user is geo-blocked, causing Medium to return a 403 error. Without any code changes, the library automatically retrieves the same article via the Freedium mirror and returns clean Markdown content.
|
|
79
|
+
|
|
80
|
+
**Acceptance Scenarios**:
|
|
81
|
+
1. Given a valid Medium article URL that returns 403 from medium.com, when `extract()` is called, then the library returns clean Markdown content retrieved via the Freedium mirror.
|
|
82
|
+
2. Given a valid Medium article URL that returns 403 from medium.com, when the Freedium mirror also fails, then the library raises an appropriate extraction error with `exc.url` set to the original Medium URL.
|
|
83
|
+
3. Given a valid Medium article URL that returns 403 from medium.com, when the library falls back to Freedium, then the caller receives the result without any knowledge of which source was used.
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
### US-008 — Automatic Fallback on Rate Limiting (P2)
|
|
88
|
+
[Source: specs/003-medium-freedium-fallback]
|
|
89
|
+
|
|
90
|
+
A developer calls the library's extract function for a Medium URL. Medium responds with 429 Too Many Requests. The library automatically uses the Freedium mirror as a fallback and returns clean Markdown without requiring the caller to retry.
|
|
91
|
+
|
|
92
|
+
**Acceptance Scenarios**:
|
|
93
|
+
1. Given a valid Medium article URL that returns 429 from medium.com, when `extract()` is called, then the library returns clean Markdown content retrieved via the Freedium mirror.
|
|
94
|
+
2. Given a valid Medium article URL that returns 429 from medium.com, when the Freedium mirror also fails, then the library raises an appropriate extraction error.
|
|
95
|
+
|
|
96
|
+
---
|
|
97
|
+
|
|
98
|
+
### US-009 — No Fallback When Primary Succeeds (P3)
|
|
99
|
+
[Source: specs/003-medium-freedium-fallback]
|
|
100
|
+
|
|
101
|
+
A developer calls the library's extract function for a publicly accessible Medium article. Medium responds successfully. The library returns the content directly without involving the Freedium mirror, preserving the existing happy-path behavior.
|
|
102
|
+
|
|
103
|
+
**Acceptance Scenarios**:
|
|
104
|
+
1. Given a valid Medium article URL that returns a successful response, when `extract()` is called, then the library returns clean Markdown content without making any request to the Freedium mirror.
|
|
105
|
+
|
|
106
|
+
---
|
|
107
|
+
|
|
108
|
+
### US-006 — Integration Tests Pass Against Real dev.to Article URLs (P3)
|
|
109
|
+
[Source: specs/002-devto-provider]
|
|
110
|
+
|
|
111
|
+
A developer runs the integration test suite and all dev.to integration tests pass against three reference articles. The tests confirm title, structural elements, and absence of HTML tags on live content.
|
|
112
|
+
|
|
113
|
+
**Acceptance Scenarios**:
|
|
114
|
+
1. When integration tests for the three reference dev.to URLs execute with a stable internet connection, all three pass without errors or assertion failures.
|
|
115
|
+
2. For each reference dev.to article, the extraction function returns non-empty Markdown free of HTML tags and containing recognisable content from the article.
|
|
116
|
+
|
|
117
|
+
---
|
|
118
|
+
|
|
119
|
+
## Functional Requirements
|
|
120
|
+
|
|
121
|
+
### Extraction
|
|
122
|
+
- **FR-001**: The library MUST expose a single primary function `extract(url: str) -> str` that accepts a URL string and returns the extracted article content as a Markdown string.
|
|
123
|
+
- **FR-002**: The library MUST enforce a provider pattern: a shared abstract base defines the extraction contract, and each supported platform is implemented as a separate, independent provider.
|
|
124
|
+
- **FR-003**: The library MUST include a Medium provider that fetches the article page, isolates the main article body, removes all non-content elements (navigation, social sharing widgets, reader interaction components, author biographies), and returns the body as Markdown.
|
|
125
|
+
- **FR-008**: The library MUST produce Markdown that preserves the structural hierarchy of the source article, including headings, code blocks, inline code, lists, and blockquotes.
|
|
126
|
+
- **FR-011**: The library MUST raise a descriptive error when the article body is located but yields no extractable text content.
|
|
127
|
+
- **FR-012**: The library MUST raise a distinct error — separate from the "unsupported platform" error — when a `medium.com` URL is provided but the page is not an article (e.g., an author profile or tag page).
|
|
128
|
+
- **FR-013**: The library MUST NOT emit any logging, print output, or diagnostic messages. All failure information is communicated exclusively through raised exceptions.
|
|
129
|
+
|
|
130
|
+
### Routing
|
|
131
|
+
- **FR-004**: The library MUST route extraction requests to the correct provider based on the URL's domain without requiring the caller to specify the provider explicitly. For v1, routing recognises only `medium.com` and its subdomains (e.g., `username.medium.com`).
|
|
132
|
+
- **FR-005**: The library MUST raise a descriptive error when given a URL whose domain has no registered provider.
|
|
133
|
+
- **FR-007**: The library MUST raise a descriptive error when given a URL that is syntactically invalid.
|
|
134
|
+
|
|
135
|
+
### Network
|
|
136
|
+
- **FR-006**: The library MUST raise a descriptive error when a network request fails (connection error, timeout, non-2xx HTTP status).
|
|
137
|
+
- **FR-014**: HTTP requests MUST use a standard browser-like User-Agent string so that web servers return readable HTML. The User-Agent MUST NOT identify the library by name or version. The library does not check or respect `robots.txt` in v1.
|
|
138
|
+
|
|
139
|
+
### Packaging & Testing
|
|
140
|
+
- **FR-009**: The library MUST be packaged and distributed via PyPI using modern Python packaging best practices, enabling installation through the standard package manager without additional steps.
|
|
141
|
+
- **FR-010**: The test suite MUST include integration tests that supply real article URLs (Medium and dev.to) to the extraction function and assert that the output matches expected Markdown structure and content.
|
|
142
|
+
|
|
143
|
+
### Freedium Fallback (Medium)
|
|
144
|
+
- **FR-020**: When a Medium article extraction results in HTTP 403, the system MUST immediately attempt extraction via the Freedium mirror (`https://freedium-mirror.cfd/{url}`) — no retries against medium.com are made first. [Source: specs/003-medium-freedium-fallback]
|
|
145
|
+
- **FR-021**: When a Medium article extraction results in HTTP 429, the system MUST immediately attempt extraction via the Freedium mirror — no retries against medium.com are made first. [Source: specs/003-medium-freedium-fallback]
|
|
146
|
+
- **FR-022**: When the Freedium mirror is used as a fallback, the system MUST return content in the same clean Markdown format as direct extraction; Freedium's `<h4>` headings are remapped to `<h3>` so both paths produce identical heading-level output. [Source: specs/003-medium-freedium-fallback]
|
|
147
|
+
- **FR-023**: When the primary Medium request succeeds (HTTP 200), the system MUST NOT make any request to the Freedium mirror. [Source: specs/003-medium-freedium-fallback]
|
|
148
|
+
- **FR-024**: When both the primary Medium request and the Freedium fallback fail, the system MUST raise an error consistent with the existing exception hierarchy; `exc.url` MUST be set to the original Medium URL (never the Freedium URL). [Source: specs/003-medium-freedium-fallback]
|
|
149
|
+
- **FR-025**: The Freedium fallback mechanism MUST require no changes to the caller's code — the public `extract()` interface remains unchanged. [Source: specs/003-medium-freedium-fallback]
|
|
150
|
+
- **FR-026**: The Freedium fallback MUST only apply to Medium provider URLs; other providers are unaffected. [Source: specs/003-medium-freedium-fallback]
|
|
151
|
+
- **FR-027**: The Freedium fallback MUST be unconditionally active for all Medium URL extractions — no caller configuration, opt-in flag, or extractor parameter is required or supported. [Source: specs/003-medium-freedium-fallback]
|
|
152
|
+
- **FR-028**: The Freedium fallback MUST be fully transparent to the caller — no warning, signal, metadata, or result field shall indicate which source (medium.com or Freedium) provided the content. [Source: specs/003-medium-freedium-fallback]
|
|
153
|
+
|
|
154
|
+
### dev.to Platform
|
|
155
|
+
- **FR-015**: The library MUST add `dev.to` to the provider router so that any URL with the `dev.to` domain is dispatched to the dev.to provider without any change to the caller's code. [Source: specs/002-devto-provider]
|
|
156
|
+
- **FR-016**: The library MUST include a dev.to provider that fetches the article page, isolates the main article body from `<div id="article-body">`, removes all non-content elements (navigation, social reaction widgets, comments, author sidebar, tag links), and returns the body as Markdown. [Source: specs/002-devto-provider]
|
|
157
|
+
- **FR-017**: The dev.to provider MUST produce Markdown that preserves both the cover image (from the article header) and all inline body images as Markdown image syntax (``). [Source: specs/002-devto-provider]
|
|
158
|
+
- **FR-018**: The dev.to provider MUST raise `UnsupportedContentTypeError` when a `dev.to` URL is provided but the page is not an article (e.g., an author profile, a tag listing, or an organisation page). [Source: specs/002-devto-provider]
|
|
159
|
+
- **FR-019**: The dev.to provider MUST replace embedded third-party content (GitHub Gists, CodePen demos, YouTube videos, liquid-tag embeds) with a plain Markdown link to the embedded resource URL. Embeds must not be silently dropped. [Source: specs/002-devto-provider]
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
## Key Entities
|
|
164
|
+
|
|
165
|
+
### URL (Input)
|
|
166
|
+
| Attribute | Type | Description |
|
|
167
|
+
|-----------|------|-------------|
|
|
168
|
+
| `raw` | `str` | The original string as supplied by the caller |
|
|
169
|
+
| `scheme` | `str` | Must be `http` or `https`; any other value triggers `InvalidURLError` |
|
|
170
|
+
| `hostname` | `str` | Lowercased hostname used for provider routing (e.g., `medium.com`) |
|
|
171
|
+
| `path` | `str` | Used to distinguish article pages from profile/tag pages |
|
|
172
|
+
|
|
173
|
+
**Validation**: syntactically parseable; scheme must be http/https; hostname must match a registered provider.
|
|
174
|
+
|
|
175
|
+
### Provider (Internal)
|
|
176
|
+
| Attribute | Type | Description |
|
|
177
|
+
|-----------|------|-------------|
|
|
178
|
+
| `DOMAINS` | `frozenset[str]` | Domain suffixes this provider handles (e.g., `{"medium.com"}`, `{"dev.to"}`) |
|
|
179
|
+
|
|
180
|
+
**Invariants**: Each domain suffix registered to exactly one provider. Stateless — every call is independent. Registered providers: `MediumExtractor` (medium.com), `DevToExtractor` (dev.to).
|
|
181
|
+
|
|
182
|
+
### ExtractionResult (Output)
|
|
183
|
+
| Attribute | Type | Description |
|
|
184
|
+
|-----------|------|-------------|
|
|
185
|
+
| `content` | `str` | The extracted article body as Markdown text |
|
|
186
|
+
|
|
187
|
+
**Validation**: Non-empty string; contains no raw HTML tags; preserves at least one heading or paragraph.
|
|
188
|
+
|
|
189
|
+
### Exception Hierarchy
|
|
190
|
+
```
|
|
191
|
+
MdfetchError (base)
|
|
192
|
+
├── InvalidURLError — syntactically invalid URL
|
|
193
|
+
├── UnsupportedPlatformError — domain not recognised by any provider
|
|
194
|
+
├── UnsupportedContentTypeError — recognised domain, but page is not an article
|
|
195
|
+
├── FetchError — network/HTTP failure
|
|
196
|
+
│ └── HTTPStatusError — non-2xx response; carries .status_code (int)
|
|
197
|
+
└── EmptyContentError — article body found but no extractable text
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
All exceptions carry `message: str` and `url: str | None`. `HTTPStatusError` additionally carries `status_code: int`.
|
|
201
|
+
|
|
202
|
+
---
|
|
203
|
+
|
|
204
|
+
## Call Lifecycle / State Transitions
|
|
205
|
+
|
|
206
|
+
```
|
|
207
|
+
caller provides URL string
|
|
208
|
+
│
|
|
209
|
+
▼
|
|
210
|
+
[VALIDATE URL] ──── invalid syntax ────► InvalidURLError
|
|
211
|
+
│
|
|
212
|
+
▼
|
|
213
|
+
[ROUTE by domain] ── no provider ──────► UnsupportedPlatformError
|
|
214
|
+
│
|
|
215
|
+
▼
|
|
216
|
+
[FETCH HTML] ─────── network failure ──► FetchError / HTTPStatusError
|
|
217
|
+
│
|
|
218
|
+
▼
|
|
219
|
+
[LOCATE ARTICLE] ─── not an article ───► UnsupportedContentTypeError
|
|
220
|
+
│
|
|
221
|
+
▼
|
|
222
|
+
[CLEAN & CONVERT] ── empty result ─────► EmptyContentError
|
|
223
|
+
│
|
|
224
|
+
▼
|
|
225
|
+
return str (Markdown)
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
---
|
|
229
|
+
|
|
230
|
+
## Edge Cases and Error Handling
|
|
231
|
+
|
|
232
|
+
- **Paywalled content (HTTP 403)**: When Medium returns HTTP 403 (paywall or geo-block), the library automatically retries via the Freedium mirror (`https://freedium-mirror.cfd/`). If Freedium also fails, the error raised carries `exc.url` set to the original Medium URL. [Source: specs/003-medium-freedium-fallback]
|
|
233
|
+
- **Freedium mirror unreachable**: If the Freedium mirror returns a network error, timeout, or non-2xx status, the fallback fails and the exception is propagated with `exc.url` set to the original Medium URL — Freedium's URL never appears in `exc.url`.
|
|
234
|
+
- **Freedium HTML structure**: Freedium uses `<div class="main-content">` (no `<article>`); if this element is absent, `UnsupportedContentTypeError` is raised with a source-agnostic message ("Fallback page missing main-content element").
|
|
235
|
+
- **Freedium heading levels**: Freedium renders section headings as `<h4>` vs. medium.com's `<h2>`/`<h3>`; `_parse_freedium()` remaps h4→h3, h5→h4, h6→h5 before conversion so snapshot tests pass regardless of which source served the content.
|
|
236
|
+
- **HTML structure changes**: If Medium changes its HTML structure and `<article>` is absent, `UnsupportedContentTypeError` is raised.
|
|
237
|
+
- **Empty article body**: If `<article>` is found but contains no extractable text, `EmptyContentError` is raised.
|
|
238
|
+
- **Network timeouts**: Covered by `FetchError` (30-second fixed timeout).
|
|
239
|
+
- **Oversized responses**: Responses exceeding 10 MB are rejected with `FetchError` to prevent OOM.
|
|
240
|
+
- **Profile/tag pages**: When a `medium.com` URL points to a non-article page, `UnsupportedContentTypeError` is raised (distinct from `UnsupportedPlatformError`).
|
|
241
|
+
- **HTTP 403 / transient failures**: Integration tests use a 3-retry helper with 2-second delay to handle transient rate limits.
|
|
242
|
+
- **dev.to profile pages**: When a `dev.to` URL points to an author profile (no `div#article-body`), `UnsupportedContentTypeError` is raised.
|
|
243
|
+
- **dev.to tag listing pages**: When a `dev.to` URL points to a tag page (e.g., `dev.to/t/kafka`), `UnsupportedContentTypeError` is raised.
|
|
244
|
+
- **dev.to liquid-tag embeds**: Embedded third-party widgets (GitHub Gists, CodePen, YouTube) serialised as `<div class="ltag__*" data-url="...">` are replaced with plain Markdown links; they are never silently dropped.
|
|
245
|
+
- **dev.to cover image**: The cover image lives in `<header class="crayons-article__header">`, not in `div#article-body` — the provider explicitly extracts and prepends it.
|
|
246
|
+
- **dev.to HTML structure changes**: If `div#article-body` is absent, `UnsupportedContentTypeError` is raised.
|
|
247
|
+
|
|
248
|
+
---
|
|
249
|
+
|
|
250
|
+
## Success Criteria
|
|
251
|
+
|
|
252
|
+
- **SC-001**: A developer can go from installing the library to extracting a real Medium article in fewer than 5 minutes with zero configuration.
|
|
253
|
+
- **SC-002**: The extraction function returns a result for a standard Medium article in under 10 seconds on a stable internet connection.
|
|
254
|
+
- **SC-003**: The returned Markdown for a standard Medium article contains no raw HTML tags.
|
|
255
|
+
- **SC-004**: The returned Markdown for an article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax.
|
|
256
|
+
- **SC-005**: 100% of integration tests pass against a curated set of real Medium article URLs at the time of initial release.
|
|
257
|
+
- **SC-006**: Adding support for a second platform requires creating exactly one new file (one new provider class decorated with `@register`). The router auto-discovers all provider modules at import time; no changes to shared library code are required.
|
|
258
|
+
- **SC-007**: A developer already using the library for Medium can extract a dev.to article without any code change — only the URL changes. [Source: specs/002-devto-provider]
|
|
259
|
+
- **SC-008**: The extraction function returns a result for a standard dev.to article in under 10 seconds on a stable internet connection. [Source: specs/002-devto-provider]
|
|
260
|
+
- **SC-009**: The returned Markdown for any of the three reference dev.to articles contains no raw HTML tags (verified by automated assertion). [Source: specs/002-devto-provider]
|
|
261
|
+
- **SC-010**: The returned Markdown for a dev.to article with headings, lists, and code blocks preserves all three structural element types in correct Markdown syntax. [Source: specs/002-devto-provider]
|
|
262
|
+
- **SC-011**: 100% of integration tests pass against the three provided reference dev.to article URLs at the time of release. [Source: specs/002-devto-provider]
|
|
263
|
+
- **SC-012**: The dev.to provider is delivered as exactly one new file; no existing source files are modified (except `test_router.py` for expected domain-example maintenance when the provider registers `dev.to`). [Source: specs/002-devto-provider]
|
|
264
|
+
|
|
265
|
+
---
|
|
266
|
+
|
|
267
|
+
## Assumptions
|
|
268
|
+
|
|
269
|
+
- The library targets developers as its primary users; no graphical interface or configuration file is required.
|
|
270
|
+
- Medium articles used in integration testing are publicly accessible (not behind a paywall).
|
|
271
|
+
- No response caching; each call performs a fresh network request.
|
|
272
|
+
- Rate limiting or authentication with Medium's servers is out of scope for v1.
|
|
273
|
+
- The library supports Python 3.12 and later.
|
|
274
|
+
- The library operates on publicly accessible HTML; it does not execute JavaScript or render dynamic content.
|
|
275
|
+
- Network timeouts use a fixed default of 30 seconds (not user-configurable in v1).
|
|
276
|
+
- **SC-013**: Articles that previously failed with a 403 paywall error are successfully extracted in at least 90% of cases where the Freedium mirror has the content available. [Source: specs/003-medium-freedium-fallback]
|
|
277
|
+
- **SC-014**: Articles that previously failed with a 429 rate-limit error are successfully extracted via fallback without requiring the caller to retry. [Source: specs/003-medium-freedium-fallback]
|
|
278
|
+
- **SC-015**: Zero changes are required in existing caller code to benefit from the Freedium fallback — existing integrations continue to work as-is. [Source: specs/003-medium-freedium-fallback]
|
|
279
|
+
- **SC-016**: When the primary Medium request succeeds, there is no additional latency attributable to the fallback mechanism. [Source: specs/003-medium-freedium-fallback]
|
|
280
|
+
- **SC-017**: All existing unit and integration tests for the Medium extractor continue to pass without modification after the fallback is introduced. [Source: specs/003-medium-freedium-fallback]
|
|
281
|
+
|
|
282
|
+
---
|
|
283
|
+
|
|
284
|
+
*Last Updated: 2026-05-15 | Sources appended: [specs/003-medium-freedium-fallback/spec.md]*
|
|
@@ -11,11 +11,18 @@ src/mdfetch/
|
|
|
11
11
|
├── base.py # BaseExtractor ABC
|
|
12
12
|
├── router.py # Domain-to-provider routing
|
|
13
13
|
└── providers/
|
|
14
|
-
|
|
14
|
+
├── __init__.py
|
|
15
|
+
├── medium.py # MediumExtractor (medium.com + *.medium.com)
|
|
16
|
+
└── devto.py # DevToExtractor (dev.to)
|
|
15
17
|
|
|
16
18
|
tests/
|
|
17
19
|
├── unit/ # pytest unit tests (no network)
|
|
18
|
-
└── integration/ #
|
|
20
|
+
└── integration/ # real network tests (Medium + dev.to URLs + snapshots)
|
|
21
|
+
|
|
22
|
+
.github/workflows/
|
|
23
|
+
├── ci.yml # lint + unit tests on push/PR (Python 3.12–3.14)
|
|
24
|
+
├── integration.yml # scheduled integration tests every Friday 23:30 UTC
|
|
25
|
+
└── publish.yml # PyPI publish on release
|
|
19
26
|
|
|
20
27
|
specs/ # Speckit feature specifications
|
|
21
28
|
pyproject.toml # hatchling build, uv package manager
|
|
@@ -38,15 +45,11 @@ make test # unit tests only
|
|
|
38
45
|
make lint # ruff check
|
|
39
46
|
make format # ruff format
|
|
40
47
|
make build # uv build (wheel + sdist)
|
|
41
|
-
make upgrade-deps # uv sync --upgrade
|
|
42
|
-
|
|
43
|
-
uv run mypy src/
|
|
48
|
+
make upgrade-deps # uv sync --all-extras --upgrade
|
|
49
|
+
make integration # integration tests (network required)
|
|
50
|
+
uv run mypy src/ # type check
|
|
44
51
|
```
|
|
45
52
|
|
|
46
|
-
## Recent Changes
|
|
47
|
-
|
|
48
|
-
- 001-mdfetch-medium-extractor: Initial library release — `extract()` API, Medium provider, typed exceptions, auto-discovery routing, PyPI packaging, snapshot integration tests
|
|
49
|
-
|
|
50
53
|
<!-- SPECKIT START -->
|
|
51
|
-
**
|
|
54
|
+
**Recent changes**: `003-medium-freedium-fallback` — Transparent Freedium mirror fallback for Medium 403/429 responses; `_no_retry_status_codes` hook on `BaseExtractor`; `_parse_freedium()` with h4→h3 heading remap on `MediumExtractor`
|
|
52
55
|
<!-- SPECKIT END -->
|