mdfetch 0.5.1__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mdfetch-0.5.1 → mdfetch-0.5.2}/PKG-INFO +1 -1
- {mdfetch-0.5.1 → mdfetch-0.5.2}/pyproject.toml +1 -1
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/base.py +33 -10
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/cli.py +23 -3
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/devto.py +9 -9
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/medium.py +11 -3
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/substack.py +10 -10
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/router.py +7 -6
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_cli.py +25 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_medium_extractor.py +23 -6
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_router.py +20 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/uv.lock +1 -1
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-analyze/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-archive-run/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-checklist/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-clarify/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-constitution/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-commit/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-feature/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-initialize/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-remote/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-git-validate/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-implement/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-plan/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-reconcile-run/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-specify/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-tasks/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.claude/skills/speckit-taskstoissues/SKILL.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.analyze.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.archive.run.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.checklist.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.clarify.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.constitution.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.implement.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.plan.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.reconcile.run.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.specify.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.tasks.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gemini/commands/speckit.taskstoissues.toml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gitattributes +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/copilot-instructions.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/ci.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/integration.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.github/workflows/publish.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.gitignore +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.python-version +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/.registry +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/LICENSE +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/README.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/commands/archive.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/archive/extension.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/README.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.commit.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.feature.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.initialize.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.remote.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/commands/speckit.git.validate.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/config-template.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/extension.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/git-config.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/auto-commit.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/git-common.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/bash/initialize-repo.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/auto-commit.ps1 +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1 +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/git-common.ps1 +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1 +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/LICENSE +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/README.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/commands/reconcile.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/reconcile/extension.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/feature.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/init-options.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integration.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/claude.manifest.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/gemini.manifest.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/integrations/speckit.manifest.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/changelog.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/constitution.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/memory/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/check-prerequisites.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/common.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/create-new-feature.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/setup-plan.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/scripts/bash/setup-tasks.sh +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/checklist-template.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/constitution-template.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/plan-template.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/spec-template.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/templates/tasks-template.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/workflows/speckit/workflow.yml +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/workflows/workflow-registry.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/.vscode/settings.json +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/CLAUDE.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/GEMINI.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/LICENSE +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/Makefile +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/README.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/contracts/api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/contracts/public-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/002-devto-provider/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/contracts/extract-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/003-medium-freedium-fallback/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/004-remove-backoff/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/contracts/extractor-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/005-substack-provider/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/contracts/public-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/006-thenewstack-provider/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/contracts/public-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/007-dzone-provider/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/contracts/cli.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/008-mdfetch-cli/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/checklists/requirements.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/contracts/formula.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/contracts/tap-update-job.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/data-model.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/plan.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/quickstart.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/research.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/spec.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/specs/009-homebrew-tap-formula/tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/__init__.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/exceptions.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/__init__.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/dzone.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/src/mdfetch/providers/thenewstack.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/__init__.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/conftest.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/__init__.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/conftest.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/architecting-the-asynchronous-agent.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-december-2025.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-july-2025.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/devto-integration-digest-march-2026.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-image-classification-pipeline-camel-djl.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-integration-patterns-fail-production.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/dzone-kiro-feature-to-requirements-design-tasks.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/from-drift-to-parity.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/integration-digest-december-2025.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/substack-api-trends-2025.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/substack-kafka-topic-types.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-api-mcp-agent.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-async-apis.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-developer-portal-api.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-json-schema-ai.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/snapshots/thenewstack-mcp-api-governance.md +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_cli_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_devto_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_dzone_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_medium_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_substack_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/integration/test_thenewstack_integration.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/__init__.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_devto_extractor.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_dzone_extractor.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_fetch_errors.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_silent.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_substack_extractor.py +0 -0
- {mdfetch-0.5.1 → mdfetch-0.5.2}/tests/unit/test_thenewstack_extractor.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mdfetch
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.2
|
|
4
4
|
Summary: Extract article content from web platforms and return it as clean Markdown.
|
|
5
5
|
Project-URL: Homepage, https://github.com/stn1slv/md-fetch
|
|
6
6
|
Project-URL: Source, https://github.com/stn1slv/md-fetch
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
import re
|
|
6
6
|
import time
|
|
7
7
|
from abc import ABC, abstractmethod
|
|
8
|
+
from collections.abc import Iterable
|
|
8
9
|
from typing import Any
|
|
9
10
|
|
|
10
11
|
import httpx
|
|
@@ -28,6 +29,10 @@ class BaseExtractor(ABC):
|
|
|
28
29
|
"""Contract all platform-specific extractors must fulfil."""
|
|
29
30
|
|
|
30
31
|
DOMAINS: frozenset[str] = frozenset()
|
|
32
|
+
# When True, the router also matches hostnames that are subdomains of any
|
|
33
|
+
# entry in DOMAINS (e.g. ``foo.medium.com`` → MediumExtractor). Leave as
|
|
34
|
+
# False for single-tenant sites where subdomains are not article URLs.
|
|
35
|
+
MATCH_SUBDOMAINS: bool = False
|
|
31
36
|
_no_retry_status_codes: frozenset[int] = frozenset()
|
|
32
37
|
|
|
33
38
|
# FR-014: use a browser-like UA (no mdfetch-specific branding) so servers serve readable HTML
|
|
@@ -139,17 +144,35 @@ class BaseExtractor(ABC):
|
|
|
139
144
|
|
|
140
145
|
return md
|
|
141
146
|
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
147
|
+
_DEFAULT_EMBED_URL_ATTRS: tuple[str, ...] = ("src", "data-src", "data-url", "href")
|
|
148
|
+
|
|
149
|
+
@classmethod
|
|
150
|
+
def _replace_embeds_with_links(
|
|
151
|
+
cls,
|
|
152
|
+
embeds: Iterable[Tag],
|
|
153
|
+
soup: BeautifulSoup,
|
|
154
|
+
*,
|
|
155
|
+
attrs: tuple[str, ...] = _DEFAULT_EMBED_URL_ATTRS,
|
|
156
|
+
) -> None:
|
|
157
|
+
"""Replace each tag in *embeds* with an anchor pointing at its URL, or decompose."""
|
|
158
|
+
for embed in embeds:
|
|
159
|
+
url = ""
|
|
160
|
+
for attr in attrs:
|
|
161
|
+
val = embed.get(attr)
|
|
162
|
+
if val:
|
|
163
|
+
url = str(val)
|
|
164
|
+
break
|
|
165
|
+
if url:
|
|
166
|
+
link = soup.new_tag("a", href=url)
|
|
167
|
+
link.string = url
|
|
168
|
+
embed.replace_with(link)
|
|
151
169
|
else:
|
|
152
|
-
|
|
170
|
+
embed.decompose()
|
|
171
|
+
|
|
172
|
+
@classmethod
|
|
173
|
+
def _replace_iframes_with_links(cls, container: Tag, soup: BeautifulSoup) -> None:
|
|
174
|
+
"""Replace ``<iframe>`` elements inside *container* with plain anchor links."""
|
|
175
|
+
cls._replace_embeds_with_links(container.find_all("iframe"), soup)
|
|
153
176
|
|
|
154
177
|
def extract(self, url: str, *, retries: int = 3, retry_delay: float = 2.0) -> str:
|
|
155
178
|
"""Orchestrate fetch → clean → convert and return Markdown."""
|
|
@@ -4,7 +4,9 @@ Command-line interface for mdfetch.
|
|
|
4
4
|
|
|
5
5
|
from __future__ import annotations
|
|
6
6
|
|
|
7
|
+
import os
|
|
7
8
|
import sys
|
|
9
|
+
from urllib.parse import urlparse
|
|
8
10
|
|
|
9
11
|
import click
|
|
10
12
|
|
|
@@ -36,8 +38,28 @@ from mdfetch.router import supported_domains
|
|
|
36
38
|
show_default=True,
|
|
37
39
|
help="Seconds to wait between retry attempts",
|
|
38
40
|
)
|
|
39
|
-
|
|
41
|
+
@click.option(
|
|
42
|
+
"-f",
|
|
43
|
+
"--force",
|
|
44
|
+
is_flag=True,
|
|
45
|
+
default=False,
|
|
46
|
+
help="Overwrite the output file if it already exists",
|
|
47
|
+
)
|
|
48
|
+
def main(
|
|
49
|
+
url: str,
|
|
50
|
+
output: str | None,
|
|
51
|
+
retries: int,
|
|
52
|
+
retry_delay: float,
|
|
53
|
+
force: bool,
|
|
54
|
+
) -> None:
|
|
40
55
|
"""Fetch and extract Markdown from the given URL."""
|
|
56
|
+
if output and os.path.exists(output) and not force:
|
|
57
|
+
click.secho(
|
|
58
|
+
f"Error: '{output}' already exists. Use --force to overwrite.",
|
|
59
|
+
err=True,
|
|
60
|
+
fg="red",
|
|
61
|
+
)
|
|
62
|
+
sys.exit(1)
|
|
41
63
|
try:
|
|
42
64
|
content = extract(url, retries=retries, retry_delay=retry_delay)
|
|
43
65
|
|
|
@@ -47,8 +69,6 @@ def main(url: str, output: str | None, retries: int, retry_delay: float) -> None
|
|
|
47
69
|
else:
|
|
48
70
|
click.echo(content)
|
|
49
71
|
except UnsupportedPlatformError as e:
|
|
50
|
-
from urllib.parse import urlparse
|
|
51
|
-
|
|
52
72
|
domain = (urlparse(e.url).hostname if e.url else None) or str(e)
|
|
53
73
|
domains = ", ".join(sorted(supported_domains()))
|
|
54
74
|
click.secho(
|
|
@@ -30,15 +30,15 @@ class DevToExtractor(BaseExtractor):
|
|
|
30
30
|
# Replace iframes with plain anchor links (FR-008)
|
|
31
31
|
self._replace_iframes_with_links(body, soup)
|
|
32
32
|
|
|
33
|
-
# Replace dev.to liquid-tag embeds with plain anchor links (FR-008)
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
33
|
+
# Replace dev.to liquid-tag embeds with plain anchor links (FR-008).
|
|
34
|
+
# Anchor at the start of the class name so unrelated classes that merely
|
|
35
|
+
# contain "ltag" (e.g. "ultraltagrelated") are not matched — bs4 matches
|
|
36
|
+
# each class independently via re.search.
|
|
37
|
+
self._replace_embeds_with_links(
|
|
38
|
+
body.find_all(class_=re.compile(r"^ltag(?:[-_]|$)", re.IGNORECASE)),
|
|
39
|
+
soup,
|
|
40
|
+
attrs=("data-url", "data-src", "src"),
|
|
41
|
+
)
|
|
42
42
|
|
|
43
43
|
# Strip empty anchor-name links inserted before headings
|
|
44
44
|
for anchor in body.find_all("a", attrs={"name": True}):
|
|
@@ -22,6 +22,7 @@ class MediumExtractor(BaseExtractor):
|
|
|
22
22
|
"""Extracts article content from medium.com and its subdomains."""
|
|
23
23
|
|
|
24
24
|
DOMAINS: frozenset[str] = frozenset({"medium.com"})
|
|
25
|
+
MATCH_SUBDOMAINS = True
|
|
25
26
|
_FREEDIUM_BASE = "https://freedium-mirror.cfd/"
|
|
26
27
|
_no_retry_status_codes: frozenset[int] = frozenset({403, 429})
|
|
27
28
|
# Smart-quote characters that Medium serves but Freedium replaces with ASCII;
|
|
@@ -91,12 +92,19 @@ class MediumExtractor(BaseExtractor):
|
|
|
91
92
|
return md
|
|
92
93
|
|
|
93
94
|
def _parse_freedium(self, soup: BeautifulSoup) -> str:
|
|
94
|
-
"""Parse Freedium mirror HTML,
|
|
95
|
-
|
|
95
|
+
"""Parse Freedium mirror HTML, whose Svelte rebuild holds the body in div.prose."""
|
|
96
|
+
# Scope to the article so a stray .prose block (e.g. a bio/summary) can't
|
|
97
|
+
# match first; fall back to a bare .prose if the <article> wrapper is absent.
|
|
98
|
+
content = soup.select_one("article .prose") or soup.find("div", class_="prose")
|
|
96
99
|
if not isinstance(content, Tag):
|
|
97
100
|
raise UnsupportedContentTypeError(
|
|
98
|
-
"Fallback page missing
|
|
101
|
+
"Fallback page missing prose element",
|
|
99
102
|
)
|
|
103
|
+
# Freedium's Shiki highlighter emits each code block twice — a light-theme
|
|
104
|
+
# and a dark-theme variant — so the visible text is duplicated. Drop the
|
|
105
|
+
# dark variant before conversion to avoid repeated fenced-code output.
|
|
106
|
+
for pre in content.select("pre.github-dark"):
|
|
107
|
+
pre.decompose()
|
|
100
108
|
# Freedium renders section headings one level deeper than medium.com (h4 vs h3).
|
|
101
109
|
# Remap so the output heading levels match the medium.com direct path.
|
|
102
110
|
for level in (4, 5, 6):
|
|
@@ -18,6 +18,7 @@ class SubstackExtractor(BaseExtractor):
|
|
|
18
18
|
"""Extracts article content from substack.com and its subdomains."""
|
|
19
19
|
|
|
20
20
|
DOMAINS: frozenset[str] = frozenset({"substack.com"})
|
|
21
|
+
MATCH_SUBDOMAINS = True
|
|
21
22
|
|
|
22
23
|
def clean_html(self, soup: BeautifulSoup) -> Tag:
|
|
23
24
|
"""Isolate the article body and strip all non-content elements."""
|
|
@@ -36,16 +37,15 @@ class SubstackExtractor(BaseExtractor):
|
|
|
36
37
|
|
|
37
38
|
# Convert other Substack embed containers to plain anchor links (FR-011)
|
|
38
39
|
_safe_components = {"SubscribeWidget", "Image2ToDOM"}
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
embed.decompose()
|
|
40
|
+
self._replace_embeds_with_links(
|
|
41
|
+
[
|
|
42
|
+
e
|
|
43
|
+
for e in body.find_all(attrs={"data-component-name": True})
|
|
44
|
+
if e.get("data-component-name") not in _safe_components
|
|
45
|
+
],
|
|
46
|
+
soup,
|
|
47
|
+
attrs=("href", "data-url", "src"),
|
|
48
|
+
)
|
|
49
49
|
|
|
50
50
|
# Prepend subtitle from post-header (FR-005)
|
|
51
51
|
header = soup.find("div", class_="post-header")
|
|
@@ -37,15 +37,16 @@ def route(url: str) -> BaseExtractor:
|
|
|
37
37
|
if parsed.scheme not in ("http", "https") or not hostname:
|
|
38
38
|
raise InvalidURLError(f"Invalid URL: {url!r}", url=url)
|
|
39
39
|
|
|
40
|
-
# Exact match first; fall back to subdomain suffix check
|
|
41
|
-
#
|
|
42
|
-
# Sort candidates by length descending so the most-specific
|
|
43
|
-
# multiple registered domains are suffixes of the same
|
|
40
|
+
# Exact match first; fall back to a subdomain suffix check only for providers
|
|
41
|
+
# that opt in via MATCH_SUBDOMAINS=True (multi-tenant sites like Medium and
|
|
42
|
+
# Substack). Sort candidates by length descending so the most-specific
|
|
43
|
+
# suffix wins when multiple registered domains are suffixes of the same host.
|
|
44
44
|
provider_cls = _REGISTRY.get(hostname)
|
|
45
45
|
if provider_cls is None:
|
|
46
46
|
for domain in sorted(_REGISTRY, key=len, reverse=True):
|
|
47
|
-
|
|
48
|
-
|
|
47
|
+
candidate = _REGISTRY[domain]
|
|
48
|
+
if candidate.MATCH_SUBDOMAINS and hostname.endswith(f".{domain}"):
|
|
49
|
+
provider_cls = candidate
|
|
49
50
|
break
|
|
50
51
|
|
|
51
52
|
if provider_cls is None:
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import pathlib
|
|
6
|
+
|
|
5
7
|
import pytest
|
|
6
8
|
import pytest_mock
|
|
7
9
|
from click.testing import CliRunner
|
|
@@ -41,3 +43,26 @@ def test_unsupported_domain_error_message(runner: CliRunner) -> None:
|
|
|
41
43
|
assert result.exit_code == 1
|
|
42
44
|
assert "'google.com' is not a supported platform." in result.output
|
|
43
45
|
assert "Supported domains:" in result.output
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_output_refuses_to_clobber_existing_file(
|
|
49
|
+
mocker: pytest_mock.MockerFixture, runner: CliRunner, tmp_path: pathlib.Path
|
|
50
|
+
) -> None:
|
|
51
|
+
mocker.patch("mdfetch.cli.extract", return_value="# Test")
|
|
52
|
+
out = tmp_path / "out.md"
|
|
53
|
+
out.write_text("existing content")
|
|
54
|
+
result = runner.invoke(main, ["https://dev.to/test", "-o", str(out)])
|
|
55
|
+
assert result.exit_code == 1
|
|
56
|
+
assert "already exists" in result.output
|
|
57
|
+
assert out.read_text() == "existing content"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_output_force_overwrites_existing_file(
|
|
61
|
+
mocker: pytest_mock.MockerFixture, runner: CliRunner, tmp_path: pathlib.Path
|
|
62
|
+
) -> None:
|
|
63
|
+
mocker.patch("mdfetch.cli.extract", return_value="# Test")
|
|
64
|
+
out = tmp_path / "out.md"
|
|
65
|
+
out.write_text("existing content")
|
|
66
|
+
result = runner.invoke(main, ["https://dev.to/test", "-o", str(out), "--force"])
|
|
67
|
+
assert result.exit_code == 0
|
|
68
|
+
assert out.read_text() == "# Test"
|
|
@@ -174,27 +174,38 @@ class TestConvertToMarkdown:
|
|
|
174
174
|
assert "line one \nline two" in md
|
|
175
175
|
|
|
176
176
|
|
|
177
|
+
# Mirrors the Svelte/Tailwind Freedium rebuild: body in div.prose, code blocks
|
|
178
|
+
# rendered twice via Shiki (light + dark theme variants).
|
|
177
179
|
_FREEDIUM_ARTICLE_HTML = """
|
|
178
180
|
<html><body>
|
|
179
|
-
<
|
|
181
|
+
<main>
|
|
182
|
+
<article>
|
|
183
|
+
<div class="prose max-w-none prose-external-links">
|
|
180
184
|
<h4>Section One</h4>
|
|
181
185
|
<p>First paragraph of the article.</p>
|
|
182
186
|
<h4>Section Two</h4>
|
|
183
187
|
<p>Second paragraph with more content.</p>
|
|
184
|
-
<
|
|
188
|
+
<div class="dark:hidden">
|
|
189
|
+
<pre class="shiki github-light"><code>print("hello")</code></pre>
|
|
190
|
+
</div>
|
|
191
|
+
<div class="hidden dark:block">
|
|
192
|
+
<pre class="shiki github-dark"><code>print("hello")</code></pre>
|
|
193
|
+
</div>
|
|
185
194
|
</div>
|
|
195
|
+
</article>
|
|
196
|
+
</main>
|
|
186
197
|
</body></html>
|
|
187
198
|
"""
|
|
188
199
|
|
|
189
200
|
_FREEDIUM_NO_CONTENT_HTML = """
|
|
190
201
|
<html><body>
|
|
191
|
-
<div class="header">No
|
|
202
|
+
<div class="header">No prose div here</div>
|
|
192
203
|
</body></html>
|
|
193
204
|
"""
|
|
194
205
|
|
|
195
206
|
|
|
196
207
|
class TestParseFreedium:
|
|
197
|
-
def
|
|
208
|
+
def test_returns_markdown_from_prose(self, extractor: MediumExtractor) -> None:
|
|
198
209
|
soup = BeautifulSoup(_FREEDIUM_ARTICLE_HTML, "lxml")
|
|
199
210
|
md = extractor._parse_freedium(soup)
|
|
200
211
|
assert "Section One" in md
|
|
@@ -206,9 +217,15 @@ class TestParseFreedium:
|
|
|
206
217
|
assert "###" in md
|
|
207
218
|
assert "####" not in md
|
|
208
219
|
|
|
209
|
-
def
|
|
220
|
+
def test_dark_theme_code_block_deduplicated(self, extractor: MediumExtractor) -> None:
|
|
221
|
+
"""Shiki renders each code block twice (light + dark); only one must survive."""
|
|
222
|
+
soup = BeautifulSoup(_FREEDIUM_ARTICLE_HTML, "lxml")
|
|
223
|
+
md = extractor._parse_freedium(soup)
|
|
224
|
+
assert md.count('print("hello")') == 1
|
|
225
|
+
|
|
226
|
+
def test_raises_when_prose_missing(self, extractor: MediumExtractor) -> None:
|
|
210
227
|
soup = BeautifulSoup(_FREEDIUM_NO_CONTENT_HTML, "lxml")
|
|
211
|
-
with pytest.raises(UnsupportedContentTypeError, match="
|
|
228
|
+
with pytest.raises(UnsupportedContentTypeError, match="prose"):
|
|
212
229
|
extractor._parse_freedium(soup)
|
|
213
230
|
|
|
214
231
|
|
|
@@ -81,6 +81,26 @@ class TestUrlValidation:
|
|
|
81
81
|
provider = route("https://dzone.com/articles/some-article")
|
|
82
82
|
assert isinstance(provider, DZoneExtractor)
|
|
83
83
|
|
|
84
|
+
def test_rejects_subdomain_of_single_tenant_domain(self) -> None:
|
|
85
|
+
# dev.to does not opt in to MATCH_SUBDOMAINS, so foo.dev.to must not
|
|
86
|
+
# be routed to DevToExtractor — it should raise UnsupportedPlatformError.
|
|
87
|
+
with pytest.raises(UnsupportedPlatformError):
|
|
88
|
+
route("https://foo.dev.to/some-article")
|
|
89
|
+
|
|
90
|
+
def test_rejects_subdomain_of_dzone(self) -> None:
|
|
91
|
+
with pytest.raises(UnsupportedPlatformError):
|
|
92
|
+
route("https://blog.dzone.com/articles/some-article")
|
|
93
|
+
|
|
94
|
+
def test_rejects_subdomain_of_thenewstack(self) -> None:
|
|
95
|
+
with pytest.raises(UnsupportedPlatformError):
|
|
96
|
+
route("https://blog.thenewstack.io/some-article")
|
|
97
|
+
|
|
98
|
+
def test_routes_substack_subdomain(self) -> None:
|
|
99
|
+
from mdfetch.providers.substack import SubstackExtractor
|
|
100
|
+
|
|
101
|
+
provider = route("https://newsletter.substack.com/p/post")
|
|
102
|
+
assert isinstance(provider, SubstackExtractor)
|
|
103
|
+
|
|
84
104
|
|
|
85
105
|
class TestSupportedDomains:
|
|
86
106
|
def test_returns_frozenset(self) -> None:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/create-new-feature.ps1
RENAMED
|
File without changes
|
|
File without changes
|
{mdfetch-0.5.1 → mdfetch-0.5.2}/.specify/extensions/git/scripts/powershell/initialize-repo.ps1
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mdfetch-0.5.1 → mdfetch-0.5.2}/specs/001-mdfetch-medium-extractor/checklists/requirements.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|