mdfetch 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mdfetch/__init__.py ADDED
@@ -0,0 +1,35 @@
1
+ """mdfetch — extract article content from web platforms as clean Markdown."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from mdfetch.exceptions import (
6
+ EmptyContentError,
7
+ FetchError,
8
+ HTTPStatusError,
9
+ InvalidURLError,
10
+ MdfetchError,
11
+ UnsupportedContentTypeError,
12
+ UnsupportedPlatformError,
13
+ )
14
+ from mdfetch.router import route
15
+
16
+ __all__ = [
17
+ "extract",
18
+ "MdfetchError",
19
+ "InvalidURLError",
20
+ "UnsupportedPlatformError",
21
+ "UnsupportedContentTypeError",
22
+ "FetchError",
23
+ "HTTPStatusError",
24
+ "EmptyContentError",
25
+ ]
26
+
27
+
28
+ def extract(url: str, *, retries: int = 3, retry_delay: float = 2.0) -> str:
29
+ """Extract article content from *url* and return it as Markdown.
30
+
31
+ On transient network failures (timeouts, connection errors, non-2xx responses)
32
+ the request is retried up to *retries* times with *retry_delay* seconds between
33
+ attempts. Set ``retries=1`` to disable retry behaviour.
34
+ """
35
+ return route(url).extract(url, retries=retries, retry_delay=retry_delay)
mdfetch/base.py ADDED
@@ -0,0 +1,93 @@
1
+ """Abstract base class for all platform extractors."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from abc import ABC, abstractmethod
7
+
8
+ import httpx
9
+ from bs4 import BeautifulSoup
10
+ from bs4.element import Tag
11
+
12
+ from mdfetch.exceptions import FetchError, HTTPStatusError, MdfetchError
13
+
14
+ _MAX_RESPONSE_BYTES = 10 * 1024 * 1024 # 10 MB — guard against runaway responses
15
+
16
+
17
+ class BaseExtractor(ABC):
18
+ """Contract all platform-specific extractors must fulfil."""
19
+
20
+ DOMAINS: frozenset[str] = frozenset()
21
+
22
+ # FR-014: use a browser-like UA (no mdfetch-specific branding) so servers serve readable HTML
23
+ _USER_AGENT = (
24
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
25
+ "AppleWebKit/537.36 (KHTML, like Gecko) "
26
+ "Chrome/120.0.0.0 Safari/537.36"
27
+ )
28
+
29
+ def fetch_html(self, url: str, *, retries: int = 3, retry_delay: float = 2.0) -> str:
30
+ """Fetch raw HTML from *url* using a 30-second timeout and a 10 MB size cap.
31
+
32
+ Retries up to *retries* times on transient :class:`FetchError` (network errors,
33
+ timeouts, non-2xx responses), waiting *retry_delay* seconds between attempts.
34
+ """
35
+ last_exc: FetchError | None = None
36
+ for attempt in range(max(1, retries)):
37
+ try:
38
+ return self._do_fetch(url)
39
+ except FetchError as exc:
40
+ last_exc = exc
41
+ if attempt < retries - 1:
42
+ time.sleep(retry_delay)
43
+ assert last_exc is not None
44
+ raise last_exc
45
+
46
+ def _do_fetch(self, url: str) -> str:
47
+ """Single HTTP fetch attempt (no retry logic)."""
48
+ headers = {"User-Agent": self._USER_AGENT}
49
+ try:
50
+ with httpx.Client(timeout=30.0, follow_redirects=True) as client:
51
+ with client.stream("GET", url, headers=headers) as response:
52
+ if not response.is_success:
53
+ raise HTTPStatusError(
54
+ f"HTTP {response.status_code} fetching {url}",
55
+ status_code=response.status_code,
56
+ url=url,
57
+ )
58
+ chunks: list[bytes] = []
59
+ total = 0
60
+ for chunk in response.iter_bytes():
61
+ total += len(chunk)
62
+ if total > _MAX_RESPONSE_BYTES:
63
+ raise FetchError(
64
+ f"Response from {url} exceeded "
65
+ f"{_MAX_RESPONSE_BYTES // (1024 * 1024)} MB limit",
66
+ url=url,
67
+ )
68
+ chunks.append(chunk)
69
+ return b"".join(chunks).decode(response.encoding or "utf-8", errors="replace")
70
+ except httpx.TimeoutException as exc:
71
+ raise FetchError(f"Request timed out: {url}", url=url) from exc
72
+ except httpx.RequestError as exc:
73
+ raise FetchError(f"Network error fetching {url}: {exc}", url=url) from exc
74
+
75
+ @abstractmethod
76
+ def clean_html(self, soup: BeautifulSoup) -> Tag:
77
+ """Isolate the article body, strip non-content elements, and return the root Tag."""
78
+
79
+ @abstractmethod
80
+ def convert_to_markdown(self, tag: Tag) -> str:
81
+ """Convert the cleaned Tag to a Markdown string."""
82
+
83
+ def extract(self, url: str, *, retries: int = 3, retry_delay: float = 2.0) -> str:
84
+ """Orchestrate fetch → clean → convert and return Markdown."""
85
+ html = self.fetch_html(url, retries=retries, retry_delay=retry_delay)
86
+ soup = BeautifulSoup(html, "lxml")
87
+ try:
88
+ cleaned = self.clean_html(soup)
89
+ return self.convert_to_markdown(cleaned)
90
+ except MdfetchError as exc:
91
+ if exc.url is None:
92
+ exc.url = url
93
+ raise
mdfetch/exceptions.py ADDED
@@ -0,0 +1,40 @@
1
+ """Custom exception hierarchy for mdfetch."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class MdfetchError(Exception):
7
+ """Base exception for all mdfetch errors."""
8
+
9
+ def __init__(self, message: str, url: str | None = None) -> None:
10
+ super().__init__(message)
11
+ self.message = message
12
+ self.url = url
13
+
14
+
15
+ class InvalidURLError(MdfetchError):
16
+ """Raised when the supplied URL is syntactically invalid."""
17
+
18
+
19
+ class UnsupportedPlatformError(MdfetchError):
20
+ """Raised when the URL domain has no registered provider."""
21
+
22
+
23
+ class UnsupportedContentTypeError(MdfetchError):
24
+ """Raised when the domain is recognised but the page is not an extractable article."""
25
+
26
+
27
+ class FetchError(MdfetchError):
28
+ """Raised when a network request fails (connection error, timeout)."""
29
+
30
+
31
+ class HTTPStatusError(FetchError):
32
+ """Raised when the server returns a non-2xx HTTP status code."""
33
+
34
+ def __init__(self, message: str, status_code: int, url: str | None = None) -> None:
35
+ super().__init__(message, url)
36
+ self.status_code = status_code
37
+
38
+
39
+ class EmptyContentError(MdfetchError):
40
+ """Raised when the article body yields no extractable text content."""
File without changes
@@ -0,0 +1,67 @@
1
+ """Medium platform extractor."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ from bs4 import BeautifulSoup
8
+ from bs4.element import Tag
9
+ from markdownify import markdownify
10
+
11
+ from mdfetch.base import BaseExtractor
12
+ from mdfetch.exceptions import EmptyContentError, UnsupportedContentTypeError
13
+ from mdfetch.router import register
14
+
15
+
16
+ @register
17
+ class MediumExtractor(BaseExtractor):
18
+ """Extracts article content from medium.com and its subdomains."""
19
+
20
+ DOMAINS: frozenset[str] = frozenset({"medium.com"})
21
+
22
+ def clean_html(self, soup: BeautifulSoup) -> Tag:
23
+ """Isolate the article body and strip all non-content elements."""
24
+ article = soup.find("article")
25
+ if not isinstance(article, Tag):
26
+ raise UnsupportedContentTypeError(
27
+ "URL is not an article page — no <article> element found",
28
+ )
29
+
30
+ for nav in article.find_all("nav"):
31
+ nav.decompose()
32
+
33
+ for button in article.find_all(
34
+ "button",
35
+ attrs={"aria-label": re.compile(r"clap|applaud", re.IGNORECASE)},
36
+ ):
37
+ button.decompose()
38
+
39
+ for element in article.find_all(attrs={"data-testid": "post-sidebar"}):
40
+ element.decompose()
41
+
42
+ for element in article.find_all(attrs={"data-testid": re.compile(r"share", re.IGNORECASE)}):
43
+ element.decompose()
44
+
45
+ for element in article.find_all(attrs={"data-testid": "post-footer"}):
46
+ element.decompose()
47
+
48
+ for section in article.find_all("section"):
49
+ raw = section.get("class")
50
+ classes: list[str] = raw if isinstance(raw, list) else ([raw] if raw else [])
51
+ if classes and any("author" in c.lower() or "bio" in c.lower() for c in classes):
52
+ section.decompose()
53
+
54
+ return article
55
+
56
+ def convert_to_markdown(self, tag: Tag) -> str:
57
+ """Convert cleaned article Tag to Markdown."""
58
+ md = markdownify(str(tag), heading_style="ATX", code_language="", strip=["script", "style"])
59
+ md = md.strip()
60
+ md = re.sub(r"\n{3,}", "\n\n", md)
61
+
62
+ if not md:
63
+ raise EmptyContentError(
64
+ "Article body contained no extractable text content",
65
+ )
66
+
67
+ return md
mdfetch/router.py ADDED
@@ -0,0 +1,65 @@
1
+ """Domain-to-provider routing."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import importlib
6
+ import pkgutil
7
+ from urllib.parse import urlparse
8
+
9
+ from mdfetch.base import BaseExtractor
10
+ from mdfetch.exceptions import InvalidURLError, UnsupportedPlatformError
11
+
12
+ _REGISTRY: dict[str, type[BaseExtractor]] = {}
13
+
14
+
15
+ def register(provider_cls: type[BaseExtractor]) -> type[BaseExtractor]:
16
+ """Register *provider_cls* for each domain it declares."""
17
+ for domain in provider_cls.DOMAINS:
18
+ existing = _REGISTRY.get(domain)
19
+ if existing is not None and existing is not provider_cls:
20
+ raise ValueError(
21
+ f"Domain {domain!r} is already registered to {existing.__name__!r}; "
22
+ f"cannot re-register to {provider_cls.__name__!r}"
23
+ )
24
+ _REGISTRY[domain] = provider_cls
25
+ return provider_cls
26
+
27
+
28
+ def route(url: str) -> BaseExtractor:
29
+ """Return a provider instance for *url*, raising typed errors on failure."""
30
+ parsed = urlparse(url)
31
+
32
+ # Use parsed.hostname (lowercased, port-stripped) for routing and validation.
33
+ # Checking hostname rather than netloc correctly rejects edge cases like
34
+ # "https://:80/" where netloc is non-empty but hostname is None/empty.
35
+ hostname = (parsed.hostname or "").lower()
36
+
37
+ if parsed.scheme not in ("http", "https") or not hostname:
38
+ raise InvalidURLError(f"Invalid URL: {url!r}", url=url)
39
+
40
+ # Exact match first; fall back to subdomain suffix check so any provider whose
41
+ # DOMAINS entry is a parent domain automatically handles its subdomains.
42
+ # Sort candidates by length descending so the most-specific suffix wins when
43
+ # multiple registered domains are suffixes of the same hostname.
44
+ provider_cls = _REGISTRY.get(hostname)
45
+ if provider_cls is None:
46
+ for domain in sorted(_REGISTRY, key=len, reverse=True):
47
+ if hostname.endswith(f".{domain}"):
48
+ provider_cls = _REGISTRY[domain]
49
+ break
50
+
51
+ if provider_cls is None:
52
+ raise UnsupportedPlatformError(f"No provider registered for domain {hostname!r}", url=url)
53
+
54
+ return provider_cls()
55
+
56
+
57
+ def _autodiscover_providers() -> None:
58
+ """Import every module in mdfetch.providers; classes decorated with @register self-enrol."""
59
+ import mdfetch.providers as _providers_pkg # noqa: PLC0415
60
+
61
+ for _, module_name, _ in pkgutil.iter_modules(_providers_pkg.__path__):
62
+ importlib.import_module(f"mdfetch.providers.{module_name}")
63
+
64
+
65
+ _autodiscover_providers()
@@ -0,0 +1,105 @@
1
+ Metadata-Version: 2.4
2
+ Name: mdfetch
3
+ Version: 0.1.0
4
+ Summary: Extract article content from web platforms and return it as clean Markdown.
5
+ Project-URL: Homepage, https://github.com/stn1slv/md-fetch
6
+ Project-URL: Source, https://github.com/stn1slv/md-fetch
7
+ Project-URL: Issues, https://github.com/stn1slv/md-fetch/issues
8
+ Author-email: Stanislav Deviatov <devyatov@gmail.com>
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: article,extraction,markdown,medium,scraping
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Topic :: Internet :: WWW/HTTP
20
+ Classifier: Topic :: Text Processing :: Markup :: Markdown
21
+ Requires-Python: >=3.12
22
+ Requires-Dist: beautifulsoup4>=4.12
23
+ Requires-Dist: httpx>=0.27
24
+ Requires-Dist: lxml>=5.0
25
+ Requires-Dist: markdownify>=0.13
26
+ Provides-Extra: dev
27
+ Requires-Dist: mypy>=1.9; extra == 'dev'
28
+ Requires-Dist: pytest>=8.0; extra == 'dev'
29
+ Requires-Dist: ruff>=0.4; extra == 'dev'
30
+ Description-Content-Type: text/markdown
31
+
32
+ # mdfetch
33
+
34
+ A Python library that extracts article content from web platforms and returns it as clean Markdown.
35
+
36
+ ## Install
37
+
38
+ ```bash
39
+ pip install mdfetch
40
+ ```
41
+
42
+ ## Usage
43
+
44
+ ```python
45
+ from mdfetch import extract
46
+
47
+ markdown = extract("https://medium.com/some-publication/article-slug-abc123")
48
+ print(markdown)
49
+ ```
50
+
51
+ ## Error handling
52
+
53
+ ```python
54
+ from mdfetch import (
55
+ extract,
56
+ InvalidURLError,
57
+ UnsupportedPlatformError,
58
+ UnsupportedContentTypeError,
59
+ FetchError,
60
+ HTTPStatusError,
61
+ EmptyContentError,
62
+ )
63
+
64
+ url = "https://medium.com/some-publication/article-slug-abc123"
65
+
66
+ try:
67
+ markdown = extract(url)
68
+ except InvalidURLError as e:
69
+ print(f"Bad URL: {e.message}")
70
+ except UnsupportedPlatformError as e:
71
+ print(f"Platform not supported: {e.message}")
72
+ except UnsupportedContentTypeError as e:
73
+ print(f"Not an article page: {e.message}")
74
+ except HTTPStatusError as e:
75
+ print(f"HTTP {e.status_code}: {e.message}")
76
+ except FetchError as e:
77
+ print(f"Network error: {e.message}")
78
+ except EmptyContentError as e:
79
+ print(f"No content: {e.message}")
80
+ ```
81
+
82
+ ## Supported platforms
83
+
84
+ | Platform | Domains |
85
+ |----------|---------|
86
+ | Medium | `medium.com`, `*.medium.com` |
87
+
88
+ ## Development
89
+
90
+ Requires [uv](https://docs.astral.sh/uv/).
91
+
92
+ ```bash
93
+ make setup # install dependencies
94
+ make test # run unit tests
95
+ make integration # run integration tests (requires network access)
96
+ make lint # ruff check
97
+ make format # ruff format
98
+ make build # build wheel + sdist
99
+ make upgrade-deps # upgrade all dependencies
100
+ make clean # remove build artifacts
101
+ ```
102
+
103
+ ## Requirements
104
+
105
+ - Python 3.12+
@@ -0,0 +1,10 @@
1
+ mdfetch/__init__.py,sha256=jYK0fy0rIPkVrBethgIwFl4SdQohoCIPYGiZpyzg7pI,1014
2
+ mdfetch/base.py,sha256=p71yd5FtsV0EPxr2f9QQ53qV-YmhRH_sV24ZWqcEuGY,3827
3
+ mdfetch/exceptions.py,sha256=kMEgh6hTL3IICnmPmJ7c2WRmTbJ75-RLcXmk1ynUFfg,1180
4
+ mdfetch/router.py,sha256=1g6vsFToMoruUkLcLKBJ2aQlHbgmP_ZnjzUBvQbW6Bg,2499
5
+ mdfetch/providers/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
6
+ mdfetch/providers/medium.py,sha256=QF1HugoOH2b_Kg5YxuSbsKIZUjcxQTerCbJafPNjROQ,2242
7
+ mdfetch-0.1.0.dist-info/METADATA,sha256=ykjqRi9cNSvMTNR6wC2o1xT9Fi7mI5e3KVJiNk4yHaE,2856
8
+ mdfetch-0.1.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
9
+ mdfetch-0.1.0.dist-info/licenses/LICENSE,sha256=erwf-XsuvrFr0bi1_8aTbCk7wu7zuxcJWqTondbCvhA,1075
10
+ mdfetch-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.29.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Stanislav Deviatov
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.