readmeta 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
readmeta/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """readmeta: check that your README will actually render on PyPI."""
2
+
3
+ __version__ = "0.1.0"
readmeta/check.py ADDED
@@ -0,0 +1,288 @@
1
+ """Core checks for readmeta.
2
+
3
+ Key insight: PyPI renders the *built artifact's* long_description (the body of
4
+ ``METADATA`` in a wheel / ``PKG-INFO`` in an sdist), not the repository's
5
+ ``README.md``. Relative images, relative links, in-page anchors and raw SVG
6
+ references that work on GitHub silently 404 on PyPI, and ``twine check``
7
+ does not catch them.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+ import tarfile
14
+ import zipfile
15
+ from dataclasses import dataclass
16
+ from email import message_from_string
17
+ from html.parser import HTMLParser
18
+
19
+
20
+ # ---------------------------------------------------------------------------
21
+ # Findings
22
+ # ---------------------------------------------------------------------------
23
+
24
+ @dataclass
25
+ class Finding:
26
+ kind: str # relative-image | relative-link | broken-anchor | relative-svg
27
+ target: str # the offending URL / anchor
28
+ context: str = "" # e.g. "dist/foo-1.0.whl:12"
29
+ hint: str = ""
30
+
31
+
32
+ HINTS = {
33
+ "relative-image": (
34
+ "PyPI renders the description standalone; relative image paths 404. "
35
+ "Use an absolute https:// URL (e.g. raw.githubusercontent.com)."
36
+ ),
37
+ "relative-link": (
38
+ "Relative links resolve against pypi.org and break. "
39
+ "Use an absolute https:// URL."
40
+ ),
41
+ "broken-anchor": (
42
+ "No heading with a matching id was found; the link goes nowhere on PyPI."
43
+ ),
44
+ "relative-svg": (
45
+ "Relative .svg references 404 on PyPI (PyPI does not serve repo files); "
46
+ "twine check does not flag these. Use an absolute https:// URL."
47
+ ),
48
+ }
49
+
50
+
51
+ def _is_external(url: str) -> bool:
52
+ u = url.strip().lower()
53
+ return u.startswith(("http://", "https://", "//", "data:", "mailto:"))
54
+
55
+
56
+ def _github_slug(text: str) -> str:
57
+ """Approximate GitHub/PyPI heading-id slugification."""
58
+ text = text.strip().lower()
59
+ text = re.sub(r"[^\w\s-]", "", text)
60
+ return re.sub(r"\s+", "-", text).strip("-")
61
+
62
+
63
+ # ---------------------------------------------------------------------------
64
+ # HTML mode (for text/html long descriptions)
65
+ # ---------------------------------------------------------------------------
66
+
67
+ class _ReadmeHTMLParser(HTMLParser):
68
+ def __init__(self) -> None:
69
+ super().__init__(convert_charrefs=True)
70
+ self.ids: set[str] = set()
71
+ self.anchors: list[tuple[str, int]] = [] # (target, line)
72
+ self.images: list[tuple[str, int]] = [] # (src, line)
73
+ self.links: list[tuple[str, int]] = [] # (href, line)
74
+ self._heading_text: str | None = None
75
+
76
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
77
+ a = dict(attrs)
78
+ line = self.getpos()[0]
79
+ ident = a.get("id")
80
+ if ident:
81
+ self.ids.add(ident)
82
+ if tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
83
+ self._heading_text = ""
84
+ elif tag == "img":
85
+ self.images.append((a.get("src") or "", line))
86
+ elif tag == "a":
87
+ href = a.get("href") or ""
88
+ self.links.append((href, line))
89
+ if href.startswith("#") and len(href) > 1:
90
+ self.anchors.append((href[1:], line))
91
+
92
+ def handle_data(self, data: str) -> None:
93
+ if self._heading_text is not None:
94
+ self._heading_text += data
95
+
96
+ def handle_endtag(self, tag: str) -> None:
97
+ if self._heading_text is not None and tag in (
98
+ "h1", "h2", "h3", "h4", "h5", "h6",
99
+ ):
100
+ slug = _github_slug(self._heading_text)
101
+ if slug:
102
+ self.ids.add(slug)
103
+ self._heading_text = None
104
+
105
+
106
+ def check_html(html: str, source: str = "<html>") -> list[Finding]:
107
+ """Check an HTML long description. Returns findings sorted by line."""
108
+ # Mask <pre>/<code> blocks: documentation *about* bad patterns must not flag.
109
+ html = _mask_spans(r"<pre\b.*?</pre>", html, re.S | re.I)
110
+ html = _mask_spans(r"<code\b.*?</code>", html, re.S | re.I)
111
+ parser = _ReadmeHTMLParser()
112
+ parser.feed(html)
113
+ out: list[Finding] = []
114
+
115
+ def ctx(line: int) -> str:
116
+ return f"{source}:{line}"
117
+
118
+ for src, line in parser.images:
119
+ s = src.strip()
120
+ if not s or _is_external(s):
121
+ continue
122
+ kind = "relative-svg" if s.lower().split("?")[0].split("#")[0].endswith(".svg") else "relative-image"
123
+ out.append(Finding(kind, s, ctx(line), HINTS[kind]))
124
+
125
+ for href, line in parser.links:
126
+ h = href.strip()
127
+ if not h or h.startswith("#") or _is_external(h):
128
+ continue
129
+ kind = "relative-svg" if h.lower().split("?")[0].split("#")[0].endswith(".svg") else "relative-link"
130
+ out.append(Finding(kind, h, ctx(line), HINTS[kind]))
131
+
132
+ for target, line in parser.anchors:
133
+ if target not in parser.ids:
134
+ out.append(Finding("broken-anchor", "#" + target, ctx(line), HINTS["broken-anchor"]))
135
+
136
+ out.sort(key=lambda f: f.context)
137
+ return out
138
+
139
+
140
+ # ---------------------------------------------------------------------------
141
+ # Text mode (Markdown / RST raw source, as stored in METADATA / PKG-INFO)
142
+ # ---------------------------------------------------------------------------
143
+
144
+ _MD_INLINE = re.compile(r"(!?)\[[^\]]*\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
145
+ _MD_REFDEF = re.compile(r"^\s*\[[^\]]+\]:\s*(\S+)", re.M)
146
+ _MD_HEADING = re.compile(r"^#{1,6}\s+(.+?)\s*#*\s*$", re.M)
147
+ _HTML_IMG_SRC = re.compile(r"<img\b[^>]*?\bsrc=[\"']([^\"']+)[\"']", re.I)
148
+ _RST_IMAGE = re.compile(r"^\s*\.\.\s+(?:image|figure)::\s*(\S+)", re.M)
149
+
150
+
151
+ def _mask_spans(pattern: str, text: str, flags: int = 0) -> str:
152
+ """Replace matches with spaces (newlines preserved) so line numbers survive."""
153
+
154
+ def _mask(m: re.Match) -> str:
155
+ return "".join("\n" if ch == "\n" else " " for ch in m.group(0))
156
+
157
+ return re.sub(pattern, _mask, text, flags=flags)
158
+
159
+
160
+ def _strip_code(text: str) -> str:
161
+ """Mask fenced code blocks and inline code spans: documentation *about*
162
+ bad patterns (like this docstring) must not be flagged."""
163
+ text = _mask_spans(r"```.*?```", text, re.S)
164
+ return _mask_spans(r"`[^`\n]+`", text)
165
+
166
+
167
+ def _line_of(text: str, pos: int) -> int:
168
+ return text.count("\n", 0, pos) + 1
169
+
170
+
171
+ def check_text(text: str, source: str = "<text>") -> list[Finding]:
172
+ """Regex-based check for raw Markdown / RST long descriptions."""
173
+ text = _strip_code(text)
174
+ out: list[Finding] = []
175
+
176
+ def ctx(pos: int) -> str:
177
+ return f"{source}:{_line_of(text, pos)}"
178
+
179
+ def add_url(url: str, pos: int, is_image: bool) -> None:
180
+ u = url.strip()
181
+ if not u or _is_external(u):
182
+ return
183
+ if u.startswith("#"):
184
+ return # handled by the anchor pass below
185
+ stem = u.lower().split("?")[0].split("#")[0]
186
+ kind = "relative-svg" if stem.endswith(".svg") else ("relative-image" if is_image else "relative-link")
187
+ out.append(Finding(kind, u, ctx(pos), HINTS[kind]))
188
+
189
+ # Inline Markdown images / links: ![alt](url), [text](url)
190
+ for m in _MD_INLINE.finditer(text):
191
+ add_url(m.group(2), m.start(), is_image=bool(m.group(1)))
192
+
193
+ # Reference-style definitions: [ref]: url
194
+ for m in _MD_REFDEF.finditer(text):
195
+ add_url(m.group(1), m.start(), is_image=False)
196
+
197
+ # Raw <img> tags embedded in Markdown
198
+ for m in _HTML_IMG_SRC.finditer(text):
199
+ add_url(m.group(1), m.start(), is_image=True)
200
+
201
+ # RST image / figure directives
202
+ for m in _RST_IMAGE.finditer(text):
203
+ add_url(m.group(1), m.start(), is_image=True)
204
+
205
+ # In-page anchors (Markdown): [text](#anchor) vs ## Headings
206
+ known = {_github_slug(m.group(1)) for m in _MD_HEADING.finditer(text)}
207
+ for m in _MD_INLINE.finditer(text):
208
+ url = m.group(2).strip()
209
+ if url.startswith("#") and len(url) > 1:
210
+ target = url[1:]
211
+ if target not in known:
212
+ out.append(Finding("broken-anchor", url, ctx(m.start()), HINTS["broken-anchor"]))
213
+
214
+ out.sort(key=lambda f: f.context)
215
+ return out
216
+
217
+
218
+ # ---------------------------------------------------------------------------
219
+ # Artifact / PyPI input
220
+ # ---------------------------------------------------------------------------
221
+
222
+ def read_artifact(path: str) -> tuple[str, str, str, str]:
223
+ """Return (name, version, content_type, long_description) for a wheel/sdist."""
224
+ if path.endswith(".whl"):
225
+ with zipfile.ZipFile(path) as z:
226
+ names = [n for n in z.namelist() if n.endswith(".dist-info/METADATA")]
227
+ if not names:
228
+ raise ValueError(f"no .dist-info/METADATA found in {path}")
229
+ raw = z.read(names[0]).decode("utf-8", "replace")
230
+ elif path.endswith((".tar.gz", ".tgz")):
231
+ with tarfile.open(path, "r:gz") as t:
232
+ members = [m for m in t.getmembers() if m.name.endswith("PKG-INFO")]
233
+ if not members:
234
+ raise ValueError(f"no PKG-INFO found in {path}")
235
+ f = t.extractfile(members[0])
236
+ assert f is not None
237
+ raw = f.read().decode("utf-8", "replace")
238
+ else:
239
+ raise ValueError(f"unsupported artifact (want .whl or .tar.gz): {path}")
240
+
241
+ msg = message_from_string(raw)
242
+ ctype = (msg.get("Description-Content-Type") or "text/plain").split(";")[0].strip().lower()
243
+ desc = msg.get_payload()
244
+ if not isinstance(desc, str):
245
+ desc = ""
246
+ return (
247
+ msg.get("Name") or "?",
248
+ msg.get("Version") or "?",
249
+ ctype,
250
+ desc,
251
+ )
252
+
253
+
254
+ def fetch_pypi(name: str, timeout: float = 20.0) -> tuple[str, str, str, str]:
255
+ """Return (name, version, content_type, description) from the PyPI JSON API.
256
+
257
+ Note: the public JSON API exposes the raw ``description``, not rendered
258
+ HTML, so text-mode scanning applies unless a ``description_html`` field
259
+ is ever present.
260
+ """
261
+ import json
262
+ import urllib.request
263
+
264
+ url = f"https://pypi.org/pypi/{name}/json"
265
+ req = urllib.request.Request(url, headers={"User-Agent": "readmeta/0.1.0"})
266
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
267
+ data = json.load(resp)
268
+ info = data.get("info", {})
269
+ if "description_html" in info and info["description_html"]:
270
+ return (
271
+ info.get("name") or name,
272
+ info.get("version") or "?",
273
+ "text/html",
274
+ info["description_html"],
275
+ )
276
+ ctype = (info.get("description_content_type") or "text/plain").split(";")[0].strip().lower()
277
+ return (
278
+ info.get("name") or name,
279
+ info.get("version") or "?",
280
+ ctype,
281
+ info.get("description") or "",
282
+ )
283
+
284
+
285
+ def check_description(content_type: str, description: str, source: str) -> list[Finding]:
286
+ if "html" in content_type:
287
+ return check_html(description, source)
288
+ return check_text(description, source)
readmeta/cli.py ADDED
@@ -0,0 +1,73 @@
1
+ """Command-line interface for readmeta."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ import urllib.error
8
+
9
+ from . import __version__
10
+ from .check import check_description, fetch_pypi, read_artifact
11
+
12
+
13
+ def _report(name: str, version: str, content_type: str, source: str, description: str) -> int:
14
+ findings = check_description(content_type, description, source)
15
+ print(f"{name} {version} [{content_type}] <- {source}")
16
+ if not findings:
17
+ print("OK: no PyPI rendering issues found.")
18
+ return 0
19
+ print(f"Found {len(findings)} issue(s):\n")
20
+ for f in findings:
21
+ print(f" [{f.kind}] {f.context}\n {f.target}\n -> {f.hint}\n")
22
+ return 1
23
+
24
+
25
+ def main(argv: list[str] | None = None) -> int:
26
+ ap = argparse.ArgumentParser(
27
+ prog="readmeta",
28
+ description=(
29
+ "Check whether a package's README will render correctly on PyPI. "
30
+ "Inspects the built artifact's long_description (what PyPI actually "
31
+ "renders), not the repo's README.md."
32
+ ),
33
+ )
34
+ ap.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
35
+ sub = ap.add_subparsers(dest="command", required=True)
36
+
37
+ chk = sub.add_parser("check", help="check artifacts or a PyPI project for rendering issues")
38
+ chk.add_argument("paths", nargs="*", help="wheel (.whl) and/or sdist (.tar.gz) files")
39
+ chk.add_argument("--pypi", metavar="NAME", help="check the description hosted on PyPI instead of local files")
40
+
41
+ args = ap.parse_args(argv)
42
+
43
+ if args.command == "check":
44
+ if args.pypi:
45
+ try:
46
+ name, version, ctype, desc = fetch_pypi(args.pypi)
47
+ except urllib.error.HTTPError as e:
48
+ print(f"error: PyPI request failed ({e.code} {e.reason})", file=sys.stderr)
49
+ return 2
50
+ except OSError as e:
51
+ print(f"error: could not reach PyPI: {e}", file=sys.stderr)
52
+ return 2
53
+ return _report(name, version, ctype, f"pypi:{args.pypi}", desc)
54
+
55
+ if not args.paths:
56
+ ap.error("check needs at least one artifact path or --pypi NAME")
57
+ worst = 0
58
+ for path in args.paths:
59
+ try:
60
+ name, version, ctype, desc = read_artifact(path)
61
+ except (OSError, ValueError) as e:
62
+ print(f"error: {e}", file=sys.stderr)
63
+ worst = max(worst, 2)
64
+ continue
65
+ rc = _report(name, version, ctype, path, desc)
66
+ worst = max(worst, rc)
67
+ return worst
68
+
69
+ return 2 # pragma: no cover
70
+
71
+
72
+ if __name__ == "__main__":
73
+ sys.exit(main())
@@ -0,0 +1,100 @@
1
+ Metadata-Version: 2.4
2
+ Name: readmeta
3
+ Version: 0.1.0
4
+ Summary: Check that your README will actually render on PyPI — inspects the built artifact, not the repo
5
+ Author: hao li
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/hahahahahahahahah6/readmeta
8
+ Keywords: pypi,readme,packaging,ci,twine
9
+ Requires-Python: >=3.9
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Dynamic: license-file
13
+
14
+ # readmeta
15
+
16
+ Check that your README will actually render on PyPI — by inspecting the **built artifact**, not the repo.
17
+
18
+ ## The problem
19
+
20
+ PyPI renders the `long_description` from your built artifact (the body of `METADATA` in a wheel / `PKG-INFO` in an sdist). It does **not** render your repo's `README.md`. Things that look perfect on GitHub silently break on PyPI:
21
+
22
+ - **Relative images** (`![shot](docs/shot.png)`) → 404, PyPI has no `docs/` folder
23
+ - **Relative links** (`[usage](docs/usage.md)`) → resolve against `pypi.org`, broken
24
+ - **Raw SVG references** (`assets/logo.svg`) → 404; PyPI doesn't serve repo files
25
+ - **In-page anchors** (`[setup](#instalation)`) → go nowhere when the heading id doesn't exist
26
+
27
+ `twine check` validates that the description *renders* — it does not check that any link or image actually *resolves*.
28
+
29
+ Real cases this catches:
30
+
31
+ - `armsmith` v1.2.1 shipped a relative SVG badge that 404'd on its PyPI page while rendering fine on GitHub.
32
+ - `aicertify` maintains a separate `README-pypi.md` with rewritten absolute URLs, precisely because the GitHub README breaks on PyPI.
33
+ - An audit of the `modern-python` GitHub org found 23 of 25 packages shipping relative asset references that can't resolve on PyPI.
34
+
35
+ ## Install
36
+
37
+ ```bash
38
+ pip install readmeta
39
+ ```
40
+
41
+ Requires Python 3.9+. No dependencies — stdlib only.
42
+
43
+ ## Usage
44
+
45
+ Check your built artifacts (build first, then check what PyPI will actually see):
46
+
47
+ ```bash
48
+ python -m build
49
+ readmeta check dist/*
50
+ ```
51
+
52
+ Or check what's currently hosted on PyPI:
53
+
54
+ ```bash
55
+ readmeta check --pypi requests
56
+ ```
57
+
58
+ Example output:
59
+
60
+ ```
61
+ fakepkg 0.1.0 [text/markdown] <- dist/fakepkg-0.1.0-py3-none-any.whl
62
+ Found 2 issue(s):
63
+
64
+ [relative-image] dist/fakepkg-0.1.0-py3-none-any.whl:12
65
+ docs/shot.png
66
+ -> PyPI renders the description standalone; relative image paths 404. Use an absolute https:// URL (e.g. raw.githubusercontent.com).
67
+
68
+ [broken-anchor] dist/fakepkg-0.1.0-py3-none-any.whl:20
69
+ #instalation
70
+ -> No heading with a matching id was found; the link goes nowhere on PyPI.
71
+ ```
72
+
73
+ Exit codes are CI-friendly: `0` = clean, `1` = issues found, `2` = error (unreadable artifact, PyPI unreachable, bad usage).
74
+
75
+ ## CI example
76
+
77
+ ```yaml
78
+ - name: Build
79
+ run: python -m build
80
+
81
+ - name: Check PyPI rendering
82
+ run: |
83
+ pip install readmeta
84
+ readmeta check dist/*
85
+ ```
86
+
87
+ ## How it works
88
+
89
+ 1. Reads `long_description` from `.whl` (`zipfile` → `.dist-info/METADATA`) or `.tar.gz` (`tarfile` → `PKG-INFO`), plus the `Description-Content-Type` header.
90
+ 2. For `text/html`, parses with `html.parser` and validates every `img[src]`, `a[href]`, and `#anchor` against collected element ids (explicit ids plus GitHub-style heading slugs).
91
+ 3. For Markdown/RST (what artifacts actually carry — the raw source, not rendered HTML), scans with regexes: inline and reference-style images/links, embedded `<img>` tags, RST `image::`/`figure::` directives, and `#anchor` links against `# Heading` slugs.
92
+ 4. `--pypi` mode fetches `https://pypi.org/pypi/<name>/json` and checks the hosted description.
93
+
94
+ ## Limitations (v0.1)
95
+
96
+ - **Check only** — no `--fix`. A build-time rewrite mode (convert relative refs to absolute URLs at build time) is planned.
97
+ - Anchor validation for RST is limited (Markdown headings and HTML ids are covered; RST `.. _target:` definitions are not yet resolved).
98
+ - Code spans and fenced code blocks are ignored (documenting a bad pattern doesn't flag it); indented code blocks are still scanned.
99
+ - Heading-slug generation approximates GitHub/PyPI's algorithm; exotic headings could produce false positives — explicit `id` attributes always win.
100
+ - `--pypi` uses the raw `description` from the JSON API (the API doesn't expose rendered HTML); findings are identical to checking a fresh local build.
@@ -0,0 +1,9 @@
1
+ readmeta/__init__.py,sha256=m79wxCNuNjTk0iJ5IIwxdCHRa-Sfi0-89aB6YW7TrCM,92
2
+ readmeta/check.py,sha256=EHHacQ6vAu3WJDo6oSoi6JviHR6ltJbKZ7IAFjWrvg0,10723
3
+ readmeta/cli.py,sha256=mhPyrGwtBg8kq8X7smi4h7j8OYLGcDy5it7SIekRHfs,2657
4
+ readmeta-0.1.0.dist-info/licenses/LICENSE,sha256=OYhQDg7nxWoFtnF6J8NkFUA8NMw5TKv9a1WTXHY-fu4,1063
5
+ readmeta-0.1.0.dist-info/METADATA,sha256=8O7qmEkXj4gFMFYuXc7He127APE6L5ILQP8C89qaBt0,4196
6
+ readmeta-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
7
+ readmeta-0.1.0.dist-info/entry_points.txt,sha256=rwamKpg_5cn4RnhaMAMZ8tijuikXf36Bc6LKY-YzWsg,47
8
+ readmeta-0.1.0.dist-info/top_level.txt,sha256=KVbdaBc_bSBeVCWPG2d4VgAkDH--2z5wBJMmWcib6Os,9
9
+ readmeta-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ readmeta = readmeta.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 hao li
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ readmeta