readmeta 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- readmeta/__init__.py +3 -0
- readmeta/check.py +288 -0
- readmeta/cli.py +73 -0
- readmeta-0.1.0.dist-info/METADATA +100 -0
- readmeta-0.1.0.dist-info/RECORD +9 -0
- readmeta-0.1.0.dist-info/WHEEL +5 -0
- readmeta-0.1.0.dist-info/entry_points.txt +2 -0
- readmeta-0.1.0.dist-info/licenses/LICENSE +21 -0
- readmeta-0.1.0.dist-info/top_level.txt +1 -0
readmeta/__init__.py
ADDED
readmeta/check.py
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""Core checks for readmeta.
|
|
2
|
+
|
|
3
|
+
Key insight: PyPI renders the *built artifact's* long_description (the body of
|
|
4
|
+
``METADATA`` in a wheel / ``PKG-INFO`` in an sdist), not the repository's
|
|
5
|
+
``README.md``. Relative images, relative links, in-page anchors and raw SVG
|
|
6
|
+
references that work on GitHub silently 404 on PyPI, and ``twine check``
|
|
7
|
+
does not catch them.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
import tarfile
|
|
14
|
+
import zipfile
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from email import message_from_string
|
|
17
|
+
from html.parser import HTMLParser
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# ---------------------------------------------------------------------------
|
|
21
|
+
# Findings
|
|
22
|
+
# ---------------------------------------------------------------------------
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class Finding:
|
|
26
|
+
kind: str # relative-image | relative-link | broken-anchor | relative-svg
|
|
27
|
+
target: str # the offending URL / anchor
|
|
28
|
+
context: str = "" # e.g. "dist/foo-1.0.whl:12"
|
|
29
|
+
hint: str = ""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
HINTS = {
|
|
33
|
+
"relative-image": (
|
|
34
|
+
"PyPI renders the description standalone; relative image paths 404. "
|
|
35
|
+
"Use an absolute https:// URL (e.g. raw.githubusercontent.com)."
|
|
36
|
+
),
|
|
37
|
+
"relative-link": (
|
|
38
|
+
"Relative links resolve against pypi.org and break. "
|
|
39
|
+
"Use an absolute https:// URL."
|
|
40
|
+
),
|
|
41
|
+
"broken-anchor": (
|
|
42
|
+
"No heading with a matching id was found; the link goes nowhere on PyPI."
|
|
43
|
+
),
|
|
44
|
+
"relative-svg": (
|
|
45
|
+
"Relative .svg references 404 on PyPI (PyPI does not serve repo files); "
|
|
46
|
+
"twine check does not flag these. Use an absolute https:// URL."
|
|
47
|
+
),
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _is_external(url: str) -> bool:
|
|
52
|
+
u = url.strip().lower()
|
|
53
|
+
return u.startswith(("http://", "https://", "//", "data:", "mailto:"))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _github_slug(text: str) -> str:
|
|
57
|
+
"""Approximate GitHub/PyPI heading-id slugification."""
|
|
58
|
+
text = text.strip().lower()
|
|
59
|
+
text = re.sub(r"[^\w\s-]", "", text)
|
|
60
|
+
return re.sub(r"\s+", "-", text).strip("-")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# ---------------------------------------------------------------------------
|
|
64
|
+
# HTML mode (for text/html long descriptions)
|
|
65
|
+
# ---------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
class _ReadmeHTMLParser(HTMLParser):
|
|
68
|
+
def __init__(self) -> None:
|
|
69
|
+
super().__init__(convert_charrefs=True)
|
|
70
|
+
self.ids: set[str] = set()
|
|
71
|
+
self.anchors: list[tuple[str, int]] = [] # (target, line)
|
|
72
|
+
self.images: list[tuple[str, int]] = [] # (src, line)
|
|
73
|
+
self.links: list[tuple[str, int]] = [] # (href, line)
|
|
74
|
+
self._heading_text: str | None = None
|
|
75
|
+
|
|
76
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
77
|
+
a = dict(attrs)
|
|
78
|
+
line = self.getpos()[0]
|
|
79
|
+
ident = a.get("id")
|
|
80
|
+
if ident:
|
|
81
|
+
self.ids.add(ident)
|
|
82
|
+
if tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
|
83
|
+
self._heading_text = ""
|
|
84
|
+
elif tag == "img":
|
|
85
|
+
self.images.append((a.get("src") or "", line))
|
|
86
|
+
elif tag == "a":
|
|
87
|
+
href = a.get("href") or ""
|
|
88
|
+
self.links.append((href, line))
|
|
89
|
+
if href.startswith("#") and len(href) > 1:
|
|
90
|
+
self.anchors.append((href[1:], line))
|
|
91
|
+
|
|
92
|
+
def handle_data(self, data: str) -> None:
|
|
93
|
+
if self._heading_text is not None:
|
|
94
|
+
self._heading_text += data
|
|
95
|
+
|
|
96
|
+
def handle_endtag(self, tag: str) -> None:
|
|
97
|
+
if self._heading_text is not None and tag in (
|
|
98
|
+
"h1", "h2", "h3", "h4", "h5", "h6",
|
|
99
|
+
):
|
|
100
|
+
slug = _github_slug(self._heading_text)
|
|
101
|
+
if slug:
|
|
102
|
+
self.ids.add(slug)
|
|
103
|
+
self._heading_text = None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def check_html(html: str, source: str = "<html>") -> list[Finding]:
|
|
107
|
+
"""Check an HTML long description. Returns findings sorted by line."""
|
|
108
|
+
# Mask <pre>/<code> blocks: documentation *about* bad patterns must not flag.
|
|
109
|
+
html = _mask_spans(r"<pre\b.*?</pre>", html, re.S | re.I)
|
|
110
|
+
html = _mask_spans(r"<code\b.*?</code>", html, re.S | re.I)
|
|
111
|
+
parser = _ReadmeHTMLParser()
|
|
112
|
+
parser.feed(html)
|
|
113
|
+
out: list[Finding] = []
|
|
114
|
+
|
|
115
|
+
def ctx(line: int) -> str:
|
|
116
|
+
return f"{source}:{line}"
|
|
117
|
+
|
|
118
|
+
for src, line in parser.images:
|
|
119
|
+
s = src.strip()
|
|
120
|
+
if not s or _is_external(s):
|
|
121
|
+
continue
|
|
122
|
+
kind = "relative-svg" if s.lower().split("?")[0].split("#")[0].endswith(".svg") else "relative-image"
|
|
123
|
+
out.append(Finding(kind, s, ctx(line), HINTS[kind]))
|
|
124
|
+
|
|
125
|
+
for href, line in parser.links:
|
|
126
|
+
h = href.strip()
|
|
127
|
+
if not h or h.startswith("#") or _is_external(h):
|
|
128
|
+
continue
|
|
129
|
+
kind = "relative-svg" if h.lower().split("?")[0].split("#")[0].endswith(".svg") else "relative-link"
|
|
130
|
+
out.append(Finding(kind, h, ctx(line), HINTS[kind]))
|
|
131
|
+
|
|
132
|
+
for target, line in parser.anchors:
|
|
133
|
+
if target not in parser.ids:
|
|
134
|
+
out.append(Finding("broken-anchor", "#" + target, ctx(line), HINTS["broken-anchor"]))
|
|
135
|
+
|
|
136
|
+
out.sort(key=lambda f: f.context)
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# ---------------------------------------------------------------------------
|
|
141
|
+
# Text mode (Markdown / RST raw source, as stored in METADATA / PKG-INFO)
|
|
142
|
+
# ---------------------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
_MD_INLINE = re.compile(r"(!?)\[[^\]]*\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
|
|
145
|
+
_MD_REFDEF = re.compile(r"^\s*\[[^\]]+\]:\s*(\S+)", re.M)
|
|
146
|
+
_MD_HEADING = re.compile(r"^#{1,6}\s+(.+?)\s*#*\s*$", re.M)
|
|
147
|
+
_HTML_IMG_SRC = re.compile(r"<img\b[^>]*?\bsrc=[\"']([^\"']+)[\"']", re.I)
|
|
148
|
+
_RST_IMAGE = re.compile(r"^\s*\.\.\s+(?:image|figure)::\s*(\S+)", re.M)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _mask_spans(pattern: str, text: str, flags: int = 0) -> str:
|
|
152
|
+
"""Replace matches with spaces (newlines preserved) so line numbers survive."""
|
|
153
|
+
|
|
154
|
+
def _mask(m: re.Match) -> str:
|
|
155
|
+
return "".join("\n" if ch == "\n" else " " for ch in m.group(0))
|
|
156
|
+
|
|
157
|
+
return re.sub(pattern, _mask, text, flags=flags)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _strip_code(text: str) -> str:
|
|
161
|
+
"""Mask fenced code blocks and inline code spans: documentation *about*
|
|
162
|
+
bad patterns (like this docstring) must not be flagged."""
|
|
163
|
+
text = _mask_spans(r"```.*?```", text, re.S)
|
|
164
|
+
return _mask_spans(r"`[^`\n]+`", text)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _line_of(text: str, pos: int) -> int:
|
|
168
|
+
return text.count("\n", 0, pos) + 1
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def check_text(text: str, source: str = "<text>") -> list[Finding]:
|
|
172
|
+
"""Regex-based check for raw Markdown / RST long descriptions."""
|
|
173
|
+
text = _strip_code(text)
|
|
174
|
+
out: list[Finding] = []
|
|
175
|
+
|
|
176
|
+
def ctx(pos: int) -> str:
|
|
177
|
+
return f"{source}:{_line_of(text, pos)}"
|
|
178
|
+
|
|
179
|
+
def add_url(url: str, pos: int, is_image: bool) -> None:
|
|
180
|
+
u = url.strip()
|
|
181
|
+
if not u or _is_external(u):
|
|
182
|
+
return
|
|
183
|
+
if u.startswith("#"):
|
|
184
|
+
return # handled by the anchor pass below
|
|
185
|
+
stem = u.lower().split("?")[0].split("#")[0]
|
|
186
|
+
kind = "relative-svg" if stem.endswith(".svg") else ("relative-image" if is_image else "relative-link")
|
|
187
|
+
out.append(Finding(kind, u, ctx(pos), HINTS[kind]))
|
|
188
|
+
|
|
189
|
+
# Inline Markdown images / links: , [text](url)
|
|
190
|
+
for m in _MD_INLINE.finditer(text):
|
|
191
|
+
add_url(m.group(2), m.start(), is_image=bool(m.group(1)))
|
|
192
|
+
|
|
193
|
+
# Reference-style definitions: [ref]: url
|
|
194
|
+
for m in _MD_REFDEF.finditer(text):
|
|
195
|
+
add_url(m.group(1), m.start(), is_image=False)
|
|
196
|
+
|
|
197
|
+
# Raw <img> tags embedded in Markdown
|
|
198
|
+
for m in _HTML_IMG_SRC.finditer(text):
|
|
199
|
+
add_url(m.group(1), m.start(), is_image=True)
|
|
200
|
+
|
|
201
|
+
# RST image / figure directives
|
|
202
|
+
for m in _RST_IMAGE.finditer(text):
|
|
203
|
+
add_url(m.group(1), m.start(), is_image=True)
|
|
204
|
+
|
|
205
|
+
# In-page anchors (Markdown): [text](#anchor) vs ## Headings
|
|
206
|
+
known = {_github_slug(m.group(1)) for m in _MD_HEADING.finditer(text)}
|
|
207
|
+
for m in _MD_INLINE.finditer(text):
|
|
208
|
+
url = m.group(2).strip()
|
|
209
|
+
if url.startswith("#") and len(url) > 1:
|
|
210
|
+
target = url[1:]
|
|
211
|
+
if target not in known:
|
|
212
|
+
out.append(Finding("broken-anchor", url, ctx(m.start()), HINTS["broken-anchor"]))
|
|
213
|
+
|
|
214
|
+
out.sort(key=lambda f: f.context)
|
|
215
|
+
return out
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
# ---------------------------------------------------------------------------
|
|
219
|
+
# Artifact / PyPI input
|
|
220
|
+
# ---------------------------------------------------------------------------
|
|
221
|
+
|
|
222
|
+
def read_artifact(path: str) -> tuple[str, str, str, str]:
|
|
223
|
+
"""Return (name, version, content_type, long_description) for a wheel/sdist."""
|
|
224
|
+
if path.endswith(".whl"):
|
|
225
|
+
with zipfile.ZipFile(path) as z:
|
|
226
|
+
names = [n for n in z.namelist() if n.endswith(".dist-info/METADATA")]
|
|
227
|
+
if not names:
|
|
228
|
+
raise ValueError(f"no .dist-info/METADATA found in {path}")
|
|
229
|
+
raw = z.read(names[0]).decode("utf-8", "replace")
|
|
230
|
+
elif path.endswith((".tar.gz", ".tgz")):
|
|
231
|
+
with tarfile.open(path, "r:gz") as t:
|
|
232
|
+
members = [m for m in t.getmembers() if m.name.endswith("PKG-INFO")]
|
|
233
|
+
if not members:
|
|
234
|
+
raise ValueError(f"no PKG-INFO found in {path}")
|
|
235
|
+
f = t.extractfile(members[0])
|
|
236
|
+
assert f is not None
|
|
237
|
+
raw = f.read().decode("utf-8", "replace")
|
|
238
|
+
else:
|
|
239
|
+
raise ValueError(f"unsupported artifact (want .whl or .tar.gz): {path}")
|
|
240
|
+
|
|
241
|
+
msg = message_from_string(raw)
|
|
242
|
+
ctype = (msg.get("Description-Content-Type") or "text/plain").split(";")[0].strip().lower()
|
|
243
|
+
desc = msg.get_payload()
|
|
244
|
+
if not isinstance(desc, str):
|
|
245
|
+
desc = ""
|
|
246
|
+
return (
|
|
247
|
+
msg.get("Name") or "?",
|
|
248
|
+
msg.get("Version") or "?",
|
|
249
|
+
ctype,
|
|
250
|
+
desc,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def fetch_pypi(name: str, timeout: float = 20.0) -> tuple[str, str, str, str]:
|
|
255
|
+
"""Return (name, version, content_type, description) from the PyPI JSON API.
|
|
256
|
+
|
|
257
|
+
Note: the public JSON API exposes the raw ``description``, not rendered
|
|
258
|
+
HTML, so text-mode scanning applies unless a ``description_html`` field
|
|
259
|
+
is ever present.
|
|
260
|
+
"""
|
|
261
|
+
import json
|
|
262
|
+
import urllib.request
|
|
263
|
+
|
|
264
|
+
url = f"https://pypi.org/pypi/{name}/json"
|
|
265
|
+
req = urllib.request.Request(url, headers={"User-Agent": "readmeta/0.1.0"})
|
|
266
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
267
|
+
data = json.load(resp)
|
|
268
|
+
info = data.get("info", {})
|
|
269
|
+
if "description_html" in info and info["description_html"]:
|
|
270
|
+
return (
|
|
271
|
+
info.get("name") or name,
|
|
272
|
+
info.get("version") or "?",
|
|
273
|
+
"text/html",
|
|
274
|
+
info["description_html"],
|
|
275
|
+
)
|
|
276
|
+
ctype = (info.get("description_content_type") or "text/plain").split(";")[0].strip().lower()
|
|
277
|
+
return (
|
|
278
|
+
info.get("name") or name,
|
|
279
|
+
info.get("version") or "?",
|
|
280
|
+
ctype,
|
|
281
|
+
info.get("description") or "",
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def check_description(content_type: str, description: str, source: str) -> list[Finding]:
|
|
286
|
+
if "html" in content_type:
|
|
287
|
+
return check_html(description, source)
|
|
288
|
+
return check_text(description, source)
|
readmeta/cli.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Command-line interface for readmeta."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
import urllib.error
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .check import check_description, fetch_pypi, read_artifact
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _report(name: str, version: str, content_type: str, source: str, description: str) -> int:
|
|
14
|
+
findings = check_description(content_type, description, source)
|
|
15
|
+
print(f"{name} {version} [{content_type}] <- {source}")
|
|
16
|
+
if not findings:
|
|
17
|
+
print("OK: no PyPI rendering issues found.")
|
|
18
|
+
return 0
|
|
19
|
+
print(f"Found {len(findings)} issue(s):\n")
|
|
20
|
+
for f in findings:
|
|
21
|
+
print(f" [{f.kind}] {f.context}\n {f.target}\n -> {f.hint}\n")
|
|
22
|
+
return 1
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def main(argv: list[str] | None = None) -> int:
|
|
26
|
+
ap = argparse.ArgumentParser(
|
|
27
|
+
prog="readmeta",
|
|
28
|
+
description=(
|
|
29
|
+
"Check whether a package's README will render correctly on PyPI. "
|
|
30
|
+
"Inspects the built artifact's long_description (what PyPI actually "
|
|
31
|
+
"renders), not the repo's README.md."
|
|
32
|
+
),
|
|
33
|
+
)
|
|
34
|
+
ap.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
35
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
36
|
+
|
|
37
|
+
chk = sub.add_parser("check", help="check artifacts or a PyPI project for rendering issues")
|
|
38
|
+
chk.add_argument("paths", nargs="*", help="wheel (.whl) and/or sdist (.tar.gz) files")
|
|
39
|
+
chk.add_argument("--pypi", metavar="NAME", help="check the description hosted on PyPI instead of local files")
|
|
40
|
+
|
|
41
|
+
args = ap.parse_args(argv)
|
|
42
|
+
|
|
43
|
+
if args.command == "check":
|
|
44
|
+
if args.pypi:
|
|
45
|
+
try:
|
|
46
|
+
name, version, ctype, desc = fetch_pypi(args.pypi)
|
|
47
|
+
except urllib.error.HTTPError as e:
|
|
48
|
+
print(f"error: PyPI request failed ({e.code} {e.reason})", file=sys.stderr)
|
|
49
|
+
return 2
|
|
50
|
+
except OSError as e:
|
|
51
|
+
print(f"error: could not reach PyPI: {e}", file=sys.stderr)
|
|
52
|
+
return 2
|
|
53
|
+
return _report(name, version, ctype, f"pypi:{args.pypi}", desc)
|
|
54
|
+
|
|
55
|
+
if not args.paths:
|
|
56
|
+
ap.error("check needs at least one artifact path or --pypi NAME")
|
|
57
|
+
worst = 0
|
|
58
|
+
for path in args.paths:
|
|
59
|
+
try:
|
|
60
|
+
name, version, ctype, desc = read_artifact(path)
|
|
61
|
+
except (OSError, ValueError) as e:
|
|
62
|
+
print(f"error: {e}", file=sys.stderr)
|
|
63
|
+
worst = max(worst, 2)
|
|
64
|
+
continue
|
|
65
|
+
rc = _report(name, version, ctype, path, desc)
|
|
66
|
+
worst = max(worst, rc)
|
|
67
|
+
return worst
|
|
68
|
+
|
|
69
|
+
return 2 # pragma: no cover
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
if __name__ == "__main__":
|
|
73
|
+
sys.exit(main())
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: readmeta
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Check that your README will actually render on PyPI — inspects the built artifact, not the repo
|
|
5
|
+
Author: hao li
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/hahahahahahahahah6/readmeta
|
|
8
|
+
Keywords: pypi,readme,packaging,ci,twine
|
|
9
|
+
Requires-Python: >=3.9
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Dynamic: license-file
|
|
13
|
+
|
|
14
|
+
# readmeta
|
|
15
|
+
|
|
16
|
+
Check that your README will actually render on PyPI — by inspecting the **built artifact**, not the repo.
|
|
17
|
+
|
|
18
|
+
## The problem
|
|
19
|
+
|
|
20
|
+
PyPI renders the `long_description` from your built artifact (the body of `METADATA` in a wheel / `PKG-INFO` in an sdist). It does **not** render your repo's `README.md`. Things that look perfect on GitHub silently break on PyPI:
|
|
21
|
+
|
|
22
|
+
- **Relative images** (``) → 404, PyPI has no `docs/` folder
|
|
23
|
+
- **Relative links** (`[usage](docs/usage.md)`) → resolve against `pypi.org`, broken
|
|
24
|
+
- **Raw SVG references** (`assets/logo.svg`) → 404; PyPI doesn't serve repo files
|
|
25
|
+
- **In-page anchors** (`[setup](#instalation)`) → go nowhere when the heading id doesn't exist
|
|
26
|
+
|
|
27
|
+
`twine check` validates that the description *renders* — it does not check that any link or image actually *resolves*.
|
|
28
|
+
|
|
29
|
+
Real cases this catches:
|
|
30
|
+
|
|
31
|
+
- `armsmith` v1.2.1 shipped a relative SVG badge that 404'd on its PyPI page while rendering fine on GitHub.
|
|
32
|
+
- `aicertify` maintains a separate `README-pypi.md` with rewritten absolute URLs, precisely because the GitHub README breaks on PyPI.
|
|
33
|
+
- An audit of the `modern-python` GitHub org found 23 of 25 packages shipping relative asset references that can't resolve on PyPI.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install readmeta
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Requires Python 3.9+. No dependencies — stdlib only.
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
Check your built artifacts (build first, then check what PyPI will actually see):
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
python -m build
|
|
49
|
+
readmeta check dist/*
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Or check what's currently hosted on PyPI:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
readmeta check --pypi requests
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Example output:
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
fakepkg 0.1.0 [text/markdown] <- dist/fakepkg-0.1.0-py3-none-any.whl
|
|
62
|
+
Found 2 issue(s):
|
|
63
|
+
|
|
64
|
+
[relative-image] dist/fakepkg-0.1.0-py3-none-any.whl:12
|
|
65
|
+
docs/shot.png
|
|
66
|
+
-> PyPI renders the description standalone; relative image paths 404. Use an absolute https:// URL (e.g. raw.githubusercontent.com).
|
|
67
|
+
|
|
68
|
+
[broken-anchor] dist/fakepkg-0.1.0-py3-none-any.whl:20
|
|
69
|
+
#instalation
|
|
70
|
+
-> No heading with a matching id was found; the link goes nowhere on PyPI.
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Exit codes are CI-friendly: `0` = clean, `1` = issues found, `2` = error (unreadable artifact, PyPI unreachable, bad usage).
|
|
74
|
+
|
|
75
|
+
## CI example
|
|
76
|
+
|
|
77
|
+
```yaml
|
|
78
|
+
- name: Build
|
|
79
|
+
run: python -m build
|
|
80
|
+
|
|
81
|
+
- name: Check PyPI rendering
|
|
82
|
+
run: |
|
|
83
|
+
pip install readmeta
|
|
84
|
+
readmeta check dist/*
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## How it works
|
|
88
|
+
|
|
89
|
+
1. Reads `long_description` from `.whl` (`zipfile` → `.dist-info/METADATA`) or `.tar.gz` (`tarfile` → `PKG-INFO`), plus the `Description-Content-Type` header.
|
|
90
|
+
2. For `text/html`, parses with `html.parser` and validates every `img[src]`, `a[href]`, and `#anchor` against collected element ids (explicit ids plus GitHub-style heading slugs).
|
|
91
|
+
3. For Markdown/RST (what artifacts actually carry — the raw source, not rendered HTML), scans with regexes: inline and reference-style images/links, embedded `<img>` tags, RST `image::`/`figure::` directives, and `#anchor` links against `# Heading` slugs.
|
|
92
|
+
4. `--pypi` mode fetches `https://pypi.org/pypi/<name>/json` and checks the hosted description.
|
|
93
|
+
|
|
94
|
+
## Limitations (v0.1)
|
|
95
|
+
|
|
96
|
+
- **Check only** — no `--fix`. A build-time rewrite mode (convert relative refs to absolute URLs at build time) is planned.
|
|
97
|
+
- Anchor validation for RST is limited (Markdown headings and HTML ids are covered; RST `.. _target:` definitions are not yet resolved).
|
|
98
|
+
- Code spans and fenced code blocks are ignored (documenting a bad pattern doesn't flag it); indented code blocks are still scanned.
|
|
99
|
+
- Heading-slug generation approximates GitHub/PyPI's algorithm; exotic headings could produce false positives — explicit `id` attributes always win.
|
|
100
|
+
- `--pypi` uses the raw `description` from the JSON API (the API doesn't expose rendered HTML); findings are identical to checking a fresh local build.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
readmeta/__init__.py,sha256=m79wxCNuNjTk0iJ5IIwxdCHRa-Sfi0-89aB6YW7TrCM,92
|
|
2
|
+
readmeta/check.py,sha256=EHHacQ6vAu3WJDo6oSoi6JviHR6ltJbKZ7IAFjWrvg0,10723
|
|
3
|
+
readmeta/cli.py,sha256=mhPyrGwtBg8kq8X7smi4h7j8OYLGcDy5it7SIekRHfs,2657
|
|
4
|
+
readmeta-0.1.0.dist-info/licenses/LICENSE,sha256=OYhQDg7nxWoFtnF6J8NkFUA8NMw5TKv9a1WTXHY-fu4,1063
|
|
5
|
+
readmeta-0.1.0.dist-info/METADATA,sha256=8O7qmEkXj4gFMFYuXc7He127APE6L5ILQP8C89qaBt0,4196
|
|
6
|
+
readmeta-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
7
|
+
readmeta-0.1.0.dist-info/entry_points.txt,sha256=rwamKpg_5cn4RnhaMAMZ8tijuikXf36Bc6LKY-YzWsg,47
|
|
8
|
+
readmeta-0.1.0.dist-info/top_level.txt,sha256=KVbdaBc_bSBeVCWPG2d4VgAkDH--2z5wBJMmWcib6Os,9
|
|
9
|
+
readmeta-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 hao li
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
readmeta
|