amethyst-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,62 @@
1
+ """YAML frontmatter: split it off the token stream, read it as metadata."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ import yaml
8
+ from markdown_it.token import Token
9
+
10
+ from amethyst.errors import InputError
11
+
12
+
13
+ def split_frontmatter(tokens: list[Token]) -> tuple[dict[str, Any], list[Token]]:
14
+ """Pull a leading ``front_matter`` token out and parse it as YAML.
15
+
16
+ The parser is given the whole file, delimiters included, so every token's
17
+ line map stays true to the source — a warning that names line 40 means line
18
+ 40 of the file the author actually wrote. The token is dropped here because
19
+ nothing downstream should have to know it was ever in the stream.
20
+ """
21
+ if not tokens or tokens[0].type != "front_matter":
22
+ return {}, tokens
23
+ return parse_frontmatter(tokens[0].content), list(tokens[1:])
24
+
25
+
26
+ def parse_frontmatter(text: str) -> dict[str, Any]:
27
+ """Parse the body of a frontmatter block into a metadata mapping.
28
+
29
+ Keys are lowercased so that ``Title:`` and ``title:`` mean the same thing.
30
+ Values are left exactly as YAML produced them — dates stay dates, lists
31
+ stay lists — and are flattened to text only where they are displayed.
32
+ """
33
+ try:
34
+ loaded = yaml.safe_load(text)
35
+ except yaml.YAMLError as exc:
36
+ raise InputError(
37
+ "The YAML frontmatter could not be parsed.", hint=_yaml_hint(exc)
38
+ ) from exc
39
+
40
+ if loaded is None:
41
+ return {}
42
+ if not isinstance(loaded, dict):
43
+ raise InputError(
44
+ f"The frontmatter is a {type(loaded).__name__}, "
45
+ "not a set of key: value pairs.",
46
+ hint="Frontmatter looks like `title: My Document`, one field per line.",
47
+ )
48
+ return {str(key).strip().lower(): value for key, value in loaded.items()}
49
+
50
+
51
+ def _yaml_hint(exc: yaml.YAMLError) -> str | None:
52
+ """Turn PyYAML's mark into a line number the user can act on.
53
+
54
+ The mark counts from the start of the frontmatter body, so add one for the
55
+ opening ``---`` to get back to a line number in the file itself.
56
+ """
57
+ problem = getattr(exc, "problem", None)
58
+ mark = getattr(exc, "problem_mark", None)
59
+ if mark is None:
60
+ return str(problem) if problem else None
61
+ where = f"line {mark.line + 2} of the file"
62
+ return f"{problem} at {where}." if problem else f"Check {where}."
@@ -0,0 +1,49 @@
1
+ """Construction of the parser both pipelines share.
2
+
3
+ One configuration serves both: the PDF path renders these tokens to HTML, the
4
+ DOCX path walks them directly. Keeping the plugin set in a single function is
5
+ what stops the two outputs from disagreeing about what the Markdown meant.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Callable
11
+
12
+ from markdown_it import MarkdownIt
13
+ from mdit_py_plugins.anchors import anchors_plugin
14
+ from mdit_py_plugins.deflist import deflist_plugin
15
+ from mdit_py_plugins.footnote import footnote_plugin
16
+ from mdit_py_plugins.front_matter import front_matter_plugin
17
+ from mdit_py_plugins.tasklists import tasklists_plugin
18
+
19
+ #: Deepest heading level that gets an ``id``. Anchors cost nothing and an
20
+ #: internal link can point at any heading, so every level gets one; --toc-depth
21
+ #: decides which of them reach the table of contents, separately and later.
22
+ ANCHOR_MAX_LEVEL = 6
23
+
24
+ #: What markdown-it hands a highlighter: the code, the language written after
25
+ #: the fence, and whatever else was on that line. Returning ``None`` — or
26
+ #: anything empty — leaves markdown-it to escape the code itself, which is
27
+ #: exactly what an unhighlightable block wants.
28
+ Highlight = Callable[[str, str, str], "str | None"]
29
+
30
+
31
+ def build_parser(highlight: Highlight | None = None) -> MarkdownIt:
32
+ """Return the parser Amethyst uses for every document.
33
+
34
+ The ``gfm-like`` preset brings tables, strikethrough and linkify. Linkify
35
+ is not self-contained: it hard-requires ``linkify-it-py`` and raises at
36
+ *render* time rather than import time when it is absent, which is why that
37
+ package is a declared dependency rather than an optional extra.
38
+
39
+ ``highlight`` is only read when the tokens are rendered to HTML, so parsing
40
+ a document needs none: it is an option of the HTML renderer that markdown-it
41
+ happens to keep on the parser.
42
+ """
43
+ md = MarkdownIt("gfm-like", {"highlight": highlight} if highlight else None)
44
+ md.use(front_matter_plugin)
45
+ md.use(footnote_plugin)
46
+ md.use(deflist_plugin)
47
+ md.use(tasklists_plugin)
48
+ md.use(anchors_plugin, max_level=ANCHOR_MAX_LEVEL)
49
+ return md
amethyst/remote.py ADDED
@@ -0,0 +1,236 @@
1
+ """Downloading the images a document points at over the network.
2
+
3
+ This is the only place in Amethyst that opens a socket, and it runs as a step
4
+ of the conversion rather than inside either renderer. That is the whole design:
5
+ by the time a renderer sees the token stream, every image it can have is a file
6
+ on disk, and neither pipeline needs an opinion about HTTP, a cache, a timeout
7
+ or a size limit. The PDF renderer still refuses a remote URL outright — that
8
+ refusal is now a backstop for the ones this step could not get, not the policy.
9
+
10
+ Downloads are cached, keyed by the URL, so converting the same document twice
11
+ — or a dozen documents that share a logo — makes one request. The cache is
12
+ never invalidated: an image at a URL is treated as the thing that URL names.
13
+ Delete the directory to be rid of it.
14
+
15
+ The limits are deliberately unfriendly. A document is a document, not a
16
+ browser: one connection at a time, ten seconds each, thirty-two megabytes at
17
+ the outside, and http or https and nothing else — checked again after
18
+ redirects, because a redirect is somebody else's choice of scheme.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import hashlib
24
+ import mimetypes
25
+ import os
26
+ from pathlib import Path
27
+ from urllib.error import HTTPError, URLError
28
+ from urllib.parse import unquote, urlsplit
29
+ from urllib.request import Request, urlopen
30
+
31
+ from amethyst import __version__
32
+ from amethyst.document import Document
33
+ from amethyst.parse.assets import REMOTE_SCHEMES, Asset, AssetKind
34
+ from amethyst.render.base import Warn, discard
35
+
36
+ #: How long to wait for a response, in seconds. A conversion is interactive:
37
+ #: a wait long enough to look like a hang is worse than a missing picture.
38
+ TIMEOUT = 10.0
39
+
40
+ #: The most a single image may be. Past this it is not an illustration, and
41
+ #: whatever it is will not lay out on a page anyway.
42
+ MAX_BYTES = 32 * 1024 * 1024
43
+
44
+ #: Read in chunks so that the size limit can be enforced while reading rather
45
+ #: than after a refusal has already been held in memory.
46
+ CHUNK = 64 * 1024
47
+
48
+ #: Sent so that a server logging its traffic can see what asked.
49
+ USER_AGENT = f"amethyst/{__version__}"
50
+
51
+ #: Where downloads are kept, under the user's cache directory.
52
+ CACHE_PARTS = ("amethyst", "images")
53
+
54
+ #: The environment variable naming the base cache directory, and the directory
55
+ #: to use when it says nothing. Both are the XDG convention, which uv, pip and
56
+ #: most of this tool's neighbours already follow on macOS as well as Linux.
57
+ CACHE_HOME = "XDG_CACHE_HOME"
58
+ DEFAULT_CACHE_HOME = Path.home() / ".cache"
59
+
60
+ #: Extensions worth trusting from a URL before anything has been fetched. A
61
+ #: path ending in something else gets its extension from the response instead.
62
+ IMAGE_SUFFIXES = frozenset(
63
+ {".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp", ".bmp", ".tif", ".tiff"}
64
+ )
65
+
66
+
67
+ def cache_directory() -> Path:
68
+ """Where downloaded images are kept between runs."""
69
+ base = os.environ.get(CACHE_HOME)
70
+ root = Path(base) if base else DEFAULT_CACHE_HOME
71
+ return root.joinpath(*CACHE_PARTS)
72
+
73
+
74
+ def fetch_remote_images(
75
+ document: Document,
76
+ *,
77
+ enabled: bool = True,
78
+ warn: Warn = discard,
79
+ cache: Path | None = None,
80
+ ) -> None:
81
+ """Download the document's remote images and point it at the local copies.
82
+
83
+ Rewrites the image tokens in place, exactly as asset resolution rewrites a
84
+ local one, so that a renderer never learns an image was ever remote. An
85
+ image that cannot be fetched is left as written: the renderer then reports
86
+ it missing, naming the URL the author typed.
87
+ """
88
+ references = _remote_images(document)
89
+ if not references or not enabled:
90
+ return
91
+
92
+ directory = cache if cache is not None else cache_directory()
93
+ downloaded: dict[str, Path] = {}
94
+ for reference in references:
95
+ path = _download(reference, directory, warn)
96
+ if path is not None:
97
+ downloaded[reference] = path
98
+ if downloaded:
99
+ _rewrite(document, downloaded)
100
+
101
+
102
+ def _remote_images(document: Document) -> list[str]:
103
+ """Every distinct remote image URL, in the order they are written."""
104
+ seen: dict[str, None] = {}
105
+ for asset in document.assets:
106
+ if asset.is_remote and asset.kind is AssetKind.image:
107
+ seen.setdefault(asset.reference, None)
108
+ return list(seen)
109
+
110
+
111
+ def _rewrite(document: Document, downloaded: dict[str, Path]) -> None:
112
+ """Point the tokens, and the recorded assets, at the downloaded files."""
113
+ for token in document.tokens:
114
+ if token.type != "inline" or not token.children:
115
+ continue
116
+ for child in token.children:
117
+ if child.type != "image":
118
+ continue
119
+ source = child.attrGet("src")
120
+ path = downloaded.get(source) if isinstance(source, str) else None
121
+ if path is not None:
122
+ child.attrSet("src", str(path))
123
+ document.assets = [
124
+ Asset(
125
+ kind=asset.kind,
126
+ reference=asset.reference,
127
+ line=asset.line,
128
+ path=downloaded[asset.reference],
129
+ is_remote=True,
130
+ )
131
+ if asset.is_remote and asset.reference in downloaded
132
+ else asset
133
+ for asset in document.assets
134
+ ]
135
+
136
+
137
+ def _download(url: str, directory: Path, warn: Warn) -> Path | None:
138
+ """Return the local copy of one remote image, fetching it if need be."""
139
+ key = _key(url)
140
+ cached = _cached(directory, key)
141
+ if cached is not None:
142
+ return cached
143
+ try:
144
+ payload, suffix = _read(url)
145
+ except (HTTPError, URLError, TimeoutError, OSError, ValueError) as exc:
146
+ warn(f"could not download {url}: {_reason(exc)}.")
147
+ return None
148
+ try:
149
+ return _store(directory, key + suffix, payload)
150
+ except OSError as exc:
151
+ detail = exc.strerror or str(exc)
152
+ warn(f"could not cache {url} in {directory}: {detail.lower()}.")
153
+ return None
154
+
155
+
156
+ def _read(url: str) -> tuple[bytes, str]:
157
+ """Fetch one URL, refusing anything too big or not over http."""
158
+ request = Request(url, headers={"User-Agent": USER_AGENT})
159
+ with urlopen(request, timeout=TIMEOUT) as response:
160
+ # A redirect is the server's choice of destination, not the author's,
161
+ # so the scheme is checked again on the URL actually opened.
162
+ if urlsplit(response.geturl()).scheme not in REMOTE_SCHEMES:
163
+ raise ValueError("it redirected somewhere that is not http")
164
+ payload = bytearray()
165
+ while chunk := response.read(CHUNK):
166
+ payload += chunk
167
+ if len(payload) > MAX_BYTES:
168
+ raise ValueError(f"it is larger than {MAX_BYTES // (1024 * 1024)}MB")
169
+ suffix = _suffix(url, response.headers.get("Content-Type"))
170
+ return bytes(payload), suffix
171
+
172
+
173
+ def _store(directory: Path, name: str, payload: bytes) -> Path:
174
+ """Write a download into the cache, whole or not at all.
175
+
176
+ Written beside the final name and moved onto it, so that a run stopped
177
+ halfway leaves no half a picture for the next one to find and trust.
178
+ """
179
+ directory.mkdir(parents=True, exist_ok=True)
180
+ destination = directory / name
181
+ partial = destination.with_name(f"{name}.part{os.getpid()}")
182
+ partial.write_bytes(payload)
183
+ partial.replace(destination)
184
+ return destination
185
+
186
+
187
+ def _cached(directory: Path, key: str) -> Path | None:
188
+ """The cached download for a key, whatever extension it ended up with."""
189
+ if not directory.is_dir():
190
+ return None
191
+ for entry in sorted(directory.glob(f"{key}*")):
192
+ if entry.is_file() and ".part" not in entry.name:
193
+ return entry
194
+ return None
195
+
196
+
197
+ def _key(url: str) -> str:
198
+ """A filename for a URL: its digest, which is stable and has no path in it."""
199
+ return hashlib.sha256(url.encode("utf-8")).hexdigest()
200
+
201
+
202
+ def _suffix(url: str, content_type: str | None) -> str:
203
+ """The extension to save a download under.
204
+
205
+ Both renderers work out an image's real type from its first bytes, so this
206
+ is for the benefit of anyone who looks in the cache directory — and for
207
+ WeasyPrint, which guesses a MIME type from the name of a ``file:`` URL
208
+ before it falls back to sniffing.
209
+ """
210
+ from_url = Path(unquote(urlsplit(url).path)).suffix.lower()
211
+ if from_url in IMAGE_SUFFIXES:
212
+ return from_url
213
+ if content_type:
214
+ guessed = mimetypes.guess_extension(content_type.split(";")[0].strip())
215
+ if guessed:
216
+ return guessed
217
+ return ""
218
+
219
+
220
+ def _reason(exc: BaseException) -> str:
221
+ """Why a download failed, in the words a person would use."""
222
+ if isinstance(exc, HTTPError):
223
+ return f"the server answered {exc.code}"
224
+ if isinstance(exc, URLError):
225
+ reason = exc.reason
226
+ if isinstance(reason, TimeoutError):
227
+ return "it timed out"
228
+ return str(reason).strip(" <>") or "the request failed"
229
+ if isinstance(exc, TimeoutError):
230
+ return "it timed out"
231
+ if isinstance(exc, OSError):
232
+ return (exc.strerror or str(exc)).lower()
233
+ return str(exc)
234
+
235
+
236
+ __all__ = ["MAX_BYTES", "TIMEOUT", "cache_directory", "fetch_remote_images"]
@@ -0,0 +1,42 @@
1
+ """Renderers: one parsed document in, one format's bytes out.
2
+
3
+ Each format gets its own pipeline rather than a shared abstraction over both.
4
+ PDF goes through HTML and CSS, because CSS paged media already solves page
5
+ geometry, running furniture and bookmarks; Word has no equivalent and needs the
6
+ token stream walked directly. What keeps the two outputs recognisably the same
7
+ document is the theme they share, not a common renderer.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from amethyst.render.base import (
13
+ DEFAULT_HIGHLIGHT_STYLE,
14
+ Renderer,
15
+ RenderOptions,
16
+ RenderResult,
17
+ Warn,
18
+ )
19
+ from amethyst.render.docx import render_docx
20
+ from amethyst.render.highlight import (
21
+ NO_HIGHLIGHTING,
22
+ Highlighter,
23
+ highlight_styles,
24
+ resolve_highlight_style,
25
+ )
26
+ from amethyst.render.html import render_html
27
+ from amethyst.render.pdf import render_pdf
28
+
29
+ __all__ = [
30
+ "DEFAULT_HIGHLIGHT_STYLE",
31
+ "NO_HIGHLIGHTING",
32
+ "Highlighter",
33
+ "RenderOptions",
34
+ "RenderResult",
35
+ "Renderer",
36
+ "Warn",
37
+ "highlight_styles",
38
+ "render_docx",
39
+ "render_html",
40
+ "render_pdf",
41
+ "resolve_highlight_style",
42
+ ]
@@ -0,0 +1,85 @@
1
+ """What a renderer is: one document and some options in, one file's bytes out.
2
+
3
+ The protocol is deliberately narrow. HTML and EPUB are out of scope for v1 but
4
+ are the obvious next outputs, and keeping the contract to "bytes, plus whatever
5
+ is worth saying about them" is what makes adding one additive rather than a
6
+ change to everything that calls a renderer.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from collections.abc import Callable
12
+ from dataclasses import dataclass, field
13
+ from pathlib import Path
14
+ from typing import Protocol
15
+
16
+ from amethyst.document import Document
17
+ from amethyst.theme import Theme, default_theme
18
+
19
+ #: How a renderer reports something the user should know but that is not fatal
20
+ #: — an image it could not load, a construct it had to approximate. Renderers
21
+ #: are given this rather than printing, so nothing in this package has an
22
+ #: opinion about where output goes or whether --quiet was passed.
23
+ Warn = Callable[[str], None]
24
+
25
+ #: The Pygments style a document is highlighted with when the user names none.
26
+ #: It lives here rather than with the highlighter because it is the default of
27
+ #: an option declared below, and :mod:`amethyst.render.highlight` reads it back
28
+ #: from here so that there is one spelling of it.
29
+ DEFAULT_HIGHLIGHT_STYLE = "default"
30
+
31
+
32
+ def discard(message: str) -> None:
33
+ """The default warning sink, for callers that do not want to hear it."""
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class RenderOptions:
38
+ """Everything a renderer needs that is not part of the document itself.
39
+
40
+ These are already resolved: the CLI has merged flags over defaults, so a
41
+ renderer never has to reason about what was and was not passed. Page
42
+ geometry is part of the theme rather than a field here — a flag that
43
+ overrides it does so by handing over an overridden theme, which leaves one
44
+ place for a renderer to read the sheet size from.
45
+ """
46
+
47
+ theme: Theme = field(default_factory=default_theme)
48
+ #: Extra stylesheet, appended after everything else so it wins. PDF only.
49
+ extra_css: Path | None = None
50
+ page_numbers: bool = True
51
+ #: Whether to open the document with a table of contents, and how deep to
52
+ #: take it. The depth is carried even when the contents is off, so that
53
+ #: nothing has to reason about which of the two flags was passed.
54
+ toc: bool = False
55
+ toc_depth: int = 3
56
+ #: Whether to open the document with a title page built from frontmatter.
57
+ title_page: bool = False
58
+ #: The Pygments style code is coloured with, or ``"none"`` for no colour.
59
+ #: Held as a name rather than as a built highlighter because a renderer
60
+ #: needs one of its own anyway: a highlighter remembers what it has already
61
+ #: warned about, and that is per conversion.
62
+ highlight_style: str = DEFAULT_HIGHLIGHT_STYLE
63
+ warn: Warn = discard
64
+
65
+
66
+ @dataclass(frozen=True)
67
+ class RenderResult:
68
+ """The finished document, and what is worth telling the user about it."""
69
+
70
+ data: bytes
71
+ #: Page count, where the format has one. DOCX does not — Word decides
72
+ #: pagination when it opens the file, so there is nothing honest to report.
73
+ pages: int | None = None
74
+
75
+
76
+ class Renderer(Protocol):
77
+ """One output format.
78
+
79
+ A callable protocol rather than a class, because a renderer has no state
80
+ worth keeping between documents; ``render_pdf`` satisfies it as written.
81
+ """
82
+
83
+ def __call__(self, document: Document, options: RenderOptions) -> RenderResult:
84
+ """Turn ``document`` into the bytes of one file."""
85
+ ...