amethyst-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- amethyst/__init__.py +5 -0
- amethyst/__main__.py +6 -0
- amethyst/cli.py +644 -0
- amethyst/config.py +387 -0
- amethyst/document.py +228 -0
- amethyst/errors.py +66 -0
- amethyst/ooxml.py +549 -0
- amethyst/parse/__init__.py +20 -0
- amethyst/parse/assets.py +106 -0
- amethyst/parse/frontmatter.py +62 -0
- amethyst/parse/markdown.py +49 -0
- amethyst/remote.py +236 -0
- amethyst/render/__init__.py +42 -0
- amethyst/render/base.py +85 -0
- amethyst/render/docx.py +1060 -0
- amethyst/render/furniture.py +112 -0
- amethyst/render/highlight.py +315 -0
- amethyst/render/html.py +266 -0
- amethyst/render/pdf.py +219 -0
- amethyst/theme/__init__.py +493 -0
- amethyst/theme/builtin/academic.toml +45 -0
- amethyst/theme/builtin/css/base.css +361 -0
- amethyst/theme/builtin/default.toml +42 -0
- amethyst/theme/builtin/github.toml +44 -0
- amethyst/theme/to_css.py +182 -0
- amethyst/theme/to_docx.py +651 -0
- amethyst_cli-0.1.0.dist-info/METADATA +293 -0
- amethyst_cli-0.1.0.dist-info/RECORD +31 -0
- amethyst_cli-0.1.0.dist-info/WHEEL +4 -0
- amethyst_cli-0.1.0.dist-info/entry_points.txt +2 -0
- amethyst_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""What goes on a page besides the document: contents, running head, cover.
|
|
2
|
+
|
|
3
|
+
The two pipelines build this furniture in completely different ways — one in
|
|
4
|
+
CSS paged media, the other in Word sections and field codes — but they must
|
|
5
|
+
agree on *what* it says, or the same source comes out as two documents. The
|
|
6
|
+
decisions that both have to make identically are made once, here, and nowhere
|
|
7
|
+
else: which headings the contents lists, which heading level the running head
|
|
8
|
+
tracks, and what the cover has on it.
|
|
9
|
+
|
|
10
|
+
Nothing in this module knows about HTML or OOXML. It reads a parsed document
|
|
11
|
+
and returns the answers.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections import Counter
|
|
17
|
+
from collections.abc import Sequence
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
from amethyst.document import Document, Heading
|
|
21
|
+
|
|
22
|
+
#: What a table of contents is called. Not read from the document, because it
|
|
23
|
+
#: is not in the document: it is furniture the renderer adds, and both formats
|
|
24
|
+
#: have to add it under the same name.
|
|
25
|
+
CONTENTS_HEADING = "Contents"
|
|
26
|
+
|
|
27
|
+
#: The heading levels a contents can list at all. ``--toc-depth`` cuts into
|
|
28
|
+
#: this; it cannot go past it, because ``h6`` is the last heading there is.
|
|
29
|
+
MAX_TOC_DEPTH = 6
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class Cover:
|
|
34
|
+
"""What a title page has on it, once the empty lines are taken out.
|
|
35
|
+
|
|
36
|
+
A cover is worth making only if it has a title. Everything else is
|
|
37
|
+
optional and simply absent when the frontmatter did not declare it.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
title: str
|
|
41
|
+
subtitle: str | None = None
|
|
42
|
+
author: str | None = None
|
|
43
|
+
date: str | None = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def contents(document: Document, depth: int) -> list[Heading]:
|
|
47
|
+
"""The headings a table of contents lists, in document order.
|
|
48
|
+
|
|
49
|
+
Deeper headings are dropped rather than flattened: a contents that lists
|
|
50
|
+
every ``h6`` in a long document is a second document, not a way into the
|
|
51
|
+
first.
|
|
52
|
+
"""
|
|
53
|
+
return [item for item in document.headings if item.level <= depth]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def section_level(document: Document) -> int | None:
|
|
57
|
+
"""Which heading level the running head should name, if any.
|
|
58
|
+
|
|
59
|
+
A running head is useful when it changes — it tells a reader which part of
|
|
60
|
+
the document the page in their hand belongs to. So it tracks the shallowest
|
|
61
|
+
level that occurs more than once, which for the ordinary shape of a
|
|
62
|
+
Markdown file (one ``h1`` naming the document, ``h2`` sections under it)
|
|
63
|
+
is the ``h2``.
|
|
64
|
+
|
|
65
|
+
``None`` when no level repeats: there is nothing to track, and a head that
|
|
66
|
+
names the one heading in the document is just the title written twice.
|
|
67
|
+
"""
|
|
68
|
+
counts = Counter(item.level for item in document.headings)
|
|
69
|
+
for level in sorted(counts):
|
|
70
|
+
if counts[level] > 1:
|
|
71
|
+
return level
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def cover(document: Document) -> Cover | None:
|
|
76
|
+
"""The title page's content, or ``None`` if there is not enough for one.
|
|
77
|
+
|
|
78
|
+
Without a title there is no cover worth printing — a page carrying nothing
|
|
79
|
+
but an author's name is a blank page with a mistake on it — so this
|
|
80
|
+
returns nothing and the caller says so rather than emitting one.
|
|
81
|
+
"""
|
|
82
|
+
title = document.title
|
|
83
|
+
if not title:
|
|
84
|
+
return None
|
|
85
|
+
return Cover(
|
|
86
|
+
title=title,
|
|
87
|
+
subtitle=document.subtitle,
|
|
88
|
+
author=document.author,
|
|
89
|
+
date=document.date,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def outline_depth(headings: Sequence[Heading], depth: int) -> int:
|
|
94
|
+
"""The deepest level actually present in a contents, for Word's ``TOC``.
|
|
95
|
+
|
|
96
|
+
Word's field takes a range of heading levels rather than a list, and a
|
|
97
|
+
range reaching past the deepest heading in the document is not wrong, only
|
|
98
|
+
untidy. One is the floor: ``TOC \\o "1-0"`` is not a range.
|
|
99
|
+
"""
|
|
100
|
+
present = [item.level for item in headings if item.level <= depth]
|
|
101
|
+
return max(present, default=1)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
__all__ = [
|
|
105
|
+
"CONTENTS_HEADING",
|
|
106
|
+
"MAX_TOC_DEPTH",
|
|
107
|
+
"Cover",
|
|
108
|
+
"contents",
|
|
109
|
+
"cover",
|
|
110
|
+
"outline_depth",
|
|
111
|
+
"section_level",
|
|
112
|
+
]
|
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""Syntax highlighting, compiled the same two ways a theme is.
|
|
2
|
+
|
|
3
|
+
Pygments knows how to colour code and how to write the HTML for it. What it
|
|
4
|
+
does not know is that this project has two pipelines which must agree, so what
|
|
5
|
+
comes out of here is not Pygments' stylesheet but a small model of it: a
|
|
6
|
+
colour, a weight and a slope per token class. The PDF path turns that into CSS
|
|
7
|
+
rules and the Word path turns it into run formatting, from the same lookup, for
|
|
8
|
+
exactly the reason :mod:`amethyst.theme` compiles two ways rather than one.
|
|
9
|
+
|
|
10
|
+
Pygments' own ``get_style_defs`` is deliberately not used. It writes rules for
|
|
11
|
+
line numbering that nothing here emits, and a ``pre { line-height: 125% }``
|
|
12
|
+
that would quietly override the stylesheet's own leading — a highlighting style
|
|
13
|
+
is meant to colour the code, not to re-typeset it.
|
|
14
|
+
|
|
15
|
+
Two decisions are worth knowing about:
|
|
16
|
+
|
|
17
|
+
A **light style keeps the theme's background.** Code then sits on the same fill
|
|
18
|
+
as the inline code and the table headings around it, which is what makes a
|
|
19
|
+
highlighted block still look like part of the document. A **dark style brings
|
|
20
|
+
its own**, because it has to: its colours are chosen against a dark ground and
|
|
21
|
+
are unreadable on a light one.
|
|
22
|
+
|
|
23
|
+
**Nothing is guessed.** A fence with no language, or with one Pygments does not
|
|
24
|
+
recognise, is set plain rather than passed to a guesser. Guessing is slow, and
|
|
25
|
+
when it is wrong it is wrong in colour. Under a dark style such a block is
|
|
26
|
+
still a dark panel, in both formats — a light box beside a dark one, a
|
|
27
|
+
paragraph apart, is worse than an uncoloured one.
|
|
28
|
+
|
|
29
|
+
Inline code is never highlighted: three words between backticks name no
|
|
30
|
+
language, and there is nothing to lex. It keeps the theme's own fill even when
|
|
31
|
+
the blocks around it are dark, because it belongs to the sentence it sits in
|
|
32
|
+
rather than to a panel.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from dataclasses import dataclass
|
|
38
|
+
|
|
39
|
+
from pygments import highlight as pygments_highlight
|
|
40
|
+
from pygments.formatters import HtmlFormatter
|
|
41
|
+
from pygments.lexer import Lexer
|
|
42
|
+
from pygments.lexers import get_lexer_by_name
|
|
43
|
+
from pygments.style import Style
|
|
44
|
+
from pygments.styles import get_all_styles, get_style_by_name
|
|
45
|
+
from pygments.token import STANDARD_TYPES, Token, _TokenType
|
|
46
|
+
from pygments.util import ClassNotFound
|
|
47
|
+
|
|
48
|
+
from amethyst.errors import UsageError
|
|
49
|
+
from amethyst.render.base import DEFAULT_HIGHLIGHT_STYLE, Warn, discard
|
|
50
|
+
|
|
51
|
+
#: What to pass to turn highlighting off. Not a Pygments style name — the point
|
|
52
|
+
#: is to have a spelling for "leave the code alone", and code set in the mono
|
|
53
|
+
#: face with no colour at all is a legitimate way to want a document to look.
|
|
54
|
+
NO_HIGHLIGHTING = "none"
|
|
55
|
+
|
|
56
|
+
#: Below this relative luminance a background counts as dark, and the style is
|
|
57
|
+
#: taken to have been designed against its own ground rather than the page's.
|
|
58
|
+
DARK_BELOW = 0.4
|
|
59
|
+
|
|
60
|
+
#: The plain-text colour for a dark style that names none, which no shipped
|
|
61
|
+
#: style does — a legible off-white rather than a guess at the style's intent.
|
|
62
|
+
FALLBACK_FOREGROUND = "#f8f8f2"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class Span:
|
|
67
|
+
"""One run of code that is all one colour."""
|
|
68
|
+
|
|
69
|
+
text: str
|
|
70
|
+
color: str | None = None
|
|
71
|
+
bold: bool = False
|
|
72
|
+
italic: bool = False
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def plain(self) -> bool:
|
|
76
|
+
"""Whether this span asks for nothing the surrounding style lacks."""
|
|
77
|
+
return self.color is None and not self.bold and not self.italic
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Highlighter:
|
|
81
|
+
"""One highlighting style, ready for either pipeline.
|
|
82
|
+
|
|
83
|
+
Built per render rather than cached, because it remembers which unknown
|
|
84
|
+
languages it has already complained about: a document that fences twenty
|
|
85
|
+
blocks as ```pseudocode`` should say so once, not twenty times.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
def __init__(self, style: str = DEFAULT_HIGHLIGHT_STYLE, *, warn: Warn = discard):
|
|
89
|
+
self.name = resolve_highlight_style(style)
|
|
90
|
+
self._warn = warn
|
|
91
|
+
self._style: type[Style] | None = (
|
|
92
|
+
None if self.name == NO_HIGHLIGHTING else get_style_by_name(self.name)
|
|
93
|
+
)
|
|
94
|
+
self._unknown: set[str] = set()
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def enabled(self) -> bool:
|
|
98
|
+
"""Whether this highlighter colours anything at all."""
|
|
99
|
+
return self._style is not None
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def background(self) -> str | None:
|
|
103
|
+
"""The ground this style needs, or ``None`` to keep the theme's fill.
|
|
104
|
+
|
|
105
|
+
Only a dark style answers with a colour. See the module docstring: a
|
|
106
|
+
light style is an ornament on the page's own background, and a dark one
|
|
107
|
+
is a panel that brings its own.
|
|
108
|
+
"""
|
|
109
|
+
if self._style is None:
|
|
110
|
+
return None
|
|
111
|
+
declared = _hex(self._style.background_color)
|
|
112
|
+
if declared is None or _luminance(declared) >= DARK_BELOW:
|
|
113
|
+
return None
|
|
114
|
+
return declared
|
|
115
|
+
|
|
116
|
+
@property
|
|
117
|
+
def foreground(self) -> str | None:
|
|
118
|
+
"""The plain-text colour that goes with :attr:`background`."""
|
|
119
|
+
if self.background is None:
|
|
120
|
+
return None
|
|
121
|
+
assert self._style is not None
|
|
122
|
+
declared = self._format(Token.Text).color
|
|
123
|
+
return declared or FALLBACK_FOREGROUND
|
|
124
|
+
|
|
125
|
+
def html(self, code: str, language: str) -> str | None:
|
|
126
|
+
"""The code as coloured ``<span>``s, or ``None`` to set it plain.
|
|
127
|
+
|
|
128
|
+
No wrapper: markdown-it puts the result inside the ``<pre><code>`` it
|
|
129
|
+
would have written anyway, so the stylesheet's idea of what a code
|
|
130
|
+
block is stays in one place.
|
|
131
|
+
"""
|
|
132
|
+
lexer = self._lexer(language)
|
|
133
|
+
if lexer is None:
|
|
134
|
+
return None
|
|
135
|
+
formatter = HtmlFormatter(nowrap=True, style=self.name)
|
|
136
|
+
return pygments_highlight(code, lexer, formatter).rstrip("\n")
|
|
137
|
+
|
|
138
|
+
def spans(self, code: str, language: str) -> list[Span] | None:
|
|
139
|
+
"""The code as coloured runs, or ``None`` to set it plain.
|
|
140
|
+
|
|
141
|
+
The Word side of :meth:`html`. Trailing newlines are dropped here
|
|
142
|
+
rather than by the caller: every lexer appends one, and in Word a
|
|
143
|
+
trailing newline is a visible empty line inside the shaded box.
|
|
144
|
+
"""
|
|
145
|
+
lexer = self._lexer(language)
|
|
146
|
+
if lexer is None:
|
|
147
|
+
return None
|
|
148
|
+
found: list[Span] = []
|
|
149
|
+
for token, text in lexer.get_tokens(code):
|
|
150
|
+
style = self._format(token)
|
|
151
|
+
span = Span(
|
|
152
|
+
text=text,
|
|
153
|
+
color=style.color or self.foreground,
|
|
154
|
+
bold=style.bold,
|
|
155
|
+
italic=style.italic,
|
|
156
|
+
)
|
|
157
|
+
# A lexer emits a token per punctuation mark, and Word stores a run
|
|
158
|
+
# per span; merging the ones that are formatted alike takes a
|
|
159
|
+
# fenced block from hundreds of runs to a handful.
|
|
160
|
+
if found and _alike(found[-1], span):
|
|
161
|
+
found[-1] = Span(
|
|
162
|
+
text=found[-1].text + span.text,
|
|
163
|
+
color=span.color,
|
|
164
|
+
bold=span.bold,
|
|
165
|
+
italic=span.italic,
|
|
166
|
+
)
|
|
167
|
+
else:
|
|
168
|
+
found.append(span)
|
|
169
|
+
return _without_trailing_newline(found)
|
|
170
|
+
|
|
171
|
+
def css(self) -> str:
|
|
172
|
+
"""The style as CSS rules, scoped to the code block they colour.
|
|
173
|
+
|
|
174
|
+
Rules are emitted shallowest token type first. They all have the same
|
|
175
|
+
specificity, so where Pygments gives a span more than one class it is
|
|
176
|
+
the later rule — the more specific token — that has to win.
|
|
177
|
+
"""
|
|
178
|
+
if self._style is None:
|
|
179
|
+
return ""
|
|
180
|
+
lines = [f"/* highlighting: {self.name} */"]
|
|
181
|
+
background = self.background
|
|
182
|
+
if background is not None:
|
|
183
|
+
lines.append(
|
|
184
|
+
f"pre {{ background: {background}; color: {self.foreground}; "
|
|
185
|
+
f"border-color: {background}; }}"
|
|
186
|
+
)
|
|
187
|
+
for token, css_class in sorted(
|
|
188
|
+
STANDARD_TYPES.items(), key=lambda item: len(item[0])
|
|
189
|
+
):
|
|
190
|
+
if not css_class:
|
|
191
|
+
continue
|
|
192
|
+
declarations = _declarations(self._format(token))
|
|
193
|
+
if declarations:
|
|
194
|
+
lines.append(f"pre .{css_class} {{ {declarations} }}")
|
|
195
|
+
return "\n".join([*lines, ""])
|
|
196
|
+
|
|
197
|
+
def _lexer(self, language: str) -> Lexer | None:
|
|
198
|
+
"""The lexer for a fence's language, warning once when there is none."""
|
|
199
|
+
if self._style is None or not language:
|
|
200
|
+
return None
|
|
201
|
+
try:
|
|
202
|
+
return get_lexer_by_name(language)
|
|
203
|
+
except ClassNotFound:
|
|
204
|
+
if language not in self._unknown:
|
|
205
|
+
self._unknown.add(language)
|
|
206
|
+
self._warn(
|
|
207
|
+
f"no syntax highlighting for {language!r}; that code is set plain."
|
|
208
|
+
)
|
|
209
|
+
return None
|
|
210
|
+
|
|
211
|
+
def _format(self, token: _TokenType) -> Span:
|
|
212
|
+
"""How this style sets one token type, with inheritance resolved."""
|
|
213
|
+
assert self._style is not None
|
|
214
|
+
declared = self._style.style_for_token(token)
|
|
215
|
+
return Span(
|
|
216
|
+
text="",
|
|
217
|
+
color=_hex(declared["color"]),
|
|
218
|
+
bold=declared["bold"],
|
|
219
|
+
italic=declared["italic"],
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def resolve_highlight_style(name: str) -> str:
|
|
224
|
+
"""Check that a highlighting style exists, returning it as given.
|
|
225
|
+
|
|
226
|
+
A name that is not a style is a mistyped invocation rather than a broken
|
|
227
|
+
document, so it exits 2 like an unknown theme does — and the hint lists
|
|
228
|
+
every style, because there is no other way to find out what they are
|
|
229
|
+
called.
|
|
230
|
+
"""
|
|
231
|
+
if name == NO_HIGHLIGHTING:
|
|
232
|
+
return name
|
|
233
|
+
try:
|
|
234
|
+
get_style_by_name(name)
|
|
235
|
+
except ClassNotFound:
|
|
236
|
+
raise UsageError(
|
|
237
|
+
f"Unknown highlighting style {name!r}.",
|
|
238
|
+
hint=f"Styles: {', '.join(highlight_styles())}.",
|
|
239
|
+
) from None
|
|
240
|
+
return name
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def highlight_styles() -> tuple[str, ...]:
|
|
244
|
+
"""Every style that can be named, sorted, with ``none`` among them."""
|
|
245
|
+
return tuple(sorted([*get_all_styles(), NO_HIGHLIGHTING]))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _declarations(span: Span) -> str:
|
|
249
|
+
"""One token class's formatting, as the body of a CSS rule."""
|
|
250
|
+
parts = []
|
|
251
|
+
if span.color is not None:
|
|
252
|
+
parts.append(f"color: {span.color};")
|
|
253
|
+
if span.bold:
|
|
254
|
+
parts.append("font-weight: 600;")
|
|
255
|
+
if span.italic:
|
|
256
|
+
parts.append("font-style: italic;")
|
|
257
|
+
return " ".join(parts)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _alike(one: Span, other: Span) -> bool:
|
|
261
|
+
"""Whether two spans are formatted identically, text aside."""
|
|
262
|
+
return (one.color, one.bold, one.italic) == (other.color, other.bold, other.italic)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _without_trailing_newline(spans: list[Span]) -> list[Span]:
|
|
266
|
+
"""Drop the newline every lexer adds, and any the author left behind."""
|
|
267
|
+
while spans:
|
|
268
|
+
trimmed = spans[-1].text.rstrip("\n")
|
|
269
|
+
if trimmed:
|
|
270
|
+
spans[-1] = Span(
|
|
271
|
+
text=trimmed,
|
|
272
|
+
color=spans[-1].color,
|
|
273
|
+
bold=spans[-1].bold,
|
|
274
|
+
italic=spans[-1].italic,
|
|
275
|
+
)
|
|
276
|
+
break
|
|
277
|
+
spans.pop()
|
|
278
|
+
return spans
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _hex(color: str | None) -> str | None:
|
|
282
|
+
"""A Pygments colour as CSS hex. It stores them without the ``#``.
|
|
283
|
+
|
|
284
|
+
A style that declares no colour for a token gives ``None``, and one that
|
|
285
|
+
declares no background gives ``""`` — both mean "say nothing here".
|
|
286
|
+
"""
|
|
287
|
+
if not color:
|
|
288
|
+
return None
|
|
289
|
+
return color if color.startswith("#") else f"#{color}"
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _luminance(color: str) -> float:
|
|
293
|
+
"""Relative luminance of a hex colour, 0 for black and 1 for white.
|
|
294
|
+
|
|
295
|
+
The sRGB coefficients, without the gamma expansion a contrast ratio would
|
|
296
|
+
need: the only question being asked is "is this a dark panel or a light
|
|
297
|
+
page", and that answer does not change between the two.
|
|
298
|
+
"""
|
|
299
|
+
digits = color.lstrip("#")
|
|
300
|
+
if len(digits) == 3:
|
|
301
|
+
digits = "".join(digit * 2 for digit in digits)
|
|
302
|
+
if len(digits) != 6:
|
|
303
|
+
return 1.0
|
|
304
|
+
red, green, blue = (int(digits[at : at + 2], 16) / 255 for at in (0, 2, 4))
|
|
305
|
+
return 0.2126 * red + 0.7152 * green + 0.0722 * blue
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
__all__ = [
|
|
309
|
+
"DEFAULT_HIGHLIGHT_STYLE",
|
|
310
|
+
"NO_HIGHLIGHTING",
|
|
311
|
+
"Highlighter",
|
|
312
|
+
"Span",
|
|
313
|
+
"highlight_styles",
|
|
314
|
+
"resolve_highlight_style",
|
|
315
|
+
]
|
amethyst/render/html.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""Tokens to one standalone HTML document — the PDF pipeline's first half.
|
|
2
|
+
|
|
3
|
+
Standalone on purpose. Everything the page needs is inlined, so the result can
|
|
4
|
+
be written out and opened in a browser, which is by far the fastest way to tell
|
|
5
|
+
a CSS mistake apart from a WeasyPrint limitation. It is also what an eventual
|
|
6
|
+
HTML output would emit, which is why this is its own module rather than a
|
|
7
|
+
private helper inside the PDF renderer.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from html import escape
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, cast
|
|
16
|
+
|
|
17
|
+
from markdown_it.renderer import RendererHTML
|
|
18
|
+
from markdown_it.token import Token
|
|
19
|
+
from markdown_it.utils import EnvType, OptionsDict
|
|
20
|
+
|
|
21
|
+
from amethyst.document import Document
|
|
22
|
+
from amethyst.errors import RenderError
|
|
23
|
+
from amethyst.parse import build_parser
|
|
24
|
+
from amethyst.parse.markdown import Highlight
|
|
25
|
+
from amethyst.render.base import RenderOptions
|
|
26
|
+
from amethyst.render.furniture import (
|
|
27
|
+
CONTENTS_HEADING,
|
|
28
|
+
contents,
|
|
29
|
+
cover,
|
|
30
|
+
section_level,
|
|
31
|
+
)
|
|
32
|
+
from amethyst.render.highlight import Highlighter
|
|
33
|
+
from amethyst.theme import base_css
|
|
34
|
+
from amethyst.theme.to_css import page_css, root_css
|
|
35
|
+
|
|
36
|
+
#: Shown in the PDF's title bar when the document declares nothing better.
|
|
37
|
+
FALLBACK_TITLE = "Untitled"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def render_html(document: Document, options: RenderOptions) -> str:
|
|
41
|
+
"""Render the document as one self-contained HTML page."""
|
|
42
|
+
# One highlighter for the whole page: it colours the code and it writes
|
|
43
|
+
# the rules that colour it, and those two have to be the same style.
|
|
44
|
+
highlighter = Highlighter(options.highlight_style, warn=options.warn)
|
|
45
|
+
sheets = stylesheets(document, options, highlighter)
|
|
46
|
+
return "\n".join(
|
|
47
|
+
[
|
|
48
|
+
"<!DOCTYPE html>",
|
|
49
|
+
'<html><head><meta charset="utf-8">',
|
|
50
|
+
f"<title>{escape(document.title or FALLBACK_TITLE)}</title>",
|
|
51
|
+
*meta_tags(document),
|
|
52
|
+
*(f"<style>\n{sheet}</style>" for sheet in sheets),
|
|
53
|
+
"</head>",
|
|
54
|
+
"<body>",
|
|
55
|
+
*front_matter(document, options),
|
|
56
|
+
render_body(document, highlighter),
|
|
57
|
+
"</body></html>",
|
|
58
|
+
"",
|
|
59
|
+
]
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def meta_tags(document: Document) -> list[str]:
|
|
64
|
+
"""The frontmatter, as the tags WeasyPrint turns into PDF metadata.
|
|
65
|
+
|
|
66
|
+
The same four fields the Word renderer writes into its core properties, so
|
|
67
|
+
that a reader looking at the document's information pane sees the same
|
|
68
|
+
thing whichever format they were sent. The date is written only when it is
|
|
69
|
+
a real one: ``dcterms.created`` is a timestamp, and a document dated
|
|
70
|
+
"Spring 2026" has a date for its title page and none for its metadata.
|
|
71
|
+
"""
|
|
72
|
+
declared = [
|
|
73
|
+
("author", document.author),
|
|
74
|
+
("description", document.subtitle),
|
|
75
|
+
("keywords", document.keywords),
|
|
76
|
+
("dcterms.created", document.created.isoformat() if document.created else None),
|
|
77
|
+
]
|
|
78
|
+
return [
|
|
79
|
+
f'<meta name="{name}" content="{escape(value, quote=True)}">'
|
|
80
|
+
for name, value in declared
|
|
81
|
+
if value
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def front_matter(document: Document, options: RenderOptions) -> list[str]:
|
|
86
|
+
"""The cover and the contents, in the order they are read."""
|
|
87
|
+
pages = []
|
|
88
|
+
if options.title_page:
|
|
89
|
+
pages += title_page_html(document, options)
|
|
90
|
+
if options.toc:
|
|
91
|
+
pages += contents_html(document, options)
|
|
92
|
+
return pages
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def title_page_html(document: Document, options: RenderOptions) -> list[str]:
|
|
96
|
+
"""A cover built from the frontmatter, or nothing when there is no title."""
|
|
97
|
+
page = cover(document)
|
|
98
|
+
if page is None:
|
|
99
|
+
options.warn(
|
|
100
|
+
"--title-page needs a title; the document declares none, so no "
|
|
101
|
+
"title page was made."
|
|
102
|
+
)
|
|
103
|
+
return []
|
|
104
|
+
lines = ['<section class="title-page">']
|
|
105
|
+
for css_class, value in (
|
|
106
|
+
("doc-title", page.title),
|
|
107
|
+
("doc-subtitle", page.subtitle),
|
|
108
|
+
("doc-author", page.author),
|
|
109
|
+
("doc-date", page.date),
|
|
110
|
+
):
|
|
111
|
+
if value:
|
|
112
|
+
lines.append(f'<p class="{css_class}">{escape(value)}</p>')
|
|
113
|
+
lines.append("</section>")
|
|
114
|
+
return lines
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def contents_html(document: Document, options: RenderOptions) -> list[str]:
|
|
118
|
+
"""The table of contents, with the page numbers left for the PDF stage.
|
|
119
|
+
|
|
120
|
+
Every entry is written as a link, and the dots and the page number are
|
|
121
|
+
added by the stylesheet through ``target-counter`` — which only resolves
|
|
122
|
+
once the document has been laid out, so there is nothing to count here.
|
|
123
|
+
"""
|
|
124
|
+
entries = contents(document, options.toc_depth)
|
|
125
|
+
if not entries:
|
|
126
|
+
options.warn(
|
|
127
|
+
"--toc needs headings; the document has none, so no contents was made."
|
|
128
|
+
)
|
|
129
|
+
return []
|
|
130
|
+
lines = [
|
|
131
|
+
'<nav class="contents">',
|
|
132
|
+
f'<h1 class="toc-heading">{CONTENTS_HEADING}</h1>',
|
|
133
|
+
"<ol>",
|
|
134
|
+
]
|
|
135
|
+
for entry in entries:
|
|
136
|
+
label = escape(entry.text)
|
|
137
|
+
# An entry with no anchor cannot be linked, and so cannot carry a page
|
|
138
|
+
# number either — target-counter has nothing to resolve. Listing it
|
|
139
|
+
# unlinked beats dropping a heading out of the contents silently.
|
|
140
|
+
body = (
|
|
141
|
+
f'<a href="#{escape(entry.anchor, quote=True)}">{label}</a>'
|
|
142
|
+
if entry.anchor
|
|
143
|
+
else label
|
|
144
|
+
)
|
|
145
|
+
lines.append(f'<li class="toc-{entry.level}">{body}</li>')
|
|
146
|
+
lines += ["</ol>", "</nav>"]
|
|
147
|
+
return lines
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def render_body(document: Document, highlighter: Highlighter | None = None) -> str:
|
|
151
|
+
"""Render just the token stream, with no page around it.
|
|
152
|
+
|
|
153
|
+
The parser is rebuilt rather than kept on the document because its options
|
|
154
|
+
— ``xhtmlOut``, the ``language-`` class prefix, the highlighter — are part
|
|
155
|
+
of how the tokens are meant to be written out, and they belong with the one
|
|
156
|
+
function that declares them.
|
|
157
|
+
"""
|
|
158
|
+
md = build_parser(_fence_highlighter(highlighter))
|
|
159
|
+
# The parser's renderer is the HTML one, which the protocol the attribute
|
|
160
|
+
# is typed as does not say; the rules table is on the concrete class.
|
|
161
|
+
renderer = cast(RendererHTML, md.renderer)
|
|
162
|
+
# markdown-it types its rules table as holding bound methods, which a
|
|
163
|
+
# replacement rule is not and never was; the table itself takes any
|
|
164
|
+
# callable of the right shape.
|
|
165
|
+
rules: dict[str, Any] = renderer.rules
|
|
166
|
+
rules["ordered_list_open"] = _ordered_list_rule(renderer)
|
|
167
|
+
# The environment the frontmatter and footnote plugins filled during
|
|
168
|
+
# parsing is not needed to render: the only thing the render rules read
|
|
169
|
+
# from it is an optional id prefix for footnote anchors, which a
|
|
170
|
+
# single-document pipeline has no use for.
|
|
171
|
+
return renderer.render(document.tokens, md.options, {}).rstrip("\n")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _fence_highlighter(highlighter: Highlighter | None) -> Highlight | None:
|
|
175
|
+
"""Adapt a highlighter to the three arguments markdown-it calls it with.
|
|
176
|
+
|
|
177
|
+
The third — whatever else was written on the fence line — is ignored:
|
|
178
|
+
markdown-it passes it along for the benefit of highlighters that take
|
|
179
|
+
options there, and nothing in this project reads one.
|
|
180
|
+
"""
|
|
181
|
+
if highlighter is None or not highlighter.enabled:
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
def highlight(code: str, language: str, _attrs: str) -> str | None:
|
|
185
|
+
return highlighter.html(code, language)
|
|
186
|
+
|
|
187
|
+
return highlight
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _ordered_list_rule(renderer: RendererHTML) -> Callable[..., str]:
|
|
191
|
+
"""Make a list start where the author said, rather than always at one.
|
|
192
|
+
|
|
193
|
+
``4. four`` opens a list numbered from four, and markdown-it writes that
|
|
194
|
+
out as ``<ol start="4">`` — which WeasyPrint 69 does not read. Verified
|
|
195
|
+
against WeasyPrint on its own, with none of Amethyst's CSS loaded, so it
|
|
196
|
+
is the renderer and not the stylesheet. Setting the list-item counter
|
|
197
|
+
directly does work, and is what this adds.
|
|
198
|
+
|
|
199
|
+
The attribute is put on for the length of one call and taken off again:
|
|
200
|
+
the token stream belongs to the document, which the Word renderer walks
|
|
201
|
+
afterwards and has no use for a CSS declaration.
|
|
202
|
+
"""
|
|
203
|
+
|
|
204
|
+
def ordered_list_open(
|
|
205
|
+
tokens: list[Token],
|
|
206
|
+
index: int,
|
|
207
|
+
options: OptionsDict,
|
|
208
|
+
env: EnvType,
|
|
209
|
+
) -> str:
|
|
210
|
+
token = tokens[index]
|
|
211
|
+
start = token.attrGet("start")
|
|
212
|
+
if start is None:
|
|
213
|
+
rendered: str = renderer.renderToken(tokens, index, options, env)
|
|
214
|
+
return rendered
|
|
215
|
+
token.attrSet("style", f"counter-reset: list-item {int(start) - 1}")
|
|
216
|
+
try:
|
|
217
|
+
rendered = renderer.renderToken(tokens, index, options, env)
|
|
218
|
+
finally:
|
|
219
|
+
token.attrs.pop("style", None)
|
|
220
|
+
return rendered
|
|
221
|
+
|
|
222
|
+
return ordered_list_open
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def stylesheets(
|
|
226
|
+
document: Document,
|
|
227
|
+
options: RenderOptions,
|
|
228
|
+
highlighter: Highlighter | None = None,
|
|
229
|
+
) -> list[str]:
|
|
230
|
+
"""The stylesheets to inline, in cascade order — last one wins.
|
|
231
|
+
|
|
232
|
+
The theme goes after the structural sheet rather than before it: both
|
|
233
|
+
declare ``:root``, both at the same specificity, so the one that wins is
|
|
234
|
+
simply the one that comes second. The highlighting comes after the theme
|
|
235
|
+
for the same reason: a dark style has to be able to take the code block's
|
|
236
|
+
background off the theme, and it says so with a rule of equal weight.
|
|
237
|
+
"""
|
|
238
|
+
if highlighter is None:
|
|
239
|
+
highlighter = Highlighter(options.highlight_style, warn=options.warn)
|
|
240
|
+
sheets = [
|
|
241
|
+
base_css(),
|
|
242
|
+
root_css(options.theme),
|
|
243
|
+
highlighter.css(),
|
|
244
|
+
page_css(
|
|
245
|
+
options.theme,
|
|
246
|
+
page_numbers=options.page_numbers,
|
|
247
|
+
running_title=document.title,
|
|
248
|
+
running_section=section_level(document),
|
|
249
|
+
front_matter=options.title_page or options.toc,
|
|
250
|
+
title_page=options.title_page,
|
|
251
|
+
),
|
|
252
|
+
]
|
|
253
|
+
if options.extra_css is not None:
|
|
254
|
+
sheets.append(read_css(options.extra_css))
|
|
255
|
+
return sheets
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def read_css(path: Path) -> str:
|
|
259
|
+
"""Read a user's extra stylesheet, reporting a failure as one clear line."""
|
|
260
|
+
try:
|
|
261
|
+
return path.read_text(encoding="utf-8")
|
|
262
|
+
except (OSError, UnicodeDecodeError) as exc:
|
|
263
|
+
detail = getattr(exc, "strerror", None) or "it is not valid UTF-8 text"
|
|
264
|
+
raise RenderError(
|
|
265
|
+
f"Could not read the CSS at {path}: {detail.lower()}."
|
|
266
|
+
) from exc
|