amethyst-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- amethyst/__init__.py +5 -0
- amethyst/__main__.py +6 -0
- amethyst/cli.py +644 -0
- amethyst/config.py +387 -0
- amethyst/document.py +228 -0
- amethyst/errors.py +66 -0
- amethyst/ooxml.py +549 -0
- amethyst/parse/__init__.py +20 -0
- amethyst/parse/assets.py +106 -0
- amethyst/parse/frontmatter.py +62 -0
- amethyst/parse/markdown.py +49 -0
- amethyst/remote.py +236 -0
- amethyst/render/__init__.py +42 -0
- amethyst/render/base.py +85 -0
- amethyst/render/docx.py +1060 -0
- amethyst/render/furniture.py +112 -0
- amethyst/render/highlight.py +315 -0
- amethyst/render/html.py +266 -0
- amethyst/render/pdf.py +219 -0
- amethyst/theme/__init__.py +493 -0
- amethyst/theme/builtin/academic.toml +45 -0
- amethyst/theme/builtin/css/base.css +361 -0
- amethyst/theme/builtin/default.toml +42 -0
- amethyst/theme/builtin/github.toml +44 -0
- amethyst/theme/to_css.py +182 -0
- amethyst/theme/to_docx.py +651 -0
- amethyst_cli-0.1.0.dist-info/METADATA +293 -0
- amethyst_cli-0.1.0.dist-info/RECORD +31 -0
- amethyst_cli-0.1.0.dist-info/WHEEL +4 -0
- amethyst_cli-0.1.0.dist-info/entry_points.txt +2 -0
- amethyst_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
amethyst/ooxml.py
ADDED
|
@@ -0,0 +1,549 @@
|
|
|
1
|
+
"""The OOXML that python-docx has no API for.
|
|
2
|
+
|
|
3
|
+
python-docx covers paragraphs, runs, styles, tables and pictures, which is
|
|
4
|
+
most of the job. What it does not cover is the furniture: hyperlinks,
|
|
5
|
+
bookmarks, shading, borders, field codes, a repeating table header, and a
|
|
6
|
+
numbering instance that starts again at one. Each of those is a handful of
|
|
7
|
+
elements built by hand, and they live here — together, and away from the
|
|
8
|
+
walker — because they are fiddly, individually testable, and will be returned
|
|
9
|
+
to.
|
|
10
|
+
|
|
11
|
+
This module sits under neither pipeline, which is the point. Both the Word
|
|
12
|
+
renderer and the theme compiler need these elements, and a module inside
|
|
13
|
+
``render/`` that ``theme/`` imported would close a circle: importing it runs
|
|
14
|
+
``render/__init__``, which imports the walker, which imports the theme
|
|
15
|
+
compiler again.
|
|
16
|
+
|
|
17
|
+
The one thing that matters everywhere below: Word validates a document against
|
|
18
|
+
the schema when it opens it, and rejects the whole file when a child element
|
|
19
|
+
is in the wrong place. So the *order* of the children of a properties element
|
|
20
|
+
is as load-bearing as their content. Every sequence the format demands is
|
|
21
|
+
written out once in ``CHILD_ORDER``, and every helper inserts through it
|
|
22
|
+
rather than appending and hoping.
|
|
23
|
+
|
|
24
|
+
Nothing here imports anything else from Amethyst. These are facts about the
|
|
25
|
+
file format, not about this program.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import re
|
|
31
|
+
from collections.abc import Iterable, Sequence
|
|
32
|
+
from typing import Any
|
|
33
|
+
|
|
34
|
+
from docx.opc.constants import RELATIONSHIP_TYPE
|
|
35
|
+
from docx.oxml import OxmlElement
|
|
36
|
+
from docx.oxml.ns import qn
|
|
37
|
+
|
|
38
|
+
#: The order the schema demands of the children of each properties element,
|
|
39
|
+
#: for the ones this project writes into. Only the elements before the one
|
|
40
|
+
#: being inserted matter, but the sequences are written out in full: a partial
|
|
41
|
+
#: list is a trap for whoever adds the next helper.
|
|
42
|
+
CHILD_ORDER: dict[str, tuple[str, ...]] = {
|
|
43
|
+
"pPr": (
|
|
44
|
+
"w:pStyle",
|
|
45
|
+
"w:keepNext",
|
|
46
|
+
"w:keepLines",
|
|
47
|
+
"w:pageBreakBefore",
|
|
48
|
+
"w:framePr",
|
|
49
|
+
"w:widowControl",
|
|
50
|
+
"w:numPr",
|
|
51
|
+
"w:suppressLineNumbers",
|
|
52
|
+
"w:pBdr",
|
|
53
|
+
"w:shd",
|
|
54
|
+
"w:tabs",
|
|
55
|
+
"w:suppressAutoHyphens",
|
|
56
|
+
"w:kinsoku",
|
|
57
|
+
"w:wordWrap",
|
|
58
|
+
"w:overflowPunct",
|
|
59
|
+
"w:topLinePunct",
|
|
60
|
+
"w:autoSpaceDE",
|
|
61
|
+
"w:autoSpaceDN",
|
|
62
|
+
"w:bidi",
|
|
63
|
+
"w:adjustRightInd",
|
|
64
|
+
"w:snapToGrid",
|
|
65
|
+
"w:spacing",
|
|
66
|
+
"w:ind",
|
|
67
|
+
"w:contextualSpacing",
|
|
68
|
+
"w:mirrorIndents",
|
|
69
|
+
"w:suppressOverlap",
|
|
70
|
+
"w:jc",
|
|
71
|
+
"w:textDirection",
|
|
72
|
+
"w:textAlignment",
|
|
73
|
+
"w:textboxTightWrap",
|
|
74
|
+
"w:outlineLvl",
|
|
75
|
+
"w:divId",
|
|
76
|
+
"w:cnfStyle",
|
|
77
|
+
"w:rPr",
|
|
78
|
+
"w:sectPr",
|
|
79
|
+
"w:pPrChange",
|
|
80
|
+
),
|
|
81
|
+
"rPr": (
|
|
82
|
+
"w:rStyle",
|
|
83
|
+
"w:rFonts",
|
|
84
|
+
"w:b",
|
|
85
|
+
"w:bCs",
|
|
86
|
+
"w:i",
|
|
87
|
+
"w:iCs",
|
|
88
|
+
"w:caps",
|
|
89
|
+
"w:smallCaps",
|
|
90
|
+
"w:strike",
|
|
91
|
+
"w:dstrike",
|
|
92
|
+
"w:outline",
|
|
93
|
+
"w:shadow",
|
|
94
|
+
"w:emboss",
|
|
95
|
+
"w:imprint",
|
|
96
|
+
"w:noProof",
|
|
97
|
+
"w:snapToGrid",
|
|
98
|
+
"w:vanish",
|
|
99
|
+
"w:webHidden",
|
|
100
|
+
"w:color",
|
|
101
|
+
"w:spacing",
|
|
102
|
+
"w:w",
|
|
103
|
+
"w:kern",
|
|
104
|
+
"w:position",
|
|
105
|
+
"w:sz",
|
|
106
|
+
"w:szCs",
|
|
107
|
+
"w:highlight",
|
|
108
|
+
"w:u",
|
|
109
|
+
"w:effect",
|
|
110
|
+
"w:bdr",
|
|
111
|
+
"w:shd",
|
|
112
|
+
"w:fitText",
|
|
113
|
+
"w:vertAlign",
|
|
114
|
+
"w:rtl",
|
|
115
|
+
"w:cs",
|
|
116
|
+
"w:em",
|
|
117
|
+
"w:lang",
|
|
118
|
+
"w:eastAsianLayout",
|
|
119
|
+
"w:specVanish",
|
|
120
|
+
"w:oMath",
|
|
121
|
+
),
|
|
122
|
+
"tblPr": (
|
|
123
|
+
"w:tblStyle",
|
|
124
|
+
"w:tblpPr",
|
|
125
|
+
"w:tblOverlap",
|
|
126
|
+
"w:bidiVisual",
|
|
127
|
+
"w:tblStyleRowBandSize",
|
|
128
|
+
"w:tblStyleColBandSize",
|
|
129
|
+
"w:tblW",
|
|
130
|
+
"w:jc",
|
|
131
|
+
"w:tblCellSpacing",
|
|
132
|
+
"w:tblInd",
|
|
133
|
+
"w:tblBorders",
|
|
134
|
+
"w:shd",
|
|
135
|
+
"w:tblLayout",
|
|
136
|
+
"w:tblCellMar",
|
|
137
|
+
"w:tblLook",
|
|
138
|
+
"w:tblCaption",
|
|
139
|
+
"w:tblDescription",
|
|
140
|
+
),
|
|
141
|
+
"tcPr": (
|
|
142
|
+
"w:cnfStyle",
|
|
143
|
+
"w:tcW",
|
|
144
|
+
"w:gridSpan",
|
|
145
|
+
"w:hMerge",
|
|
146
|
+
"w:vMerge",
|
|
147
|
+
"w:tcBorders",
|
|
148
|
+
"w:shd",
|
|
149
|
+
"w:noWrap",
|
|
150
|
+
"w:tcMar",
|
|
151
|
+
"w:textDirection",
|
|
152
|
+
"w:tcFitText",
|
|
153
|
+
"w:vAlign",
|
|
154
|
+
"w:hideMark",
|
|
155
|
+
),
|
|
156
|
+
"trPr": (
|
|
157
|
+
"w:cnfStyle",
|
|
158
|
+
"w:divId",
|
|
159
|
+
"w:gridBefore",
|
|
160
|
+
"w:gridAfter",
|
|
161
|
+
"w:wBefore",
|
|
162
|
+
"w:wAfter",
|
|
163
|
+
"w:cantSplit",
|
|
164
|
+
"w:trHeight",
|
|
165
|
+
"w:tblHeader",
|
|
166
|
+
"w:tblCellSpacing",
|
|
167
|
+
"w:jc",
|
|
168
|
+
"w:hidden",
|
|
169
|
+
),
|
|
170
|
+
"pBdr": ("w:top", "w:left", "w:bottom", "w:right", "w:between", "w:bar"),
|
|
171
|
+
"tblBorders": (
|
|
172
|
+
"w:top",
|
|
173
|
+
"w:left",
|
|
174
|
+
"w:bottom",
|
|
175
|
+
"w:right",
|
|
176
|
+
"w:insideH",
|
|
177
|
+
"w:insideV",
|
|
178
|
+
),
|
|
179
|
+
"tcBorders": (
|
|
180
|
+
"w:top",
|
|
181
|
+
"w:left",
|
|
182
|
+
"w:bottom",
|
|
183
|
+
"w:right",
|
|
184
|
+
"w:insideH",
|
|
185
|
+
"w:insideV",
|
|
186
|
+
"w:tl2br",
|
|
187
|
+
"w:tr2bl",
|
|
188
|
+
),
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
#: Which element holds the borders, for each properties element that can have
|
|
192
|
+
#: them. The three are the same shape and spelled three different ways, which
|
|
193
|
+
#: is the sort of thing worth stating once.
|
|
194
|
+
BORDER_CONTAINER = {
|
|
195
|
+
"pPr": "w:pBdr",
|
|
196
|
+
"tblPr": "w:tblBorders",
|
|
197
|
+
"tcPr": "w:tcBorders",
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
#: Border width, in the eighths of a point the format measures it in. Six is
|
|
201
|
+
#: the 0.75pt that a CSS ``1px`` rule comes to on paper.
|
|
202
|
+
HAIRLINE = 6
|
|
203
|
+
|
|
204
|
+
#: What Word allows in a bookmark name: letters, digits and underscores, up to
|
|
205
|
+
#: forty characters, not starting with a digit. Heading anchors are slugs with
|
|
206
|
+
#: hyphens in them, so every one of them has to be translated.
|
|
207
|
+
BOOKMARK_UNSAFE = re.compile(r"[^0-9A-Za-z_]+")
|
|
208
|
+
BOOKMARK_MAX_LENGTH = 40
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
# --- properties -----------------------------------------------------------
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def properties_child(properties: Any, tag: str) -> Any:
|
|
215
|
+
"""Return a child of a properties element, adding it in schema order."""
|
|
216
|
+
existing = properties.find(qn(tag))
|
|
217
|
+
if existing is not None:
|
|
218
|
+
return existing
|
|
219
|
+
element = OxmlElement(tag)
|
|
220
|
+
_insert(properties, element, tag)
|
|
221
|
+
return element
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _insert(parent: Any, element: Any, tag: str) -> None:
|
|
225
|
+
"""Put ``element`` in front of the first sibling that must follow it.
|
|
226
|
+
|
|
227
|
+
Insertion is done here rather than through python-docx's own
|
|
228
|
+
``insert_element_before`` because that method belongs to the element
|
|
229
|
+
classes python-docx registers, and half the elements built in this module
|
|
230
|
+
— ``w:pBdr``, ``w:tblBorders`` — are not among them and come back as plain
|
|
231
|
+
lxml.
|
|
232
|
+
"""
|
|
233
|
+
sequence = CHILD_ORDER[_local_name(parent)]
|
|
234
|
+
following = {qn(name) for name in sequence[sequence.index(tag) + 1 :]}
|
|
235
|
+
for child in parent:
|
|
236
|
+
if child.tag in following:
|
|
237
|
+
child.addprevious(element)
|
|
238
|
+
return
|
|
239
|
+
parent.append(element)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def shade(properties: Any, fill: str) -> None:
|
|
243
|
+
"""Fill the background of whatever ``properties`` describes.
|
|
244
|
+
|
|
245
|
+
Serves a paragraph, a run, a table and a table cell alike: ``w:shd`` is
|
|
246
|
+
spelled the same in all four, and only its position among its siblings
|
|
247
|
+
changes.
|
|
248
|
+
"""
|
|
249
|
+
element = properties_child(properties, "w:shd")
|
|
250
|
+
element.set(qn("w:val"), "clear")
|
|
251
|
+
element.set(qn("w:color"), "auto")
|
|
252
|
+
element.set(qn("w:fill"), _hex(fill))
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def set_borders(
|
|
256
|
+
properties: Any,
|
|
257
|
+
edges: Iterable[str],
|
|
258
|
+
*,
|
|
259
|
+
color: str,
|
|
260
|
+
size: int = HAIRLINE,
|
|
261
|
+
space: int = 0,
|
|
262
|
+
style: str = "single",
|
|
263
|
+
) -> None:
|
|
264
|
+
"""Draw ``edges`` — "top", "left", "bottom", "right", "insideH", … .
|
|
265
|
+
|
|
266
|
+
``size`` is in eighths of a point and ``space`` in points, because that is
|
|
267
|
+
what the format measures them in and translating here would only hide it.
|
|
268
|
+
"""
|
|
269
|
+
container = properties_child(properties, BORDER_CONTAINER[_local_name(properties)])
|
|
270
|
+
for edge in edges:
|
|
271
|
+
tag = f"w:{edge}"
|
|
272
|
+
element = OxmlElement(tag)
|
|
273
|
+
element.set(qn("w:val"), style)
|
|
274
|
+
element.set(qn("w:sz"), str(size))
|
|
275
|
+
element.set(qn("w:space"), str(space))
|
|
276
|
+
element.set(qn("w:color"), _hex(color))
|
|
277
|
+
_insert(container, element, tag)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def clear_properties_child(properties: Any, tag: str) -> None:
|
|
281
|
+
"""Take a child off a properties element, if it is there at all.
|
|
282
|
+
|
|
283
|
+
The counterpart of :func:`properties_child`, and needed because Word's
|
|
284
|
+
built-in styles arrive carrying things a theme has to undo rather than
|
|
285
|
+
override — a rule drawn under ``Title`` in an accent colour that belongs
|
|
286
|
+
to no theme here, or the stray numbering reference on ``Subtitle``.
|
|
287
|
+
"""
|
|
288
|
+
element = properties.find(qn(tag))
|
|
289
|
+
if element is not None:
|
|
290
|
+
properties.remove(element)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def repeat_as_header(row: Any) -> None:
|
|
294
|
+
"""Mark a table row as the header, so it repeats on every page.
|
|
295
|
+
|
|
296
|
+
The counterpart of ``thead { display: table-header-group }`` in the PDF
|
|
297
|
+
stylesheet: a table split across a page break is only readable because its
|
|
298
|
+
column headings come back at the top of the next one.
|
|
299
|
+
"""
|
|
300
|
+
properties_child(row._tr.get_or_add_trPr(), "w:tblHeader")
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
# --- links and bookmarks --------------------------------------------------
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def bookmark_name(anchor: str) -> str:
|
|
307
|
+
"""Translate a heading's HTML id into a name Word will accept.
|
|
308
|
+
|
|
309
|
+
Word's rules are much tighter than HTML's — no hyphens, no leading digit,
|
|
310
|
+
forty characters — so this is lossy by necessity. It is deterministic,
|
|
311
|
+
which is the property that matters: the heading and the link that points
|
|
312
|
+
at it are translated by the same function and so agree.
|
|
313
|
+
"""
|
|
314
|
+
cleaned = BOOKMARK_UNSAFE.sub("_", anchor).strip("_")
|
|
315
|
+
if not cleaned or cleaned[0].isdigit():
|
|
316
|
+
cleaned = f"_{cleaned}"
|
|
317
|
+
return cleaned[:BOOKMARK_MAX_LENGTH]
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def bookmark(paragraph: Any, name: str, identifier: int) -> None:
|
|
321
|
+
"""Mark a paragraph as the destination of an internal link."""
|
|
322
|
+
start = OxmlElement("w:bookmarkStart")
|
|
323
|
+
start.set(qn("w:id"), str(identifier))
|
|
324
|
+
start.set(qn("w:name"), name)
|
|
325
|
+
end = OxmlElement("w:bookmarkEnd")
|
|
326
|
+
end.set(qn("w:id"), str(identifier))
|
|
327
|
+
|
|
328
|
+
paragraph_element = paragraph._p
|
|
329
|
+
properties = paragraph_element.find(qn("w:pPr"))
|
|
330
|
+
if properties is None:
|
|
331
|
+
paragraph_element.insert(0, start)
|
|
332
|
+
else:
|
|
333
|
+
properties.addnext(start)
|
|
334
|
+
paragraph_element.append(end)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def link(
|
|
338
|
+
paragraph: Any,
|
|
339
|
+
elements: Sequence[Any],
|
|
340
|
+
*,
|
|
341
|
+
url: str | None = None,
|
|
342
|
+
anchor: str | None = None,
|
|
343
|
+
) -> None:
|
|
344
|
+
"""Wrap already-written runs in a hyperlink, external or internal.
|
|
345
|
+
|
|
346
|
+
The runs are built first and moved in afterwards rather than the other way
|
|
347
|
+
round, so the code that renders a link's text is the same code that
|
|
348
|
+
renders every other run — a link's label is Markdown like any other, and
|
|
349
|
+
can hold bold, code and an image.
|
|
350
|
+
"""
|
|
351
|
+
if not elements:
|
|
352
|
+
return
|
|
353
|
+
element = OxmlElement("w:hyperlink")
|
|
354
|
+
if url is not None:
|
|
355
|
+
relationship = paragraph.part.relate_to(
|
|
356
|
+
url, RELATIONSHIP_TYPE.HYPERLINK, is_external=True
|
|
357
|
+
)
|
|
358
|
+
element.set(qn("r:id"), relationship)
|
|
359
|
+
if anchor is not None:
|
|
360
|
+
element.set(qn("w:anchor"), anchor)
|
|
361
|
+
elements[0].addprevious(element)
|
|
362
|
+
for run in elements:
|
|
363
|
+
element.append(run)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
# --- fields ---------------------------------------------------------------
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def field(
|
|
370
|
+
paragraph: Any, instruction: str, placeholder: str = "", *, dirty: bool = False
|
|
371
|
+
) -> None:
|
|
372
|
+
"""Append a field — ``PAGE``, ``PAGEREF``, ``STYLEREF`` — to a paragraph.
|
|
373
|
+
|
|
374
|
+
A field is not a value but an instruction that Word evaluates when it
|
|
375
|
+
opens the document, which is the only way to write something that has to
|
|
376
|
+
know a page number the writer cannot know. The placeholder is what a
|
|
377
|
+
reader that does not evaluate fields shows instead.
|
|
378
|
+
|
|
379
|
+
``dirty`` asks Word to evaluate the field as soon as it opens the file
|
|
380
|
+
rather than showing the placeholder until someone presses F9. It is what a
|
|
381
|
+
field with no useful placeholder — a page number nothing here can compute
|
|
382
|
+
— needs to arrive filled in.
|
|
383
|
+
"""
|
|
384
|
+
for child in (
|
|
385
|
+
*field_start(instruction, dirty=dirty),
|
|
386
|
+
_text_run(placeholder),
|
|
387
|
+
field_end(),
|
|
388
|
+
):
|
|
389
|
+
paragraph._p.append(child)
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def field_start(instruction: str, *, dirty: bool = False) -> tuple[Any, ...]:
|
|
393
|
+
"""The runs that open a field, up to where its result begins.
|
|
394
|
+
|
|
395
|
+
Separate from :func:`field` because a field's result is not always one
|
|
396
|
+
run in one paragraph: a table of contents is a run of paragraphs, and the
|
|
397
|
+
only way to write one is to open the field in the first and close it in
|
|
398
|
+
the last.
|
|
399
|
+
"""
|
|
400
|
+
return (
|
|
401
|
+
_field_char("begin", dirty=dirty),
|
|
402
|
+
_instruction(instruction),
|
|
403
|
+
_field_char("separate"),
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def field_end() -> Any:
|
|
408
|
+
"""The run that closes a field opened by :func:`field_start`."""
|
|
409
|
+
return _field_char("end")
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _field_char(kind: str, *, dirty: bool = False) -> Any:
|
|
413
|
+
"""One of a field's three markers: ``begin``, ``separate`` or ``end``.
|
|
414
|
+
|
|
415
|
+
``dirty`` is what makes Word evaluate the field when the file opens rather
|
|
416
|
+
than showing whatever result was written beside it — needed by ``PAGEREF``
|
|
417
|
+
and ``TOC``, and not by ``PAGE`` or ``STYLEREF``, which Word computes while
|
|
418
|
+
it lays the page out. It is not free: a document containing any dirty field
|
|
419
|
+
greets the reader with an update-fields dialog.
|
|
420
|
+
"""
|
|
421
|
+
run = OxmlElement("w:r")
|
|
422
|
+
char = OxmlElement("w:fldChar")
|
|
423
|
+
char.set(qn("w:fldCharType"), kind)
|
|
424
|
+
if dirty:
|
|
425
|
+
char.set(qn("w:dirty"), "true")
|
|
426
|
+
run.append(char)
|
|
427
|
+
return run
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _instruction(instruction: str) -> Any:
|
|
431
|
+
"""The run carrying a field's instruction text, spaces and all."""
|
|
432
|
+
run = OxmlElement("w:r")
|
|
433
|
+
text = OxmlElement("w:instrText")
|
|
434
|
+
# Without this the leading and trailing spaces a field instruction needs
|
|
435
|
+
# are collapsed away, and Word reads a different instruction than the one
|
|
436
|
+
# that was written.
|
|
437
|
+
text.set(qn("xml:space"), "preserve")
|
|
438
|
+
text.text = f" {instruction} "
|
|
439
|
+
run.append(text)
|
|
440
|
+
return run
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _text_run(content: str) -> Any:
|
|
444
|
+
"""A plain run of text that keeps the whitespace it was given."""
|
|
445
|
+
run = OxmlElement("w:r")
|
|
446
|
+
text = OxmlElement("w:t")
|
|
447
|
+
text.set(qn("xml:space"), "preserve")
|
|
448
|
+
text.text = content
|
|
449
|
+
run.append(text)
|
|
450
|
+
return run
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
# --- numbering ------------------------------------------------------------
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def numbering_instance(document: Any, style_name: str, *, start: int = 1) -> int:
|
|
457
|
+
"""Give one list its own counter, and return the id to point at it with.
|
|
458
|
+
|
|
459
|
+
Word's numbering styles carry a counter each, not a counter per list, so
|
|
460
|
+
every ordered list in a document that uses ``List Number`` shares one — and
|
|
461
|
+
the second list in a document carries on from where the first stopped
|
|
462
|
+
instead of starting again at one. The cure is an instance of its own per
|
|
463
|
+
list, pointing at the same shape and overriding where it starts.
|
|
464
|
+
"""
|
|
465
|
+
numbering = document.part.numbering_part.element
|
|
466
|
+
abstract_id = _abstract_num_id(document, style_name)
|
|
467
|
+
|
|
468
|
+
identifier = 1 + max(
|
|
469
|
+
(
|
|
470
|
+
int(existing.get(qn("w:numId")) or 0)
|
|
471
|
+
for existing in numbering.findall(qn("w:num"))
|
|
472
|
+
),
|
|
473
|
+
default=0,
|
|
474
|
+
)
|
|
475
|
+
instance = OxmlElement("w:num")
|
|
476
|
+
instance.set(qn("w:numId"), str(identifier))
|
|
477
|
+
reference = OxmlElement("w:abstractNumId")
|
|
478
|
+
reference.set(qn("w:val"), str(abstract_id))
|
|
479
|
+
instance.append(reference)
|
|
480
|
+
|
|
481
|
+
override = OxmlElement("w:lvlOverride")
|
|
482
|
+
override.set(qn("w:ilvl"), "0")
|
|
483
|
+
first = OxmlElement("w:startOverride")
|
|
484
|
+
first.set(qn("w:val"), str(start))
|
|
485
|
+
override.append(first)
|
|
486
|
+
instance.append(override)
|
|
487
|
+
|
|
488
|
+
# Every w:num follows every w:abstractNum, so the end of the part is the
|
|
489
|
+
# one position that is always right.
|
|
490
|
+
numbering.append(instance)
|
|
491
|
+
return identifier
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def set_numbering(paragraph: Any, identifier: int, level: int = 0) -> None:
|
|
495
|
+
"""Point a paragraph at a numbering instance, overriding its style's."""
|
|
496
|
+
properties = paragraph._p.get_or_add_pPr().get_or_add_numPr()
|
|
497
|
+
properties.get_or_add_ilvl().val = level
|
|
498
|
+
properties.get_or_add_numId().val = identifier
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def _abstract_num_id(document: Any, style_name: str) -> int:
|
|
502
|
+
"""The numbering shape a built-in list style draws its bullets from."""
|
|
503
|
+
style = document.styles[style_name]
|
|
504
|
+
reference = style.element.find(f"{qn('w:pPr')}/{qn('w:numPr')}/{qn('w:numId')}")
|
|
505
|
+
if reference is None:
|
|
506
|
+
raise KeyError(f"style {style_name!r} carries no numbering")
|
|
507
|
+
wanted = reference.get(qn("w:val"))
|
|
508
|
+
numbering = document.part.numbering_part.element
|
|
509
|
+
for instance in numbering.findall(qn("w:num")):
|
|
510
|
+
if instance.get(qn("w:numId")) == wanted:
|
|
511
|
+
abstract = instance.find(qn("w:abstractNumId"))
|
|
512
|
+
if abstract is not None:
|
|
513
|
+
return int(abstract.get(qn("w:val")))
|
|
514
|
+
raise KeyError(
|
|
515
|
+
f"style {style_name!r} points at numbering {wanted!r}, which is not there"
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
# --- shared ---------------------------------------------------------------
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _local_name(element: Any) -> str:
|
|
523
|
+
"""The tag without its namespace: ``{...}pPr`` is ``pPr``."""
|
|
524
|
+
return str(element.tag).rpartition("}")[2]
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _hex(color: str) -> str:
|
|
528
|
+
"""A theme colour as OOXML writes it: six digits, no leading hash."""
|
|
529
|
+
return color.lstrip("#").upper()
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
__all__ = [
|
|
533
|
+
"BOOKMARK_MAX_LENGTH",
|
|
534
|
+
"CHILD_ORDER",
|
|
535
|
+
"HAIRLINE",
|
|
536
|
+
"bookmark",
|
|
537
|
+
"bookmark_name",
|
|
538
|
+
"clear_properties_child",
|
|
539
|
+
"field",
|
|
540
|
+
"field_end",
|
|
541
|
+
"field_start",
|
|
542
|
+
"link",
|
|
543
|
+
"numbering_instance",
|
|
544
|
+
"properties_child",
|
|
545
|
+
"repeat_as_header",
|
|
546
|
+
"set_borders",
|
|
547
|
+
"set_numbering",
|
|
548
|
+
"shade",
|
|
549
|
+
]
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""The parse layer: Markdown text in, a token stream and metadata out.
|
|
2
|
+
|
|
3
|
+
Nothing here knows about PDF or DOCX. :mod:`amethyst.document` assembles these
|
|
4
|
+
pieces into the ``Document`` the renderers are handed.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from amethyst.parse.assets import Asset, AssetKind, resolve_assets
|
|
10
|
+
from amethyst.parse.frontmatter import parse_frontmatter, split_frontmatter
|
|
11
|
+
from amethyst.parse.markdown import build_parser
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"Asset",
|
|
15
|
+
"AssetKind",
|
|
16
|
+
"build_parser",
|
|
17
|
+
"parse_frontmatter",
|
|
18
|
+
"resolve_assets",
|
|
19
|
+
"split_frontmatter",
|
|
20
|
+
]
|
amethyst/parse/assets.py
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Resolution of the files a document points at.
|
|
2
|
+
|
|
3
|
+
Markdown says ```` and means "next to this file". Both
|
|
4
|
+
renderers need that turned into something they can open — WeasyPrint fetches a
|
|
5
|
+
URL, python-docx opens a path — and resolving it once, here, is what stops the
|
|
6
|
+
two from disagreeing about where the file was.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from enum import Enum
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from urllib.parse import unquote, urlsplit
|
|
15
|
+
|
|
16
|
+
from markdown_it.token import Token
|
|
17
|
+
|
|
18
|
+
#: Schemes that name something to be fetched over the network. Anything else
|
|
19
|
+
#: with a scheme (``data:``, ``mailto:``, ``file:``) needs no resolving and is
|
|
20
|
+
#: passed through to the renderer exactly as written.
|
|
21
|
+
REMOTE_SCHEMES = frozenset({"http", "https"})
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class AssetKind(str, Enum):
|
|
25
|
+
"""What the reference was written as. The value doubles as its noun."""
|
|
26
|
+
|
|
27
|
+
image = "image"
|
|
28
|
+
link = "link"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Asset:
|
|
33
|
+
"""One outward reference from the document, and where it landed."""
|
|
34
|
+
|
|
35
|
+
kind: AssetKind
|
|
36
|
+
#: The target exactly as written in the Markdown, for error messages.
|
|
37
|
+
reference: str
|
|
38
|
+
#: 1-based line in the source file, or ``None`` if the token carried no map.
|
|
39
|
+
line: int | None = None
|
|
40
|
+
#: The local file the reference resolves to, or ``None`` if it is not one.
|
|
41
|
+
path: Path | None = None
|
|
42
|
+
is_remote: bool = False
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def is_missing(self) -> bool:
|
|
46
|
+
"""True for a local reference with no file at the resolved path."""
|
|
47
|
+
return self.path is not None and not self.path.is_file()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def resolve_assets(tokens: list[Token], base_dir: Path) -> list[Asset]:
|
|
51
|
+
"""Resolve local references against ``base_dir`` and rewrite image sources.
|
|
52
|
+
|
|
53
|
+
An image whose source names a local file that exists is rewritten in place
|
|
54
|
+
to an absolute path, which is what both renderers want. Everything else —
|
|
55
|
+
remote URLs, data URIs, files that are not there — is left exactly as the
|
|
56
|
+
author typed it, so the eventual error names their reference rather than
|
|
57
|
+
one this function invented.
|
|
58
|
+
|
|
59
|
+
Returns every reference worth knowing about later: images of all kinds, and
|
|
60
|
+
links that point at a local file. Bare fragments (``#section``) and
|
|
61
|
+
``mailto:`` links resolve to nothing and are left out.
|
|
62
|
+
"""
|
|
63
|
+
assets: list[Asset] = []
|
|
64
|
+
for token in tokens:
|
|
65
|
+
if token.type != "inline" or not token.children:
|
|
66
|
+
continue
|
|
67
|
+
line = token.map[0] + 1 if token.map else None
|
|
68
|
+
for child in token.children:
|
|
69
|
+
if child.type == "image":
|
|
70
|
+
asset = _resolve(AssetKind.image, child.attrGet("src"), base_dir, line)
|
|
71
|
+
if asset is None:
|
|
72
|
+
continue
|
|
73
|
+
assets.append(asset)
|
|
74
|
+
if asset.path is not None and not asset.is_missing:
|
|
75
|
+
child.attrSet("src", str(asset.path))
|
|
76
|
+
elif child.type == "link_open":
|
|
77
|
+
asset = _resolve(AssetKind.link, child.attrGet("href"), base_dir, line)
|
|
78
|
+
# Links are recorded, never rewritten: an absolute filesystem
|
|
79
|
+
# path in an href would be a worse link than the relative one.
|
|
80
|
+
if asset is not None and asset.path is not None:
|
|
81
|
+
assets.append(asset)
|
|
82
|
+
return assets
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _resolve(
|
|
86
|
+
kind: AssetKind, reference: object, base_dir: Path, line: int | None
|
|
87
|
+
) -> Asset | None:
|
|
88
|
+
"""Classify one reference, and locate it if it names a local file."""
|
|
89
|
+
if not isinstance(reference, str) or not reference:
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
parts = urlsplit(reference)
|
|
93
|
+
if parts.scheme or parts.netloc:
|
|
94
|
+
if parts.scheme not in REMOTE_SCHEMES:
|
|
95
|
+
return None
|
|
96
|
+
return Asset(kind, reference, line, is_remote=True)
|
|
97
|
+
|
|
98
|
+
# Strip the fragment and query the URL syntax allows, then undo the
|
|
99
|
+
# percent-encoding: `my%20image.png` on the page is `my image.png` on disk.
|
|
100
|
+
target = unquote(parts.path)
|
|
101
|
+
if not target:
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
candidate = Path(target)
|
|
105
|
+
resolved = candidate if candidate.is_absolute() else base_dir / candidate
|
|
106
|
+
return Asset(kind, reference, line, path=resolved.resolve())
|