amethyst-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- amethyst/__init__.py +5 -0
- amethyst/__main__.py +6 -0
- amethyst/cli.py +644 -0
- amethyst/config.py +387 -0
- amethyst/document.py +228 -0
- amethyst/errors.py +66 -0
- amethyst/ooxml.py +549 -0
- amethyst/parse/__init__.py +20 -0
- amethyst/parse/assets.py +106 -0
- amethyst/parse/frontmatter.py +62 -0
- amethyst/parse/markdown.py +49 -0
- amethyst/remote.py +236 -0
- amethyst/render/__init__.py +42 -0
- amethyst/render/base.py +85 -0
- amethyst/render/docx.py +1060 -0
- amethyst/render/furniture.py +112 -0
- amethyst/render/highlight.py +315 -0
- amethyst/render/html.py +266 -0
- amethyst/render/pdf.py +219 -0
- amethyst/theme/__init__.py +493 -0
- amethyst/theme/builtin/academic.toml +45 -0
- amethyst/theme/builtin/css/base.css +361 -0
- amethyst/theme/builtin/default.toml +42 -0
- amethyst/theme/builtin/github.toml +44 -0
- amethyst/theme/to_css.py +182 -0
- amethyst/theme/to_docx.py +651 -0
- amethyst_cli-0.1.0.dist-info/METADATA +293 -0
- amethyst_cli-0.1.0.dist-info/RECORD +31 -0
- amethyst_cli-0.1.0.dist-info/WHEEL +4 -0
- amethyst_cli-0.1.0.dist-info/entry_points.txt +2 -0
- amethyst_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
amethyst/render/docx.py
ADDED
|
@@ -0,0 +1,1060 @@
|
|
|
1
|
+
"""The token stream, walked into a Word document.
|
|
2
|
+
|
|
3
|
+
Word has no equivalent of CSS paged media, so there is nothing here to hand
|
|
4
|
+
the job to: a DOCX file is a flat sequence of styled paragraphs and runs, and
|
|
5
|
+
this module is the walk that produces it. That is why the two pipelines are
|
|
6
|
+
separate. What keeps their output the same document is the theme — every style
|
|
7
|
+
this module names was defined by :mod:`amethyst.theme.to_docx` from the same
|
|
8
|
+
declaration the stylesheet was generated from, and nothing below chooses a
|
|
9
|
+
font, a size or a colour.
|
|
10
|
+
|
|
11
|
+
Three things about the walk itself.
|
|
12
|
+
|
|
13
|
+
markdown-it's stream is flat, with ``_open`` and ``_close`` tokens marking
|
|
14
|
+
nesting, so containers are handled by finding the matching close and recursing
|
|
15
|
+
over the span between. The alternative — a state machine over a flat loop —
|
|
16
|
+
is the same program with the structure hidden.
|
|
17
|
+
|
|
18
|
+
Where a paragraph goes is context, not content. A paragraph inside a list item
|
|
19
|
+
carries the item's bullet; inside a blockquote it is indented and set in the
|
|
20
|
+
quoted style; inside a footnote it is small and muted. All of that is decided
|
|
21
|
+
in one place, :meth:`_Builder._paragraph`, so no handler has to know what it
|
|
22
|
+
is nested inside.
|
|
23
|
+
|
|
24
|
+
And several things Markdown says have no Word equivalent at all. Raw HTML is
|
|
25
|
+
skipped with a warning naming its line, footnotes become an endnote-style list
|
|
26
|
+
at the end rather than real Word footnotes, and a task list gets a printed
|
|
27
|
+
checkbox rather than a real one. Each is a deliberate approximation from the
|
|
28
|
+
feature matrix, not an oversight.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import io
|
|
34
|
+
import re
|
|
35
|
+
from dataclasses import dataclass, replace
|
|
36
|
+
from datetime import datetime, time, timezone
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Any
|
|
39
|
+
from urllib.parse import unquote, urlsplit
|
|
40
|
+
|
|
41
|
+
from docx import Document as new_docx
|
|
42
|
+
from docx.enum.section import WD_SECTION
|
|
43
|
+
from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_TAB_ALIGNMENT
|
|
44
|
+
from docx.image.exceptions import (
|
|
45
|
+
InvalidImageStreamError,
|
|
46
|
+
UnexpectedEndOfFileError,
|
|
47
|
+
UnrecognizedImageError,
|
|
48
|
+
)
|
|
49
|
+
from docx.oxml.ns import qn
|
|
50
|
+
from docx.shared import Emu, Length, Pt
|
|
51
|
+
from markdown_it.token import Token
|
|
52
|
+
|
|
53
|
+
from amethyst.document import Document
|
|
54
|
+
from amethyst.errors import RenderError
|
|
55
|
+
from amethyst.ooxml import (
|
|
56
|
+
bookmark,
|
|
57
|
+
bookmark_name,
|
|
58
|
+
field,
|
|
59
|
+
field_end,
|
|
60
|
+
field_start,
|
|
61
|
+
link,
|
|
62
|
+
numbering_instance,
|
|
63
|
+
repeat_as_header,
|
|
64
|
+
set_borders,
|
|
65
|
+
set_numbering,
|
|
66
|
+
shade,
|
|
67
|
+
)
|
|
68
|
+
from amethyst.parse.assets import REMOTE_SCHEMES
|
|
69
|
+
from amethyst.render.base import RenderOptions, RenderResult
|
|
70
|
+
from amethyst.render.furniture import (
|
|
71
|
+
CONTENTS_HEADING,
|
|
72
|
+
contents,
|
|
73
|
+
cover,
|
|
74
|
+
outline_depth,
|
|
75
|
+
section_level,
|
|
76
|
+
)
|
|
77
|
+
from amethyst.render.highlight import Highlighter, Span
|
|
78
|
+
from amethyst.theme.to_docx import (
|
|
79
|
+
BODY_STYLE,
|
|
80
|
+
BULLET_STYLES,
|
|
81
|
+
CODE_BOX_PADDING,
|
|
82
|
+
CODE_BOX_SIDES,
|
|
83
|
+
CODE_INLINE_STYLE,
|
|
84
|
+
CODE_STYLE,
|
|
85
|
+
COVER_DATE_STYLE,
|
|
86
|
+
COVER_SPACE_AFTER,
|
|
87
|
+
FOOTER_STYLE,
|
|
88
|
+
FOOTNOTE_STYLE,
|
|
89
|
+
HEADER_STYLE,
|
|
90
|
+
HEADING_STYLES,
|
|
91
|
+
LINK_STYLE,
|
|
92
|
+
LIST_INDENT_STEP,
|
|
93
|
+
NUMBER_STYLES,
|
|
94
|
+
QUOTE_STYLE,
|
|
95
|
+
SUBTITLE_STYLE,
|
|
96
|
+
TABLE_STYLE,
|
|
97
|
+
TABLE_TEXT_STYLE,
|
|
98
|
+
TITLE_PAGE_OFFSET,
|
|
99
|
+
TITLE_STYLE,
|
|
100
|
+
TOC_HEADING_STYLE,
|
|
101
|
+
TOC_STYLES,
|
|
102
|
+
apply_page,
|
|
103
|
+
apply_theme,
|
|
104
|
+
quote_indent,
|
|
105
|
+
rgb,
|
|
106
|
+
text_width,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
#: python-docx's three ways of saying "that is not a picture I can embed".
|
|
110
|
+
#: They share no base class, so all three have to be named — and all three are
|
|
111
|
+
#: reachable from a document: a file that is not an image, one that is an image
|
|
112
|
+
#: format Word has no part type for, and one that was cut off. The last is why
|
|
113
|
+
#: this matters more since images are downloaded: a truncated file is what a
|
|
114
|
+
#: dropped connection leaves behind.
|
|
115
|
+
UNUSABLE_IMAGE = (
|
|
116
|
+
InvalidImageStreamError,
|
|
117
|
+
UnexpectedEndOfFileError,
|
|
118
|
+
UnrecognizedImageError,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
#: What a task list item is printed as. Word has real checkbox controls, but
|
|
122
|
+
#: they are form fields tied to a content control, and a document that has to
|
|
123
|
+
#: be filled in is not what a converted Markdown list is.
|
|
124
|
+
CHECKED = "☑"
|
|
125
|
+
UNCHECKED = "☐"
|
|
126
|
+
|
|
127
|
+
#: The instructions Word evaluates for itself: the number of the page a field
|
|
128
|
+
#: lands on, the page a bookmark is on, the text of the nearest heading of one
|
|
129
|
+
#: style — Word's answer to a CSS named string — and the table of contents
|
|
130
|
+
#: itself, built from heading levels 1 to N.
|
|
131
|
+
#:
|
|
132
|
+
#: The first and third are worked out while Word lays the pages out and need
|
|
133
|
+
#: nothing asked of the reader. The other two are not, and are marked dirty so
|
|
134
|
+
#: that Word fills them in on open — which is what raises its "update the
|
|
135
|
+
#: fields in this document?" prompt, and why only ``--toc`` raises it.
|
|
136
|
+
PAGE_FIELD = "PAGE"
|
|
137
|
+
PAGE_REFERENCE_FIELD = "PAGEREF {name} \\h"
|
|
138
|
+
SECTION_FIELD = 'STYLEREF "{style}" \\* MERGEFORMAT'
|
|
139
|
+
CONTENTS_FIELD = 'TOC \\o "1-{depth}" \\h \\z \\u'
|
|
140
|
+
|
|
141
|
+
#: The air around a horizontal rule, and above the line that opens the
|
|
142
|
+
#: footnotes, as multiples of the body size. Both mirror the margins the
|
|
143
|
+
#: structural stylesheet gives the same two elements, halved for the rule
|
|
144
|
+
#: because CSS collapses a margin against its neighbour's and Word adds them.
|
|
145
|
+
RULE_GAP = 0.8
|
|
146
|
+
FOOTNOTES_GAP = 2.5
|
|
147
|
+
|
|
148
|
+
#: Cell alignment as markdown-it writes it: an inline style on the cell.
|
|
149
|
+
ALIGNMENTS = {
|
|
150
|
+
"text-align:left": WD_ALIGN_PARAGRAPH.LEFT,
|
|
151
|
+
"text-align:center": WD_ALIGN_PARAGRAPH.CENTER,
|
|
152
|
+
"text-align:right": WD_ALIGN_PARAGRAPH.RIGHT,
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
#: An HTML block that is nothing but a comment. It was never going to be
|
|
156
|
+
#: visible in either format, so warning that it was skipped is noise.
|
|
157
|
+
HTML_COMMENT = re.compile(r"\A\s*(?:<!--.*?-->\s*)+\Z", re.DOTALL)
|
|
158
|
+
|
|
159
|
+
#: The checkbox the tasklist plugin emits. It arrives as raw HTML inside the
|
|
160
|
+
#: item's inline token rather than as a token of its own, so without this the
|
|
161
|
+
#: rule that skips raw HTML would quietly eat every checkbox in the document.
|
|
162
|
+
TASK_CHECKBOX = re.compile(r"<input[^>]*\bclass=\"task-list-item-checkbox\"", re.I)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def render_docx(document: Document, options: RenderOptions) -> RenderResult:
|
|
166
|
+
"""Convert a document to the bytes of a Word file."""
|
|
167
|
+
return _Builder(document, options).build()
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
@dataclass(frozen=True)
|
|
171
|
+
class _Run:
|
|
172
|
+
"""The character formatting in force at one point in an inline walk."""
|
|
173
|
+
|
|
174
|
+
bold: bool = False
|
|
175
|
+
italic: bool = False
|
|
176
|
+
strike: bool = False
|
|
177
|
+
code: bool = False
|
|
178
|
+
link: bool = False
|
|
179
|
+
superscript: bool = False
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
#: No emphasis, no link, no code: what an inline walk starts from unless the
|
|
183
|
+
#: block it sits in says otherwise.
|
|
184
|
+
_PLAIN = _Run()
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@dataclass(frozen=True)
|
|
188
|
+
class _ListLevel:
|
|
189
|
+
"""One open list, and how the items inside it are marked and indented."""
|
|
190
|
+
|
|
191
|
+
style: str
|
|
192
|
+
#: The list's own counter, or ``None`` for a list that is not numbered.
|
|
193
|
+
number: int | None
|
|
194
|
+
indent: Length
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class _Builder:
|
|
198
|
+
"""One conversion. Not reused: the state below belongs to one document."""
|
|
199
|
+
|
|
200
|
+
def __init__(self, document: Document, options: RenderOptions) -> None:
|
|
201
|
+
"""Start an empty document with the theme already compiled into it."""
|
|
202
|
+
self._document = document
|
|
203
|
+
self._options = options
|
|
204
|
+
self._theme = options.theme
|
|
205
|
+
self._warn = options.warn
|
|
206
|
+
self._docx = new_docx()
|
|
207
|
+
self._highlighter = Highlighter(options.highlight_style, warn=self._warn)
|
|
208
|
+
apply_theme(self._docx, self._theme, warn=self._warn)
|
|
209
|
+
|
|
210
|
+
self._lists: list[_ListLevel] = []
|
|
211
|
+
self._pending: _ListLevel | None = None
|
|
212
|
+
self._quotes = 0
|
|
213
|
+
self._footnotes = 0
|
|
214
|
+
self._prefix: str | None = None
|
|
215
|
+
self._bookmarks = 0
|
|
216
|
+
self._after_table = False
|
|
217
|
+
self._extra_indent: Length | None = None
|
|
218
|
+
self._warned_html: set[int | None] = set()
|
|
219
|
+
|
|
220
|
+
# --- the document ------------------------------------------------------
|
|
221
|
+
|
|
222
|
+
def build(self) -> RenderResult:
|
|
223
|
+
"""Walk the whole document and return the file's bytes."""
|
|
224
|
+
self._properties()
|
|
225
|
+
if self._front_matter():
|
|
226
|
+
# The front matter becomes a section of its own so that it can
|
|
227
|
+
# carry different furniture from the body — no running head, and
|
|
228
|
+
# no page number on a cover. It is also what starts the document
|
|
229
|
+
# proper on a fresh page, which is the break the stylesheet gets
|
|
230
|
+
# from `break-after: page`.
|
|
231
|
+
self._docx.add_section(WD_SECTION.NEW_PAGE)
|
|
232
|
+
apply_page(self._docx.sections[-1], self._theme, warn=self._warn)
|
|
233
|
+
self._blocks(self._document.tokens, 0, len(self._document.tokens))
|
|
234
|
+
self._furniture()
|
|
235
|
+
return RenderResult(data=self._save(), pages=None)
|
|
236
|
+
|
|
237
|
+
def _properties(self) -> None:
|
|
238
|
+
"""Say what the document says about itself, and nothing else.
|
|
239
|
+
|
|
240
|
+
python-docx builds every document from a template whose properties
|
|
241
|
+
claim it was written by "python-docx" in 2013, and Word shows those in
|
|
242
|
+
the file's info pane. They are replaced by the frontmatter — the same
|
|
243
|
+
four fields the PDF carries as its metadata — so that a reader who
|
|
244
|
+
opens the information pane sees the same thing in either format.
|
|
245
|
+
"""
|
|
246
|
+
properties = self._docx.core_properties
|
|
247
|
+
properties.author = self._document.author or ""
|
|
248
|
+
properties.title = self._document.title or ""
|
|
249
|
+
properties.subject = self._document.subtitle or ""
|
|
250
|
+
properties.keywords = self._document.keywords or ""
|
|
251
|
+
properties.last_modified_by = ""
|
|
252
|
+
properties.comments = ""
|
|
253
|
+
properties.revision = 1
|
|
254
|
+
now = datetime.now(tz=timezone.utc)
|
|
255
|
+
declared = self._document.created
|
|
256
|
+
# A declared date is the document's date, which is what the format
|
|
257
|
+
# means by "created". A date it could not read — "Spring 2026" — stays
|
|
258
|
+
# on the title page and out of the timestamp.
|
|
259
|
+
properties.created = (
|
|
260
|
+
datetime.combine(declared, time(), tzinfo=timezone.utc)
|
|
261
|
+
if declared is not None
|
|
262
|
+
else now
|
|
263
|
+
)
|
|
264
|
+
properties.modified = now
|
|
265
|
+
|
|
266
|
+
def _save(self) -> bytes:
|
|
267
|
+
"""Serialise the finished document, without ever touching the disk."""
|
|
268
|
+
buffer = io.BytesIO()
|
|
269
|
+
try:
|
|
270
|
+
self._docx.save(buffer)
|
|
271
|
+
except OSError as exc: # pragma: no cover - an in-memory write
|
|
272
|
+
raise RenderError(f"Could not build the Word document: {exc}.") from exc
|
|
273
|
+
return buffer.getvalue()
|
|
274
|
+
|
|
275
|
+
# --- front matter ------------------------------------------------------
|
|
276
|
+
|
|
277
|
+
def _front_matter(self) -> bool:
|
|
278
|
+
"""Write the cover and the contents, and say whether either happened."""
|
|
279
|
+
written = False
|
|
280
|
+
if self._options.title_page:
|
|
281
|
+
written |= self._title_page()
|
|
282
|
+
if self._options.toc:
|
|
283
|
+
written |= self._contents()
|
|
284
|
+
return written
|
|
285
|
+
|
|
286
|
+
def _title_page(self) -> bool:
|
|
287
|
+
"""A cover built from the frontmatter, in the styles the theme set."""
|
|
288
|
+
page = cover(self._document)
|
|
289
|
+
if page is None:
|
|
290
|
+
self._warn(
|
|
291
|
+
"--title-page needs a title; the document declares none, so "
|
|
292
|
+
"no title page was made."
|
|
293
|
+
)
|
|
294
|
+
return False
|
|
295
|
+
title = self._docx.add_paragraph(style=self._docx.styles[TITLE_STYLE])
|
|
296
|
+
# The stylesheet pads the cover down the sheet; Word has no padding, so
|
|
297
|
+
# the same gap is set above the one paragraph it would have pushed.
|
|
298
|
+
title.paragraph_format.space_before = Pt(
|
|
299
|
+
self._theme.type.size * TITLE_PAGE_OFFSET
|
|
300
|
+
)
|
|
301
|
+
title.add_run(page.title)
|
|
302
|
+
for style, value in (
|
|
303
|
+
(SUBTITLE_STYLE, page.subtitle),
|
|
304
|
+
(BODY_STYLE, page.author),
|
|
305
|
+
(COVER_DATE_STYLE, page.date),
|
|
306
|
+
):
|
|
307
|
+
if value:
|
|
308
|
+
line = self._docx.add_paragraph(value, style=self._docx.styles[style])
|
|
309
|
+
if style == BODY_STYLE:
|
|
310
|
+
# The author sits on the date rather than a paragraph's gap
|
|
311
|
+
# away from it, which is the one thing `Normal` gets wrong
|
|
312
|
+
# on a cover.
|
|
313
|
+
line.paragraph_format.space_after = Pt(
|
|
314
|
+
self._theme.type.size * COVER_SPACE_AFTER
|
|
315
|
+
)
|
|
316
|
+
return True
|
|
317
|
+
|
|
318
|
+
def _contents(self) -> bool:
|
|
319
|
+
"""The contents, as a TOC field whose result is already filled in.
|
|
320
|
+
|
|
321
|
+
The field is what makes this a real Word table of contents: it is
|
|
322
|
+
marked dirty, so Word rebuilds it against its own pagination the
|
|
323
|
+
moment the file opens, and it stays right when the document is edited.
|
|
324
|
+
Writing the entries out inside it as well costs little and means every
|
|
325
|
+
other reader — one that shows a field's stored result rather than
|
|
326
|
+
evaluating it — has a contents rather than a blank page.
|
|
327
|
+
"""
|
|
328
|
+
entries = contents(self._document, self._options.toc_depth)
|
|
329
|
+
if not entries:
|
|
330
|
+
self._warn(
|
|
331
|
+
"--toc needs headings; the document has none, so no contents was made."
|
|
332
|
+
)
|
|
333
|
+
return False
|
|
334
|
+
|
|
335
|
+
heading = self._docx.add_paragraph(style=self._docx.styles[TOC_HEADING_STYLE])
|
|
336
|
+
heading.add_run(CONTENTS_HEADING)
|
|
337
|
+
|
|
338
|
+
instruction = CONTENTS_FIELD.format(
|
|
339
|
+
depth=outline_depth(entries, self._options.toc_depth)
|
|
340
|
+
)
|
|
341
|
+
paragraph = None
|
|
342
|
+
for index, entry in enumerate(entries):
|
|
343
|
+
level = min(entry.level, len(TOC_STYLES))
|
|
344
|
+
paragraph = self._docx.add_paragraph(
|
|
345
|
+
style=self._docx.styles[TOC_STYLES[level - 1]]
|
|
346
|
+
)
|
|
347
|
+
if index == 0:
|
|
348
|
+
for element in field_start(instruction, dirty=True):
|
|
349
|
+
paragraph._p.append(element)
|
|
350
|
+
self._contents_entry(paragraph, entry.text, entry.anchor)
|
|
351
|
+
if paragraph is not None:
|
|
352
|
+
paragraph._p.append(field_end())
|
|
353
|
+
return True
|
|
354
|
+
|
|
355
|
+
def _contents_entry(self, paragraph: Any, text: str, anchor: str | None) -> None:
|
|
356
|
+
"""One line of the contents: the heading, a leader, and its page.
|
|
357
|
+
|
|
358
|
+
An entry with no anchor gets neither the link nor the number, because
|
|
359
|
+
both are the same bookmark by two names. Listing it unlinked beats
|
|
360
|
+
dropping a heading out of the contents without saying so.
|
|
361
|
+
"""
|
|
362
|
+
run = paragraph.add_run(text)
|
|
363
|
+
if anchor is None:
|
|
364
|
+
return
|
|
365
|
+
name = bookmark_name(anchor)
|
|
366
|
+
link(paragraph, [run._r], anchor=name)
|
|
367
|
+
paragraph.add_run().add_tab()
|
|
368
|
+
# Dirty, because the placeholder is empty: nothing here knows which
|
|
369
|
+
# page a heading will land on, and Word does as soon as it opens.
|
|
370
|
+
field(paragraph, PAGE_REFERENCE_FIELD.format(name=name), dirty=True)
|
|
371
|
+
|
|
372
|
+
# --- page furniture ----------------------------------------------------
|
|
373
|
+
|
|
374
|
+
def _furniture(self) -> None:
|
|
375
|
+
"""Put the running head and the page number on the pages that get them.
|
|
376
|
+
|
|
377
|
+
The two formats are made to agree here, and the agreement is worth
|
|
378
|
+
stating: the opening page of a document carries no head, whatever is
|
|
379
|
+
on it; a cover carries no page number either; and the front matter,
|
|
380
|
+
which belongs to no section, carries no head at all. In CSS that is
|
|
381
|
+
``@page :first`` and a named page. In Word there is no such selector,
|
|
382
|
+
so it is a section for the front matter, and Word's own "different
|
|
383
|
+
first page" for a document that has none.
|
|
384
|
+
"""
|
|
385
|
+
sections = self._docx.sections
|
|
386
|
+
front = sections[0] if len(sections) > 1 else None
|
|
387
|
+
body = sections[-1]
|
|
388
|
+
head = self._running_head()
|
|
389
|
+
|
|
390
|
+
if front is not None:
|
|
391
|
+
# A section with no header part simply has none, which is what the
|
|
392
|
+
# front matter wants — so the only thing to say is that the cover
|
|
393
|
+
# is not to be numbered.
|
|
394
|
+
front.different_first_page_header_footer = self._options.title_page
|
|
395
|
+
if self._options.title_page:
|
|
396
|
+
# Explicit rather than inherited: a section marked "different
|
|
397
|
+
# first page" with nothing defined for it is empty by
|
|
398
|
+
# inheritance, and inheriting from nothing is a fact about the
|
|
399
|
+
# format that is cheaper to state than to rely on.
|
|
400
|
+
front.first_page_footer.is_linked_to_previous = False
|
|
401
|
+
self._number(front.footer)
|
|
402
|
+
body.different_first_page_header_footer = False
|
|
403
|
+
else:
|
|
404
|
+
# Nothing precedes the body, so its first page is the document's
|
|
405
|
+
# opening page and takes the head off in the only way Word offers.
|
|
406
|
+
body.different_first_page_header_footer = head is not None
|
|
407
|
+
if head is not None:
|
|
408
|
+
self._number(body.first_page_footer)
|
|
409
|
+
if head is not None:
|
|
410
|
+
self._head(body.header, head)
|
|
411
|
+
self._number(body.footer)
|
|
412
|
+
|
|
413
|
+
def _running_head(self) -> tuple[str | None, int | None] | None:
|
|
414
|
+
"""What the head says: the title, and the level it tracks. Or nothing."""
|
|
415
|
+
title = self._document.title
|
|
416
|
+
level = section_level(self._document)
|
|
417
|
+
if not title and level is None:
|
|
418
|
+
return None
|
|
419
|
+
return (title, level)
|
|
420
|
+
|
|
421
|
+
def _head(self, header: Any, head: tuple[str | None, int | None]) -> None:
|
|
422
|
+
"""Write the running head: the title left, the current section right.
|
|
423
|
+
|
|
424
|
+
Word's answer to the stylesheet's named string is ``STYLEREF``, which
|
|
425
|
+
names the nearest heading of one style. The tab stop is set here rather
|
|
426
|
+
than left to the ``Header`` style's own, which sits where a US Letter
|
|
427
|
+
sheet with one-inch margins puts it and nowhere near the edge of any
|
|
428
|
+
other column.
|
|
429
|
+
"""
|
|
430
|
+
title, level = head
|
|
431
|
+
header.is_linked_to_previous = False
|
|
432
|
+
paragraph = header.paragraphs[0]
|
|
433
|
+
paragraph.style = self._docx.styles[HEADER_STYLE]
|
|
434
|
+
width = text_width(self._docx.sections[-1])
|
|
435
|
+
if width is not None:
|
|
436
|
+
paragraph.paragraph_format.tab_stops.add_tab_stop(
|
|
437
|
+
width, WD_TAB_ALIGNMENT.RIGHT
|
|
438
|
+
)
|
|
439
|
+
if title:
|
|
440
|
+
paragraph.add_run(title)
|
|
441
|
+
if level is not None:
|
|
442
|
+
paragraph.add_run().add_tab()
|
|
443
|
+
# Not marked dirty, and deliberately: Word works a STYLEREF out
|
|
444
|
+
# while it lays the page out, the same way it works out a PAGE, so
|
|
445
|
+
# the head fills itself in with nothing asked of the reader.
|
|
446
|
+
# Marking it would cost a "do you want to update the fields in this
|
|
447
|
+
# document?" on every open and buy nothing. Verified in Word.
|
|
448
|
+
field(paragraph, SECTION_FIELD.format(style=HEADING_STYLES[level - 1]))
|
|
449
|
+
|
|
450
|
+
def _number(self, footer: Any) -> None:
|
|
451
|
+
"""Put the page number in a footer, as the PDF puts it in the margin.
|
|
452
|
+
|
|
453
|
+
A field rather than a number: nothing here knows how many pages Word
|
|
454
|
+
will decide the document has, and a field is how the format says "the
|
|
455
|
+
number of the page this lands on". With ``--no-page-numbers`` no
|
|
456
|
+
footer is defined at all, which is a page with nothing at the foot of
|
|
457
|
+
it rather than a page with an empty line there.
|
|
458
|
+
"""
|
|
459
|
+
if not self._options.page_numbers:
|
|
460
|
+
return
|
|
461
|
+
footer.is_linked_to_previous = False
|
|
462
|
+
paragraph = footer.paragraphs[0]
|
|
463
|
+
paragraph.style = self._docx.styles[FOOTER_STYLE]
|
|
464
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
465
|
+
field(paragraph, PAGE_FIELD, "1")
|
|
466
|
+
|
|
467
|
+
# --- blocks ------------------------------------------------------------
|
|
468
|
+
|
|
469
|
+
def _blocks(self, tokens: list[Token], start: int, end: int) -> None:
|
|
470
|
+
"""Walk a run of block tokens, in order, until the end of the range."""
|
|
471
|
+
index = start
|
|
472
|
+
while index < end:
|
|
473
|
+
index = self._block(tokens, index, end)
|
|
474
|
+
|
|
475
|
+
def _block(self, tokens: list[Token], index: int, end: int) -> int:
|
|
476
|
+
"""Handle one block, and return the index of the next one."""
|
|
477
|
+
token = tokens[index]
|
|
478
|
+
handler = _BLOCKS.get(token.type)
|
|
479
|
+
if handler is None:
|
|
480
|
+
# Everything the walk does not act on is either a close tag whose
|
|
481
|
+
# open tag consumed the span, or a token that carries no content.
|
|
482
|
+
return index + 1
|
|
483
|
+
return handler(self, tokens, index, end)
|
|
484
|
+
|
|
485
|
+
def _heading(self, tokens: list[Token], index: int, end: int) -> int:
|
|
486
|
+
"""A heading, styled by level and bookmarked so the contents can link."""
|
|
487
|
+
token = tokens[index]
|
|
488
|
+
level = min(int(token.tag[1:]), len(HEADING_STYLES))
|
|
489
|
+
paragraph = self._paragraph(HEADING_STYLES[level - 1])
|
|
490
|
+
anchor = token.attrGet("id")
|
|
491
|
+
if isinstance(anchor, str) and anchor:
|
|
492
|
+
self._bookmarks += 1
|
|
493
|
+
bookmark(paragraph, bookmark_name(anchor), self._bookmarks)
|
|
494
|
+
close = _closing(tokens, index, end)
|
|
495
|
+
self._inline_span(paragraph, tokens, index + 1, close)
|
|
496
|
+
return close + 1
|
|
497
|
+
|
|
498
|
+
def _paragraph_block(self, tokens: list[Token], index: int, end: int) -> int:
|
|
499
|
+
"""An ordinary paragraph, dropped again if nothing survived the walk."""
|
|
500
|
+
close = _closing(tokens, index, end)
|
|
501
|
+
paragraph = self._paragraph()
|
|
502
|
+
self._inline_span(paragraph, tokens, index + 1, close)
|
|
503
|
+
# A paragraph whose whole content was dropped — one holding nothing
|
|
504
|
+
# but an image that could not be loaded, say — would otherwise be a
|
|
505
|
+
# blank line where the author wrote something.
|
|
506
|
+
if not paragraph.runs and paragraph._p.find(qn("w:hyperlink")) is None:
|
|
507
|
+
paragraph._p.getparent().remove(paragraph._p)
|
|
508
|
+
return close + 1
|
|
509
|
+
|
|
510
|
+
def _code(self, tokens: list[Token], index: int, _end: int) -> int:
|
|
511
|
+
"""A fenced block: one shaded paragraph, highlighted run by run."""
|
|
512
|
+
token = tokens[index]
|
|
513
|
+
paragraph = self._paragraph(CODE_STYLE)
|
|
514
|
+
background = self._highlighter.background
|
|
515
|
+
if background is not None:
|
|
516
|
+
# A dark highlighting style is a panel of its own, and every code
|
|
517
|
+
# block in the document is one — including the blocks it could not
|
|
518
|
+
# colour, which would otherwise be the theme's light box a
|
|
519
|
+
# paragraph away from a dark one. The stylesheet says the same
|
|
520
|
+
# thing with a bare `pre` rule.
|
|
521
|
+
properties = paragraph._p.get_or_add_pPr()
|
|
522
|
+
shade(properties, background)
|
|
523
|
+
set_borders(
|
|
524
|
+
properties,
|
|
525
|
+
CODE_BOX_SIDES,
|
|
526
|
+
color=background,
|
|
527
|
+
space=CODE_BOX_PADDING,
|
|
528
|
+
)
|
|
529
|
+
# `info` is what was written after the fence — the language, and
|
|
530
|
+
# anything else on the line, which nothing here reads. An indented
|
|
531
|
+
# block has no info at all, and so is never highlighted.
|
|
532
|
+
language = (token.info or "").split(maxsplit=1)
|
|
533
|
+
spans = self._highlighter.spans(token.content, language[0] if language else "")
|
|
534
|
+
if spans is None:
|
|
535
|
+
spans = [
|
|
536
|
+
Span(
|
|
537
|
+
text=token.content.rstrip("\n"),
|
|
538
|
+
color=self._highlighter.foreground,
|
|
539
|
+
)
|
|
540
|
+
]
|
|
541
|
+
for span in spans:
|
|
542
|
+
self._span(paragraph, span)
|
|
543
|
+
return index + 1
|
|
544
|
+
|
|
545
|
+
def _span(self, paragraph: Any, span: Span) -> None:
|
|
546
|
+
"""One coloured run of code, inside the paragraph holding the block."""
|
|
547
|
+
run = paragraph.add_run()
|
|
548
|
+
# The setter turns each newline into a line break rather than a new
|
|
549
|
+
# paragraph, which keeps one fenced block inside one shaded box.
|
|
550
|
+
run.text = span.text
|
|
551
|
+
if span.color is not None:
|
|
552
|
+
run.font.color.rgb = rgb(span.color)
|
|
553
|
+
if span.bold:
|
|
554
|
+
run.bold = True
|
|
555
|
+
if span.italic:
|
|
556
|
+
run.italic = True
|
|
557
|
+
|
|
558
|
+
def _quote(self, tokens: list[Token], index: int, end: int) -> int:
|
|
559
|
+
"""A blockquote, which is a depth rather than a container: what it holds
|
|
560
|
+
is ordinary blocks, indented by how many quotes are open around them."""
|
|
561
|
+
close = _closing(tokens, index, end)
|
|
562
|
+
self._quotes += 1
|
|
563
|
+
self._blocks(tokens, index + 1, close)
|
|
564
|
+
self._quotes -= 1
|
|
565
|
+
return close + 1
|
|
566
|
+
|
|
567
|
+
def _list(self, tokens: list[Token], index: int, end: int) -> int:
|
|
568
|
+
"""A list, given a counter of its own so it does not continue the last."""
|
|
569
|
+
token = tokens[index]
|
|
570
|
+
close = _closing(tokens, index, end)
|
|
571
|
+
ordered = token.type == "ordered_list_open"
|
|
572
|
+
# A task list carries its own marker in the checkbox, so it gets the
|
|
573
|
+
# indent of a list and none of the bullets — which is exactly what the
|
|
574
|
+
# stylesheet does with `.contains-task-list` on the PDF side.
|
|
575
|
+
tasks = "task-list" in str(token.attrGet("class") or "")
|
|
576
|
+
level = min(len(self._lists), len(BULLET_STYLES) - 1)
|
|
577
|
+
indent = Emu(int(LIST_INDENT_STEP) * (level + 1))
|
|
578
|
+
|
|
579
|
+
if tasks:
|
|
580
|
+
entry = _ListLevel(style=BODY_STYLE, number=None, indent=indent)
|
|
581
|
+
else:
|
|
582
|
+
style = (NUMBER_STYLES if ordered else BULLET_STYLES)[level]
|
|
583
|
+
entry = _ListLevel(
|
|
584
|
+
style=style,
|
|
585
|
+
number=numbering_instance(self._docx, style, start=_start(token)),
|
|
586
|
+
indent=indent,
|
|
587
|
+
)
|
|
588
|
+
|
|
589
|
+
self._lists.append(entry)
|
|
590
|
+
self._blocks(tokens, index + 1, close)
|
|
591
|
+
self._lists.pop()
|
|
592
|
+
return close + 1
|
|
593
|
+
|
|
594
|
+
def _item(self, tokens: list[Token], index: int, end: int) -> int:
|
|
595
|
+
"""One item, whose marker goes on the first paragraph inside it."""
|
|
596
|
+
close = _closing(tokens, index, end)
|
|
597
|
+
self._pending = self._lists[-1] if self._lists else None
|
|
598
|
+
self._blocks(tokens, index + 1, close)
|
|
599
|
+
# An item whose only content was something that starts no paragraph
|
|
600
|
+
# would otherwise leak its marker onto whatever comes next.
|
|
601
|
+
self._pending = None
|
|
602
|
+
return close + 1
|
|
603
|
+
|
|
604
|
+
def _rule(self, _tokens: list[Token], index: int, _end: int) -> int:
|
|
605
|
+
"""A horizontal rule: an empty paragraph with a border underneath it."""
|
|
606
|
+
paragraph = self._paragraph()
|
|
607
|
+
set_borders(
|
|
608
|
+
paragraph._p.get_or_add_pPr(), ["bottom"], color=self._theme.colors.rule
|
|
609
|
+
)
|
|
610
|
+
gap = Pt(self._theme.type.size * RULE_GAP)
|
|
611
|
+
paragraph.paragraph_format.space_before = gap
|
|
612
|
+
paragraph.paragraph_format.space_after = gap
|
|
613
|
+
return index + 1
|
|
614
|
+
|
|
615
|
+
def _term(self, tokens: list[Token], index: int, end: int) -> int:
|
|
616
|
+
"""The term of a definition list, set bold above its definition."""
|
|
617
|
+
close = _closing(tokens, index, end)
|
|
618
|
+
paragraph = self._paragraph()
|
|
619
|
+
self._inline_span(paragraph, tokens, index + 1, close, base=_Run(bold=True))
|
|
620
|
+
return close + 1
|
|
621
|
+
|
|
622
|
+
def _definition(self, tokens: list[Token], index: int, end: int) -> int:
|
|
623
|
+
"""A definition is blocks of its own, set in under the term above it."""
|
|
624
|
+
close = _closing(tokens, index, end)
|
|
625
|
+
outer = self._extra_indent
|
|
626
|
+
self._extra_indent = Pt(self._theme.type.size * self._theme.spacing.indent)
|
|
627
|
+
self._blocks(tokens, index + 1, close)
|
|
628
|
+
self._extra_indent = outer
|
|
629
|
+
return close + 1
|
|
630
|
+
|
|
631
|
+
def _html(self, tokens: list[Token], index: int, _end: int) -> int:
|
|
632
|
+
"""A raw HTML block, which Word has no way to hold: skip it and warn."""
|
|
633
|
+
self._skip_html(tokens[index])
|
|
634
|
+
return index + 1
|
|
635
|
+
|
|
636
|
+
def _footnote_block(self, tokens: list[Token], index: int, end: int) -> int:
|
|
637
|
+
"""Open the endnote-style list the footnote bodies are collected into."""
|
|
638
|
+
close = _closing(tokens, index, end)
|
|
639
|
+
separator = self._paragraph(FOOTNOTE_STYLE)
|
|
640
|
+
set_borders(
|
|
641
|
+
separator._p.get_or_add_pPr(), ["bottom"], color=self._theme.colors.rule
|
|
642
|
+
)
|
|
643
|
+
separator.paragraph_format.space_before = Pt(
|
|
644
|
+
self._theme.type.size * FOOTNOTES_GAP
|
|
645
|
+
)
|
|
646
|
+
self._footnotes += 1
|
|
647
|
+
self._blocks(tokens, index + 1, close)
|
|
648
|
+
self._footnotes -= 1
|
|
649
|
+
return close + 1
|
|
650
|
+
|
|
651
|
+
def _footnote(self, tokens: list[Token], index: int, end: int) -> int:
|
|
652
|
+
"""One footnote body, numbered to match the reference that points at it."""
|
|
653
|
+
close = _closing(tokens, index, end)
|
|
654
|
+
self._prefix = f"{_footnote_number(tokens[index])}. "
|
|
655
|
+
self._blocks(tokens, index + 1, close)
|
|
656
|
+
self._prefix = None
|
|
657
|
+
return close + 1
|
|
658
|
+
|
|
659
|
+
# --- tables ------------------------------------------------------------
|
|
660
|
+
|
|
661
|
+
def _table(self, tokens: list[Token], index: int, end: int) -> int:
|
|
662
|
+
"""A GFM table, sized to its widest row and given a repeating header."""
|
|
663
|
+
close = _closing(tokens, index, end)
|
|
664
|
+
rows = _table_rows(tokens, index + 1, close)
|
|
665
|
+
if not rows:
|
|
666
|
+
return close + 1
|
|
667
|
+
|
|
668
|
+
width = max(len(cells) for _, cells in rows)
|
|
669
|
+
table = self._docx.add_table(rows=len(rows), cols=width, style=TABLE_STYLE)
|
|
670
|
+
table.autofit = True
|
|
671
|
+
set_borders(
|
|
672
|
+
table._tbl.tblPr,
|
|
673
|
+
["top", "left", "bottom", "right", "insideH", "insideV"],
|
|
674
|
+
color=self._theme.colors.rule,
|
|
675
|
+
)
|
|
676
|
+
|
|
677
|
+
for row_index, (header, cells) in enumerate(rows):
|
|
678
|
+
if header:
|
|
679
|
+
repeat_as_header(table.rows[row_index])
|
|
680
|
+
for column, (content, alignment) in enumerate(cells):
|
|
681
|
+
self._cell(table.cell(row_index, column), content, alignment, header)
|
|
682
|
+
|
|
683
|
+
# Word puts nothing between a table and what follows it, and this is
|
|
684
|
+
# cheaper than the empty paragraph the usual workaround adds.
|
|
685
|
+
self._after_table = True
|
|
686
|
+
return close + 1
|
|
687
|
+
|
|
688
|
+
def _cell(
|
|
689
|
+
self, cell: Any, content: Token | None, alignment: str | None, header: bool
|
|
690
|
+
) -> None:
|
|
691
|
+
"""One cell: the column's alignment, and a fill if it is a header."""
|
|
692
|
+
paragraph = cell.paragraphs[0]
|
|
693
|
+
paragraph.style = self._docx.styles[TABLE_TEXT_STYLE]
|
|
694
|
+
if alignment in ALIGNMENTS:
|
|
695
|
+
paragraph.alignment = ALIGNMENTS[alignment]
|
|
696
|
+
if header:
|
|
697
|
+
shade(cell._tc.get_or_add_tcPr(), self._theme.colors.fill)
|
|
698
|
+
if content is not None:
|
|
699
|
+
self._inline(paragraph, content, base=_Run(bold=header))
|
|
700
|
+
|
|
701
|
+
# --- inline ------------------------------------------------------------
|
|
702
|
+
|
|
703
|
+
def _inline_span(
|
|
704
|
+
self,
|
|
705
|
+
paragraph: Any,
|
|
706
|
+
tokens: list[Token],
|
|
707
|
+
start: int,
|
|
708
|
+
end: int,
|
|
709
|
+
*,
|
|
710
|
+
base: _Run = _PLAIN,
|
|
711
|
+
) -> None:
|
|
712
|
+
"""Render whichever of the tokens in a span carries the inline text."""
|
|
713
|
+
for index in range(start, end):
|
|
714
|
+
if tokens[index].type == "inline":
|
|
715
|
+
self._inline(paragraph, tokens[index], base=base)
|
|
716
|
+
|
|
717
|
+
def _inline(self, paragraph: Any, token: Token, *, base: _Run = _PLAIN) -> None:
|
|
718
|
+
"""Walk one inline token into runs, carrying the formatting as it opens
|
|
719
|
+
and closes. Links are the awkward case: their runs are written into the
|
|
720
|
+
paragraph first and moved inside the hyperlink element on close."""
|
|
721
|
+
style = base
|
|
722
|
+
line = token.map[0] + 1 if token.map else None
|
|
723
|
+
# Where each open link points, and the runs written since it opened,
|
|
724
|
+
# so that they can be moved inside it when it closes. The target is
|
|
725
|
+
# kept from the opening token because the closing one does not carry
|
|
726
|
+
# it — an easy thing to get wrong, and silent when you do: the runs
|
|
727
|
+
# come out styled as a link that goes nowhere.
|
|
728
|
+
collected: list[tuple[str, list[Any]]] = []
|
|
729
|
+
|
|
730
|
+
for child in token.children or []:
|
|
731
|
+
kind = child.type
|
|
732
|
+
if kind == "text":
|
|
733
|
+
# markdown-it leaves empty text tokens behind where it merged
|
|
734
|
+
# adjacent ones, and a run with nothing in it is still a run.
|
|
735
|
+
if child.content:
|
|
736
|
+
self._add(paragraph, child.content, style, collected)
|
|
737
|
+
elif kind == "softbreak":
|
|
738
|
+
self._add(paragraph, " ", style, collected)
|
|
739
|
+
elif kind == "hardbreak":
|
|
740
|
+
self._add(paragraph, "", style, collected).add_break()
|
|
741
|
+
elif kind == "code_inline":
|
|
742
|
+
self._add(
|
|
743
|
+
paragraph, child.content, replace(style, code=True), collected
|
|
744
|
+
)
|
|
745
|
+
elif kind in _OPENS:
|
|
746
|
+
style = replace(style, **{_OPENS[kind]: True})
|
|
747
|
+
elif kind in _CLOSES:
|
|
748
|
+
style = replace(style, **{_CLOSES[kind]: False})
|
|
749
|
+
elif kind == "link_open":
|
|
750
|
+
style = replace(style, link=True)
|
|
751
|
+
collected.append((_href(child), []))
|
|
752
|
+
elif kind == "link_close":
|
|
753
|
+
style = replace(style, link=False)
|
|
754
|
+
self._close_link(paragraph, collected)
|
|
755
|
+
elif kind == "image":
|
|
756
|
+
self._image(paragraph, child, line, collected)
|
|
757
|
+
elif kind == "footnote_ref":
|
|
758
|
+
number = str(_footnote_number(child))
|
|
759
|
+
self._add(
|
|
760
|
+
paragraph, number, replace(style, superscript=True), collected
|
|
761
|
+
)
|
|
762
|
+
elif kind == "html_inline":
|
|
763
|
+
self._inline_html(paragraph, child, style, collected, line)
|
|
764
|
+
elif kind == "footnote_anchor":
|
|
765
|
+
# The back-reference to where the footnote was cited. The PDF
|
|
766
|
+
# hides it too: it is a browser affordance, and on paper the
|
|
767
|
+
# link it offers goes nowhere.
|
|
768
|
+
continue
|
|
769
|
+
|
|
770
|
+
def _add(
|
|
771
|
+
self,
|
|
772
|
+
paragraph: Any,
|
|
773
|
+
text: str,
|
|
774
|
+
style: _Run,
|
|
775
|
+
collected: list[tuple[str, list[Any]]],
|
|
776
|
+
) -> Any:
|
|
777
|
+
"""Add one run, formatted as the walk currently says, and record it."""
|
|
778
|
+
run = paragraph.add_run(text)
|
|
779
|
+
if style.code:
|
|
780
|
+
run.style = self._docx.styles[CODE_INLINE_STYLE]
|
|
781
|
+
if style.link:
|
|
782
|
+
# A run carries one character style, so a link that is also
|
|
783
|
+
# code keeps the code style and takes the link's colour.
|
|
784
|
+
run.font.color.rgb = rgb(self._theme.colors.accent)
|
|
785
|
+
elif style.link:
|
|
786
|
+
run.style = self._docx.styles[LINK_STYLE]
|
|
787
|
+
if style.bold:
|
|
788
|
+
run.bold = True
|
|
789
|
+
if style.italic:
|
|
790
|
+
run.italic = True
|
|
791
|
+
if style.strike:
|
|
792
|
+
run.font.strike = True
|
|
793
|
+
if style.superscript:
|
|
794
|
+
run.font.superscript = True
|
|
795
|
+
for _, pending in collected:
|
|
796
|
+
pending.append(run._r)
|
|
797
|
+
return run
|
|
798
|
+
|
|
799
|
+
def _close_link(
|
|
800
|
+
self, paragraph: Any, collected: list[tuple[str, list[Any]]]
|
|
801
|
+
) -> None:
|
|
802
|
+
"""Wrap the runs written since a link opened in the link itself."""
|
|
803
|
+
if not collected:
|
|
804
|
+
return
|
|
805
|
+
href, elements = collected.pop()
|
|
806
|
+
# Links are recorded but never rewritten by asset resolution, so this
|
|
807
|
+
# is the author's own reference: a URL, or a fragment naming a heading
|
|
808
|
+
# in this document, which is a bookmark by the time it gets here.
|
|
809
|
+
if href.startswith("#"):
|
|
810
|
+
link(paragraph, elements, anchor=bookmark_name(href[1:]))
|
|
811
|
+
elif href:
|
|
812
|
+
link(paragraph, elements, url=href)
|
|
813
|
+
|
|
814
|
+
def _image(
|
|
815
|
+
self,
|
|
816
|
+
paragraph: Any,
|
|
817
|
+
token: Token,
|
|
818
|
+
line: int | None,
|
|
819
|
+
collected: list[tuple[str, list[Any]]],
|
|
820
|
+
) -> None:
|
|
821
|
+
"""An inline image, scaled to the text column, or a warning naming it."""
|
|
822
|
+
source = token.attrGet("src")
|
|
823
|
+
source = source if isinstance(source, str) else ""
|
|
824
|
+
where = f" (line {line})" if line is not None else ""
|
|
825
|
+
if urlsplit(source).scheme in REMOTE_SCHEMES:
|
|
826
|
+
self._warn(f"remote image not available locally: {source}{where}")
|
|
827
|
+
return
|
|
828
|
+
|
|
829
|
+
path = Path(unquote(urlsplit(source).path))
|
|
830
|
+
if not path.is_absolute():
|
|
831
|
+
path = self._document.base_dir / path
|
|
832
|
+
run = paragraph.add_run()
|
|
833
|
+
try:
|
|
834
|
+
picture = run.add_picture(str(path))
|
|
835
|
+
except (OSError, ValueError, *UNUSABLE_IMAGE) as exc:
|
|
836
|
+
self._warn(f"image skipped: {source}{where} — {_why_unusable(exc)}.")
|
|
837
|
+
run._r.getparent().remove(run._r)
|
|
838
|
+
return
|
|
839
|
+
_fit(picture, self._column_width())
|
|
840
|
+
for _, pending in collected:
|
|
841
|
+
pending.append(run._r)
|
|
842
|
+
|
|
843
|
+
def _inline_html(
|
|
844
|
+
self,
|
|
845
|
+
paragraph: Any,
|
|
846
|
+
token: Token,
|
|
847
|
+
style: _Run,
|
|
848
|
+
collected: list[tuple[str, list[Any]]],
|
|
849
|
+
line: int | None,
|
|
850
|
+
) -> None:
|
|
851
|
+
"""Raw HTML inside a paragraph — which is where a checkbox arrives."""
|
|
852
|
+
if TASK_CHECKBOX.search(token.content):
|
|
853
|
+
checked = "checked" in token.content.lower()
|
|
854
|
+
self._add(paragraph, CHECKED if checked else UNCHECKED, style, collected)
|
|
855
|
+
return
|
|
856
|
+
# An inline token carries no line map of its own; the paragraph it sits
|
|
857
|
+
# in does, and naming that line is what makes the warning actionable.
|
|
858
|
+
self._skip_html(token, line)
|
|
859
|
+
|
|
860
|
+
# --- shared ------------------------------------------------------------
|
|
861
|
+
|
|
862
|
+
def _paragraph(self, style: str = BODY_STYLE) -> Any:
|
|
863
|
+
"""Start a paragraph, in whatever the walk is currently inside.
|
|
864
|
+
|
|
865
|
+
This is the one place that knows what nesting means, so that no
|
|
866
|
+
handler above has to: a list item's marker, a blockquote's indent and
|
|
867
|
+
style, a footnote's smaller type and its number, and the gap Word
|
|
868
|
+
would otherwise not leave under a table all land here.
|
|
869
|
+
"""
|
|
870
|
+
item, self._pending = self._pending, None
|
|
871
|
+
prefix, self._prefix = self._prefix, None
|
|
872
|
+
plain = style == BODY_STYLE
|
|
873
|
+
|
|
874
|
+
if plain and item is not None:
|
|
875
|
+
style = item.style
|
|
876
|
+
elif plain and self._footnotes:
|
|
877
|
+
style = FOOTNOTE_STYLE
|
|
878
|
+
elif plain and self._quotes:
|
|
879
|
+
style = QUOTE_STYLE
|
|
880
|
+
|
|
881
|
+
paragraph = self._docx.add_paragraph(style=self._docx.styles[style])
|
|
882
|
+
if item is not None and item.number is not None:
|
|
883
|
+
set_numbering(paragraph, item.number)
|
|
884
|
+
indent = self._indent(item)
|
|
885
|
+
if indent:
|
|
886
|
+
paragraph.paragraph_format.left_indent = indent
|
|
887
|
+
if self._after_table:
|
|
888
|
+
self._open_after_table(paragraph, style)
|
|
889
|
+
self._after_table = False
|
|
890
|
+
if prefix:
|
|
891
|
+
paragraph.add_run(prefix)
|
|
892
|
+
return paragraph
|
|
893
|
+
|
|
894
|
+
def _open_after_table(self, paragraph: Any, style: str) -> None:
|
|
895
|
+
"""Leave the gap Word does not leave of its own accord after a table.
|
|
896
|
+
|
|
897
|
+
A table has no space below it, so whatever follows sits flush against
|
|
898
|
+
its bottom rule. The gap belongs on the paragraph that follows rather
|
|
899
|
+
than on an empty paragraph of its own, which would be a blank line —
|
|
900
|
+
and a style that already asks for more room than that keeps its own.
|
|
901
|
+
"""
|
|
902
|
+
gap = Pt(self._theme.type.size * self._theme.spacing.block)
|
|
903
|
+
declared = self._docx.styles[style].paragraph_format.space_before
|
|
904
|
+
if declared is None or declared < gap:
|
|
905
|
+
paragraph.paragraph_format.space_before = gap
|
|
906
|
+
|
|
907
|
+
def _indent(self, item: _ListLevel | None) -> Length | None:
|
|
908
|
+
"""How far in this paragraph starts, adding up every reason there is.
|
|
909
|
+
|
|
910
|
+
A numbered item is the one case where nothing has to be written: its
|
|
911
|
+
indent comes from the numbering definition. That stops being true the
|
|
912
|
+
moment anything else indents it too, because an explicit indent
|
|
913
|
+
*replaces* the numbering's rather than adding to it — so once there is
|
|
914
|
+
a quote or a definition in the way, the list's own step has to be
|
|
915
|
+
counted back in by hand.
|
|
916
|
+
"""
|
|
917
|
+
quoted = int(quote_indent(self._theme)) * self._quotes
|
|
918
|
+
extra = int(self._extra_indent or 0)
|
|
919
|
+
listed = int(self._lists[-1].indent) if self._lists else 0
|
|
920
|
+
if item is not None and item.number is not None and not quoted and not extra:
|
|
921
|
+
return None
|
|
922
|
+
total = quoted + extra + listed
|
|
923
|
+
return Emu(total) if total else None
|
|
924
|
+
|
|
925
|
+
def _column_width(self) -> Length | None:
|
|
926
|
+
"""The width of the text column, which an image may not exceed.
|
|
927
|
+
|
|
928
|
+
The last section rather than the first: front matter opens a section
|
|
929
|
+
of its own, and the body an image sits in is the one after it.
|
|
930
|
+
"""
|
|
931
|
+
return text_width(self._docx.sections[-1])
|
|
932
|
+
|
|
933
|
+
def _skip_html(self, token: Token, line: int | None = None) -> None:
|
|
934
|
+
"""Warn once per line that raw HTML was dropped, and carry on.
|
|
935
|
+
|
|
936
|
+
The PDF passes HTML through, because its pipeline is HTML. Word has
|
|
937
|
+
nowhere to put it. Warning once per line rather than once per token
|
|
938
|
+
keeps a paragraph with an opening and a closing tag in it from
|
|
939
|
+
producing two warnings about one construct.
|
|
940
|
+
"""
|
|
941
|
+
if HTML_COMMENT.match(token.content):
|
|
942
|
+
return
|
|
943
|
+
if line is None and token.map:
|
|
944
|
+
line = token.map[0] + 1
|
|
945
|
+
if line in self._warned_html:
|
|
946
|
+
return
|
|
947
|
+
self._warned_html.add(line)
|
|
948
|
+
where = f" (line {line})" if line is not None else ""
|
|
949
|
+
self._warn(f"raw HTML is not converted to Word; skipped it{where}")
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
# --- token helpers --------------------------------------------------------
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
def _closing(tokens: list[Token], index: int, end: int) -> int:
|
|
956
|
+
"""The index of the token that closes the container opened at ``index``."""
|
|
957
|
+
depth = 0
|
|
958
|
+
for offset in range(index, end):
|
|
959
|
+
depth += tokens[offset].nesting
|
|
960
|
+
if depth == 0:
|
|
961
|
+
return offset
|
|
962
|
+
return end - 1
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
def _href(token: Token) -> str:
|
|
966
|
+
"""Where a link points, read off the token that opened it."""
|
|
967
|
+
value = token.attrGet("href")
|
|
968
|
+
return value if isinstance(value, str) else ""
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
def _start(token: Token) -> int:
|
|
972
|
+
"""Where an ordered list is told to start counting."""
|
|
973
|
+
value = token.attrGet("start")
|
|
974
|
+
try:
|
|
975
|
+
return int(str(value))
|
|
976
|
+
except (TypeError, ValueError):
|
|
977
|
+
return 1
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def _footnote_number(token: Token) -> int:
|
|
981
|
+
"""Footnotes are numbered from zero in the token stream and one on paper."""
|
|
982
|
+
return int(token.meta.get("id", 0)) + 1
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
def _table_rows(
|
|
986
|
+
tokens: list[Token], start: int, end: int
|
|
987
|
+
) -> list[tuple[bool, list[tuple[Token | None, str | None]]]]:
|
|
988
|
+
"""Read a table's tokens into rows of cells, with the header row marked."""
|
|
989
|
+
rows: list[tuple[bool, list[tuple[Token | None, str | None]]]] = []
|
|
990
|
+
header = False
|
|
991
|
+
cells: list[tuple[Token | None, str | None]] = []
|
|
992
|
+
for index in range(start, end):
|
|
993
|
+
token = tokens[index]
|
|
994
|
+
if token.type == "thead_open":
|
|
995
|
+
header = True
|
|
996
|
+
elif token.type == "thead_close":
|
|
997
|
+
header = False
|
|
998
|
+
elif token.type == "tr_open":
|
|
999
|
+
cells = []
|
|
1000
|
+
elif token.type == "tr_close":
|
|
1001
|
+
rows.append((header, cells))
|
|
1002
|
+
elif token.type in {"th_open", "td_open"}:
|
|
1003
|
+
content = tokens[index + 1] if index + 1 < end else None
|
|
1004
|
+
alignment = token.attrGet("style")
|
|
1005
|
+
cells.append(
|
|
1006
|
+
(
|
|
1007
|
+
content
|
|
1008
|
+
if content is not None and content.type == "inline"
|
|
1009
|
+
else None,
|
|
1010
|
+
alignment if isinstance(alignment, str) else None,
|
|
1011
|
+
)
|
|
1012
|
+
)
|
|
1013
|
+
return rows
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def _why_unusable(exc: BaseException) -> str:
|
|
1017
|
+
"""Why an image could not be embedded, in the words a person would use."""
|
|
1018
|
+
if isinstance(exc, OSError):
|
|
1019
|
+
return "it could not be read"
|
|
1020
|
+
if isinstance(exc, InvalidImageStreamError | UnexpectedEndOfFileError):
|
|
1021
|
+
return "the file is damaged, or is not the picture it claims to be"
|
|
1022
|
+
return "Word has no way to hold that image"
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
def _fit(picture: Any, available: Length | None) -> None:
|
|
1026
|
+
"""Scale a picture down to the text column, keeping its proportions."""
|
|
1027
|
+
if available is None or picture.width <= available:
|
|
1028
|
+
return
|
|
1029
|
+
height = int(picture.height * int(available) / int(picture.width))
|
|
1030
|
+
picture.width = available
|
|
1031
|
+
picture.height = Emu(height)
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
#: Which character formatting each inline token turns on and off.
|
|
1035
|
+
_OPENS = {"strong_open": "bold", "em_open": "italic", "s_open": "strike"}
|
|
1036
|
+
_CLOSES = {"strong_close": "bold", "em_close": "italic", "s_close": "strike"}
|
|
1037
|
+
|
|
1038
|
+
#: The block tokens the walk acts on. Everything else — the closing halves,
|
|
1039
|
+
#: and the tokens that only mark structure another handler already consumed —
|
|
1040
|
+
#: is stepped over.
|
|
1041
|
+
_BLOCKS = {
|
|
1042
|
+
"heading_open": _Builder._heading,
|
|
1043
|
+
"paragraph_open": _Builder._paragraph_block,
|
|
1044
|
+
"fence": _Builder._code,
|
|
1045
|
+
"code_block": _Builder._code,
|
|
1046
|
+
"blockquote_open": _Builder._quote,
|
|
1047
|
+
"bullet_list_open": _Builder._list,
|
|
1048
|
+
"ordered_list_open": _Builder._list,
|
|
1049
|
+
"list_item_open": _Builder._item,
|
|
1050
|
+
"hr": _Builder._rule,
|
|
1051
|
+
"dt_open": _Builder._term,
|
|
1052
|
+
"dd_open": _Builder._definition,
|
|
1053
|
+
"table_open": _Builder._table,
|
|
1054
|
+
"html_block": _Builder._html,
|
|
1055
|
+
"footnote_block_open": _Builder._footnote_block,
|
|
1056
|
+
"footnote_open": _Builder._footnote,
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
|
|
1060
|
+
__all__ = ["render_docx"]
|