amethyst-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1060 @@
1
+ """The token stream, walked into a Word document.
2
+
3
+ Word has no equivalent of CSS paged media, so there is nothing here to hand
4
+ the job to: a DOCX file is a flat sequence of styled paragraphs and runs, and
5
+ this module is the walk that produces it. That is why the two pipelines are
6
+ separate. What keeps their output the same document is the theme — every style
7
+ this module names was defined by :mod:`amethyst.theme.to_docx` from the same
8
+ declaration the stylesheet was generated from, and nothing below chooses a
9
+ font, a size or a colour.
10
+
11
+ Three things about the walk itself.
12
+
13
+ markdown-it's stream is flat, with ``_open`` and ``_close`` tokens marking
14
+ nesting, so containers are handled by finding the matching close and recursing
15
+ over the span between. The alternative — a state machine over a flat loop —
16
+ is the same program with the structure hidden.
17
+
18
+ Where a paragraph goes is context, not content. A paragraph inside a list item
19
+ carries the item's bullet; inside a blockquote it is indented and set in the
20
+ quoted style; inside a footnote it is small and muted. All of that is decided
21
+ in one place, :meth:`_Builder._paragraph`, so no handler has to know what it
22
+ is nested inside.
23
+
24
+ And several things Markdown says have no Word equivalent at all. Raw HTML is
25
+ skipped with a warning naming its line, footnotes become an endnote-style list
26
+ at the end rather than real Word footnotes, and a task list gets a printed
27
+ checkbox rather than a real one. Each is a deliberate approximation from the
28
+ feature matrix, not an oversight.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import io
34
+ import re
35
+ from dataclasses import dataclass, replace
36
+ from datetime import datetime, time, timezone
37
+ from pathlib import Path
38
+ from typing import Any
39
+ from urllib.parse import unquote, urlsplit
40
+
41
+ from docx import Document as new_docx
42
+ from docx.enum.section import WD_SECTION
43
+ from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_TAB_ALIGNMENT
44
+ from docx.image.exceptions import (
45
+ InvalidImageStreamError,
46
+ UnexpectedEndOfFileError,
47
+ UnrecognizedImageError,
48
+ )
49
+ from docx.oxml.ns import qn
50
+ from docx.shared import Emu, Length, Pt
51
+ from markdown_it.token import Token
52
+
53
+ from amethyst.document import Document
54
+ from amethyst.errors import RenderError
55
+ from amethyst.ooxml import (
56
+ bookmark,
57
+ bookmark_name,
58
+ field,
59
+ field_end,
60
+ field_start,
61
+ link,
62
+ numbering_instance,
63
+ repeat_as_header,
64
+ set_borders,
65
+ set_numbering,
66
+ shade,
67
+ )
68
+ from amethyst.parse.assets import REMOTE_SCHEMES
69
+ from amethyst.render.base import RenderOptions, RenderResult
70
+ from amethyst.render.furniture import (
71
+ CONTENTS_HEADING,
72
+ contents,
73
+ cover,
74
+ outline_depth,
75
+ section_level,
76
+ )
77
+ from amethyst.render.highlight import Highlighter, Span
78
+ from amethyst.theme.to_docx import (
79
+ BODY_STYLE,
80
+ BULLET_STYLES,
81
+ CODE_BOX_PADDING,
82
+ CODE_BOX_SIDES,
83
+ CODE_INLINE_STYLE,
84
+ CODE_STYLE,
85
+ COVER_DATE_STYLE,
86
+ COVER_SPACE_AFTER,
87
+ FOOTER_STYLE,
88
+ FOOTNOTE_STYLE,
89
+ HEADER_STYLE,
90
+ HEADING_STYLES,
91
+ LINK_STYLE,
92
+ LIST_INDENT_STEP,
93
+ NUMBER_STYLES,
94
+ QUOTE_STYLE,
95
+ SUBTITLE_STYLE,
96
+ TABLE_STYLE,
97
+ TABLE_TEXT_STYLE,
98
+ TITLE_PAGE_OFFSET,
99
+ TITLE_STYLE,
100
+ TOC_HEADING_STYLE,
101
+ TOC_STYLES,
102
+ apply_page,
103
+ apply_theme,
104
+ quote_indent,
105
+ rgb,
106
+ text_width,
107
+ )
108
+
109
+ #: python-docx's three ways of saying "that is not a picture I can embed".
110
+ #: They share no base class, so all three have to be named — and all three are
111
+ #: reachable from a document: a file that is not an image, one that is an image
112
+ #: format Word has no part type for, and one that was cut off. The last is why
113
+ #: this matters more since images are downloaded: a truncated file is what a
114
+ #: dropped connection leaves behind.
115
+ UNUSABLE_IMAGE = (
116
+ InvalidImageStreamError,
117
+ UnexpectedEndOfFileError,
118
+ UnrecognizedImageError,
119
+ )
120
+
121
+ #: What a task list item is printed as. Word has real checkbox controls, but
122
+ #: they are form fields tied to a content control, and a document that has to
123
+ #: be filled in is not what a converted Markdown list is.
124
+ CHECKED = "☑"
125
+ UNCHECKED = "☐"
126
+
127
+ #: The instructions Word evaluates for itself: the number of the page a field
128
+ #: lands on, the page a bookmark is on, the text of the nearest heading of one
129
+ #: style — Word's answer to a CSS named string — and the table of contents
130
+ #: itself, built from heading levels 1 to N.
131
+ #:
132
+ #: The first and third are worked out while Word lays the pages out and need
133
+ #: nothing asked of the reader. The other two are not, and are marked dirty so
134
+ #: that Word fills them in on open — which is what raises its "update the
135
+ #: fields in this document?" prompt, and why only ``--toc`` raises it.
136
+ PAGE_FIELD = "PAGE"
137
+ PAGE_REFERENCE_FIELD = "PAGEREF {name} \\h"
138
+ SECTION_FIELD = 'STYLEREF "{style}" \\* MERGEFORMAT'
139
+ CONTENTS_FIELD = 'TOC \\o "1-{depth}" \\h \\z \\u'
140
+
141
+ #: The air around a horizontal rule, and above the line that opens the
142
+ #: footnotes, as multiples of the body size. Both mirror the margins the
143
+ #: structural stylesheet gives the same two elements, halved for the rule
144
+ #: because CSS collapses a margin against its neighbour's and Word adds them.
145
+ RULE_GAP = 0.8
146
+ FOOTNOTES_GAP = 2.5
147
+
148
+ #: Cell alignment as markdown-it writes it: an inline style on the cell.
149
+ ALIGNMENTS = {
150
+ "text-align:left": WD_ALIGN_PARAGRAPH.LEFT,
151
+ "text-align:center": WD_ALIGN_PARAGRAPH.CENTER,
152
+ "text-align:right": WD_ALIGN_PARAGRAPH.RIGHT,
153
+ }
154
+
155
+ #: An HTML block that is nothing but a comment. It was never going to be
156
+ #: visible in either format, so warning that it was skipped is noise.
157
+ HTML_COMMENT = re.compile(r"\A\s*(?:<!--.*?-->\s*)+\Z", re.DOTALL)
158
+
159
+ #: The checkbox the tasklist plugin emits. It arrives as raw HTML inside the
160
+ #: item's inline token rather than as a token of its own, so without this the
161
+ #: rule that skips raw HTML would quietly eat every checkbox in the document.
162
+ TASK_CHECKBOX = re.compile(r"<input[^>]*\bclass=\"task-list-item-checkbox\"", re.I)
163
+
164
+
165
+ def render_docx(document: Document, options: RenderOptions) -> RenderResult:
166
+ """Convert a document to the bytes of a Word file."""
167
+ return _Builder(document, options).build()
168
+
169
+
170
+ @dataclass(frozen=True)
171
+ class _Run:
172
+ """The character formatting in force at one point in an inline walk."""
173
+
174
+ bold: bool = False
175
+ italic: bool = False
176
+ strike: bool = False
177
+ code: bool = False
178
+ link: bool = False
179
+ superscript: bool = False
180
+
181
+
182
+ #: No emphasis, no link, no code: what an inline walk starts from unless the
183
+ #: block it sits in says otherwise.
184
+ _PLAIN = _Run()
185
+
186
+
187
+ @dataclass(frozen=True)
188
+ class _ListLevel:
189
+ """One open list, and how the items inside it are marked and indented."""
190
+
191
+ style: str
192
+ #: The list's own counter, or ``None`` for a list that is not numbered.
193
+ number: int | None
194
+ indent: Length
195
+
196
+
197
+ class _Builder:
198
+ """One conversion. Not reused: the state below belongs to one document."""
199
+
200
+ def __init__(self, document: Document, options: RenderOptions) -> None:
201
+ """Start an empty document with the theme already compiled into it."""
202
+ self._document = document
203
+ self._options = options
204
+ self._theme = options.theme
205
+ self._warn = options.warn
206
+ self._docx = new_docx()
207
+ self._highlighter = Highlighter(options.highlight_style, warn=self._warn)
208
+ apply_theme(self._docx, self._theme, warn=self._warn)
209
+
210
+ self._lists: list[_ListLevel] = []
211
+ self._pending: _ListLevel | None = None
212
+ self._quotes = 0
213
+ self._footnotes = 0
214
+ self._prefix: str | None = None
215
+ self._bookmarks = 0
216
+ self._after_table = False
217
+ self._extra_indent: Length | None = None
218
+ self._warned_html: set[int | None] = set()
219
+
220
+ # --- the document ------------------------------------------------------
221
+
222
+ def build(self) -> RenderResult:
223
+ """Walk the whole document and return the file's bytes."""
224
+ self._properties()
225
+ if self._front_matter():
226
+ # The front matter becomes a section of its own so that it can
227
+ # carry different furniture from the body — no running head, and
228
+ # no page number on a cover. It is also what starts the document
229
+ # proper on a fresh page, which is the break the stylesheet gets
230
+ # from `break-after: page`.
231
+ self._docx.add_section(WD_SECTION.NEW_PAGE)
232
+ apply_page(self._docx.sections[-1], self._theme, warn=self._warn)
233
+ self._blocks(self._document.tokens, 0, len(self._document.tokens))
234
+ self._furniture()
235
+ return RenderResult(data=self._save(), pages=None)
236
+
237
+ def _properties(self) -> None:
238
+ """Say what the document says about itself, and nothing else.
239
+
240
+ python-docx builds every document from a template whose properties
241
+ claim it was written by "python-docx" in 2013, and Word shows those in
242
+ the file's info pane. They are replaced by the frontmatter — the same
243
+ four fields the PDF carries as its metadata — so that a reader who
244
+ opens the information pane sees the same thing in either format.
245
+ """
246
+ properties = self._docx.core_properties
247
+ properties.author = self._document.author or ""
248
+ properties.title = self._document.title or ""
249
+ properties.subject = self._document.subtitle or ""
250
+ properties.keywords = self._document.keywords or ""
251
+ properties.last_modified_by = ""
252
+ properties.comments = ""
253
+ properties.revision = 1
254
+ now = datetime.now(tz=timezone.utc)
255
+ declared = self._document.created
256
+ # A declared date is the document's date, which is what the format
257
+ # means by "created". A date it could not read — "Spring 2026" — stays
258
+ # on the title page and out of the timestamp.
259
+ properties.created = (
260
+ datetime.combine(declared, time(), tzinfo=timezone.utc)
261
+ if declared is not None
262
+ else now
263
+ )
264
+ properties.modified = now
265
+
266
+ def _save(self) -> bytes:
267
+ """Serialise the finished document, without ever touching the disk."""
268
+ buffer = io.BytesIO()
269
+ try:
270
+ self._docx.save(buffer)
271
+ except OSError as exc: # pragma: no cover - an in-memory write
272
+ raise RenderError(f"Could not build the Word document: {exc}.") from exc
273
+ return buffer.getvalue()
274
+
275
+ # --- front matter ------------------------------------------------------
276
+
277
+ def _front_matter(self) -> bool:
278
+ """Write the cover and the contents, and say whether either happened."""
279
+ written = False
280
+ if self._options.title_page:
281
+ written |= self._title_page()
282
+ if self._options.toc:
283
+ written |= self._contents()
284
+ return written
285
+
286
+ def _title_page(self) -> bool:
287
+ """A cover built from the frontmatter, in the styles the theme set."""
288
+ page = cover(self._document)
289
+ if page is None:
290
+ self._warn(
291
+ "--title-page needs a title; the document declares none, so "
292
+ "no title page was made."
293
+ )
294
+ return False
295
+ title = self._docx.add_paragraph(style=self._docx.styles[TITLE_STYLE])
296
+ # The stylesheet pads the cover down the sheet; Word has no padding, so
297
+ # the same gap is set above the one paragraph it would have pushed.
298
+ title.paragraph_format.space_before = Pt(
299
+ self._theme.type.size * TITLE_PAGE_OFFSET
300
+ )
301
+ title.add_run(page.title)
302
+ for style, value in (
303
+ (SUBTITLE_STYLE, page.subtitle),
304
+ (BODY_STYLE, page.author),
305
+ (COVER_DATE_STYLE, page.date),
306
+ ):
307
+ if value:
308
+ line = self._docx.add_paragraph(value, style=self._docx.styles[style])
309
+ if style == BODY_STYLE:
310
+ # The author sits on the date rather than a paragraph's gap
311
+ # away from it, which is the one thing `Normal` gets wrong
312
+ # on a cover.
313
+ line.paragraph_format.space_after = Pt(
314
+ self._theme.type.size * COVER_SPACE_AFTER
315
+ )
316
+ return True
317
+
318
+ def _contents(self) -> bool:
319
+ """The contents, as a TOC field whose result is already filled in.
320
+
321
+ The field is what makes this a real Word table of contents: it is
322
+ marked dirty, so Word rebuilds it against its own pagination the
323
+ moment the file opens, and it stays right when the document is edited.
324
+ Writing the entries out inside it as well costs little and means every
325
+ other reader — one that shows a field's stored result rather than
326
+ evaluating it — has a contents rather than a blank page.
327
+ """
328
+ entries = contents(self._document, self._options.toc_depth)
329
+ if not entries:
330
+ self._warn(
331
+ "--toc needs headings; the document has none, so no contents was made."
332
+ )
333
+ return False
334
+
335
+ heading = self._docx.add_paragraph(style=self._docx.styles[TOC_HEADING_STYLE])
336
+ heading.add_run(CONTENTS_HEADING)
337
+
338
+ instruction = CONTENTS_FIELD.format(
339
+ depth=outline_depth(entries, self._options.toc_depth)
340
+ )
341
+ paragraph = None
342
+ for index, entry in enumerate(entries):
343
+ level = min(entry.level, len(TOC_STYLES))
344
+ paragraph = self._docx.add_paragraph(
345
+ style=self._docx.styles[TOC_STYLES[level - 1]]
346
+ )
347
+ if index == 0:
348
+ for element in field_start(instruction, dirty=True):
349
+ paragraph._p.append(element)
350
+ self._contents_entry(paragraph, entry.text, entry.anchor)
351
+ if paragraph is not None:
352
+ paragraph._p.append(field_end())
353
+ return True
354
+
355
+ def _contents_entry(self, paragraph: Any, text: str, anchor: str | None) -> None:
356
+ """One line of the contents: the heading, a leader, and its page.
357
+
358
+ An entry with no anchor gets neither the link nor the number, because
359
+ both are the same bookmark by two names. Listing it unlinked beats
360
+ dropping a heading out of the contents without saying so.
361
+ """
362
+ run = paragraph.add_run(text)
363
+ if anchor is None:
364
+ return
365
+ name = bookmark_name(anchor)
366
+ link(paragraph, [run._r], anchor=name)
367
+ paragraph.add_run().add_tab()
368
+ # Dirty, because the placeholder is empty: nothing here knows which
369
+ # page a heading will land on, and Word does as soon as it opens.
370
+ field(paragraph, PAGE_REFERENCE_FIELD.format(name=name), dirty=True)
371
+
372
+ # --- page furniture ----------------------------------------------------
373
+
374
+ def _furniture(self) -> None:
375
+ """Put the running head and the page number on the pages that get them.
376
+
377
+ The two formats are made to agree here, and the agreement is worth
378
+ stating: the opening page of a document carries no head, whatever is
379
+ on it; a cover carries no page number either; and the front matter,
380
+ which belongs to no section, carries no head at all. In CSS that is
381
+ ``@page :first`` and a named page. In Word there is no such selector,
382
+ so it is a section for the front matter, and Word's own "different
383
+ first page" for a document that has none.
384
+ """
385
+ sections = self._docx.sections
386
+ front = sections[0] if len(sections) > 1 else None
387
+ body = sections[-1]
388
+ head = self._running_head()
389
+
390
+ if front is not None:
391
+ # A section with no header part simply has none, which is what the
392
+ # front matter wants — so the only thing to say is that the cover
393
+ # is not to be numbered.
394
+ front.different_first_page_header_footer = self._options.title_page
395
+ if self._options.title_page:
396
+ # Explicit rather than inherited: a section marked "different
397
+ # first page" with nothing defined for it is empty by
398
+ # inheritance, and inheriting from nothing is a fact about the
399
+ # format that is cheaper to state than to rely on.
400
+ front.first_page_footer.is_linked_to_previous = False
401
+ self._number(front.footer)
402
+ body.different_first_page_header_footer = False
403
+ else:
404
+ # Nothing precedes the body, so its first page is the document's
405
+ # opening page and takes the head off in the only way Word offers.
406
+ body.different_first_page_header_footer = head is not None
407
+ if head is not None:
408
+ self._number(body.first_page_footer)
409
+ if head is not None:
410
+ self._head(body.header, head)
411
+ self._number(body.footer)
412
+
413
+ def _running_head(self) -> tuple[str | None, int | None] | None:
414
+ """What the head says: the title, and the level it tracks. Or nothing."""
415
+ title = self._document.title
416
+ level = section_level(self._document)
417
+ if not title and level is None:
418
+ return None
419
+ return (title, level)
420
+
421
+ def _head(self, header: Any, head: tuple[str | None, int | None]) -> None:
422
+ """Write the running head: the title left, the current section right.
423
+
424
+ Word's answer to the stylesheet's named string is ``STYLEREF``, which
425
+ names the nearest heading of one style. The tab stop is set here rather
426
+ than left to the ``Header`` style's own, which sits where a US Letter
427
+ sheet with one-inch margins puts it and nowhere near the edge of any
428
+ other column.
429
+ """
430
+ title, level = head
431
+ header.is_linked_to_previous = False
432
+ paragraph = header.paragraphs[0]
433
+ paragraph.style = self._docx.styles[HEADER_STYLE]
434
+ width = text_width(self._docx.sections[-1])
435
+ if width is not None:
436
+ paragraph.paragraph_format.tab_stops.add_tab_stop(
437
+ width, WD_TAB_ALIGNMENT.RIGHT
438
+ )
439
+ if title:
440
+ paragraph.add_run(title)
441
+ if level is not None:
442
+ paragraph.add_run().add_tab()
443
+ # Not marked dirty, and deliberately: Word works a STYLEREF out
444
+ # while it lays the page out, the same way it works out a PAGE, so
445
+ # the head fills itself in with nothing asked of the reader.
446
+ # Marking it would cost a "do you want to update the fields in this
447
+ # document?" on every open and buy nothing. Verified in Word.
448
+ field(paragraph, SECTION_FIELD.format(style=HEADING_STYLES[level - 1]))
449
+
450
+ def _number(self, footer: Any) -> None:
451
+ """Put the page number in a footer, as the PDF puts it in the margin.
452
+
453
+ A field rather than a number: nothing here knows how many pages Word
454
+ will decide the document has, and a field is how the format says "the
455
+ number of the page this lands on". With ``--no-page-numbers`` no
456
+ footer is defined at all, which is a page with nothing at the foot of
457
+ it rather than a page with an empty line there.
458
+ """
459
+ if not self._options.page_numbers:
460
+ return
461
+ footer.is_linked_to_previous = False
462
+ paragraph = footer.paragraphs[0]
463
+ paragraph.style = self._docx.styles[FOOTER_STYLE]
464
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
465
+ field(paragraph, PAGE_FIELD, "1")
466
+
467
+ # --- blocks ------------------------------------------------------------
468
+
469
+ def _blocks(self, tokens: list[Token], start: int, end: int) -> None:
470
+ """Walk a run of block tokens, in order, until the end of the range."""
471
+ index = start
472
+ while index < end:
473
+ index = self._block(tokens, index, end)
474
+
475
+ def _block(self, tokens: list[Token], index: int, end: int) -> int:
476
+ """Handle one block, and return the index of the next one."""
477
+ token = tokens[index]
478
+ handler = _BLOCKS.get(token.type)
479
+ if handler is None:
480
+ # Everything the walk does not act on is either a close tag whose
481
+ # open tag consumed the span, or a token that carries no content.
482
+ return index + 1
483
+ return handler(self, tokens, index, end)
484
+
485
+ def _heading(self, tokens: list[Token], index: int, end: int) -> int:
486
+ """A heading, styled by level and bookmarked so the contents can link."""
487
+ token = tokens[index]
488
+ level = min(int(token.tag[1:]), len(HEADING_STYLES))
489
+ paragraph = self._paragraph(HEADING_STYLES[level - 1])
490
+ anchor = token.attrGet("id")
491
+ if isinstance(anchor, str) and anchor:
492
+ self._bookmarks += 1
493
+ bookmark(paragraph, bookmark_name(anchor), self._bookmarks)
494
+ close = _closing(tokens, index, end)
495
+ self._inline_span(paragraph, tokens, index + 1, close)
496
+ return close + 1
497
+
498
+ def _paragraph_block(self, tokens: list[Token], index: int, end: int) -> int:
499
+ """An ordinary paragraph, dropped again if nothing survived the walk."""
500
+ close = _closing(tokens, index, end)
501
+ paragraph = self._paragraph()
502
+ self._inline_span(paragraph, tokens, index + 1, close)
503
+ # A paragraph whose whole content was dropped — one holding nothing
504
+ # but an image that could not be loaded, say — would otherwise be a
505
+ # blank line where the author wrote something.
506
+ if not paragraph.runs and paragraph._p.find(qn("w:hyperlink")) is None:
507
+ paragraph._p.getparent().remove(paragraph._p)
508
+ return close + 1
509
+
510
+ def _code(self, tokens: list[Token], index: int, _end: int) -> int:
511
+ """A fenced block: one shaded paragraph, highlighted run by run."""
512
+ token = tokens[index]
513
+ paragraph = self._paragraph(CODE_STYLE)
514
+ background = self._highlighter.background
515
+ if background is not None:
516
+ # A dark highlighting style is a panel of its own, and every code
517
+ # block in the document is one — including the blocks it could not
518
+ # colour, which would otherwise be the theme's light box a
519
+ # paragraph away from a dark one. The stylesheet says the same
520
+ # thing with a bare `pre` rule.
521
+ properties = paragraph._p.get_or_add_pPr()
522
+ shade(properties, background)
523
+ set_borders(
524
+ properties,
525
+ CODE_BOX_SIDES,
526
+ color=background,
527
+ space=CODE_BOX_PADDING,
528
+ )
529
+ # `info` is what was written after the fence — the language, and
530
+ # anything else on the line, which nothing here reads. An indented
531
+ # block has no info at all, and so is never highlighted.
532
+ language = (token.info or "").split(maxsplit=1)
533
+ spans = self._highlighter.spans(token.content, language[0] if language else "")
534
+ if spans is None:
535
+ spans = [
536
+ Span(
537
+ text=token.content.rstrip("\n"),
538
+ color=self._highlighter.foreground,
539
+ )
540
+ ]
541
+ for span in spans:
542
+ self._span(paragraph, span)
543
+ return index + 1
544
+
545
+ def _span(self, paragraph: Any, span: Span) -> None:
546
+ """One coloured run of code, inside the paragraph holding the block."""
547
+ run = paragraph.add_run()
548
+ # The setter turns each newline into a line break rather than a new
549
+ # paragraph, which keeps one fenced block inside one shaded box.
550
+ run.text = span.text
551
+ if span.color is not None:
552
+ run.font.color.rgb = rgb(span.color)
553
+ if span.bold:
554
+ run.bold = True
555
+ if span.italic:
556
+ run.italic = True
557
+
558
+ def _quote(self, tokens: list[Token], index: int, end: int) -> int:
559
+ """A blockquote, which is a depth rather than a container: what it holds
560
+ is ordinary blocks, indented by how many quotes are open around them."""
561
+ close = _closing(tokens, index, end)
562
+ self._quotes += 1
563
+ self._blocks(tokens, index + 1, close)
564
+ self._quotes -= 1
565
+ return close + 1
566
+
567
+ def _list(self, tokens: list[Token], index: int, end: int) -> int:
568
+ """A list, given a counter of its own so it does not continue the last."""
569
+ token = tokens[index]
570
+ close = _closing(tokens, index, end)
571
+ ordered = token.type == "ordered_list_open"
572
+ # A task list carries its own marker in the checkbox, so it gets the
573
+ # indent of a list and none of the bullets — which is exactly what the
574
+ # stylesheet does with `.contains-task-list` on the PDF side.
575
+ tasks = "task-list" in str(token.attrGet("class") or "")
576
+ level = min(len(self._lists), len(BULLET_STYLES) - 1)
577
+ indent = Emu(int(LIST_INDENT_STEP) * (level + 1))
578
+
579
+ if tasks:
580
+ entry = _ListLevel(style=BODY_STYLE, number=None, indent=indent)
581
+ else:
582
+ style = (NUMBER_STYLES if ordered else BULLET_STYLES)[level]
583
+ entry = _ListLevel(
584
+ style=style,
585
+ number=numbering_instance(self._docx, style, start=_start(token)),
586
+ indent=indent,
587
+ )
588
+
589
+ self._lists.append(entry)
590
+ self._blocks(tokens, index + 1, close)
591
+ self._lists.pop()
592
+ return close + 1
593
+
594
+ def _item(self, tokens: list[Token], index: int, end: int) -> int:
595
+ """One item, whose marker goes on the first paragraph inside it."""
596
+ close = _closing(tokens, index, end)
597
+ self._pending = self._lists[-1] if self._lists else None
598
+ self._blocks(tokens, index + 1, close)
599
+ # An item whose only content was something that starts no paragraph
600
+ # would otherwise leak its marker onto whatever comes next.
601
+ self._pending = None
602
+ return close + 1
603
+
604
+ def _rule(self, _tokens: list[Token], index: int, _end: int) -> int:
605
+ """A horizontal rule: an empty paragraph with a border underneath it."""
606
+ paragraph = self._paragraph()
607
+ set_borders(
608
+ paragraph._p.get_or_add_pPr(), ["bottom"], color=self._theme.colors.rule
609
+ )
610
+ gap = Pt(self._theme.type.size * RULE_GAP)
611
+ paragraph.paragraph_format.space_before = gap
612
+ paragraph.paragraph_format.space_after = gap
613
+ return index + 1
614
+
615
+ def _term(self, tokens: list[Token], index: int, end: int) -> int:
616
+ """The term of a definition list, set bold above its definition."""
617
+ close = _closing(tokens, index, end)
618
+ paragraph = self._paragraph()
619
+ self._inline_span(paragraph, tokens, index + 1, close, base=_Run(bold=True))
620
+ return close + 1
621
+
622
+ def _definition(self, tokens: list[Token], index: int, end: int) -> int:
623
+ """A definition is blocks of its own, set in under the term above it."""
624
+ close = _closing(tokens, index, end)
625
+ outer = self._extra_indent
626
+ self._extra_indent = Pt(self._theme.type.size * self._theme.spacing.indent)
627
+ self._blocks(tokens, index + 1, close)
628
+ self._extra_indent = outer
629
+ return close + 1
630
+
631
+ def _html(self, tokens: list[Token], index: int, _end: int) -> int:
632
+ """A raw HTML block, which Word has no way to hold: skip it and warn."""
633
+ self._skip_html(tokens[index])
634
+ return index + 1
635
+
636
+ def _footnote_block(self, tokens: list[Token], index: int, end: int) -> int:
637
+ """Open the endnote-style list the footnote bodies are collected into."""
638
+ close = _closing(tokens, index, end)
639
+ separator = self._paragraph(FOOTNOTE_STYLE)
640
+ set_borders(
641
+ separator._p.get_or_add_pPr(), ["bottom"], color=self._theme.colors.rule
642
+ )
643
+ separator.paragraph_format.space_before = Pt(
644
+ self._theme.type.size * FOOTNOTES_GAP
645
+ )
646
+ self._footnotes += 1
647
+ self._blocks(tokens, index + 1, close)
648
+ self._footnotes -= 1
649
+ return close + 1
650
+
651
+ def _footnote(self, tokens: list[Token], index: int, end: int) -> int:
652
+ """One footnote body, numbered to match the reference that points at it."""
653
+ close = _closing(tokens, index, end)
654
+ self._prefix = f"{_footnote_number(tokens[index])}. "
655
+ self._blocks(tokens, index + 1, close)
656
+ self._prefix = None
657
+ return close + 1
658
+
659
+ # --- tables ------------------------------------------------------------
660
+
661
+ def _table(self, tokens: list[Token], index: int, end: int) -> int:
662
+ """A GFM table, sized to its widest row and given a repeating header."""
663
+ close = _closing(tokens, index, end)
664
+ rows = _table_rows(tokens, index + 1, close)
665
+ if not rows:
666
+ return close + 1
667
+
668
+ width = max(len(cells) for _, cells in rows)
669
+ table = self._docx.add_table(rows=len(rows), cols=width, style=TABLE_STYLE)
670
+ table.autofit = True
671
+ set_borders(
672
+ table._tbl.tblPr,
673
+ ["top", "left", "bottom", "right", "insideH", "insideV"],
674
+ color=self._theme.colors.rule,
675
+ )
676
+
677
+ for row_index, (header, cells) in enumerate(rows):
678
+ if header:
679
+ repeat_as_header(table.rows[row_index])
680
+ for column, (content, alignment) in enumerate(cells):
681
+ self._cell(table.cell(row_index, column), content, alignment, header)
682
+
683
+ # Word puts nothing between a table and what follows it, and this is
684
+ # cheaper than the empty paragraph the usual workaround adds.
685
+ self._after_table = True
686
+ return close + 1
687
+
688
+ def _cell(
689
+ self, cell: Any, content: Token | None, alignment: str | None, header: bool
690
+ ) -> None:
691
+ """One cell: the column's alignment, and a fill if it is a header."""
692
+ paragraph = cell.paragraphs[0]
693
+ paragraph.style = self._docx.styles[TABLE_TEXT_STYLE]
694
+ if alignment in ALIGNMENTS:
695
+ paragraph.alignment = ALIGNMENTS[alignment]
696
+ if header:
697
+ shade(cell._tc.get_or_add_tcPr(), self._theme.colors.fill)
698
+ if content is not None:
699
+ self._inline(paragraph, content, base=_Run(bold=header))
700
+
701
+ # --- inline ------------------------------------------------------------
702
+
703
+ def _inline_span(
704
+ self,
705
+ paragraph: Any,
706
+ tokens: list[Token],
707
+ start: int,
708
+ end: int,
709
+ *,
710
+ base: _Run = _PLAIN,
711
+ ) -> None:
712
+ """Render whichever of the tokens in a span carries the inline text."""
713
+ for index in range(start, end):
714
+ if tokens[index].type == "inline":
715
+ self._inline(paragraph, tokens[index], base=base)
716
+
717
+ def _inline(self, paragraph: Any, token: Token, *, base: _Run = _PLAIN) -> None:
718
+ """Walk one inline token into runs, carrying the formatting as it opens
719
+ and closes. Links are the awkward case: their runs are written into the
720
+ paragraph first and moved inside the hyperlink element on close."""
721
+ style = base
722
+ line = token.map[0] + 1 if token.map else None
723
+ # Where each open link points, and the runs written since it opened,
724
+ # so that they can be moved inside it when it closes. The target is
725
+ # kept from the opening token because the closing one does not carry
726
+ # it — an easy thing to get wrong, and silent when you do: the runs
727
+ # come out styled as a link that goes nowhere.
728
+ collected: list[tuple[str, list[Any]]] = []
729
+
730
+ for child in token.children or []:
731
+ kind = child.type
732
+ if kind == "text":
733
+ # markdown-it leaves empty text tokens behind where it merged
734
+ # adjacent ones, and a run with nothing in it is still a run.
735
+ if child.content:
736
+ self._add(paragraph, child.content, style, collected)
737
+ elif kind == "softbreak":
738
+ self._add(paragraph, " ", style, collected)
739
+ elif kind == "hardbreak":
740
+ self._add(paragraph, "", style, collected).add_break()
741
+ elif kind == "code_inline":
742
+ self._add(
743
+ paragraph, child.content, replace(style, code=True), collected
744
+ )
745
+ elif kind in _OPENS:
746
+ style = replace(style, **{_OPENS[kind]: True})
747
+ elif kind in _CLOSES:
748
+ style = replace(style, **{_CLOSES[kind]: False})
749
+ elif kind == "link_open":
750
+ style = replace(style, link=True)
751
+ collected.append((_href(child), []))
752
+ elif kind == "link_close":
753
+ style = replace(style, link=False)
754
+ self._close_link(paragraph, collected)
755
+ elif kind == "image":
756
+ self._image(paragraph, child, line, collected)
757
+ elif kind == "footnote_ref":
758
+ number = str(_footnote_number(child))
759
+ self._add(
760
+ paragraph, number, replace(style, superscript=True), collected
761
+ )
762
+ elif kind == "html_inline":
763
+ self._inline_html(paragraph, child, style, collected, line)
764
+ elif kind == "footnote_anchor":
765
+ # The back-reference to where the footnote was cited. The PDF
766
+ # hides it too: it is a browser affordance, and on paper the
767
+ # link it offers goes nowhere.
768
+ continue
769
+
770
+ def _add(
771
+ self,
772
+ paragraph: Any,
773
+ text: str,
774
+ style: _Run,
775
+ collected: list[tuple[str, list[Any]]],
776
+ ) -> Any:
777
+ """Add one run, formatted as the walk currently says, and record it."""
778
+ run = paragraph.add_run(text)
779
+ if style.code:
780
+ run.style = self._docx.styles[CODE_INLINE_STYLE]
781
+ if style.link:
782
+ # A run carries one character style, so a link that is also
783
+ # code keeps the code style and takes the link's colour.
784
+ run.font.color.rgb = rgb(self._theme.colors.accent)
785
+ elif style.link:
786
+ run.style = self._docx.styles[LINK_STYLE]
787
+ if style.bold:
788
+ run.bold = True
789
+ if style.italic:
790
+ run.italic = True
791
+ if style.strike:
792
+ run.font.strike = True
793
+ if style.superscript:
794
+ run.font.superscript = True
795
+ for _, pending in collected:
796
+ pending.append(run._r)
797
+ return run
798
+
799
+ def _close_link(
800
+ self, paragraph: Any, collected: list[tuple[str, list[Any]]]
801
+ ) -> None:
802
+ """Wrap the runs written since a link opened in the link itself."""
803
+ if not collected:
804
+ return
805
+ href, elements = collected.pop()
806
+ # Links are recorded but never rewritten by asset resolution, so this
807
+ # is the author's own reference: a URL, or a fragment naming a heading
808
+ # in this document, which is a bookmark by the time it gets here.
809
+ if href.startswith("#"):
810
+ link(paragraph, elements, anchor=bookmark_name(href[1:]))
811
+ elif href:
812
+ link(paragraph, elements, url=href)
813
+
814
+ def _image(
815
+ self,
816
+ paragraph: Any,
817
+ token: Token,
818
+ line: int | None,
819
+ collected: list[tuple[str, list[Any]]],
820
+ ) -> None:
821
+ """An inline image, scaled to the text column, or a warning naming it."""
822
+ source = token.attrGet("src")
823
+ source = source if isinstance(source, str) else ""
824
+ where = f" (line {line})" if line is not None else ""
825
+ if urlsplit(source).scheme in REMOTE_SCHEMES:
826
+ self._warn(f"remote image not available locally: {source}{where}")
827
+ return
828
+
829
+ path = Path(unquote(urlsplit(source).path))
830
+ if not path.is_absolute():
831
+ path = self._document.base_dir / path
832
+ run = paragraph.add_run()
833
+ try:
834
+ picture = run.add_picture(str(path))
835
+ except (OSError, ValueError, *UNUSABLE_IMAGE) as exc:
836
+ self._warn(f"image skipped: {source}{where} — {_why_unusable(exc)}.")
837
+ run._r.getparent().remove(run._r)
838
+ return
839
+ _fit(picture, self._column_width())
840
+ for _, pending in collected:
841
+ pending.append(run._r)
842
+
843
+ def _inline_html(
844
+ self,
845
+ paragraph: Any,
846
+ token: Token,
847
+ style: _Run,
848
+ collected: list[tuple[str, list[Any]]],
849
+ line: int | None,
850
+ ) -> None:
851
+ """Raw HTML inside a paragraph — which is where a checkbox arrives."""
852
+ if TASK_CHECKBOX.search(token.content):
853
+ checked = "checked" in token.content.lower()
854
+ self._add(paragraph, CHECKED if checked else UNCHECKED, style, collected)
855
+ return
856
+ # An inline token carries no line map of its own; the paragraph it sits
857
+ # in does, and naming that line is what makes the warning actionable.
858
+ self._skip_html(token, line)
859
+
860
+ # --- shared ------------------------------------------------------------
861
+
862
+ def _paragraph(self, style: str = BODY_STYLE) -> Any:
863
+ """Start a paragraph, in whatever the walk is currently inside.
864
+
865
+ This is the one place that knows what nesting means, so that no
866
+ handler above has to: a list item's marker, a blockquote's indent and
867
+ style, a footnote's smaller type and its number, and the gap Word
868
+ would otherwise not leave under a table all land here.
869
+ """
870
+ item, self._pending = self._pending, None
871
+ prefix, self._prefix = self._prefix, None
872
+ plain = style == BODY_STYLE
873
+
874
+ if plain and item is not None:
875
+ style = item.style
876
+ elif plain and self._footnotes:
877
+ style = FOOTNOTE_STYLE
878
+ elif plain and self._quotes:
879
+ style = QUOTE_STYLE
880
+
881
+ paragraph = self._docx.add_paragraph(style=self._docx.styles[style])
882
+ if item is not None and item.number is not None:
883
+ set_numbering(paragraph, item.number)
884
+ indent = self._indent(item)
885
+ if indent:
886
+ paragraph.paragraph_format.left_indent = indent
887
+ if self._after_table:
888
+ self._open_after_table(paragraph, style)
889
+ self._after_table = False
890
+ if prefix:
891
+ paragraph.add_run(prefix)
892
+ return paragraph
893
+
894
+ def _open_after_table(self, paragraph: Any, style: str) -> None:
895
+ """Leave the gap Word does not leave of its own accord after a table.
896
+
897
+ A table has no space below it, so whatever follows sits flush against
898
+ its bottom rule. The gap belongs on the paragraph that follows rather
899
+ than on an empty paragraph of its own, which would be a blank line —
900
+ and a style that already asks for more room than that keeps its own.
901
+ """
902
+ gap = Pt(self._theme.type.size * self._theme.spacing.block)
903
+ declared = self._docx.styles[style].paragraph_format.space_before
904
+ if declared is None or declared < gap:
905
+ paragraph.paragraph_format.space_before = gap
906
+
907
+ def _indent(self, item: _ListLevel | None) -> Length | None:
908
+ """How far in this paragraph starts, adding up every reason there is.
909
+
910
+ A numbered item is the one case where nothing has to be written: its
911
+ indent comes from the numbering definition. That stops being true the
912
+ moment anything else indents it too, because an explicit indent
913
+ *replaces* the numbering's rather than adding to it — so once there is
914
+ a quote or a definition in the way, the list's own step has to be
915
+ counted back in by hand.
916
+ """
917
+ quoted = int(quote_indent(self._theme)) * self._quotes
918
+ extra = int(self._extra_indent or 0)
919
+ listed = int(self._lists[-1].indent) if self._lists else 0
920
+ if item is not None and item.number is not None and not quoted and not extra:
921
+ return None
922
+ total = quoted + extra + listed
923
+ return Emu(total) if total else None
924
+
925
+ def _column_width(self) -> Length | None:
926
+ """The width of the text column, which an image may not exceed.
927
+
928
+ The last section rather than the first: front matter opens a section
929
+ of its own, and the body an image sits in is the one after it.
930
+ """
931
+ return text_width(self._docx.sections[-1])
932
+
933
+ def _skip_html(self, token: Token, line: int | None = None) -> None:
934
+ """Warn once per line that raw HTML was dropped, and carry on.
935
+
936
+ The PDF passes HTML through, because its pipeline is HTML. Word has
937
+ nowhere to put it. Warning once per line rather than once per token
938
+ keeps a paragraph with an opening and a closing tag in it from
939
+ producing two warnings about one construct.
940
+ """
941
+ if HTML_COMMENT.match(token.content):
942
+ return
943
+ if line is None and token.map:
944
+ line = token.map[0] + 1
945
+ if line in self._warned_html:
946
+ return
947
+ self._warned_html.add(line)
948
+ where = f" (line {line})" if line is not None else ""
949
+ self._warn(f"raw HTML is not converted to Word; skipped it{where}")
950
+
951
+
952
+ # --- token helpers --------------------------------------------------------
953
+
954
+
955
+ def _closing(tokens: list[Token], index: int, end: int) -> int:
956
+ """The index of the token that closes the container opened at ``index``."""
957
+ depth = 0
958
+ for offset in range(index, end):
959
+ depth += tokens[offset].nesting
960
+ if depth == 0:
961
+ return offset
962
+ return end - 1
963
+
964
+
965
+ def _href(token: Token) -> str:
966
+ """Where a link points, read off the token that opened it."""
967
+ value = token.attrGet("href")
968
+ return value if isinstance(value, str) else ""
969
+
970
+
971
+ def _start(token: Token) -> int:
972
+ """Where an ordered list is told to start counting."""
973
+ value = token.attrGet("start")
974
+ try:
975
+ return int(str(value))
976
+ except (TypeError, ValueError):
977
+ return 1
978
+
979
+
980
+ def _footnote_number(token: Token) -> int:
981
+ """Footnotes are numbered from zero in the token stream and one on paper."""
982
+ return int(token.meta.get("id", 0)) + 1
983
+
984
+
985
+ def _table_rows(
986
+ tokens: list[Token], start: int, end: int
987
+ ) -> list[tuple[bool, list[tuple[Token | None, str | None]]]]:
988
+ """Read a table's tokens into rows of cells, with the header row marked."""
989
+ rows: list[tuple[bool, list[tuple[Token | None, str | None]]]] = []
990
+ header = False
991
+ cells: list[tuple[Token | None, str | None]] = []
992
+ for index in range(start, end):
993
+ token = tokens[index]
994
+ if token.type == "thead_open":
995
+ header = True
996
+ elif token.type == "thead_close":
997
+ header = False
998
+ elif token.type == "tr_open":
999
+ cells = []
1000
+ elif token.type == "tr_close":
1001
+ rows.append((header, cells))
1002
+ elif token.type in {"th_open", "td_open"}:
1003
+ content = tokens[index + 1] if index + 1 < end else None
1004
+ alignment = token.attrGet("style")
1005
+ cells.append(
1006
+ (
1007
+ content
1008
+ if content is not None and content.type == "inline"
1009
+ else None,
1010
+ alignment if isinstance(alignment, str) else None,
1011
+ )
1012
+ )
1013
+ return rows
1014
+
1015
+
1016
+ def _why_unusable(exc: BaseException) -> str:
1017
+ """Why an image could not be embedded, in the words a person would use."""
1018
+ if isinstance(exc, OSError):
1019
+ return "it could not be read"
1020
+ if isinstance(exc, InvalidImageStreamError | UnexpectedEndOfFileError):
1021
+ return "the file is damaged, or is not the picture it claims to be"
1022
+ return "Word has no way to hold that image"
1023
+
1024
+
1025
+ def _fit(picture: Any, available: Length | None) -> None:
1026
+ """Scale a picture down to the text column, keeping its proportions."""
1027
+ if available is None or picture.width <= available:
1028
+ return
1029
+ height = int(picture.height * int(available) / int(picture.width))
1030
+ picture.width = available
1031
+ picture.height = Emu(height)
1032
+
1033
+
1034
+ #: Which character formatting each inline token turns on and off.
1035
+ _OPENS = {"strong_open": "bold", "em_open": "italic", "s_open": "strike"}
1036
+ _CLOSES = {"strong_close": "bold", "em_close": "italic", "s_close": "strike"}
1037
+
1038
+ #: The block tokens the walk acts on. Everything else — the closing halves,
1039
+ #: and the tokens that only mark structure another handler already consumed —
1040
+ #: is stepped over.
1041
+ _BLOCKS = {
1042
+ "heading_open": _Builder._heading,
1043
+ "paragraph_open": _Builder._paragraph_block,
1044
+ "fence": _Builder._code,
1045
+ "code_block": _Builder._code,
1046
+ "blockquote_open": _Builder._quote,
1047
+ "bullet_list_open": _Builder._list,
1048
+ "ordered_list_open": _Builder._list,
1049
+ "list_item_open": _Builder._item,
1050
+ "hr": _Builder._rule,
1051
+ "dt_open": _Builder._term,
1052
+ "dd_open": _Builder._definition,
1053
+ "table_open": _Builder._table,
1054
+ "html_block": _Builder._html,
1055
+ "footnote_block_open": _Builder._footnote_block,
1056
+ "footnote_open": _Builder._footnote,
1057
+ }
1058
+
1059
+
1060
+ __all__ = ["render_docx"]