amethyst-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
amethyst/ooxml.py ADDED
@@ -0,0 +1,549 @@
1
+ """The OOXML that python-docx has no API for.
2
+
3
+ python-docx covers paragraphs, runs, styles, tables and pictures, which is
4
+ most of the job. What it does not cover is the furniture: hyperlinks,
5
+ bookmarks, shading, borders, field codes, a repeating table header, and a
6
+ numbering instance that starts again at one. Each of those is a handful of
7
+ elements built by hand, and they live here — together, and away from the
8
+ walker — because they are fiddly, individually testable, and will be returned
9
+ to.
10
+
11
+ This module sits under neither pipeline, which is the point. Both the Word
12
+ renderer and the theme compiler need these elements, and a module inside
13
+ ``render/`` that ``theme/`` imported would close a circle: importing it runs
14
+ ``render/__init__``, which imports the walker, which imports the theme
15
+ compiler again.
16
+
17
+ The one thing that matters everywhere below: Word validates a document against
18
+ the schema when it opens it, and rejects the whole file when a child element
19
+ is in the wrong place. So the *order* of the children of a properties element
20
+ is as load-bearing as their content. Every sequence the format demands is
21
+ written out once in ``CHILD_ORDER``, and every helper inserts through it
22
+ rather than appending and hoping.
23
+
24
+ Nothing here imports anything else from Amethyst. These are facts about the
25
+ file format, not about this program.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import re
31
+ from collections.abc import Iterable, Sequence
32
+ from typing import Any
33
+
34
+ from docx.opc.constants import RELATIONSHIP_TYPE
35
+ from docx.oxml import OxmlElement
36
+ from docx.oxml.ns import qn
37
+
38
+ #: The order the schema demands of the children of each properties element,
39
+ #: for the ones this project writes into. Only the elements before the one
40
+ #: being inserted matter, but the sequences are written out in full: a partial
41
+ #: list is a trap for whoever adds the next helper.
42
+ CHILD_ORDER: dict[str, tuple[str, ...]] = {
43
+ "pPr": (
44
+ "w:pStyle",
45
+ "w:keepNext",
46
+ "w:keepLines",
47
+ "w:pageBreakBefore",
48
+ "w:framePr",
49
+ "w:widowControl",
50
+ "w:numPr",
51
+ "w:suppressLineNumbers",
52
+ "w:pBdr",
53
+ "w:shd",
54
+ "w:tabs",
55
+ "w:suppressAutoHyphens",
56
+ "w:kinsoku",
57
+ "w:wordWrap",
58
+ "w:overflowPunct",
59
+ "w:topLinePunct",
60
+ "w:autoSpaceDE",
61
+ "w:autoSpaceDN",
62
+ "w:bidi",
63
+ "w:adjustRightInd",
64
+ "w:snapToGrid",
65
+ "w:spacing",
66
+ "w:ind",
67
+ "w:contextualSpacing",
68
+ "w:mirrorIndents",
69
+ "w:suppressOverlap",
70
+ "w:jc",
71
+ "w:textDirection",
72
+ "w:textAlignment",
73
+ "w:textboxTightWrap",
74
+ "w:outlineLvl",
75
+ "w:divId",
76
+ "w:cnfStyle",
77
+ "w:rPr",
78
+ "w:sectPr",
79
+ "w:pPrChange",
80
+ ),
81
+ "rPr": (
82
+ "w:rStyle",
83
+ "w:rFonts",
84
+ "w:b",
85
+ "w:bCs",
86
+ "w:i",
87
+ "w:iCs",
88
+ "w:caps",
89
+ "w:smallCaps",
90
+ "w:strike",
91
+ "w:dstrike",
92
+ "w:outline",
93
+ "w:shadow",
94
+ "w:emboss",
95
+ "w:imprint",
96
+ "w:noProof",
97
+ "w:snapToGrid",
98
+ "w:vanish",
99
+ "w:webHidden",
100
+ "w:color",
101
+ "w:spacing",
102
+ "w:w",
103
+ "w:kern",
104
+ "w:position",
105
+ "w:sz",
106
+ "w:szCs",
107
+ "w:highlight",
108
+ "w:u",
109
+ "w:effect",
110
+ "w:bdr",
111
+ "w:shd",
112
+ "w:fitText",
113
+ "w:vertAlign",
114
+ "w:rtl",
115
+ "w:cs",
116
+ "w:em",
117
+ "w:lang",
118
+ "w:eastAsianLayout",
119
+ "w:specVanish",
120
+ "w:oMath",
121
+ ),
122
+ "tblPr": (
123
+ "w:tblStyle",
124
+ "w:tblpPr",
125
+ "w:tblOverlap",
126
+ "w:bidiVisual",
127
+ "w:tblStyleRowBandSize",
128
+ "w:tblStyleColBandSize",
129
+ "w:tblW",
130
+ "w:jc",
131
+ "w:tblCellSpacing",
132
+ "w:tblInd",
133
+ "w:tblBorders",
134
+ "w:shd",
135
+ "w:tblLayout",
136
+ "w:tblCellMar",
137
+ "w:tblLook",
138
+ "w:tblCaption",
139
+ "w:tblDescription",
140
+ ),
141
+ "tcPr": (
142
+ "w:cnfStyle",
143
+ "w:tcW",
144
+ "w:gridSpan",
145
+ "w:hMerge",
146
+ "w:vMerge",
147
+ "w:tcBorders",
148
+ "w:shd",
149
+ "w:noWrap",
150
+ "w:tcMar",
151
+ "w:textDirection",
152
+ "w:tcFitText",
153
+ "w:vAlign",
154
+ "w:hideMark",
155
+ ),
156
+ "trPr": (
157
+ "w:cnfStyle",
158
+ "w:divId",
159
+ "w:gridBefore",
160
+ "w:gridAfter",
161
+ "w:wBefore",
162
+ "w:wAfter",
163
+ "w:cantSplit",
164
+ "w:trHeight",
165
+ "w:tblHeader",
166
+ "w:tblCellSpacing",
167
+ "w:jc",
168
+ "w:hidden",
169
+ ),
170
+ "pBdr": ("w:top", "w:left", "w:bottom", "w:right", "w:between", "w:bar"),
171
+ "tblBorders": (
172
+ "w:top",
173
+ "w:left",
174
+ "w:bottom",
175
+ "w:right",
176
+ "w:insideH",
177
+ "w:insideV",
178
+ ),
179
+ "tcBorders": (
180
+ "w:top",
181
+ "w:left",
182
+ "w:bottom",
183
+ "w:right",
184
+ "w:insideH",
185
+ "w:insideV",
186
+ "w:tl2br",
187
+ "w:tr2bl",
188
+ ),
189
+ }
190
+
191
+ #: Which element holds the borders, for each properties element that can have
192
+ #: them. The three are the same shape and spelled three different ways, which
193
+ #: is the sort of thing worth stating once.
194
+ BORDER_CONTAINER = {
195
+ "pPr": "w:pBdr",
196
+ "tblPr": "w:tblBorders",
197
+ "tcPr": "w:tcBorders",
198
+ }
199
+
200
+ #: Border width, in the eighths of a point the format measures it in. Six is
201
+ #: the 0.75pt that a CSS ``1px`` rule comes to on paper.
202
+ HAIRLINE = 6
203
+
204
+ #: What Word allows in a bookmark name: letters, digits and underscores, up to
205
+ #: forty characters, not starting with a digit. Heading anchors are slugs with
206
+ #: hyphens in them, so every one of them has to be translated.
207
+ BOOKMARK_UNSAFE = re.compile(r"[^0-9A-Za-z_]+")
208
+ BOOKMARK_MAX_LENGTH = 40
209
+
210
+
211
+ # --- properties -----------------------------------------------------------
212
+
213
+
214
+ def properties_child(properties: Any, tag: str) -> Any:
215
+ """Return a child of a properties element, adding it in schema order."""
216
+ existing = properties.find(qn(tag))
217
+ if existing is not None:
218
+ return existing
219
+ element = OxmlElement(tag)
220
+ _insert(properties, element, tag)
221
+ return element
222
+
223
+
224
+ def _insert(parent: Any, element: Any, tag: str) -> None:
225
+ """Put ``element`` in front of the first sibling that must follow it.
226
+
227
+ Insertion is done here rather than through python-docx's own
228
+ ``insert_element_before`` because that method belongs to the element
229
+ classes python-docx registers, and half the elements built in this module
230
+ — ``w:pBdr``, ``w:tblBorders`` — are not among them and come back as plain
231
+ lxml.
232
+ """
233
+ sequence = CHILD_ORDER[_local_name(parent)]
234
+ following = {qn(name) for name in sequence[sequence.index(tag) + 1 :]}
235
+ for child in parent:
236
+ if child.tag in following:
237
+ child.addprevious(element)
238
+ return
239
+ parent.append(element)
240
+
241
+
242
+ def shade(properties: Any, fill: str) -> None:
243
+ """Fill the background of whatever ``properties`` describes.
244
+
245
+ Serves a paragraph, a run, a table and a table cell alike: ``w:shd`` is
246
+ spelled the same in all four, and only its position among its siblings
247
+ changes.
248
+ """
249
+ element = properties_child(properties, "w:shd")
250
+ element.set(qn("w:val"), "clear")
251
+ element.set(qn("w:color"), "auto")
252
+ element.set(qn("w:fill"), _hex(fill))
253
+
254
+
255
+ def set_borders(
256
+ properties: Any,
257
+ edges: Iterable[str],
258
+ *,
259
+ color: str,
260
+ size: int = HAIRLINE,
261
+ space: int = 0,
262
+ style: str = "single",
263
+ ) -> None:
264
+ """Draw ``edges`` — "top", "left", "bottom", "right", "insideH", … .
265
+
266
+ ``size`` is in eighths of a point and ``space`` in points, because that is
267
+ what the format measures them in and translating here would only hide it.
268
+ """
269
+ container = properties_child(properties, BORDER_CONTAINER[_local_name(properties)])
270
+ for edge in edges:
271
+ tag = f"w:{edge}"
272
+ element = OxmlElement(tag)
273
+ element.set(qn("w:val"), style)
274
+ element.set(qn("w:sz"), str(size))
275
+ element.set(qn("w:space"), str(space))
276
+ element.set(qn("w:color"), _hex(color))
277
+ _insert(container, element, tag)
278
+
279
+
280
+ def clear_properties_child(properties: Any, tag: str) -> None:
281
+ """Take a child off a properties element, if it is there at all.
282
+
283
+ The counterpart of :func:`properties_child`, and needed because Word's
284
+ built-in styles arrive carrying things a theme has to undo rather than
285
+ override — a rule drawn under ``Title`` in an accent colour that belongs
286
+ to no theme here, or the stray numbering reference on ``Subtitle``.
287
+ """
288
+ element = properties.find(qn(tag))
289
+ if element is not None:
290
+ properties.remove(element)
291
+
292
+
293
+ def repeat_as_header(row: Any) -> None:
294
+ """Mark a table row as the header, so it repeats on every page.
295
+
296
+ The counterpart of ``thead { display: table-header-group }`` in the PDF
297
+ stylesheet: a table split across a page break is only readable because its
298
+ column headings come back at the top of the next one.
299
+ """
300
+ properties_child(row._tr.get_or_add_trPr(), "w:tblHeader")
301
+
302
+
303
+ # --- links and bookmarks --------------------------------------------------
304
+
305
+
306
+ def bookmark_name(anchor: str) -> str:
307
+ """Translate a heading's HTML id into a name Word will accept.
308
+
309
+ Word's rules are much tighter than HTML's — no hyphens, no leading digit,
310
+ forty characters — so this is lossy by necessity. It is deterministic,
311
+ which is the property that matters: the heading and the link that points
312
+ at it are translated by the same function and so agree.
313
+ """
314
+ cleaned = BOOKMARK_UNSAFE.sub("_", anchor).strip("_")
315
+ if not cleaned or cleaned[0].isdigit():
316
+ cleaned = f"_{cleaned}"
317
+ return cleaned[:BOOKMARK_MAX_LENGTH]
318
+
319
+
320
+ def bookmark(paragraph: Any, name: str, identifier: int) -> None:
321
+ """Mark a paragraph as the destination of an internal link."""
322
+ start = OxmlElement("w:bookmarkStart")
323
+ start.set(qn("w:id"), str(identifier))
324
+ start.set(qn("w:name"), name)
325
+ end = OxmlElement("w:bookmarkEnd")
326
+ end.set(qn("w:id"), str(identifier))
327
+
328
+ paragraph_element = paragraph._p
329
+ properties = paragraph_element.find(qn("w:pPr"))
330
+ if properties is None:
331
+ paragraph_element.insert(0, start)
332
+ else:
333
+ properties.addnext(start)
334
+ paragraph_element.append(end)
335
+
336
+
337
+ def link(
338
+ paragraph: Any,
339
+ elements: Sequence[Any],
340
+ *,
341
+ url: str | None = None,
342
+ anchor: str | None = None,
343
+ ) -> None:
344
+ """Wrap already-written runs in a hyperlink, external or internal.
345
+
346
+ The runs are built first and moved in afterwards rather than the other way
347
+ round, so the code that renders a link's text is the same code that
348
+ renders every other run — a link's label is Markdown like any other, and
349
+ can hold bold, code and an image.
350
+ """
351
+ if not elements:
352
+ return
353
+ element = OxmlElement("w:hyperlink")
354
+ if url is not None:
355
+ relationship = paragraph.part.relate_to(
356
+ url, RELATIONSHIP_TYPE.HYPERLINK, is_external=True
357
+ )
358
+ element.set(qn("r:id"), relationship)
359
+ if anchor is not None:
360
+ element.set(qn("w:anchor"), anchor)
361
+ elements[0].addprevious(element)
362
+ for run in elements:
363
+ element.append(run)
364
+
365
+
366
+ # --- fields ---------------------------------------------------------------
367
+
368
+
369
+ def field(
370
+ paragraph: Any, instruction: str, placeholder: str = "", *, dirty: bool = False
371
+ ) -> None:
372
+ """Append a field — ``PAGE``, ``PAGEREF``, ``STYLEREF`` — to a paragraph.
373
+
374
+ A field is not a value but an instruction that Word evaluates when it
375
+ opens the document, which is the only way to write something that has to
376
+ know a page number the writer cannot know. The placeholder is what a
377
+ reader that does not evaluate fields shows instead.
378
+
379
+ ``dirty`` asks Word to evaluate the field as soon as it opens the file
380
+ rather than showing the placeholder until someone presses F9. It is what a
381
+ field with no useful placeholder — a page number nothing here can compute
382
+ — needs to arrive filled in.
383
+ """
384
+ for child in (
385
+ *field_start(instruction, dirty=dirty),
386
+ _text_run(placeholder),
387
+ field_end(),
388
+ ):
389
+ paragraph._p.append(child)
390
+
391
+
392
+ def field_start(instruction: str, *, dirty: bool = False) -> tuple[Any, ...]:
393
+ """The runs that open a field, up to where its result begins.
394
+
395
+ Separate from :func:`field` because a field's result is not always one
396
+ run in one paragraph: a table of contents is a run of paragraphs, and the
397
+ only way to write one is to open the field in the first and close it in
398
+ the last.
399
+ """
400
+ return (
401
+ _field_char("begin", dirty=dirty),
402
+ _instruction(instruction),
403
+ _field_char("separate"),
404
+ )
405
+
406
+
407
+ def field_end() -> Any:
408
+ """The run that closes a field opened by :func:`field_start`."""
409
+ return _field_char("end")
410
+
411
+
412
+ def _field_char(kind: str, *, dirty: bool = False) -> Any:
413
+ """One of a field's three markers: ``begin``, ``separate`` or ``end``.
414
+
415
+ ``dirty`` is what makes Word evaluate the field when the file opens rather
416
+ than showing whatever result was written beside it — needed by ``PAGEREF``
417
+ and ``TOC``, and not by ``PAGE`` or ``STYLEREF``, which Word computes while
418
+ it lays the page out. It is not free: a document containing any dirty field
419
+ greets the reader with an update-fields dialog.
420
+ """
421
+ run = OxmlElement("w:r")
422
+ char = OxmlElement("w:fldChar")
423
+ char.set(qn("w:fldCharType"), kind)
424
+ if dirty:
425
+ char.set(qn("w:dirty"), "true")
426
+ run.append(char)
427
+ return run
428
+
429
+
430
+ def _instruction(instruction: str) -> Any:
431
+ """The run carrying a field's instruction text, spaces and all."""
432
+ run = OxmlElement("w:r")
433
+ text = OxmlElement("w:instrText")
434
+ # Without this the leading and trailing spaces a field instruction needs
435
+ # are collapsed away, and Word reads a different instruction than the one
436
+ # that was written.
437
+ text.set(qn("xml:space"), "preserve")
438
+ text.text = f" {instruction} "
439
+ run.append(text)
440
+ return run
441
+
442
+
443
+ def _text_run(content: str) -> Any:
444
+ """A plain run of text that keeps the whitespace it was given."""
445
+ run = OxmlElement("w:r")
446
+ text = OxmlElement("w:t")
447
+ text.set(qn("xml:space"), "preserve")
448
+ text.text = content
449
+ run.append(text)
450
+ return run
451
+
452
+
453
+ # --- numbering ------------------------------------------------------------
454
+
455
+
456
+ def numbering_instance(document: Any, style_name: str, *, start: int = 1) -> int:
457
+ """Give one list its own counter, and return the id to point at it with.
458
+
459
+ Word's numbering styles carry a counter each, not a counter per list, so
460
+ every ordered list in a document that uses ``List Number`` shares one — and
461
+ the second list in a document carries on from where the first stopped
462
+ instead of starting again at one. The cure is an instance of its own per
463
+ list, pointing at the same shape and overriding where it starts.
464
+ """
465
+ numbering = document.part.numbering_part.element
466
+ abstract_id = _abstract_num_id(document, style_name)
467
+
468
+ identifier = 1 + max(
469
+ (
470
+ int(existing.get(qn("w:numId")) or 0)
471
+ for existing in numbering.findall(qn("w:num"))
472
+ ),
473
+ default=0,
474
+ )
475
+ instance = OxmlElement("w:num")
476
+ instance.set(qn("w:numId"), str(identifier))
477
+ reference = OxmlElement("w:abstractNumId")
478
+ reference.set(qn("w:val"), str(abstract_id))
479
+ instance.append(reference)
480
+
481
+ override = OxmlElement("w:lvlOverride")
482
+ override.set(qn("w:ilvl"), "0")
483
+ first = OxmlElement("w:startOverride")
484
+ first.set(qn("w:val"), str(start))
485
+ override.append(first)
486
+ instance.append(override)
487
+
488
+ # Every w:num follows every w:abstractNum, so the end of the part is the
489
+ # one position that is always right.
490
+ numbering.append(instance)
491
+ return identifier
492
+
493
+
494
+ def set_numbering(paragraph: Any, identifier: int, level: int = 0) -> None:
495
+ """Point a paragraph at a numbering instance, overriding its style's."""
496
+ properties = paragraph._p.get_or_add_pPr().get_or_add_numPr()
497
+ properties.get_or_add_ilvl().val = level
498
+ properties.get_or_add_numId().val = identifier
499
+
500
+
501
+ def _abstract_num_id(document: Any, style_name: str) -> int:
502
+ """The numbering shape a built-in list style draws its bullets from."""
503
+ style = document.styles[style_name]
504
+ reference = style.element.find(f"{qn('w:pPr')}/{qn('w:numPr')}/{qn('w:numId')}")
505
+ if reference is None:
506
+ raise KeyError(f"style {style_name!r} carries no numbering")
507
+ wanted = reference.get(qn("w:val"))
508
+ numbering = document.part.numbering_part.element
509
+ for instance in numbering.findall(qn("w:num")):
510
+ if instance.get(qn("w:numId")) == wanted:
511
+ abstract = instance.find(qn("w:abstractNumId"))
512
+ if abstract is not None:
513
+ return int(abstract.get(qn("w:val")))
514
+ raise KeyError(
515
+ f"style {style_name!r} points at numbering {wanted!r}, which is not there"
516
+ )
517
+
518
+
519
+ # --- shared ---------------------------------------------------------------
520
+
521
+
522
+ def _local_name(element: Any) -> str:
523
+ """The tag without its namespace: ``{...}pPr`` is ``pPr``."""
524
+ return str(element.tag).rpartition("}")[2]
525
+
526
+
527
+ def _hex(color: str) -> str:
528
+ """A theme colour as OOXML writes it: six digits, no leading hash."""
529
+ return color.lstrip("#").upper()
530
+
531
+
532
+ __all__ = [
533
+ "BOOKMARK_MAX_LENGTH",
534
+ "CHILD_ORDER",
535
+ "HAIRLINE",
536
+ "bookmark",
537
+ "bookmark_name",
538
+ "clear_properties_child",
539
+ "field",
540
+ "field_end",
541
+ "field_start",
542
+ "link",
543
+ "numbering_instance",
544
+ "properties_child",
545
+ "repeat_as_header",
546
+ "set_borders",
547
+ "set_numbering",
548
+ "shade",
549
+ ]
@@ -0,0 +1,20 @@
1
+ """The parse layer: Markdown text in, a token stream and metadata out.
2
+
3
+ Nothing here knows about PDF or DOCX. :mod:`amethyst.document` assembles these
4
+ pieces into the ``Document`` the renderers are handed.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from amethyst.parse.assets import Asset, AssetKind, resolve_assets
10
+ from amethyst.parse.frontmatter import parse_frontmatter, split_frontmatter
11
+ from amethyst.parse.markdown import build_parser
12
+
13
+ __all__ = [
14
+ "Asset",
15
+ "AssetKind",
16
+ "build_parser",
17
+ "parse_frontmatter",
18
+ "resolve_assets",
19
+ "split_frontmatter",
20
+ ]
@@ -0,0 +1,106 @@
1
+ """Resolution of the files a document points at.
2
+
3
+ Markdown says ``![](diagram.png)`` and means "next to this file". Both
4
+ renderers need that turned into something they can open — WeasyPrint fetches a
5
+ URL, python-docx opens a path — and resolving it once, here, is what stops the
6
+ two from disagreeing about where the file was.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass
12
+ from enum import Enum
13
+ from pathlib import Path
14
+ from urllib.parse import unquote, urlsplit
15
+
16
+ from markdown_it.token import Token
17
+
18
+ #: Schemes that name something to be fetched over the network. Anything else
19
+ #: with a scheme (``data:``, ``mailto:``, ``file:``) needs no resolving and is
20
+ #: passed through to the renderer exactly as written.
21
+ REMOTE_SCHEMES = frozenset({"http", "https"})
22
+
23
+
24
+ class AssetKind(str, Enum):
25
+ """What the reference was written as. The value doubles as its noun."""
26
+
27
+ image = "image"
28
+ link = "link"
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class Asset:
33
+ """One outward reference from the document, and where it landed."""
34
+
35
+ kind: AssetKind
36
+ #: The target exactly as written in the Markdown, for error messages.
37
+ reference: str
38
+ #: 1-based line in the source file, or ``None`` if the token carried no map.
39
+ line: int | None = None
40
+ #: The local file the reference resolves to, or ``None`` if it is not one.
41
+ path: Path | None = None
42
+ is_remote: bool = False
43
+
44
+ @property
45
+ def is_missing(self) -> bool:
46
+ """True for a local reference with no file at the resolved path."""
47
+ return self.path is not None and not self.path.is_file()
48
+
49
+
50
+ def resolve_assets(tokens: list[Token], base_dir: Path) -> list[Asset]:
51
+ """Resolve local references against ``base_dir`` and rewrite image sources.
52
+
53
+ An image whose source names a local file that exists is rewritten in place
54
+ to an absolute path, which is what both renderers want. Everything else —
55
+ remote URLs, data URIs, files that are not there — is left exactly as the
56
+ author typed it, so the eventual error names their reference rather than
57
+ one this function invented.
58
+
59
+ Returns every reference worth knowing about later: images of all kinds, and
60
+ links that point at a local file. Bare fragments (``#section``) and
61
+ ``mailto:`` links resolve to nothing and are left out.
62
+ """
63
+ assets: list[Asset] = []
64
+ for token in tokens:
65
+ if token.type != "inline" or not token.children:
66
+ continue
67
+ line = token.map[0] + 1 if token.map else None
68
+ for child in token.children:
69
+ if child.type == "image":
70
+ asset = _resolve(AssetKind.image, child.attrGet("src"), base_dir, line)
71
+ if asset is None:
72
+ continue
73
+ assets.append(asset)
74
+ if asset.path is not None and not asset.is_missing:
75
+ child.attrSet("src", str(asset.path))
76
+ elif child.type == "link_open":
77
+ asset = _resolve(AssetKind.link, child.attrGet("href"), base_dir, line)
78
+ # Links are recorded, never rewritten: an absolute filesystem
79
+ # path in an href would be a worse link than the relative one.
80
+ if asset is not None and asset.path is not None:
81
+ assets.append(asset)
82
+ return assets
83
+
84
+
85
+ def _resolve(
86
+ kind: AssetKind, reference: object, base_dir: Path, line: int | None
87
+ ) -> Asset | None:
88
+ """Classify one reference, and locate it if it names a local file."""
89
+ if not isinstance(reference, str) or not reference:
90
+ return None
91
+
92
+ parts = urlsplit(reference)
93
+ if parts.scheme or parts.netloc:
94
+ if parts.scheme not in REMOTE_SCHEMES:
95
+ return None
96
+ return Asset(kind, reference, line, is_remote=True)
97
+
98
+ # Strip the fragment and query the URL syntax allows, then undo the
99
+ # percent-encoding: `my%20image.png` on the page is `my image.png` on disk.
100
+ target = unquote(parts.path)
101
+ if not target:
102
+ return None
103
+
104
+ candidate = Path(target)
105
+ resolved = candidate if candidate.is_absolute() else base_dir / candidate
106
+ return Asset(kind, reference, line, path=resolved.resolve())