mdhtml2docx 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mdhtml2docx/__init__.py +3 -0
- mdhtml2docx/asvocab.py +88 -0
- mdhtml2docx/convert.py +919 -0
- mdhtml2docx/hilite.py +27 -0
- mdhtml2docx/schemas/SOURCES.md +7 -0
- mdhtml2docx/schemas/dml-chart.xsd +1499 -0
- mdhtml2docx/schemas/dml-chartDrawing.xsd +146 -0
- mdhtml2docx/schemas/dml-diagram.xsd +1085 -0
- mdhtml2docx/schemas/dml-lockedCanvas.xsd +11 -0
- mdhtml2docx/schemas/dml-main.xsd +3081 -0
- mdhtml2docx/schemas/dml-picture.xsd +23 -0
- mdhtml2docx/schemas/dml-wordprocessingDrawing.xsd +287 -0
- mdhtml2docx/schemas/shared-commonSimpleTypes.xsd +172 -0
- mdhtml2docx/schemas/shared-customXmlSchemaProperties.xsd +18 -0
- mdhtml2docx/schemas/shared-math.xsd +582 -0
- mdhtml2docx/schemas/shared-relationshipReference.xsd +25 -0
- mdhtml2docx/schemas/wml.xsd +3643 -0
- mdhtml2docx/schemas/xml.xsd +287 -0
- mdhtml2docx/styles.py +85 -0
- mdhtml2docx/templates/reference.docx +0 -0
- mdhtml2docx/validate.py +38 -0
- mdhtml2docx/wml.py +110 -0
- mdhtml2docx/word.py +136 -0
- mdhtml2docx/word.sdef +9661 -0
- mdhtml2docx-0.1.0.dist-info/METADATA +96 -0
- mdhtml2docx-0.1.0.dist-info/RECORD +29 -0
- mdhtml2docx-0.1.0.dist-info/WHEEL +5 -0
- mdhtml2docx-0.1.0.dist-info/licenses/LICENSE +202 -0
- mdhtml2docx-0.1.0.dist-info/top_level.txt +1 -0
mdhtml2docx/convert.py
ADDED
|
@@ -0,0 +1,919 @@
|
|
|
1
|
+
"""Convert MDHTML fragments to docx.
|
|
2
|
+
|
|
3
|
+
Write-only, reference-archive architecture: the reference template supplies styles/theme/fonts,
|
|
4
|
+
we generate word/document.xml (plus footnotes/numbering/media parts as needed) into a copy of its
|
|
5
|
+
archive. Block and inline walkers mirror the MDHTML element inventory; STYLE_MAP names every
|
|
6
|
+
style we emit."""
|
|
7
|
+
import posixpath, re, zipfile
|
|
8
|
+
from copy import deepcopy
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from fast5ever import Comment, Element, Node, Text
|
|
11
|
+
from lxml import etree
|
|
12
|
+
from mdhtml import parse_mdhtml
|
|
13
|
+
from mdhtml.export import REFTYPES, SCHEMES, decode_raw, group_plan, mustache_kind, ref_tokens, ref_variant
|
|
14
|
+
from .styles import STYLE_MAP, style_id, theme_styles
|
|
15
|
+
from .styles import ref_path as _refpath
|
|
16
|
+
from .wml import *
|
|
17
|
+
from .wml import qn
|
|
18
|
+
from .hilite import segments, tokenize
|
|
19
|
+
|
|
20
|
+
__all__ = ['convert', 'mustache_fields', 'jinja_literal']
|
|
21
|
+
|
|
22
|
+
def _sid(key): return style_id(STYLE_MAP[key])
|
|
23
|
+
|
|
24
|
+
BLOCK_TAGS = set(('address article aside blockquote details dialog div dl fieldset figure footer form h1 h2 h3 h4 h5 h6 '
|
|
25
|
+
'header hgroup hr main menu nav ol p pre search section table ul').split())
|
|
26
|
+
INERT_TAGS = {'base', 'link', 'meta', 'style', 'template', 'title'}
|
|
27
|
+
|
|
28
|
+
def _tag(el): return el.name
|
|
29
|
+
def _get(el, key, default=None): return el.attrs.get(key, default)
|
|
30
|
+
def _els(el): return [n for n in el.children if isinstance(n, Element)]
|
|
31
|
+
def _walk(el):
|
|
32
|
+
yield el
|
|
33
|
+
for child in _els(el): yield from _walk(child)
|
|
34
|
+
def _classes(el): return (_get(el, 'class') or '').split()
|
|
35
|
+
def _is_raw(el): return _tag(el) == 'script' and _get(el, 'type') == 'application/vnd.mdhtml.raw'
|
|
36
|
+
|
|
37
|
+
def parse_frag(src):
|
|
38
|
+
"Parse an MDHTML body fragment, or return an existing mutable fragment"
|
|
39
|
+
if isinstance(src, Node): return src
|
|
40
|
+
if not isinstance(src, str): raise TypeError('input must be an MDHTML string or fast5ever node')
|
|
41
|
+
return parse_mdhtml(src)
|
|
42
|
+
|
|
43
|
+
class Converter:
|
|
44
|
+
def __init__(self, reference=None, base=None, reftypes=None, number_headings=None, tmpl=None):
|
|
45
|
+
self.tmpl = tmpl
|
|
46
|
+
if reference is None: reference = [_refpath()] + (['github_light'] if tokenize else [])
|
|
47
|
+
elif not isinstance(reference, (list, tuple)): reference = [reference]
|
|
48
|
+
self.refz = zipfile.ZipFile(reference[0] or _refpath())
|
|
49
|
+
self.contribs = reference[1:] # styles-only contributors: docx paths, styles/numbering .xml paths, theme names
|
|
50
|
+
self.base = Path(base or '.')
|
|
51
|
+
self.rels = [] # (rId, type, target, external?) beyond the template's
|
|
52
|
+
self.fn_rels = [] # same, but for the footnotes part
|
|
53
|
+
self.media = {} # archive name -> bytes
|
|
54
|
+
self.warnings = []
|
|
55
|
+
self._rid = 1000 # clear of template rIds
|
|
56
|
+
self.bq = 0 # blockquote nesting depth
|
|
57
|
+
self._bkid = 0 # bookmark id counter
|
|
58
|
+
self._imgn = 0 # image part counter (doubles as docPr id)
|
|
59
|
+
self._urlrids = {} # hyperlink URL -> rId
|
|
60
|
+
self.nums = [] # (numId, abstractNumId, start) per list instance
|
|
61
|
+
self.fndefs = {} # endnote li elements by id, harvested before the walk
|
|
62
|
+
self.fnids = {} # endnote id -> footnote w:id
|
|
63
|
+
self.fnotes = [] # (w:id, [footnote blocks])
|
|
64
|
+
self.stubs = {} # undefined custom-style name -> (kind, styleId)
|
|
65
|
+
self.first = True # next body paragraph is a 'First Paragraph' (doc start; reset after FIRST_AFTER blocks)
|
|
66
|
+
self._bknames = {} # element id -> Word-legal bookmark name
|
|
67
|
+
self.has_fields = False
|
|
68
|
+
self.has_controls = False
|
|
69
|
+
self.bound = [] # distinct bound-control names, in first-appearance order
|
|
70
|
+
self.contrib_numels, self.contrib_styleels = [], []
|
|
71
|
+
self.reftypes = REFTYPES | (reftypes or {})
|
|
72
|
+
tdoc = etree.fromstring(self.refz.read('word/document.xml'))
|
|
73
|
+
self.sectpr = deepcopy(tdoc.find(f'{{{W}}}body/{{{W}}}sectPr'))
|
|
74
|
+
pg, mar = self.sectpr.find(qn('w:pgSz')), self.sectpr.find(qn('w:pgMar'))
|
|
75
|
+
self.content_w = int(pg.get(qn('w:w'))) - int(mar.get(qn('w:left'))) - int(mar.get(qn('w:right')))
|
|
76
|
+
self.sroot = etree.fromstring(self.refz.read('word/styles.xml'))
|
|
77
|
+
for r in self.contribs: self._merge_styles(r)
|
|
78
|
+
self.refstyles = {s.find(qn('w:name')).get(qn('w:val')).lower(): s.get(qn('w:styleId')) for s in self.sroot.iter(qn('w:style'))}
|
|
79
|
+
if missing := [n for n in STYLE_MAP.values() if n.lower() not in self.refstyles]:
|
|
80
|
+
raise ValueError(f'reference doc lacks dialect styles (map/template drift?): {missing}')
|
|
81
|
+
self.hlstyles = {n.removeprefix('hl ').replace(' ', '.'): sid for n, sid in self.refstyles.items() if n.startswith('hl ')}
|
|
82
|
+
self.tmplnum = (etree.fromstring(self.refz.read('word/numbering.xml')) if 'word/numbering.xml' in self.refz.namelist() else None)
|
|
83
|
+
used = [int(v) for e in ([] if self.tmplnum is None else self.tmplnum.iter())
|
|
84
|
+
for a in ('w:numId', 'w:abstractNumId') if (v := e.get(qn(a))) and v.lstrip('-').isdigit()]
|
|
85
|
+
self._numid = self._absbase = max(used, default=-1) + 1 # num and abstract ids both offset past the template's
|
|
86
|
+
if isinstance(number_headings, str):
|
|
87
|
+
if number_headings not in SCHEMES: raise ValueError(f'unknown numbering scheme {number_headings!r}')
|
|
88
|
+
number_headings = SCHEMES[number_headings]
|
|
89
|
+
self.scheme = list(number_headings.items()) if number_headings else None
|
|
90
|
+
self.headnum = None
|
|
91
|
+
if number_headings and self.sroot.find(f'{{{W}}}style[@{{{W}}}styleId="Heading1"]/{{{W}}}pPr/{{{W}}}numPr') is None:
|
|
92
|
+
self._numid += 1
|
|
93
|
+
self.headnum = self._numid
|
|
94
|
+
self._adopt_contrib_nums()
|
|
95
|
+
|
|
96
|
+
def _merge_styles(self, ref):
|
|
97
|
+
"Merge a contributor's w:style elements into self.sroot, later-wins on style id or name"
|
|
98
|
+
def keys(e):
|
|
99
|
+
nm = e.find(qn('w:name'))
|
|
100
|
+
return {(e.get(qn('w:styleId')) or '').lower(), '' if nm is None else nm.get(qn('w:val')).lower()} - {''}
|
|
101
|
+
new = (etree.fromstring(zipfile.ZipFile(ref).read('word/styles.xml')).findall(qn('w:style'))
|
|
102
|
+
if isinstance(ref, Path) and ref.suffix == '.docx' or str(ref).endswith('.docx')
|
|
103
|
+
else self._xml_contrib(ref) if str(ref).endswith('.xml') else theme_styles(ref))
|
|
104
|
+
for s in new:
|
|
105
|
+
ks = keys(s)
|
|
106
|
+
for old in [o for o in self.sroot.findall(qn('w:style')) if keys(o) & ks]: self.sroot.remove(old)
|
|
107
|
+
self.sroot.append(s)
|
|
108
|
+
|
|
109
|
+
def _xml_contrib(self, ref):
|
|
110
|
+
"Styles from a raw .xml contributor, stashing its abstractNum/num elements for numbering adoption"
|
|
111
|
+
root = etree.parse(str(ref)).getroot()
|
|
112
|
+
self.contrib_numels += root.findall(qn('w:abstractNum')) + root.findall(qn('w:num'))
|
|
113
|
+
styles = root.findall(qn('w:style'))
|
|
114
|
+
self.contrib_styleels += styles
|
|
115
|
+
return styles
|
|
116
|
+
|
|
117
|
+
def _adopt_contrib_nums(self):
|
|
118
|
+
"Renumber .xml contributors' numbering ids past ours, remapping their styles' numPr references"
|
|
119
|
+
self.xabs = [e for e in self.contrib_numels if etree.QName(e).localname == 'abstractNum']
|
|
120
|
+
self.xnums = [e for e in self.contrib_numels if etree.QName(e).localname == 'num']
|
|
121
|
+
amap, nmap = {}, {}
|
|
122
|
+
for e in self.xabs:
|
|
123
|
+
self._numid += 1
|
|
124
|
+
amap[e.get(qn('w:abstractNumId'))] = self._numid
|
|
125
|
+
e.set(qn('w:abstractNumId'), str(self._numid))
|
|
126
|
+
for e in self.xnums:
|
|
127
|
+
self._numid += 1
|
|
128
|
+
nmap[e.get(qn('w:numId'))] = self._numid
|
|
129
|
+
e.set(qn('w:numId'), str(self._numid))
|
|
130
|
+
ref = e.find(qn('w:abstractNumId'))
|
|
131
|
+
if ref is not None and ref.get(qn('w:val')) in amap: ref.set(qn('w:val'), str(amap[ref.get(qn('w:val'))]))
|
|
132
|
+
for s in self.contrib_styleels:
|
|
133
|
+
for nid in s.iter(qn('w:numId')):
|
|
134
|
+
if nid.get(qn('w:val')) in nmap: nid.set(qn('w:val'), str(nmap[nid.get(qn('w:val'))]))
|
|
135
|
+
|
|
136
|
+
def hlsid(self, scope):
|
|
137
|
+
"Hl* style id for a dotted scope: exact, else progressively shorter prefixes (tree-sitter resolution)"
|
|
138
|
+
parts = (scope or '').split('.')
|
|
139
|
+
while parts:
|
|
140
|
+
if sid := self.hlstyles.get('.'.join(parts)): return sid
|
|
141
|
+
parts.pop()
|
|
142
|
+
|
|
143
|
+
def rid(self):
|
|
144
|
+
self._rid += 1
|
|
145
|
+
return f'rId{self._rid}'
|
|
146
|
+
def warn(self, msg): self.warnings.append(msg)
|
|
147
|
+
|
|
148
|
+
# ---- inline level -------------------------------------------------------
|
|
149
|
+
def rpr(self, fmt):
|
|
150
|
+
"w:rPr for a formatting context dict, in CT_RPr child order; None if empty"
|
|
151
|
+
kids = []
|
|
152
|
+
if s := fmt.get('rstyle'): kids.append(E('w:rStyle', {'w:val': s}))
|
|
153
|
+
if fmt.get('b'): kids += [E('w:b'), E('w:bCs')]
|
|
154
|
+
if fmt.get('i'): kids += [E('w:i'), E('w:iCs')]
|
|
155
|
+
if fmt.get('strike'): kids.append(E('w:strike'))
|
|
156
|
+
if fmt.get('mark'): kids.append(E('w:highlight', {'w:val': 'yellow'}))
|
|
157
|
+
if fmt.get('u'): kids.append(E('w:u', {'w:val': 'single'}))
|
|
158
|
+
if v := fmt.get('vert'): kids.append(E('w:vertAlign', {'w:val': v}))
|
|
159
|
+
return E('w:rPr', *kids) if kids else None
|
|
160
|
+
|
|
161
|
+
def text_runs(self, text, fmt):
|
|
162
|
+
"Runs for a text node; newlines collapse to spaces (pre-context text never comes here)"
|
|
163
|
+
text = re.sub(r'\s+', ' ', text)
|
|
164
|
+
if not text: return []
|
|
165
|
+
t = E('w:t', text)
|
|
166
|
+
if text != text.strip(): t.set(qn('xml:space'), 'preserve')
|
|
167
|
+
return [E('w:r', self.rpr(fmt), t)]
|
|
168
|
+
|
|
169
|
+
def link(self, el, fmt):
|
|
170
|
+
"w:hyperlink for `a`: internal '#x' -> anchor, external -> relationship (deduped per URL); data-ref -> field"
|
|
171
|
+
if _get(el, 'data-ref') is not None:
|
|
172
|
+
fld = self.ref_fld(el, fmt)
|
|
173
|
+
return self.ref_prefix(el, fmt) + fld
|
|
174
|
+
href = _get(el, 'href')
|
|
175
|
+
if not href: return self.runs(el, fmt)
|
|
176
|
+
runs = self.runs(el, fmt | {'rstyle': _sid('hyperlink')})
|
|
177
|
+
if href.startswith('#'): return [E('w:hyperlink', {'w:anchor': self.bkname(href[1:])}, *runs)]
|
|
178
|
+
return self.external_link(href, runs)
|
|
179
|
+
|
|
180
|
+
def external_link(self, href, runs):
|
|
181
|
+
if href not in self._urlrids:
|
|
182
|
+
self._urlrids[href] = self.rid()
|
|
183
|
+
self.rels.append((self._urlrids[href], f'{R}/hyperlink', href, True))
|
|
184
|
+
return [E('w:hyperlink', {'r:id': self._urlrids[href]}, *runs)]
|
|
185
|
+
|
|
186
|
+
REFSWITCH = dict(full=r'\w', rel=r'\r', leaf=r'\n', text='', page=None)
|
|
187
|
+
|
|
188
|
+
def bkname(self, id):
|
|
189
|
+
"Word-legal bookmark name for `id` (letter first, word chars only), stable within the document"
|
|
190
|
+
if id not in self._bknames:
|
|
191
|
+
nm = re.sub(r'\W', '_', id)
|
|
192
|
+
if not nm[:1].isalpha(): nm = 'B' + nm
|
|
193
|
+
while nm in self._bknames.values(): nm += '_'
|
|
194
|
+
self._bknames[id] = nm
|
|
195
|
+
return self._bknames[id]
|
|
196
|
+
|
|
197
|
+
def ref_prefix(self, el, fmt, plural=False):
|
|
198
|
+
"Literal runs before a reference field: override text, the type prefix word, or nothing for bare and caption refs"
|
|
199
|
+
pre = el.to_text().strip()
|
|
200
|
+
if not pre:
|
|
201
|
+
tgt = (_get(el, 'href') or '#')[1:]
|
|
202
|
+
if 'bare' in ref_tokens(_get(el, 'data-ref')) or self.reftarget.get(tgt) == 'caption': return []
|
|
203
|
+
t = tgt.split('-')[0]
|
|
204
|
+
if t not in self.reftypes: raise ValueError(f'unknown reference type {t!r}; pass reftypes= to define its prefix')
|
|
205
|
+
pre = self.reftypes[t][plural]
|
|
206
|
+
return self.text_runs(pre + ' ', fmt)
|
|
207
|
+
|
|
208
|
+
def ref_fld(self, el, fmt):
|
|
209
|
+
"""REF/PAGEREF field for a cross-reference `a`, with a cached placeholder Word replaces on update.
|
|
210
|
+
Heading/paragraph targets number via `\\w`; caption targets return their bookmarked 'Label N' text
|
|
211
|
+
(or the number-only `_n` bookmark for bare/leaf/rel refs), so `\\w` never applies to them."""
|
|
212
|
+
tgt = (_get(el, 'href') or '#')[1:]
|
|
213
|
+
tokens = ref_tokens(_get(el, 'data-ref'))
|
|
214
|
+
if tgt not in self.reftarget:
|
|
215
|
+
raise ValueError(f'cross-reference target #{tgt} not found (targets are headings, paragraphs, figures, and tables with ids)')
|
|
216
|
+
kind = ref_variant(tokens)
|
|
217
|
+
nm, self.has_fields = self.bkname(tgt), True
|
|
218
|
+
if kind == 'page': instr, cached = rf' PAGEREF {nm} \h ', '#'
|
|
219
|
+
elif self.reftarget[tgt] == 'caption':
|
|
220
|
+
bare = 'bare' in tokens or kind in ('leaf', 'rel')
|
|
221
|
+
instr, cached = (rf' REF {nm}_n \h ' if bare else rf' REF {nm} \h '), '#'
|
|
222
|
+
else:
|
|
223
|
+
sw = self.REFSWITCH[kind]
|
|
224
|
+
instr = rf' REF {nm} {sw} \h ' if sw else rf' REF {nm} \h '
|
|
225
|
+
cached = self.idtext.get(tgt, '#') if kind == 'text' else '#'
|
|
226
|
+
return [E('w:fldSimple', {'w:instr': instr}, E('w:r', self.rpr(fmt), E('w:t', cached)))]
|
|
227
|
+
|
|
228
|
+
def ref_group(self, el, fmt):
|
|
229
|
+
"data-refs span: one pluralized prefix for a same-type group, per-item singular prefixes for mixed types; never range-collapsed"
|
|
230
|
+
refs = [c for c in _els(el) if _tag(c) == 'a']
|
|
231
|
+
types = [(_get(a, 'href') or '#')[1:].split('-')[0] for a in refs]
|
|
232
|
+
out = []
|
|
233
|
+
for (sep, pre, plural), a in zip(group_plan(types), refs):
|
|
234
|
+
if sep: out += self.text_runs(sep, fmt)
|
|
235
|
+
if pre: out += self.ref_prefix(a, fmt, plural=plural)
|
|
236
|
+
out += self.ref_fld(a, fmt)
|
|
237
|
+
return out
|
|
238
|
+
|
|
239
|
+
def custom_style(self, el, kind):
|
|
240
|
+
"Style id for an explicit custom-style attr (stubbed + warned if undefined), else a class matching a reference style name"
|
|
241
|
+
if cs := _get(el, 'custom-style'):
|
|
242
|
+
if cs.lower() in self.refstyles: return self.refstyles[cs.lower()]
|
|
243
|
+
if cs not in self.stubs:
|
|
244
|
+
self.stubs[cs] = (kind, re.sub(r'\W', '', cs) or f'Custom{len(self.stubs)}')
|
|
245
|
+
self.warn(f'custom style {cs!r} not in reference doc; stub injected')
|
|
246
|
+
return self.stubs[cs][1]
|
|
247
|
+
return next((self.refstyles[c.lower()] for c in _classes(el) if c.lower() in self.refstyles), None)
|
|
248
|
+
|
|
249
|
+
def span(self, el, fmt):
|
|
250
|
+
"Inline span: math -> inline m:oMath zone (linear source, dialect-agnostic), custom style -> rStyle, else transparent"
|
|
251
|
+
if _get(el, 'data-refs') is not None: return self.ref_group(el, fmt)
|
|
252
|
+
if 'math' in _classes(el): return [self.omath(el)]
|
|
253
|
+
if sid := self.custom_style(el, 'character'): return self.runs(el, fmt | {'rstyle': sid})
|
|
254
|
+
return self.runs(el, fmt)
|
|
255
|
+
|
|
256
|
+
def omath(self, el):
|
|
257
|
+
"An m:oMath zone holding `el`'s text as linear-format math runs"
|
|
258
|
+
return E('m:oMath', E('m:r', E('m:t', el.to_text(), {'xml:space': 'preserve'})))
|
|
259
|
+
|
|
260
|
+
def fnref(self, el, fmt):
|
|
261
|
+
"Footnote-reference run for a sup>a.footnote-ref, or None when `el` is an ordinary sup"
|
|
262
|
+
children = _els(el)
|
|
263
|
+
a = children[0] if len(children) == 1 and _tag(children[0]) == 'a' else None
|
|
264
|
+
if a is None or 'footnote-ref' not in _classes(a): return None
|
|
265
|
+
key = (_get(a, 'href') or '#')[1:]
|
|
266
|
+
if key not in self.fndefs:
|
|
267
|
+
self.warn(f'footnote reference #{key} has no definition; dropped')
|
|
268
|
+
return []
|
|
269
|
+
if key not in self.fnids:
|
|
270
|
+
self.fnids[key] = len(self.fnids) + 1
|
|
271
|
+
self.fnotes.append((self.fnids[key], self.fn_blocks(self.fndefs[key])))
|
|
272
|
+
return [E('w:r', E('w:rPr', E('w:rStyle', {'w:val': _sid('footnoteref')})),
|
|
273
|
+
E('w:footnoteReference', {'w:id': self.fnids[key]}))]
|
|
274
|
+
|
|
275
|
+
def fn_blocks(self, li):
|
|
276
|
+
"Footnote body: the li's blocks in footnote-text style, backref stripped, reference mark prepended"
|
|
277
|
+
save = self.rels, self._urlrids, self.first
|
|
278
|
+
self.rels, self._urlrids = self.fn_rels, {} # rel ids are per-part (see fn_relsxml)
|
|
279
|
+
try:
|
|
280
|
+
blks = [b for kind, val in self.li_parts(li) # chkstyle: ignore-node
|
|
281
|
+
for b in ([self.para(self.group_runs(val, {}), 'footnotetext')] if kind == 'inline'
|
|
282
|
+
else self.block(val, 'footnotetext'))]
|
|
283
|
+
finally: self.rels, self._urlrids, self.first = save
|
|
284
|
+
if not blks: blks = [self.para([], 'footnotetext')]
|
|
285
|
+
mark = E('w:r', E('w:rPr', E('w:rStyle', {'w:val': _sid('footnoteref')})), E('w:footnoteRef'))
|
|
286
|
+
blks[0].insert(1, E('w:r', E('w:t', ' ', {'xml:space': 'preserve'})))
|
|
287
|
+
blks[0].insert(1, mark)
|
|
288
|
+
return blks
|
|
289
|
+
|
|
290
|
+
def image(self, el, fmt, alt=None):
|
|
291
|
+
"Embed a local image (dimensions sniffed, width/height px attrs override); remote srcs degrade to a link"
|
|
292
|
+
src = _get(el, 'src') or ''
|
|
293
|
+
if alt is None: alt = _get(el, 'alt') or src
|
|
294
|
+
if re.match(r'[a-z][a-z0-9+.-]*://', src):
|
|
295
|
+
self.warn(f'remote image not embedded: {src}')
|
|
296
|
+
return self.external_link(src, self.text_runs(alt, fmt | {'rstyle': _sid('hyperlink')}))
|
|
297
|
+
try: data = (self.base/src).read_bytes()
|
|
298
|
+
except OSError:
|
|
299
|
+
self.warn(f'image not found: {src}; alt text emitted')
|
|
300
|
+
return self.text_runs(alt, fmt)
|
|
301
|
+
pw, ph, dx, dy = imgsize(data) or (300, 200, 96, 96)
|
|
302
|
+
cx, cy = round(pw * 914400 / dx), round(ph * 914400 / dy)
|
|
303
|
+
w_, h_ = _get(el, 'width'), _get(el, 'height')
|
|
304
|
+
if w_: cx = round(float(w_) * EMU_PER_PX)
|
|
305
|
+
if h_: cy = round(float(h_) * EMU_PER_PX)
|
|
306
|
+
if w_ and not h_: cy = round(cx * ph / pw)
|
|
307
|
+
if h_ and not w_: cx = round(cy * pw / ph)
|
|
308
|
+
self._imgn += 1
|
|
309
|
+
name = f'word/media/image{self._imgn}{Path(src).suffix.lower() or ".bin"}'
|
|
310
|
+
self.media[name] = data
|
|
311
|
+
rid = self.rid()
|
|
312
|
+
self.rels.append((rid, f'{R}/image', name.removeprefix('word/'), False))
|
|
313
|
+
return [E('w:r', drawing(rid, self._imgn, cx, cy, alt))]
|
|
314
|
+
|
|
315
|
+
def inline(self, el, fmt):
|
|
316
|
+
"Run-level elements for inline `el` under formatting context `fmt` (dict; copied on change)"
|
|
317
|
+
tag = _tag(el)
|
|
318
|
+
if tag == 'em': out = self.runs(el, fmt | {'i': True})
|
|
319
|
+
elif tag == 'strong': out = self.runs(el, fmt | {'b': True})
|
|
320
|
+
elif tag == 'code': out = self.runs(el, fmt | {'rstyle': _sid('codeinline')})
|
|
321
|
+
elif tag == 'a': out = self.link(el, fmt)
|
|
322
|
+
elif tag == 'del': out = self.runs(el, fmt | {'strike': True})
|
|
323
|
+
elif tag == 'mark': out = self.runs(el, fmt | {'mark': True})
|
|
324
|
+
elif tag == 'u': out = self.runs(el, fmt | {'u': True})
|
|
325
|
+
elif tag == 'sub': out = self.runs(el, fmt | {'vert': 'subscript'})
|
|
326
|
+
elif tag == 'sup':
|
|
327
|
+
fn = self.fnref(el, fmt)
|
|
328
|
+
out = fn if fn is not None else self.runs(el, fmt | {'vert': 'superscript'})
|
|
329
|
+
elif tag == 'span': out = self.span(el, fmt)
|
|
330
|
+
elif tag == 'img': out = self.image(el, fmt)
|
|
331
|
+
elif tag == 'br': out = [E('w:r', self.rpr(fmt), E('w:br'))]
|
|
332
|
+
elif tag == 'input': # task-list checkbox
|
|
333
|
+
if _get(el, 'type') != 'checkbox':
|
|
334
|
+
self.warn(f'unhandled inline <input type={_get(el, "type")!r}>; dropped')
|
|
335
|
+
return []
|
|
336
|
+
g = '☒' if _get(el, 'checked') is not None else '☐'
|
|
337
|
+
out = [E('w:r', self.rpr(fmt), E('w:t', g + ' ', {'xml:space': 'preserve'}))]
|
|
338
|
+
elif tag == 'script': out = self.rawxml(el)
|
|
339
|
+
elif tag == 'template': out = self.tmpl_runs(el, fmt, 'inline')
|
|
340
|
+
elif tag in INERT_TAGS: out = []
|
|
341
|
+
else: # unknown inline (abbr etc): recurse transparently
|
|
342
|
+
out = self.runs(el, fmt)
|
|
343
|
+
return out
|
|
344
|
+
|
|
345
|
+
def runs(self, el, fmt):
|
|
346
|
+
"Run-level elements for an element's ordered text and element children"
|
|
347
|
+
return [r for node in el.children for r in self.inline_node(node, fmt)]
|
|
348
|
+
|
|
349
|
+
def inline_node(self, node, fmt):
|
|
350
|
+
if isinstance(node, Text): return self.text_runs(node.text, fmt)
|
|
351
|
+
if isinstance(node, Element):
|
|
352
|
+
if _tag(node) == 'a' and 'footnote-backref' in _classes(node): return []
|
|
353
|
+
return self.inline(node, fmt)
|
|
354
|
+
return []
|
|
355
|
+
|
|
356
|
+
# ---- block level --------------------------------------------------------
|
|
357
|
+
def para(self, runs, style='body', extra=None, sid=None):
|
|
358
|
+
"A w:p with `style` (STYLE_MAP key, or `sid` style-id override) and optional extra pPr children (schema order!)"
|
|
359
|
+
ppr = E('w:pPr', E('w:pStyle', {'w:val': sid or _sid(style)}), *(extra or []))
|
|
360
|
+
return E('w:p', ppr, *runs)
|
|
361
|
+
|
|
362
|
+
def bookmark(self, el, runs):
|
|
363
|
+
"Wrap `runs` in a bookmark when `el` carries an id (target for internal links)"
|
|
364
|
+
if not (i := _get(el, 'id')): return runs
|
|
365
|
+
self._bkid += 1
|
|
366
|
+
return [E('w:bookmarkStart', {'w:id': self._bkid, 'w:name': self.bkname(i)}),
|
|
367
|
+
*runs, E('w:bookmarkEnd', {'w:id': self._bkid})]
|
|
368
|
+
|
|
369
|
+
def codeblock(self, el):
|
|
370
|
+
"Source Code paragraph, lines joined with w:br; Hl* character styles when a language class names one"
|
|
371
|
+
children = _els(el)
|
|
372
|
+
code = children[0] if children and _tag(children[0]) == 'code' else el
|
|
373
|
+
lang = next((c.removeprefix('language-') for c in _classes(code) if c.startswith('language-')), None)
|
|
374
|
+
text = code.to_text().rstrip('\n')
|
|
375
|
+
segs = (segments(text, lang) if self.hlstyles else None) or [(text, None)]
|
|
376
|
+
runs = []
|
|
377
|
+
for txt, scope in segs:
|
|
378
|
+
for j, part in enumerate(txt.split('\n')):
|
|
379
|
+
if j: runs.append(E('w:r', E('w:br')))
|
|
380
|
+
if not part: continue
|
|
381
|
+
sid = self.hlsid(scope)
|
|
382
|
+
runs.append(E('w:r', E('w:rPr', E('w:rStyle', {'w:val': sid})) if sid else None,
|
|
383
|
+
E('w:t', part, {'xml:space': 'preserve'})))
|
|
384
|
+
return [self.para(runs, 'codeblock')]
|
|
385
|
+
|
|
386
|
+
def qindent(self):
|
|
387
|
+
"Extra indent for paragraphs in nested blockquotes (Quote style itself carries the first level)"
|
|
388
|
+
return [E('w:ind', {'w:left': 720 * self.bq})] if self.bq > 1 else None
|
|
389
|
+
|
|
390
|
+
# ---- lists --------------------------------------------------------------
|
|
391
|
+
def list_el(self, el, ilvl=0):
|
|
392
|
+
"A ul/ol: fresh num instance (so each ordered list restarts), items at level `ilvl`"
|
|
393
|
+
self._numid += 1
|
|
394
|
+
nid = self._numid
|
|
395
|
+
self.nums.append((nid, 0 if _tag(el) == 'ul' else 1, int(_get(el, 'start', 1))))
|
|
396
|
+
return [b for li in _els(el) if _tag(li) == 'li' for b in self.li(li, nid, min(ilvl, 8))]
|
|
397
|
+
|
|
398
|
+
def li_parts(self, el):
|
|
399
|
+
"Split mixed li content into ('inline', [nodes]) groups and ('block', child) items, in order"
|
|
400
|
+
parts = []
|
|
401
|
+
def add(x):
|
|
402
|
+
if isinstance(x, Text) and not x.text.strip() and '\n' in x.text: return
|
|
403
|
+
if parts and parts[-1][0] == 'inline': parts[-1][1].append(x)
|
|
404
|
+
else: parts.append(('inline', [x]))
|
|
405
|
+
for node in el.children:
|
|
406
|
+
if isinstance(node, Element) and _tag(node) in BLOCK_TAGS: parts.append(('block', node))
|
|
407
|
+
elif isinstance(node, (Text, Element)): add(node)
|
|
408
|
+
return parts
|
|
409
|
+
|
|
410
|
+
def group_runs(self, nodes, fmt):
|
|
411
|
+
"Runs for a mixed list of text and inline nodes"
|
|
412
|
+
return [r for node in nodes for r in self.inline_node(node, fmt)]
|
|
413
|
+
|
|
414
|
+
def li(self, li, nid, ilvl):
|
|
415
|
+
"Blocks for one list item: the first paragraph carries the number, the rest continue indented"
|
|
416
|
+
numpr = [E('w:numPr', E('w:ilvl', {'w:val': ilvl}), E('w:numId', {'w:val': nid}))]
|
|
417
|
+
cont = [E('w:ind', {'w:left': 720 * (ilvl + 1)})]
|
|
418
|
+
out = []
|
|
419
|
+
for kind, val in self.li_parts(li):
|
|
420
|
+
if kind == 'inline': out.append(self.para(self.group_runs(val, {}), 'list', numpr if not out else cont))
|
|
421
|
+
elif _tag(val) in ('ul', 'ol'): out += self.list_el(val, ilvl + 1)
|
|
422
|
+
elif _tag(val) == 'p': out.append(self.para(self.runs(val, {}), 'list', numpr if not out else cont))
|
|
423
|
+
else: out += self.block(val, 'list')
|
|
424
|
+
return out or [self.para([], 'list', numpr)]
|
|
425
|
+
|
|
426
|
+
# ---- tables -------------------------------------------------------------
|
|
427
|
+
def table_grid(self, rows):
|
|
428
|
+
"Resolve row/colspans into per-row cell placements: ('cell', ci, el, cs, rs) / ('cont', ci, width)"
|
|
429
|
+
spans, placed, ncols = {}, [], 0
|
|
430
|
+
for ri, tr in enumerate(rows):
|
|
431
|
+
ci, rowcells = 0, []
|
|
432
|
+
def _skip(ci):
|
|
433
|
+
while (ri, ci) in spans:
|
|
434
|
+
wd = spans.pop((ri, ci))
|
|
435
|
+
rowcells.append(('cont', ci, wd))
|
|
436
|
+
ci += wd
|
|
437
|
+
return ci
|
|
438
|
+
ci = _skip(ci)
|
|
439
|
+
for cell in _els(tr):
|
|
440
|
+
if _tag(cell) not in ('td', 'th'): continue
|
|
441
|
+
cs, rs = int(_get(cell, 'colspan', 1)), int(_get(cell, 'rowspan', 1))
|
|
442
|
+
rowcells.append(('cell', ci, cell, cs, rs))
|
|
443
|
+
for k in range(1, rs): spans[(ri + k, ci)] = cs
|
|
444
|
+
ci = _skip(ci + cs)
|
|
445
|
+
placed.append(rowcells)
|
|
446
|
+
ncols = max(ncols, ci)
|
|
447
|
+
return placed, ncols
|
|
448
|
+
|
|
449
|
+
def col_widths(self, el, ncols):
|
|
450
|
+
"colwidths tracks -> (dxa list, all_fr?) or (None, False) when absent"
|
|
451
|
+
s = _get(el, 'colwidths') or _get(el, 'data-colwidths')
|
|
452
|
+
if not s: return None, False
|
|
453
|
+
tracks = parse_tracks(s)
|
|
454
|
+
if len(tracks) != ncols:
|
|
455
|
+
self.warn(f'colwidths has {len(tracks)} tracks for {ncols} columns; padding with 1fr')
|
|
456
|
+
tracks = tracks[:ncols] + [('fr', 1.0)] * (ncols - len(tracks))
|
|
457
|
+
fixed = sum(v for k, v in tracks if k == 'dxa')
|
|
458
|
+
frs = sum(v for k, v in tracks if k == 'fr')
|
|
459
|
+
rem = max(self.content_w - fixed, 0)
|
|
460
|
+
dxa = [round(v) if k == 'dxa' else round(v * rem / frs) for k, v in tracks]
|
|
461
|
+
return dxa, fixed == 0
|
|
462
|
+
|
|
463
|
+
def cell_blocks(self, cell, header):
|
|
464
|
+
"Block content of one table cell; header cells bold, align attr honored for inline cells"
|
|
465
|
+
if any(_tag(c) in BLOCK_TAGS for c in _els(cell)): return self.blocks(cell)
|
|
466
|
+
jc = [E('w:jc', {'w:val': _get(cell, 'align')})] if _get(cell, 'align') in ('center', 'right') else None
|
|
467
|
+
return [self.para(self.runs(cell, {'b': True} if header else {}), 'compact', jc)]
|
|
468
|
+
|
|
469
|
+
def table(self, el):
|
|
470
|
+
"w:tbl (+ caption paragraph before, spacer paragraph after)"
|
|
471
|
+
cap = None
|
|
472
|
+
rows, nhead = [], 0
|
|
473
|
+
for sec in _els(el):
|
|
474
|
+
t = _tag(sec)
|
|
475
|
+
if t == 'caption': cap = sec
|
|
476
|
+
elif t == 'thead':
|
|
477
|
+
rows += _els(sec)
|
|
478
|
+
nhead = len(rows)
|
|
479
|
+
elif t in ('tbody', 'tfoot'): rows += _els(sec)
|
|
480
|
+
elif t == 'tr': rows.append(sec)
|
|
481
|
+
placed, ncols = self.table_grid(rows)
|
|
482
|
+
dxa, all_fr = self.col_widths(el, ncols)
|
|
483
|
+
def _tcw(ci, cs):
|
|
484
|
+
if not dxa: return None
|
|
485
|
+
wd = sum(dxa[ci:ci + cs])
|
|
486
|
+
if all_fr: return E('w:tcW', {'w:type': 'pct', 'w:w': round(wd / self.content_w * 5000)})
|
|
487
|
+
return E('w:tcW', {'w:type': 'dxa', 'w:w': wd})
|
|
488
|
+
tblw = (E('w:tblW', {'w:type': 'auto', 'w:w': 0}) if not dxa # chkstyle: ignore-node
|
|
489
|
+
else E('w:tblW', {'w:type': 'pct', 'w:w': 5000}) if all_fr
|
|
490
|
+
else E('w:tblW', {'w:type': 'dxa', 'w:w': sum(dxa)}))
|
|
491
|
+
tblpr = E('w:tblPr', E('w:tblStyle', {'w:val': self.custom_style(el, 'table') or _sid('table')}), tblw, # chkstyle: ignore-node
|
|
492
|
+
E('w:tblLayout', {'w:type': 'fixed'}) if dxa and not all_fr else None,
|
|
493
|
+
E('w:tblLook', {'w:val': '04A0', 'w:firstRow': 1, 'w:lastRow': 0,
|
|
494
|
+
'w:firstColumn': 0, 'w:lastColumn': 0, 'w:noHBand': 0, 'w:noVBand': 1}))
|
|
495
|
+
gw = dxa or [self.content_w // ncols] * ncols # pandoc's docx reader drops tables whose gridCols lack w:w
|
|
496
|
+
grid = E('w:tblGrid', *[E('w:gridCol', {'w:w': gw[i]}) for i in range(ncols)])
|
|
497
|
+
trs = []
|
|
498
|
+
for ri, rowcells in enumerate(placed):
|
|
499
|
+
tcs = []
|
|
500
|
+
for item in rowcells:
|
|
501
|
+
if item[0] == 'cont':
|
|
502
|
+
_, ci, wd = item
|
|
503
|
+
tcs.append(E('w:tc', E('w:tcPr', _tcw(ci, wd), # chkstyle: ignore-node
|
|
504
|
+
E('w:gridSpan', {'w:val': wd}) if wd > 1 else None,
|
|
505
|
+
E('w:vMerge')), E('w:p')))
|
|
506
|
+
else:
|
|
507
|
+
_, ci, cell, cs, rs = item
|
|
508
|
+
tcpr = E('w:tcPr', _tcw(ci, cs), # chkstyle: ignore-node
|
|
509
|
+
E('w:gridSpan', {'w:val': cs}) if cs > 1 else None,
|
|
510
|
+
E('w:vMerge', {'w:val': 'restart'}) if rs > 1 else None)
|
|
511
|
+
body = self.cell_blocks(cell, ri < nhead)
|
|
512
|
+
if not len(body) or etree.QName(body[-1]).localname != 'p': body.append(E('w:p'))
|
|
513
|
+
tcs.append(E('w:tc', tcpr, *body))
|
|
514
|
+
trs.append(E('w:tr', E('w:trPr', E('w:tblHeader')) if ri < nhead else None, *tcs))
|
|
515
|
+
out = self.caption_para(el, 'tbl', cap)
|
|
516
|
+
return out + [E('w:tbl', tblpr, grid, *trs), E('w:p')]
|
|
517
|
+
|
|
518
|
+
def caption_para(self, el, typ, capel, fmt={}):
|
|
519
|
+
"""Numbered caption paragraph: 'Label N: text' with a SEQ field as N. When `el` has an id, the
|
|
520
|
+
label+number span is bookmarked under it (REF target) and the number alone under `<name>_n`.
|
|
521
|
+
Emitted whenever there is a caption or an id; the label word comes from reftypes[typ]."""
|
|
522
|
+
if capel is None and not _get(el, 'id'): return []
|
|
523
|
+
label = self.reftypes[typ][0]
|
|
524
|
+
seq = [E('w:fldSimple', {'w:instr': rf' SEQ {label} \* ARABIC '}, E('w:r', self.rpr(fmt), E('w:t', '#')))]
|
|
525
|
+
self.has_fields = True
|
|
526
|
+
if i := _get(el, 'id'):
|
|
527
|
+
nm = self.bkname(i)
|
|
528
|
+
self._bknames[i + '\0n'] = nm + '_n' # reserve the number-only name too
|
|
529
|
+
self._bkid += 2
|
|
530
|
+
seq = [E('w:bookmarkStart', {'w:id': self._bkid, 'w:name': nm + '_n'}), *seq,
|
|
531
|
+
E('w:bookmarkEnd', {'w:id': self._bkid})]
|
|
532
|
+
runs = [E('w:bookmarkStart', {'w:id': self._bkid - 1, 'w:name': nm}), # chkstyle: ignore-node
|
|
533
|
+
*self.text_runs(label + ' ', fmt), *seq,
|
|
534
|
+
E('w:bookmarkEnd', {'w:id': self._bkid - 1})]
|
|
535
|
+
else: runs = self.text_runs(label + ' ', fmt) + seq
|
|
536
|
+
cap = [] if capel is None else self.runs(capel, fmt)
|
|
537
|
+
if cap: runs += self.text_runs(': ', fmt) + cap
|
|
538
|
+
return [self.para(runs, 'caption')]
|
|
539
|
+
|
|
540
|
+
def figure(self, el):
|
|
541
|
+
"Figure: image paragraph, then its numbered caption paragraph below (Word convention)"
|
|
542
|
+
img = next((c for c in _walk(el) if _tag(c) == 'img'), None)
|
|
543
|
+
capel = next((c for c in _els(el) if _tag(c) == 'figcaption'), None)
|
|
544
|
+
alt = capel.to_text().strip() if capel is not None else None
|
|
545
|
+
out = [] if img is None else [self.para(self.image(img, {}, alt), 'body')]
|
|
546
|
+
return out + self.caption_para(el, 'fig', capel)
|
|
547
|
+
|
|
548
|
+
# Paragraphs directly after these blocks (or at document start) take First Paragraph rather
|
|
549
|
+
# than Body Text, matching pandoc's docx writer exactly, so the two agree on which is "first".
|
|
550
|
+
FIRST_AFTER = {'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'pre', 'ul', 'ol', 'table', 'dl', 'hr'}
|
|
551
|
+
|
|
552
|
+
def block(self, el, style='body', sid=None):
|
|
553
|
+
"Block elements for `el` (one element may yield several); `sid` is a custom-style id override for paragraphs"
|
|
554
|
+
out = self._block(el, style, sid)
|
|
555
|
+
tag = _tag(el)
|
|
556
|
+
if tag in self.FIRST_AFTER or (tag == 'div' and 'display' in _classes(el)): self.first = True
|
|
557
|
+
return out
|
|
558
|
+
|
|
559
|
+
def _block(self, el, style, sid):
|
|
560
|
+
tag = _tag(el)
|
|
561
|
+
if tag == 'p':
|
|
562
|
+
ex = self.qindent() if style == 'blockquote' else None
|
|
563
|
+
psid = self.custom_style(el, 'paragraph') or sid
|
|
564
|
+
use = 'firstpara' if self.first and style == 'body' and not psid else style
|
|
565
|
+
self.first = False
|
|
566
|
+
return [self.para(self.bookmark(el, self.runs(el, {})), use, ex, psid)]
|
|
567
|
+
if tag in ('h1', 'h2', 'h3', 'h4', 'h5', 'h6'): return [self.para(self.bookmark(el, self.runs(el, {})), tag)]
|
|
568
|
+
if tag == 'blockquote':
|
|
569
|
+
self.bq += 1
|
|
570
|
+
try: return self.blocks(el, 'blockquote')
|
|
571
|
+
finally: self.bq -= 1
|
|
572
|
+
if tag == 'pre': return self.codeblock(el)
|
|
573
|
+
if tag in ('ul', 'ol'): return self.list_el(el)
|
|
574
|
+
if tag == 'table': return self.table(el)
|
|
575
|
+
if tag == 'hr':
|
|
576
|
+
return [E('w:p', E('w:pPr', E('w:pBdr',
|
|
577
|
+
E('w:bottom', {'w:val': 'single', 'w:sz': 6, 'w:space': 1, 'w:color': 'auto'}))))]
|
|
578
|
+
if tag == 'dl': return self.dl(el)
|
|
579
|
+
if tag == 'script': return self.rawxml(el)
|
|
580
|
+
if tag == 'figure': return self.figure(el)
|
|
581
|
+
if tag == 'template': return [self.para(runs)] if (runs := self.tmpl_runs(el, {}, 'block')) else []
|
|
582
|
+
if tag == 'div':
|
|
583
|
+
cls = _classes(el)
|
|
584
|
+
if 'math' in cls and 'display' in cls: return [E('w:p', E('m:oMathPara', self.omath(el)))]
|
|
585
|
+
return self.blocks(el, style, self.custom_style(el, 'paragraph') or sid)
|
|
586
|
+
if tag in BLOCK_TAGS and any(_tag(c) in BLOCK_TAGS for c in _els(el)):
|
|
587
|
+
return self.blocks(el, style, sid) # unknown container: recurse
|
|
588
|
+
self.warn(f'unhandled block <{tag}>; emitted as plain paragraph')
|
|
589
|
+
return [self.para(self.runs(el, {}), style, None, sid)]
|
|
590
|
+
|
|
591
|
+
RAWNS = ' '.join(f'xmlns:{k}="{v}"' for k, v in NS.items() if k != 'xml')
|
|
592
|
+
|
|
593
|
+
BIND_NS = 'urn:mdhtml:fields'
|
|
594
|
+
BIND_ID = '{8E2C9A44-7D31-4E5B-9C0D-1A6F2B3C4D5E}' # fixed datastore id, so builds are reproducible
|
|
595
|
+
|
|
596
|
+
def tmpl_runs(self, el, fmt, form):
|
|
597
|
+
"""Template-token runs via the `tmpl` callable: str is a literal text run, ('field', instr) a live
|
|
598
|
+
field, ('control', name) an interactive plain-text content control, None dropped"""
|
|
599
|
+
if self.tmpl is None: return []
|
|
600
|
+
res = self.tmpl(el.to_text(), _get(el, 'data-template', ''), form)
|
|
601
|
+
if res is None: return []
|
|
602
|
+
if isinstance(res, str): return self.text_runs(res, fmt)
|
|
603
|
+
kind, val = res
|
|
604
|
+
if kind == 'field':
|
|
605
|
+
self.has_fields = True
|
|
606
|
+
return [E('w:fldSimple', {'w:instr': f' {val.strip()} '}, E('w:r', self.rpr(fmt), E('w:t', f'«{el.to_text().strip()}»')))]
|
|
607
|
+
if kind in ('control', 'bound'):
|
|
608
|
+
self.has_controls = True
|
|
609
|
+
sdtpr = E('w:sdtPr', E('w:alias', {'w:val': val}), E('w:tag', {'w:val': val}), E('w:showingPlcHdr'))
|
|
610
|
+
if kind == 'bound':
|
|
611
|
+
if val not in self.bound: self.bound.append(val)
|
|
612
|
+
sdtpr.append(E('w:dataBinding', {'w:prefixMappings': f"xmlns:ns0='{self.BIND_NS}'",
|
|
613
|
+
'w:xpath': f'/ns0:fields[1]/ns0:{val}[1]', 'w:storeItemID': self.BIND_ID}))
|
|
614
|
+
sdtpr.append(E('w:text'))
|
|
615
|
+
return [E('w:sdt', sdtpr,
|
|
616
|
+
E('w:sdtContent', E('w:r', self.rpr(fmt | {'rstyle': 'PlaceholderText'}), E('w:t', val))))]
|
|
617
|
+
raise ValueError(f'unknown template rendering {res!r}')
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def rawxml(self, el):
|
|
621
|
+
"Elements parsed from a raw docx payload (`{=docx}` in Markdown); other formats skip silently"
|
|
622
|
+
if _get(el, 'type') != 'application/vnd.mdhtml.raw' or _get(el, 'data-format') != 'docx': return []
|
|
623
|
+
payload, warn = decode_raw(el)
|
|
624
|
+
if warn:
|
|
625
|
+
self.warn(warn)
|
|
626
|
+
return []
|
|
627
|
+
try: return list(etree.fromstring(f'<m2d {self.RAWNS}>{payload}</m2d>'))
|
|
628
|
+
except etree.XMLSyntaxError as e:
|
|
629
|
+
self.warn(f'malformed docx raw payload: {e}')
|
|
630
|
+
return []
|
|
631
|
+
|
|
632
|
+
def dl(self, el):
|
|
633
|
+
"Definition list: dt/dd paragraphs in their dialect styles"
|
|
634
|
+
out = []
|
|
635
|
+
for c in _els(el):
|
|
636
|
+
t = _tag(c)
|
|
637
|
+
if t == 'dt': out.append(self.para(self.runs(c, {}), 'dt'))
|
|
638
|
+
elif t == 'dd':
|
|
639
|
+
blocky = any(_tag(k) in BLOCK_TAGS for k in _els(c))
|
|
640
|
+
out += self.blocks(c, 'dd') if blocky else [self.para(self.runs(c, {}), 'dd')]
|
|
641
|
+
return out
|
|
642
|
+
|
|
643
|
+
def blocks(self, parent, style='body', sid=None): return self.block_nodes(parent.children, style, sid)
|
|
644
|
+
|
|
645
|
+
def block_nodes(self, nodes, style='body', sid=None):
|
|
646
|
+
out, inline = [], []
|
|
647
|
+
def flush():
|
|
648
|
+
meaningful = [n for n in inline if not isinstance(n, Text) or n.text.strip()]
|
|
649
|
+
if not meaningful:
|
|
650
|
+
inline.clear()
|
|
651
|
+
return
|
|
652
|
+
raw = all(isinstance(n, Element) and _is_raw(n) for n in meaningful)
|
|
653
|
+
if raw:
|
|
654
|
+
for node in meaningful: out.extend(self.block(node, style, sid))
|
|
655
|
+
else:
|
|
656
|
+
extra = self.qindent() if style == 'blockquote' else None
|
|
657
|
+
use = 'firstpara' if self.first and style == 'body' and not sid else style
|
|
658
|
+
self.first = False
|
|
659
|
+
out.append(self.para(self.group_runs(inline, {}), use, extra, sid))
|
|
660
|
+
inline.clear()
|
|
661
|
+
for node in nodes:
|
|
662
|
+
if isinstance(node, Comment): continue
|
|
663
|
+
if isinstance(node, Element) and _tag(node) == 'template':
|
|
664
|
+
flush()
|
|
665
|
+
if runs := self.tmpl_runs(node, {}, 'block'): out.append(self.para(runs))
|
|
666
|
+
continue
|
|
667
|
+
if isinstance(node, Element) and (_tag(node) in INERT_TAGS or _tag(node) == 'script' and not _is_raw(node)): continue
|
|
668
|
+
if isinstance(node, Element) and _tag(node) in BLOCK_TAGS:
|
|
669
|
+
flush()
|
|
670
|
+
out.extend(self.block(node, style, sid))
|
|
671
|
+
elif isinstance(node, (Text, Element)): inline.append(node)
|
|
672
|
+
flush()
|
|
673
|
+
return out
|
|
674
|
+
|
|
675
|
+
# ---- assembly -----------------------------------------------------------
|
|
676
|
+
def document(self, body_blocks):
|
|
677
|
+
"word/document.xml bytes: our blocks + the template's sectPr"
|
|
678
|
+
root = etree.Element(qn('w:document'), nsmap=NS)
|
|
679
|
+
body = etree.SubElement(root, qn('w:body'))
|
|
680
|
+
for b in body_blocks: body.append(b)
|
|
681
|
+
body.append(self.sectpr)
|
|
682
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
683
|
+
|
|
684
|
+
BULLETS = ['•', '◦', '▪']
|
|
685
|
+
NUMFMTS = ['decimal', 'lowerLetter', 'lowerRoman']
|
|
686
|
+
|
|
687
|
+
def numbering_xml(self):
|
|
688
|
+
"""word/numbering.xml: the template's part (if any) merged with our list definitions (bullet and
|
|
689
|
+
decimal abstracts, one num per list with startOverrides so each restarts), the heading scheme,
|
|
690
|
+
and .xml contributors' definitions - all abstractNum before all num, as the schema requires"""
|
|
691
|
+
root = self.tmplnum if self.tmplnum is not None else etree.Element(qn('w:numbering'), nsmap=dict(w=W))
|
|
692
|
+
tnums = [e for e in root if etree.QName(e).localname == 'num']
|
|
693
|
+
for e in tnums: root.remove(e)
|
|
694
|
+
if self.nums:
|
|
695
|
+
for aid in (0, 1):
|
|
696
|
+
an = E('w:abstractNum', {'w:abstractNumId': self._absbase + aid},
|
|
697
|
+
E('w:multiLevelType', {'w:val': 'hybridMultilevel'}))
|
|
698
|
+
for i in range(9):
|
|
699
|
+
fmt, txt = ('bullet', self.BULLETS[i % 3]) if aid == 0 else (self.NUMFMTS[i % 3], f'%{i+1}.')
|
|
700
|
+
an.append(E('w:lvl', {'w:ilvl': i}, # chkstyle: ignore-node
|
|
701
|
+
E('w:start', {'w:val': 1}), E('w:numFmt', {'w:val': fmt}),
|
|
702
|
+
E('w:lvlText', {'w:val': txt}), E('w:lvlJc', {'w:val': 'left'}),
|
|
703
|
+
E('w:pPr', E('w:ind', {'w:left': 720 * (i + 1), 'w:hanging': 360}))))
|
|
704
|
+
root.append(an)
|
|
705
|
+
if self.headnum:
|
|
706
|
+
an = E('w:abstractNum', {'w:abstractNumId': self._absbase + 2}, E('w:multiLevelType', {'w:val': 'multilevel'}))
|
|
707
|
+
for i in range(9):
|
|
708
|
+
txt, fmt = self.scheme[i] if i < len(self.scheme) else (f'%{i+1}.', 'decimal')
|
|
709
|
+
an.append(E('w:lvl', {'w:ilvl': i}, # chkstyle: ignore-node
|
|
710
|
+
E('w:start', {'w:val': 1}), E('w:numFmt', {'w:val': fmt}),
|
|
711
|
+
E('w:pStyle', {'w:val': f'Heading{i + 1}'}) if i < 6 else None,
|
|
712
|
+
E('w:lvlText', {'w:val': txt}), E('w:lvlJc', {'w:val': 'left'})))
|
|
713
|
+
root.append(an)
|
|
714
|
+
for e in self.xabs: root.append(e)
|
|
715
|
+
for e in tnums: root.append(e)
|
|
716
|
+
if self.headnum: root.append(E('w:num', {'w:numId': self.headnum}, E('w:abstractNumId', {'w:val': self._absbase + 2})))
|
|
717
|
+
for e in self.xnums: root.append(e)
|
|
718
|
+
for nid, aid, start in self.nums:
|
|
719
|
+
root.append(E('w:num', {'w:numId': nid}, E('w:abstractNumId', {'w:val': self._absbase + aid}), # chkstyle: ignore-node
|
|
720
|
+
*[E('w:lvlOverride', {'w:ilvl': i}, E('w:startOverride', {'w:val': start if i == 0 else 1}))
|
|
721
|
+
for i in range(9)]))
|
|
722
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
723
|
+
|
|
724
|
+
RNS = 'http://schemas.openxmlformats.org/package/2006/relationships'
|
|
725
|
+
|
|
726
|
+
def _add_rels(self, root, rels):
|
|
727
|
+
for rid, typ, target, ext in rels:
|
|
728
|
+
rel = etree.SubElement(root, f'{{{self.RNS}}}Relationship', Id=rid, Type=typ, Target=target)
|
|
729
|
+
if ext: rel.set('TargetMode', 'External')
|
|
730
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
731
|
+
|
|
732
|
+
def relsxml(self):
|
|
733
|
+
"word/_rels/document.xml.rels: template rels plus ours"
|
|
734
|
+
return self._add_rels(etree.fromstring(self.refz.read('word/_rels/document.xml.rels')), self.rels)
|
|
735
|
+
|
|
736
|
+
def fn_relsxml(self):
|
|
737
|
+
"word/_rels/footnotes.xml.rels: relationship ids are per-part, so footnote links/images get their own file"
|
|
738
|
+
return self._add_rels(etree.Element(f'{{{self.RNS}}}Relationships', nsmap={None: self.RNS}), self.fn_rels)
|
|
739
|
+
|
|
740
|
+
DSNS = 'http://schemas.openxmlformats.org/officeDocument/2006/customXml'
|
|
741
|
+
|
|
742
|
+
def bind_item_xml(self):
|
|
743
|
+
"customXml/item1.xml: one empty element per bound variable; every same-name control is a live view of it"
|
|
744
|
+
root = etree.Element(f'{{{self.BIND_NS}}}fields', nsmap=dict(ns0=self.BIND_NS))
|
|
745
|
+
for name in self.bound: etree.SubElement(root, f'{{{self.BIND_NS}}}{name}')
|
|
746
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
747
|
+
|
|
748
|
+
def bind_props_xml(self):
|
|
749
|
+
"customXml/itemProps1.xml: the datastore id that the controls' `storeItemID` points at"
|
|
750
|
+
root = etree.Element(f'{{{self.DSNS}}}datastoreItem', nsmap=dict(ds=self.DSNS))
|
|
751
|
+
root.set(f'{{{self.DSNS}}}itemID', self.BIND_ID)
|
|
752
|
+
srs = etree.SubElement(root, f'{{{self.DSNS}}}schemaRefs')
|
|
753
|
+
etree.SubElement(srs, f'{{{self.DSNS}}}schemaRef').set(f'{{{self.DSNS}}}uri', self.BIND_NS)
|
|
754
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
755
|
+
|
|
756
|
+
def bind_rels_xml(self):
|
|
757
|
+
"customXml/_rels/item1.xml.rels: the item's link to its datastore properties"
|
|
758
|
+
root = etree.Element(f'{{{self.RNS}}}Relationships', nsmap={None: self.RNS})
|
|
759
|
+
return self._add_rels(root, [('rId1', f'{R}/customXmlProps', 'itemProps1.xml', False)])
|
|
760
|
+
|
|
761
|
+
|
|
762
|
+
def footnotes_xml(self):
|
|
763
|
+
"word/footnotes.xml: the two Word-required separator notes plus our harvested ones"
|
|
764
|
+
root = etree.Element(qn('w:footnotes'), nsmap=dict(w=W, r=NS['r']))
|
|
765
|
+
for typ, wid in (('separator', -1), ('continuationSeparator', 0)):
|
|
766
|
+
root.append(E('w:footnote', {'w:type': typ, 'w:id': wid},
|
|
767
|
+
E('w:p', E('w:pPr', E('w:spacing', {'w:after': 0})), E('w:r', E(f'w:{typ}')))))
|
|
768
|
+
for wid, blks in self.fnotes: root.append(E('w:footnote', {'w:id': wid}, *blks))
|
|
769
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
770
|
+
|
|
771
|
+
def content_types(self, extra_parts):
|
|
772
|
+
"[Content_Types].xml with defaults for media extensions and overrides for added parts"
|
|
773
|
+
CT = 'http://schemas.openxmlformats.org/package/2006/content-types'
|
|
774
|
+
root = etree.fromstring(self.refz.read('[Content_Types].xml'))
|
|
775
|
+
have = {d.get('Extension') for d in root if d.tag == f'{{{CT}}}Default'}
|
|
776
|
+
MIME = dict(png='image/png', jpeg='image/jpeg', jpg='image/jpeg', gif='image/gif', tiff='image/tiff')
|
|
777
|
+
for name in self.media:
|
|
778
|
+
ext = posixpath.splitext(name)[1][1:].lower()
|
|
779
|
+
if ext not in have:
|
|
780
|
+
etree.SubElement(root, f'{{{CT}}}Default', Extension=ext, ContentType=MIME.get(ext, 'application/octet-stream'))
|
|
781
|
+
have.add(ext)
|
|
782
|
+
WPML = 'application/vnd.openxmlformats-officedocument.wordprocessingml'
|
|
783
|
+
for part, kind in extra_parts:
|
|
784
|
+
if any(d.get('PartName') == '/' + part for d in root if d.tag == f'{{{CT}}}Override'): continue
|
|
785
|
+
etree.SubElement(root, f'{{{CT}}}Override', PartName='/'+part, ContentType=f'{WPML}.{kind}+xml')
|
|
786
|
+
if self.bound:
|
|
787
|
+
if 'xml' not in have: etree.SubElement(root, f'{{{CT}}}Default', Extension='xml', ContentType='application/xml')
|
|
788
|
+
pn = '/customXml/itemProps1.xml'
|
|
789
|
+
if not any(d.get('PartName') == pn for d in root if d.tag == f'{{{CT}}}Override'):
|
|
790
|
+
etree.SubElement(root, f'{{{CT}}}Override', PartName=pn,
|
|
791
|
+
ContentType='application/vnd.openxmlformats-officedocument.customXmlProperties+xml')
|
|
792
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
793
|
+
|
|
794
|
+
PPR_PRE_NUMPR = {'pStyle', 'keepNext', 'keepLines', 'pageBreakBefore', 'framePr', 'widowControl'}
|
|
795
|
+
|
|
796
|
+
def _number_heading_styles(self):
|
|
797
|
+
"Patch w:numPr into Heading1-6 styles, binding them to the generated heading numbering"
|
|
798
|
+
for i in range(6):
|
|
799
|
+
st = self.sroot.find(f'{{{W}}}style[@{{{W}}}styleId="Heading{i + 1}"]')
|
|
800
|
+
if st is None: continue
|
|
801
|
+
ppr = st.find(qn('w:pPr'))
|
|
802
|
+
if ppr is None:
|
|
803
|
+
ppr = E('w:pPr')
|
|
804
|
+
rpr = st.find(qn('w:rPr'))
|
|
805
|
+
st.insert(list(st).index(rpr) if rpr is not None else len(st), ppr)
|
|
806
|
+
np = E('w:numPr', E('w:ilvl', {'w:val': i}), E('w:numId', {'w:val': self.headnum}))
|
|
807
|
+
pos = next((j for j, c in enumerate(ppr) if etree.QName(c).localname not in self.PPR_PRE_NUMPR), len(ppr))
|
|
808
|
+
ppr.insert(pos, np)
|
|
809
|
+
|
|
810
|
+
def styles_xml(self):
|
|
811
|
+
"word/styles.xml: the merged reference styles, plus stub definitions for undefined custom styles (pandoc's mechanism, plus our warning)"
|
|
812
|
+
if self.headnum: self._number_heading_styles()
|
|
813
|
+
root = self.sroot
|
|
814
|
+
for name, (kind, sid) in self.stubs.items():
|
|
815
|
+
root.append(E('w:style', {'w:type': kind, 'w:customStyle': 1, 'w:styleId': sid}, # chkstyle: ignore-node
|
|
816
|
+
E('w:name', {'w:val': name}),
|
|
817
|
+
E('w:basedOn', {'w:val': 'BodyText' if kind == 'paragraph' else 'DefaultParagraphFont'}),
|
|
818
|
+
E('w:qFormat')))
|
|
819
|
+
if self.has_controls and 'placeholder text' not in self.refstyles:
|
|
820
|
+
root.append(E('w:style', {'w:type': 'character', 'w:styleId': 'PlaceholderText'}, # chkstyle: ignore-node
|
|
821
|
+
E('w:name', {'w:val': 'Placeholder Text'}),
|
|
822
|
+
E('w:basedOn', {'w:val': 'DefaultParagraphFont'}), E('w:semiHidden'),
|
|
823
|
+
E('w:rPr', E('w:color', {'w:val': '808080'}))))
|
|
824
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
825
|
+
|
|
826
|
+
def harvest_footnotes(self, els):
|
|
827
|
+
"Split out footnote endnote sections, indexing their li definitions by id; returns body elements"
|
|
828
|
+
body, fn = [], []
|
|
829
|
+
for el in els:
|
|
830
|
+
is_notes = isinstance(el, Element) and _tag(el) == 'section' and 'footnotes' in _classes(el)
|
|
831
|
+
(fn if is_notes else body).append(el)
|
|
832
|
+
self.fndefs.update({_get(li, 'id'): li for sec in fn for li in _walk(sec) if _tag(li) == 'li' and _get(li, 'id')})
|
|
833
|
+
return body
|
|
834
|
+
|
|
835
|
+
BOOKMARKABLE = {'p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'}
|
|
836
|
+
|
|
837
|
+
def to_docx(self, mdhtml, dest):
|
|
838
|
+
root = parse_frag(mdhtml)
|
|
839
|
+
nodes = self.harvest_footnotes(root.children)
|
|
840
|
+
self.idtext, self.reftarget = {}, {}
|
|
841
|
+
for e in (e for node in nodes if isinstance(node, Element) for e in _walk(node)):
|
|
842
|
+
if not (i := _get(e, 'id')): continue
|
|
843
|
+
self.idtext[i] = ' '.join(e.to_text().split())
|
|
844
|
+
t = _tag(e)
|
|
845
|
+
if t in ('figure', 'table'): self.reftarget[i] = 'caption'
|
|
846
|
+
elif t in self.BOOKMARKABLE: self.reftarget[i] = 'block'
|
|
847
|
+
self.ids = set(self.idtext)
|
|
848
|
+
blocks = self.block_nodes(nodes)
|
|
849
|
+
docxml = self.document(blocks)
|
|
850
|
+
parts = {} # archive name -> (bytes, content-type kind); each also gets a document rel
|
|
851
|
+
if self.nums or self.headnum or self.xnums:
|
|
852
|
+
parts['word/numbering.xml'] = (self.numbering_xml(), 'numbering')
|
|
853
|
+
if self.tmplnum is None: self.rels.append((self.rid(), f'{R}/numbering', 'numbering.xml', False))
|
|
854
|
+
if self.fnotes:
|
|
855
|
+
parts['word/footnotes.xml'] = (self.footnotes_xml(), 'footnotes')
|
|
856
|
+
self.rels.append((self.rid(), f'{R}/footnotes', 'footnotes.xml', False))
|
|
857
|
+
if self.bound: self.rels.append((self.rid(), f'{R}/customXml', '../customXml/item1.xml', False))
|
|
858
|
+
extra = [(name, kind) for name, (data, kind) in parts.items()]
|
|
859
|
+
with zipfile.ZipFile(dest, 'w', zipfile.ZIP_DEFLATED) as zo:
|
|
860
|
+
for i in self.refz.infolist():
|
|
861
|
+
if i.filename == 'word/document.xml': zo.writestr(i.filename, docxml)
|
|
862
|
+
elif i.filename in parts: pass # replaced below (e.g. a template numbering.xml we merged)
|
|
863
|
+
elif i.filename == 'word/_rels/document.xml.rels': zo.writestr(i.filename, self.relsxml())
|
|
864
|
+
elif i.filename == '[Content_Types].xml': zo.writestr(i.filename, self.content_types(extra))
|
|
865
|
+
elif i.filename == 'word/styles.xml': zo.writestr(i.filename, self.styles_xml())
|
|
866
|
+
elif i.filename == 'word/settings.xml' and self.has_fields: zo.writestr(i.filename, self.settings_xml())
|
|
867
|
+
else: zo.writestr(i.filename, self.refz.read(i.filename))
|
|
868
|
+
for name, (data, kind) in parts.items(): zo.writestr(name, data)
|
|
869
|
+
if self.fn_rels: zo.writestr('word/_rels/footnotes.xml.rels', self.fn_relsxml())
|
|
870
|
+
if self.bound:
|
|
871
|
+
zo.writestr('customXml/item1.xml', self.bind_item_xml())
|
|
872
|
+
zo.writestr('customXml/itemProps1.xml', self.bind_props_xml())
|
|
873
|
+
zo.writestr('customXml/_rels/item1.xml.rels', self.bind_rels_xml())
|
|
874
|
+
for name, data in self.media.items(): zo.writestr(name, data)
|
|
875
|
+
return self.warnings
|
|
876
|
+
|
|
877
|
+
SETT_AFTER_UPDATE = {'footnotePr', 'endnotePr', 'compat', 'rsids', 'mathPr', 'attachedSchema', # chkstyle: ignore-node
|
|
878
|
+
'themeFontLang', 'clrSchemeMapping', 'doNotIncludeSubdocsInStats',
|
|
879
|
+
'doNotAutoCompressPictures', 'forceUpgrade', 'captions', 'readModeInkLockDown',
|
|
880
|
+
'smartTagType', 'shapeDefaults', 'doNotEmbedSmartTags', 'decimalSymbol',
|
|
881
|
+
'listSeparator', 'docId', 'discardImageEditingData', 'defaultImageDpi',
|
|
882
|
+
'docVars', 'chartTrackingRefBased'}
|
|
883
|
+
|
|
884
|
+
def settings_xml(self):
|
|
885
|
+
"word/settings.xml with w:updateFields added (schema-ordered), so Word refreshes REF fields on open"
|
|
886
|
+
root = etree.fromstring(self.refz.read('word/settings.xml'))
|
|
887
|
+
pos = next((i for i, c in enumerate(root) if etree.QName(c).localname in self.SETT_AFTER_UPDATE), len(root))
|
|
888
|
+
root.insert(pos, E('w:updateFields', {'w:val': 'true'}))
|
|
889
|
+
return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
|
|
890
|
+
|
|
891
|
+
def mustache_fields(body, syntax, form):
|
|
892
|
+
"Mustache variables as live Word `MERGEFIELD`s; section markers (`#`/`/`/`^` sigils) stay literal text"
|
|
893
|
+
if mustache_kind(body) == 'section': return '{{' + body + '}}'
|
|
894
|
+
return 'field', f'MERGEFIELD {body.strip()}'
|
|
895
|
+
|
|
896
|
+
|
|
897
|
+
def jinja_literal(body, syntax, form):
|
|
898
|
+
"Jinja tokens re-spelled canonically (`{{ x }}`/`{% x %}`) as text, for docxtpl-style downstream pipelines"
|
|
899
|
+
o, c = ('{%', '%}') if syntax == 'jinja-stmt' else ('{{', '}}')
|
|
900
|
+
return f'{o} {body.strip()} {c}'
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def convert(mdhtml, dest, reference=None, base=None, reftypes=None, number_headings=None, tmpl=None):
|
|
904
|
+
"""Convert an MDHTML string or mutable fast5ever DOM to a docx file at `dest`; returns warnings.
|
|
905
|
+
`reference` is a reference docx path, or a list of them: the first supplies the whole archive
|
|
906
|
+
(default, or when None: the built-in template), later entries contribute styles only, later-wins -
|
|
907
|
+
each a .docx path, a raw styles/numbering .xml path, or a fastpylight theme name (whose Hl*/Source
|
|
908
|
+
Code styles are generated; see styles.theme_ref). Default adds 'github_light' when fastpylight is
|
|
909
|
+
installed, so code blocks are colored; pass a bare reference for plain code. Relative image srcs
|
|
910
|
+
resolve against `base` ('.'). Cross-references (`data-ref` anchors from Markdown `[@sec-x]`) become
|
|
911
|
+
live REF fields; `reftypes` maps type tokens to (singular, plural) prefix words beyond the built-in
|
|
912
|
+
`sec`, and `number_headings` (a styles.SCHEMES name such as 'legal', or a {lvlText: numFmt} dict, one entry per heading level)
|
|
913
|
+
numbers the headings via a multilevel list so `\\w` fields resolve. Template tokens are dropped
|
|
914
|
+
unless `tmpl` is given: a callable `(body, syntax, form) -> str` for a literal text run,
|
|
915
|
+
`('field', instr)` for a live field, `('control', name)` for an interactive plain-text content
|
|
916
|
+
control, `('bound', name)` for a content control data-bound to a shared per-variable XML node
|
|
917
|
+
(same-name controls stay in sync as one is filled), or None to drop - `mustache_fields` and
|
|
918
|
+
`jinja_literal` are ready-made recipes."""
|
|
919
|
+
return Converter(reference, base, reftypes, number_headings, tmpl).to_docx(mdhtml, dest)
|