mdhtml2docx 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mdhtml2docx/convert.py ADDED
@@ -0,0 +1,919 @@
1
+ """Convert MDHTML fragments to docx.
2
+
3
+ Write-only, reference-archive architecture: the reference template supplies styles/theme/fonts,
4
+ we generate word/document.xml (plus footnotes/numbering/media parts as needed) into a copy of its
5
+ archive. Block and inline walkers mirror the MDHTML element inventory; STYLE_MAP names every
6
+ style we emit."""
7
+ import posixpath, re, zipfile
8
+ from copy import deepcopy
9
+ from pathlib import Path
10
+ from fast5ever import Comment, Element, Node, Text
11
+ from lxml import etree
12
+ from mdhtml import parse_mdhtml
13
+ from mdhtml.export import REFTYPES, SCHEMES, decode_raw, group_plan, mustache_kind, ref_tokens, ref_variant
14
+ from .styles import STYLE_MAP, style_id, theme_styles
15
+ from .styles import ref_path as _refpath
16
+ from .wml import *
17
+ from .wml import qn
18
+ from .hilite import segments, tokenize
19
+
20
+ __all__ = ['convert', 'mustache_fields', 'jinja_literal']
21
+
22
+ def _sid(key): return style_id(STYLE_MAP[key])
23
+
24
+ BLOCK_TAGS = set(('address article aside blockquote details dialog div dl fieldset figure footer form h1 h2 h3 h4 h5 h6 '
25
+ 'header hgroup hr main menu nav ol p pre search section table ul').split())
26
+ INERT_TAGS = {'base', 'link', 'meta', 'style', 'template', 'title'}
27
+
28
+ def _tag(el): return el.name
29
+ def _get(el, key, default=None): return el.attrs.get(key, default)
30
+ def _els(el): return [n for n in el.children if isinstance(n, Element)]
31
+ def _walk(el):
32
+ yield el
33
+ for child in _els(el): yield from _walk(child)
34
+ def _classes(el): return (_get(el, 'class') or '').split()
35
+ def _is_raw(el): return _tag(el) == 'script' and _get(el, 'type') == 'application/vnd.mdhtml.raw'
36
+
37
+ def parse_frag(src):
38
+ "Parse an MDHTML body fragment, or return an existing mutable fragment"
39
+ if isinstance(src, Node): return src
40
+ if not isinstance(src, str): raise TypeError('input must be an MDHTML string or fast5ever node')
41
+ return parse_mdhtml(src)
42
+
43
+ class Converter:
44
+ def __init__(self, reference=None, base=None, reftypes=None, number_headings=None, tmpl=None):
45
+ self.tmpl = tmpl
46
+ if reference is None: reference = [_refpath()] + (['github_light'] if tokenize else [])
47
+ elif not isinstance(reference, (list, tuple)): reference = [reference]
48
+ self.refz = zipfile.ZipFile(reference[0] or _refpath())
49
+ self.contribs = reference[1:] # styles-only contributors: docx paths, styles/numbering .xml paths, theme names
50
+ self.base = Path(base or '.')
51
+ self.rels = [] # (rId, type, target, external?) beyond the template's
52
+ self.fn_rels = [] # same, but for the footnotes part
53
+ self.media = {} # archive name -> bytes
54
+ self.warnings = []
55
+ self._rid = 1000 # clear of template rIds
56
+ self.bq = 0 # blockquote nesting depth
57
+ self._bkid = 0 # bookmark id counter
58
+ self._imgn = 0 # image part counter (doubles as docPr id)
59
+ self._urlrids = {} # hyperlink URL -> rId
60
+ self.nums = [] # (numId, abstractNumId, start) per list instance
61
+ self.fndefs = {} # endnote li elements by id, harvested before the walk
62
+ self.fnids = {} # endnote id -> footnote w:id
63
+ self.fnotes = [] # (w:id, [footnote blocks])
64
+ self.stubs = {} # undefined custom-style name -> (kind, styleId)
65
+ self.first = True # next body paragraph is a 'First Paragraph' (doc start; reset after FIRST_AFTER blocks)
66
+ self._bknames = {} # element id -> Word-legal bookmark name
67
+ self.has_fields = False
68
+ self.has_controls = False
69
+ self.bound = [] # distinct bound-control names, in first-appearance order
70
+ self.contrib_numels, self.contrib_styleels = [], []
71
+ self.reftypes = REFTYPES | (reftypes or {})
72
+ tdoc = etree.fromstring(self.refz.read('word/document.xml'))
73
+ self.sectpr = deepcopy(tdoc.find(f'{{{W}}}body/{{{W}}}sectPr'))
74
+ pg, mar = self.sectpr.find(qn('w:pgSz')), self.sectpr.find(qn('w:pgMar'))
75
+ self.content_w = int(pg.get(qn('w:w'))) - int(mar.get(qn('w:left'))) - int(mar.get(qn('w:right')))
76
+ self.sroot = etree.fromstring(self.refz.read('word/styles.xml'))
77
+ for r in self.contribs: self._merge_styles(r)
78
+ self.refstyles = {s.find(qn('w:name')).get(qn('w:val')).lower(): s.get(qn('w:styleId')) for s in self.sroot.iter(qn('w:style'))}
79
+ if missing := [n for n in STYLE_MAP.values() if n.lower() not in self.refstyles]:
80
+ raise ValueError(f'reference doc lacks dialect styles (map/template drift?): {missing}')
81
+ self.hlstyles = {n.removeprefix('hl ').replace(' ', '.'): sid for n, sid in self.refstyles.items() if n.startswith('hl ')}
82
+ self.tmplnum = (etree.fromstring(self.refz.read('word/numbering.xml')) if 'word/numbering.xml' in self.refz.namelist() else None)
83
+ used = [int(v) for e in ([] if self.tmplnum is None else self.tmplnum.iter())
84
+ for a in ('w:numId', 'w:abstractNumId') if (v := e.get(qn(a))) and v.lstrip('-').isdigit()]
85
+ self._numid = self._absbase = max(used, default=-1) + 1 # num and abstract ids both offset past the template's
86
+ if isinstance(number_headings, str):
87
+ if number_headings not in SCHEMES: raise ValueError(f'unknown numbering scheme {number_headings!r}')
88
+ number_headings = SCHEMES[number_headings]
89
+ self.scheme = list(number_headings.items()) if number_headings else None
90
+ self.headnum = None
91
+ if number_headings and self.sroot.find(f'{{{W}}}style[@{{{W}}}styleId="Heading1"]/{{{W}}}pPr/{{{W}}}numPr') is None:
92
+ self._numid += 1
93
+ self.headnum = self._numid
94
+ self._adopt_contrib_nums()
95
+
96
+ def _merge_styles(self, ref):
97
+ "Merge a contributor's w:style elements into self.sroot, later-wins on style id or name"
98
+ def keys(e):
99
+ nm = e.find(qn('w:name'))
100
+ return {(e.get(qn('w:styleId')) or '').lower(), '' if nm is None else nm.get(qn('w:val')).lower()} - {''}
101
+ new = (etree.fromstring(zipfile.ZipFile(ref).read('word/styles.xml')).findall(qn('w:style'))
102
+ if isinstance(ref, Path) and ref.suffix == '.docx' or str(ref).endswith('.docx')
103
+ else self._xml_contrib(ref) if str(ref).endswith('.xml') else theme_styles(ref))
104
+ for s in new:
105
+ ks = keys(s)
106
+ for old in [o for o in self.sroot.findall(qn('w:style')) if keys(o) & ks]: self.sroot.remove(old)
107
+ self.sroot.append(s)
108
+
109
+ def _xml_contrib(self, ref):
110
+ "Styles from a raw .xml contributor, stashing its abstractNum/num elements for numbering adoption"
111
+ root = etree.parse(str(ref)).getroot()
112
+ self.contrib_numels += root.findall(qn('w:abstractNum')) + root.findall(qn('w:num'))
113
+ styles = root.findall(qn('w:style'))
114
+ self.contrib_styleels += styles
115
+ return styles
116
+
117
+ def _adopt_contrib_nums(self):
118
+ "Renumber .xml contributors' numbering ids past ours, remapping their styles' numPr references"
119
+ self.xabs = [e for e in self.contrib_numels if etree.QName(e).localname == 'abstractNum']
120
+ self.xnums = [e for e in self.contrib_numels if etree.QName(e).localname == 'num']
121
+ amap, nmap = {}, {}
122
+ for e in self.xabs:
123
+ self._numid += 1
124
+ amap[e.get(qn('w:abstractNumId'))] = self._numid
125
+ e.set(qn('w:abstractNumId'), str(self._numid))
126
+ for e in self.xnums:
127
+ self._numid += 1
128
+ nmap[e.get(qn('w:numId'))] = self._numid
129
+ e.set(qn('w:numId'), str(self._numid))
130
+ ref = e.find(qn('w:abstractNumId'))
131
+ if ref is not None and ref.get(qn('w:val')) in amap: ref.set(qn('w:val'), str(amap[ref.get(qn('w:val'))]))
132
+ for s in self.contrib_styleels:
133
+ for nid in s.iter(qn('w:numId')):
134
+ if nid.get(qn('w:val')) in nmap: nid.set(qn('w:val'), str(nmap[nid.get(qn('w:val'))]))
135
+
136
+ def hlsid(self, scope):
137
+ "Hl* style id for a dotted scope: exact, else progressively shorter prefixes (tree-sitter resolution)"
138
+ parts = (scope or '').split('.')
139
+ while parts:
140
+ if sid := self.hlstyles.get('.'.join(parts)): return sid
141
+ parts.pop()
142
+
143
+ def rid(self):
144
+ self._rid += 1
145
+ return f'rId{self._rid}'
146
+ def warn(self, msg): self.warnings.append(msg)
147
+
148
+ # ---- inline level -------------------------------------------------------
149
+ def rpr(self, fmt):
150
+ "w:rPr for a formatting context dict, in CT_RPr child order; None if empty"
151
+ kids = []
152
+ if s := fmt.get('rstyle'): kids.append(E('w:rStyle', {'w:val': s}))
153
+ if fmt.get('b'): kids += [E('w:b'), E('w:bCs')]
154
+ if fmt.get('i'): kids += [E('w:i'), E('w:iCs')]
155
+ if fmt.get('strike'): kids.append(E('w:strike'))
156
+ if fmt.get('mark'): kids.append(E('w:highlight', {'w:val': 'yellow'}))
157
+ if fmt.get('u'): kids.append(E('w:u', {'w:val': 'single'}))
158
+ if v := fmt.get('vert'): kids.append(E('w:vertAlign', {'w:val': v}))
159
+ return E('w:rPr', *kids) if kids else None
160
+
161
+ def text_runs(self, text, fmt):
162
+ "Runs for a text node; newlines collapse to spaces (pre-context text never comes here)"
163
+ text = re.sub(r'\s+', ' ', text)
164
+ if not text: return []
165
+ t = E('w:t', text)
166
+ if text != text.strip(): t.set(qn('xml:space'), 'preserve')
167
+ return [E('w:r', self.rpr(fmt), t)]
168
+
169
+ def link(self, el, fmt):
170
+ "w:hyperlink for `a`: internal '#x' -> anchor, external -> relationship (deduped per URL); data-ref -> field"
171
+ if _get(el, 'data-ref') is not None:
172
+ fld = self.ref_fld(el, fmt)
173
+ return self.ref_prefix(el, fmt) + fld
174
+ href = _get(el, 'href')
175
+ if not href: return self.runs(el, fmt)
176
+ runs = self.runs(el, fmt | {'rstyle': _sid('hyperlink')})
177
+ if href.startswith('#'): return [E('w:hyperlink', {'w:anchor': self.bkname(href[1:])}, *runs)]
178
+ return self.external_link(href, runs)
179
+
180
+ def external_link(self, href, runs):
181
+ if href not in self._urlrids:
182
+ self._urlrids[href] = self.rid()
183
+ self.rels.append((self._urlrids[href], f'{R}/hyperlink', href, True))
184
+ return [E('w:hyperlink', {'r:id': self._urlrids[href]}, *runs)]
185
+
186
+ REFSWITCH = dict(full=r'\w', rel=r'\r', leaf=r'\n', text='', page=None)
187
+
188
+ def bkname(self, id):
189
+ "Word-legal bookmark name for `id` (letter first, word chars only), stable within the document"
190
+ if id not in self._bknames:
191
+ nm = re.sub(r'\W', '_', id)
192
+ if not nm[:1].isalpha(): nm = 'B' + nm
193
+ while nm in self._bknames.values(): nm += '_'
194
+ self._bknames[id] = nm
195
+ return self._bknames[id]
196
+
197
+ def ref_prefix(self, el, fmt, plural=False):
198
+ "Literal runs before a reference field: override text, the type prefix word, or nothing for bare and caption refs"
199
+ pre = el.to_text().strip()
200
+ if not pre:
201
+ tgt = (_get(el, 'href') or '#')[1:]
202
+ if 'bare' in ref_tokens(_get(el, 'data-ref')) or self.reftarget.get(tgt) == 'caption': return []
203
+ t = tgt.split('-')[0]
204
+ if t not in self.reftypes: raise ValueError(f'unknown reference type {t!r}; pass reftypes= to define its prefix')
205
+ pre = self.reftypes[t][plural]
206
+ return self.text_runs(pre + ' ', fmt)
207
+
208
+ def ref_fld(self, el, fmt):
209
+ """REF/PAGEREF field for a cross-reference `a`, with a cached placeholder Word replaces on update.
210
+ Heading/paragraph targets number via `\\w`; caption targets return their bookmarked 'Label N' text
211
+ (or the number-only `_n` bookmark for bare/leaf/rel refs), so `\\w` never applies to them."""
212
+ tgt = (_get(el, 'href') or '#')[1:]
213
+ tokens = ref_tokens(_get(el, 'data-ref'))
214
+ if tgt not in self.reftarget:
215
+ raise ValueError(f'cross-reference target #{tgt} not found (targets are headings, paragraphs, figures, and tables with ids)')
216
+ kind = ref_variant(tokens)
217
+ nm, self.has_fields = self.bkname(tgt), True
218
+ if kind == 'page': instr, cached = rf' PAGEREF {nm} \h ', '#'
219
+ elif self.reftarget[tgt] == 'caption':
220
+ bare = 'bare' in tokens or kind in ('leaf', 'rel')
221
+ instr, cached = (rf' REF {nm}_n \h ' if bare else rf' REF {nm} \h '), '#'
222
+ else:
223
+ sw = self.REFSWITCH[kind]
224
+ instr = rf' REF {nm} {sw} \h ' if sw else rf' REF {nm} \h '
225
+ cached = self.idtext.get(tgt, '#') if kind == 'text' else '#'
226
+ return [E('w:fldSimple', {'w:instr': instr}, E('w:r', self.rpr(fmt), E('w:t', cached)))]
227
+
228
+ def ref_group(self, el, fmt):
229
+ "data-refs span: one pluralized prefix for a same-type group, per-item singular prefixes for mixed types; never range-collapsed"
230
+ refs = [c for c in _els(el) if _tag(c) == 'a']
231
+ types = [(_get(a, 'href') or '#')[1:].split('-')[0] for a in refs]
232
+ out = []
233
+ for (sep, pre, plural), a in zip(group_plan(types), refs):
234
+ if sep: out += self.text_runs(sep, fmt)
235
+ if pre: out += self.ref_prefix(a, fmt, plural=plural)
236
+ out += self.ref_fld(a, fmt)
237
+ return out
238
+
239
+ def custom_style(self, el, kind):
240
+ "Style id for an explicit custom-style attr (stubbed + warned if undefined), else a class matching a reference style name"
241
+ if cs := _get(el, 'custom-style'):
242
+ if cs.lower() in self.refstyles: return self.refstyles[cs.lower()]
243
+ if cs not in self.stubs:
244
+ self.stubs[cs] = (kind, re.sub(r'\W', '', cs) or f'Custom{len(self.stubs)}')
245
+ self.warn(f'custom style {cs!r} not in reference doc; stub injected')
246
+ return self.stubs[cs][1]
247
+ return next((self.refstyles[c.lower()] for c in _classes(el) if c.lower() in self.refstyles), None)
248
+
249
+ def span(self, el, fmt):
250
+ "Inline span: math -> inline m:oMath zone (linear source, dialect-agnostic), custom style -> rStyle, else transparent"
251
+ if _get(el, 'data-refs') is not None: return self.ref_group(el, fmt)
252
+ if 'math' in _classes(el): return [self.omath(el)]
253
+ if sid := self.custom_style(el, 'character'): return self.runs(el, fmt | {'rstyle': sid})
254
+ return self.runs(el, fmt)
255
+
256
+ def omath(self, el):
257
+ "An m:oMath zone holding `el`'s text as linear-format math runs"
258
+ return E('m:oMath', E('m:r', E('m:t', el.to_text(), {'xml:space': 'preserve'})))
259
+
260
+ def fnref(self, el, fmt):
261
+ "Footnote-reference run for a sup>a.footnote-ref, or None when `el` is an ordinary sup"
262
+ children = _els(el)
263
+ a = children[0] if len(children) == 1 and _tag(children[0]) == 'a' else None
264
+ if a is None or 'footnote-ref' not in _classes(a): return None
265
+ key = (_get(a, 'href') or '#')[1:]
266
+ if key not in self.fndefs:
267
+ self.warn(f'footnote reference #{key} has no definition; dropped')
268
+ return []
269
+ if key not in self.fnids:
270
+ self.fnids[key] = len(self.fnids) + 1
271
+ self.fnotes.append((self.fnids[key], self.fn_blocks(self.fndefs[key])))
272
+ return [E('w:r', E('w:rPr', E('w:rStyle', {'w:val': _sid('footnoteref')})),
273
+ E('w:footnoteReference', {'w:id': self.fnids[key]}))]
274
+
275
+ def fn_blocks(self, li):
276
+ "Footnote body: the li's blocks in footnote-text style, backref stripped, reference mark prepended"
277
+ save = self.rels, self._urlrids, self.first
278
+ self.rels, self._urlrids = self.fn_rels, {} # rel ids are per-part (see fn_relsxml)
279
+ try:
280
+ blks = [b for kind, val in self.li_parts(li) # chkstyle: ignore-node
281
+ for b in ([self.para(self.group_runs(val, {}), 'footnotetext')] if kind == 'inline'
282
+ else self.block(val, 'footnotetext'))]
283
+ finally: self.rels, self._urlrids, self.first = save
284
+ if not blks: blks = [self.para([], 'footnotetext')]
285
+ mark = E('w:r', E('w:rPr', E('w:rStyle', {'w:val': _sid('footnoteref')})), E('w:footnoteRef'))
286
+ blks[0].insert(1, E('w:r', E('w:t', ' ', {'xml:space': 'preserve'})))
287
+ blks[0].insert(1, mark)
288
+ return blks
289
+
290
+ def image(self, el, fmt, alt=None):
291
+ "Embed a local image (dimensions sniffed, width/height px attrs override); remote srcs degrade to a link"
292
+ src = _get(el, 'src') or ''
293
+ if alt is None: alt = _get(el, 'alt') or src
294
+ if re.match(r'[a-z][a-z0-9+.-]*://', src):
295
+ self.warn(f'remote image not embedded: {src}')
296
+ return self.external_link(src, self.text_runs(alt, fmt | {'rstyle': _sid('hyperlink')}))
297
+ try: data = (self.base/src).read_bytes()
298
+ except OSError:
299
+ self.warn(f'image not found: {src}; alt text emitted')
300
+ return self.text_runs(alt, fmt)
301
+ pw, ph, dx, dy = imgsize(data) or (300, 200, 96, 96)
302
+ cx, cy = round(pw * 914400 / dx), round(ph * 914400 / dy)
303
+ w_, h_ = _get(el, 'width'), _get(el, 'height')
304
+ if w_: cx = round(float(w_) * EMU_PER_PX)
305
+ if h_: cy = round(float(h_) * EMU_PER_PX)
306
+ if w_ and not h_: cy = round(cx * ph / pw)
307
+ if h_ and not w_: cx = round(cy * pw / ph)
308
+ self._imgn += 1
309
+ name = f'word/media/image{self._imgn}{Path(src).suffix.lower() or ".bin"}'
310
+ self.media[name] = data
311
+ rid = self.rid()
312
+ self.rels.append((rid, f'{R}/image', name.removeprefix('word/'), False))
313
+ return [E('w:r', drawing(rid, self._imgn, cx, cy, alt))]
314
+
315
+ def inline(self, el, fmt):
316
+ "Run-level elements for inline `el` under formatting context `fmt` (dict; copied on change)"
317
+ tag = _tag(el)
318
+ if tag == 'em': out = self.runs(el, fmt | {'i': True})
319
+ elif tag == 'strong': out = self.runs(el, fmt | {'b': True})
320
+ elif tag == 'code': out = self.runs(el, fmt | {'rstyle': _sid('codeinline')})
321
+ elif tag == 'a': out = self.link(el, fmt)
322
+ elif tag == 'del': out = self.runs(el, fmt | {'strike': True})
323
+ elif tag == 'mark': out = self.runs(el, fmt | {'mark': True})
324
+ elif tag == 'u': out = self.runs(el, fmt | {'u': True})
325
+ elif tag == 'sub': out = self.runs(el, fmt | {'vert': 'subscript'})
326
+ elif tag == 'sup':
327
+ fn = self.fnref(el, fmt)
328
+ out = fn if fn is not None else self.runs(el, fmt | {'vert': 'superscript'})
329
+ elif tag == 'span': out = self.span(el, fmt)
330
+ elif tag == 'img': out = self.image(el, fmt)
331
+ elif tag == 'br': out = [E('w:r', self.rpr(fmt), E('w:br'))]
332
+ elif tag == 'input': # task-list checkbox
333
+ if _get(el, 'type') != 'checkbox':
334
+ self.warn(f'unhandled inline <input type={_get(el, "type")!r}>; dropped')
335
+ return []
336
+ g = '☒' if _get(el, 'checked') is not None else '☐'
337
+ out = [E('w:r', self.rpr(fmt), E('w:t', g + ' ', {'xml:space': 'preserve'}))]
338
+ elif tag == 'script': out = self.rawxml(el)
339
+ elif tag == 'template': out = self.tmpl_runs(el, fmt, 'inline')
340
+ elif tag in INERT_TAGS: out = []
341
+ else: # unknown inline (abbr etc): recurse transparently
342
+ out = self.runs(el, fmt)
343
+ return out
344
+
345
+ def runs(self, el, fmt):
346
+ "Run-level elements for an element's ordered text and element children"
347
+ return [r for node in el.children for r in self.inline_node(node, fmt)]
348
+
349
+ def inline_node(self, node, fmt):
350
+ if isinstance(node, Text): return self.text_runs(node.text, fmt)
351
+ if isinstance(node, Element):
352
+ if _tag(node) == 'a' and 'footnote-backref' in _classes(node): return []
353
+ return self.inline(node, fmt)
354
+ return []
355
+
356
+ # ---- block level --------------------------------------------------------
357
+ def para(self, runs, style='body', extra=None, sid=None):
358
+ "A w:p with `style` (STYLE_MAP key, or `sid` style-id override) and optional extra pPr children (schema order!)"
359
+ ppr = E('w:pPr', E('w:pStyle', {'w:val': sid or _sid(style)}), *(extra or []))
360
+ return E('w:p', ppr, *runs)
361
+
362
+ def bookmark(self, el, runs):
363
+ "Wrap `runs` in a bookmark when `el` carries an id (target for internal links)"
364
+ if not (i := _get(el, 'id')): return runs
365
+ self._bkid += 1
366
+ return [E('w:bookmarkStart', {'w:id': self._bkid, 'w:name': self.bkname(i)}),
367
+ *runs, E('w:bookmarkEnd', {'w:id': self._bkid})]
368
+
369
+ def codeblock(self, el):
370
+ "Source Code paragraph, lines joined with w:br; Hl* character styles when a language class names one"
371
+ children = _els(el)
372
+ code = children[0] if children and _tag(children[0]) == 'code' else el
373
+ lang = next((c.removeprefix('language-') for c in _classes(code) if c.startswith('language-')), None)
374
+ text = code.to_text().rstrip('\n')
375
+ segs = (segments(text, lang) if self.hlstyles else None) or [(text, None)]
376
+ runs = []
377
+ for txt, scope in segs:
378
+ for j, part in enumerate(txt.split('\n')):
379
+ if j: runs.append(E('w:r', E('w:br')))
380
+ if not part: continue
381
+ sid = self.hlsid(scope)
382
+ runs.append(E('w:r', E('w:rPr', E('w:rStyle', {'w:val': sid})) if sid else None,
383
+ E('w:t', part, {'xml:space': 'preserve'})))
384
+ return [self.para(runs, 'codeblock')]
385
+
386
+ def qindent(self):
387
+ "Extra indent for paragraphs in nested blockquotes (Quote style itself carries the first level)"
388
+ return [E('w:ind', {'w:left': 720 * self.bq})] if self.bq > 1 else None
389
+
390
+ # ---- lists --------------------------------------------------------------
391
+ def list_el(self, el, ilvl=0):
392
+ "A ul/ol: fresh num instance (so each ordered list restarts), items at level `ilvl`"
393
+ self._numid += 1
394
+ nid = self._numid
395
+ self.nums.append((nid, 0 if _tag(el) == 'ul' else 1, int(_get(el, 'start', 1))))
396
+ return [b for li in _els(el) if _tag(li) == 'li' for b in self.li(li, nid, min(ilvl, 8))]
397
+
398
+ def li_parts(self, el):
399
+ "Split mixed li content into ('inline', [nodes]) groups and ('block', child) items, in order"
400
+ parts = []
401
+ def add(x):
402
+ if isinstance(x, Text) and not x.text.strip() and '\n' in x.text: return
403
+ if parts and parts[-1][0] == 'inline': parts[-1][1].append(x)
404
+ else: parts.append(('inline', [x]))
405
+ for node in el.children:
406
+ if isinstance(node, Element) and _tag(node) in BLOCK_TAGS: parts.append(('block', node))
407
+ elif isinstance(node, (Text, Element)): add(node)
408
+ return parts
409
+
410
+ def group_runs(self, nodes, fmt):
411
+ "Runs for a mixed list of text and inline nodes"
412
+ return [r for node in nodes for r in self.inline_node(node, fmt)]
413
+
414
+ def li(self, li, nid, ilvl):
415
+ "Blocks for one list item: the first paragraph carries the number, the rest continue indented"
416
+ numpr = [E('w:numPr', E('w:ilvl', {'w:val': ilvl}), E('w:numId', {'w:val': nid}))]
417
+ cont = [E('w:ind', {'w:left': 720 * (ilvl + 1)})]
418
+ out = []
419
+ for kind, val in self.li_parts(li):
420
+ if kind == 'inline': out.append(self.para(self.group_runs(val, {}), 'list', numpr if not out else cont))
421
+ elif _tag(val) in ('ul', 'ol'): out += self.list_el(val, ilvl + 1)
422
+ elif _tag(val) == 'p': out.append(self.para(self.runs(val, {}), 'list', numpr if not out else cont))
423
+ else: out += self.block(val, 'list')
424
+ return out or [self.para([], 'list', numpr)]
425
+
426
+ # ---- tables -------------------------------------------------------------
427
+ def table_grid(self, rows):
428
+ "Resolve row/colspans into per-row cell placements: ('cell', ci, el, cs, rs) / ('cont', ci, width)"
429
+ spans, placed, ncols = {}, [], 0
430
+ for ri, tr in enumerate(rows):
431
+ ci, rowcells = 0, []
432
+ def _skip(ci):
433
+ while (ri, ci) in spans:
434
+ wd = spans.pop((ri, ci))
435
+ rowcells.append(('cont', ci, wd))
436
+ ci += wd
437
+ return ci
438
+ ci = _skip(ci)
439
+ for cell in _els(tr):
440
+ if _tag(cell) not in ('td', 'th'): continue
441
+ cs, rs = int(_get(cell, 'colspan', 1)), int(_get(cell, 'rowspan', 1))
442
+ rowcells.append(('cell', ci, cell, cs, rs))
443
+ for k in range(1, rs): spans[(ri + k, ci)] = cs
444
+ ci = _skip(ci + cs)
445
+ placed.append(rowcells)
446
+ ncols = max(ncols, ci)
447
+ return placed, ncols
448
+
449
+ def col_widths(self, el, ncols):
450
+ "colwidths tracks -> (dxa list, all_fr?) or (None, False) when absent"
451
+ s = _get(el, 'colwidths') or _get(el, 'data-colwidths')
452
+ if not s: return None, False
453
+ tracks = parse_tracks(s)
454
+ if len(tracks) != ncols:
455
+ self.warn(f'colwidths has {len(tracks)} tracks for {ncols} columns; padding with 1fr')
456
+ tracks = tracks[:ncols] + [('fr', 1.0)] * (ncols - len(tracks))
457
+ fixed = sum(v for k, v in tracks if k == 'dxa')
458
+ frs = sum(v for k, v in tracks if k == 'fr')
459
+ rem = max(self.content_w - fixed, 0)
460
+ dxa = [round(v) if k == 'dxa' else round(v * rem / frs) for k, v in tracks]
461
+ return dxa, fixed == 0
462
+
463
+ def cell_blocks(self, cell, header):
464
+ "Block content of one table cell; header cells bold, align attr honored for inline cells"
465
+ if any(_tag(c) in BLOCK_TAGS for c in _els(cell)): return self.blocks(cell)
466
+ jc = [E('w:jc', {'w:val': _get(cell, 'align')})] if _get(cell, 'align') in ('center', 'right') else None
467
+ return [self.para(self.runs(cell, {'b': True} if header else {}), 'compact', jc)]
468
+
469
+ def table(self, el):
470
+ "w:tbl (+ caption paragraph before, spacer paragraph after)"
471
+ cap = None
472
+ rows, nhead = [], 0
473
+ for sec in _els(el):
474
+ t = _tag(sec)
475
+ if t == 'caption': cap = sec
476
+ elif t == 'thead':
477
+ rows += _els(sec)
478
+ nhead = len(rows)
479
+ elif t in ('tbody', 'tfoot'): rows += _els(sec)
480
+ elif t == 'tr': rows.append(sec)
481
+ placed, ncols = self.table_grid(rows)
482
+ dxa, all_fr = self.col_widths(el, ncols)
483
+ def _tcw(ci, cs):
484
+ if not dxa: return None
485
+ wd = sum(dxa[ci:ci + cs])
486
+ if all_fr: return E('w:tcW', {'w:type': 'pct', 'w:w': round(wd / self.content_w * 5000)})
487
+ return E('w:tcW', {'w:type': 'dxa', 'w:w': wd})
488
+ tblw = (E('w:tblW', {'w:type': 'auto', 'w:w': 0}) if not dxa # chkstyle: ignore-node
489
+ else E('w:tblW', {'w:type': 'pct', 'w:w': 5000}) if all_fr
490
+ else E('w:tblW', {'w:type': 'dxa', 'w:w': sum(dxa)}))
491
+ tblpr = E('w:tblPr', E('w:tblStyle', {'w:val': self.custom_style(el, 'table') or _sid('table')}), tblw, # chkstyle: ignore-node
492
+ E('w:tblLayout', {'w:type': 'fixed'}) if dxa and not all_fr else None,
493
+ E('w:tblLook', {'w:val': '04A0', 'w:firstRow': 1, 'w:lastRow': 0,
494
+ 'w:firstColumn': 0, 'w:lastColumn': 0, 'w:noHBand': 0, 'w:noVBand': 1}))
495
+ gw = dxa or [self.content_w // ncols] * ncols # pandoc's docx reader drops tables whose gridCols lack w:w
496
+ grid = E('w:tblGrid', *[E('w:gridCol', {'w:w': gw[i]}) for i in range(ncols)])
497
+ trs = []
498
+ for ri, rowcells in enumerate(placed):
499
+ tcs = []
500
+ for item in rowcells:
501
+ if item[0] == 'cont':
502
+ _, ci, wd = item
503
+ tcs.append(E('w:tc', E('w:tcPr', _tcw(ci, wd), # chkstyle: ignore-node
504
+ E('w:gridSpan', {'w:val': wd}) if wd > 1 else None,
505
+ E('w:vMerge')), E('w:p')))
506
+ else:
507
+ _, ci, cell, cs, rs = item
508
+ tcpr = E('w:tcPr', _tcw(ci, cs), # chkstyle: ignore-node
509
+ E('w:gridSpan', {'w:val': cs}) if cs > 1 else None,
510
+ E('w:vMerge', {'w:val': 'restart'}) if rs > 1 else None)
511
+ body = self.cell_blocks(cell, ri < nhead)
512
+ if not len(body) or etree.QName(body[-1]).localname != 'p': body.append(E('w:p'))
513
+ tcs.append(E('w:tc', tcpr, *body))
514
+ trs.append(E('w:tr', E('w:trPr', E('w:tblHeader')) if ri < nhead else None, *tcs))
515
+ out = self.caption_para(el, 'tbl', cap)
516
+ return out + [E('w:tbl', tblpr, grid, *trs), E('w:p')]
517
+
518
+ def caption_para(self, el, typ, capel, fmt={}):
519
+ """Numbered caption paragraph: 'Label N: text' with a SEQ field as N. When `el` has an id, the
520
+ label+number span is bookmarked under it (REF target) and the number alone under `<name>_n`.
521
+ Emitted whenever there is a caption or an id; the label word comes from reftypes[typ]."""
522
+ if capel is None and not _get(el, 'id'): return []
523
+ label = self.reftypes[typ][0]
524
+ seq = [E('w:fldSimple', {'w:instr': rf' SEQ {label} \* ARABIC '}, E('w:r', self.rpr(fmt), E('w:t', '#')))]
525
+ self.has_fields = True
526
+ if i := _get(el, 'id'):
527
+ nm = self.bkname(i)
528
+ self._bknames[i + '\0n'] = nm + '_n' # reserve the number-only name too
529
+ self._bkid += 2
530
+ seq = [E('w:bookmarkStart', {'w:id': self._bkid, 'w:name': nm + '_n'}), *seq,
531
+ E('w:bookmarkEnd', {'w:id': self._bkid})]
532
+ runs = [E('w:bookmarkStart', {'w:id': self._bkid - 1, 'w:name': nm}), # chkstyle: ignore-node
533
+ *self.text_runs(label + ' ', fmt), *seq,
534
+ E('w:bookmarkEnd', {'w:id': self._bkid - 1})]
535
+ else: runs = self.text_runs(label + ' ', fmt) + seq
536
+ cap = [] if capel is None else self.runs(capel, fmt)
537
+ if cap: runs += self.text_runs(': ', fmt) + cap
538
+ return [self.para(runs, 'caption')]
539
+
540
+ def figure(self, el):
541
+ "Figure: image paragraph, then its numbered caption paragraph below (Word convention)"
542
+ img = next((c for c in _walk(el) if _tag(c) == 'img'), None)
543
+ capel = next((c for c in _els(el) if _tag(c) == 'figcaption'), None)
544
+ alt = capel.to_text().strip() if capel is not None else None
545
+ out = [] if img is None else [self.para(self.image(img, {}, alt), 'body')]
546
+ return out + self.caption_para(el, 'fig', capel)
547
+
548
+ # Paragraphs directly after these blocks (or at document start) take First Paragraph rather
549
+ # than Body Text, matching pandoc's docx writer exactly, so the two agree on which is "first".
550
+ FIRST_AFTER = {'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'pre', 'ul', 'ol', 'table', 'dl', 'hr'}
551
+
552
+ def block(self, el, style='body', sid=None):
553
+ "Block elements for `el` (one element may yield several); `sid` is a custom-style id override for paragraphs"
554
+ out = self._block(el, style, sid)
555
+ tag = _tag(el)
556
+ if tag in self.FIRST_AFTER or (tag == 'div' and 'display' in _classes(el)): self.first = True
557
+ return out
558
+
559
+ def _block(self, el, style, sid):
560
+ tag = _tag(el)
561
+ if tag == 'p':
562
+ ex = self.qindent() if style == 'blockquote' else None
563
+ psid = self.custom_style(el, 'paragraph') or sid
564
+ use = 'firstpara' if self.first and style == 'body' and not psid else style
565
+ self.first = False
566
+ return [self.para(self.bookmark(el, self.runs(el, {})), use, ex, psid)]
567
+ if tag in ('h1', 'h2', 'h3', 'h4', 'h5', 'h6'): return [self.para(self.bookmark(el, self.runs(el, {})), tag)]
568
+ if tag == 'blockquote':
569
+ self.bq += 1
570
+ try: return self.blocks(el, 'blockquote')
571
+ finally: self.bq -= 1
572
+ if tag == 'pre': return self.codeblock(el)
573
+ if tag in ('ul', 'ol'): return self.list_el(el)
574
+ if tag == 'table': return self.table(el)
575
+ if tag == 'hr':
576
+ return [E('w:p', E('w:pPr', E('w:pBdr',
577
+ E('w:bottom', {'w:val': 'single', 'w:sz': 6, 'w:space': 1, 'w:color': 'auto'}))))]
578
+ if tag == 'dl': return self.dl(el)
579
+ if tag == 'script': return self.rawxml(el)
580
+ if tag == 'figure': return self.figure(el)
581
+ if tag == 'template': return [self.para(runs)] if (runs := self.tmpl_runs(el, {}, 'block')) else []
582
+ if tag == 'div':
583
+ cls = _classes(el)
584
+ if 'math' in cls and 'display' in cls: return [E('w:p', E('m:oMathPara', self.omath(el)))]
585
+ return self.blocks(el, style, self.custom_style(el, 'paragraph') or sid)
586
+ if tag in BLOCK_TAGS and any(_tag(c) in BLOCK_TAGS for c in _els(el)):
587
+ return self.blocks(el, style, sid) # unknown container: recurse
588
+ self.warn(f'unhandled block <{tag}>; emitted as plain paragraph')
589
+ return [self.para(self.runs(el, {}), style, None, sid)]
590
+
591
+ RAWNS = ' '.join(f'xmlns:{k}="{v}"' for k, v in NS.items() if k != 'xml')
592
+
593
+ BIND_NS = 'urn:mdhtml:fields'
594
+ BIND_ID = '{8E2C9A44-7D31-4E5B-9C0D-1A6F2B3C4D5E}' # fixed datastore id, so builds are reproducible
595
+
596
+ def tmpl_runs(self, el, fmt, form):
597
+ """Template-token runs via the `tmpl` callable: str is a literal text run, ('field', instr) a live
598
+ field, ('control', name) an interactive plain-text content control, None dropped"""
599
+ if self.tmpl is None: return []
600
+ res = self.tmpl(el.to_text(), _get(el, 'data-template', ''), form)
601
+ if res is None: return []
602
+ if isinstance(res, str): return self.text_runs(res, fmt)
603
+ kind, val = res
604
+ if kind == 'field':
605
+ self.has_fields = True
606
+ return [E('w:fldSimple', {'w:instr': f' {val.strip()} '}, E('w:r', self.rpr(fmt), E('w:t', f'«{el.to_text().strip()}»')))]
607
+ if kind in ('control', 'bound'):
608
+ self.has_controls = True
609
+ sdtpr = E('w:sdtPr', E('w:alias', {'w:val': val}), E('w:tag', {'w:val': val}), E('w:showingPlcHdr'))
610
+ if kind == 'bound':
611
+ if val not in self.bound: self.bound.append(val)
612
+ sdtpr.append(E('w:dataBinding', {'w:prefixMappings': f"xmlns:ns0='{self.BIND_NS}'",
613
+ 'w:xpath': f'/ns0:fields[1]/ns0:{val}[1]', 'w:storeItemID': self.BIND_ID}))
614
+ sdtpr.append(E('w:text'))
615
+ return [E('w:sdt', sdtpr,
616
+ E('w:sdtContent', E('w:r', self.rpr(fmt | {'rstyle': 'PlaceholderText'}), E('w:t', val))))]
617
+ raise ValueError(f'unknown template rendering {res!r}')
618
+
619
+
620
+ def rawxml(self, el):
621
+ "Elements parsed from a raw docx payload (`{=docx}` in Markdown); other formats skip silently"
622
+ if _get(el, 'type') != 'application/vnd.mdhtml.raw' or _get(el, 'data-format') != 'docx': return []
623
+ payload, warn = decode_raw(el)
624
+ if warn:
625
+ self.warn(warn)
626
+ return []
627
+ try: return list(etree.fromstring(f'<m2d {self.RAWNS}>{payload}</m2d>'))
628
+ except etree.XMLSyntaxError as e:
629
+ self.warn(f'malformed docx raw payload: {e}')
630
+ return []
631
+
632
+ def dl(self, el):
633
+ "Definition list: dt/dd paragraphs in their dialect styles"
634
+ out = []
635
+ for c in _els(el):
636
+ t = _tag(c)
637
+ if t == 'dt': out.append(self.para(self.runs(c, {}), 'dt'))
638
+ elif t == 'dd':
639
+ blocky = any(_tag(k) in BLOCK_TAGS for k in _els(c))
640
+ out += self.blocks(c, 'dd') if blocky else [self.para(self.runs(c, {}), 'dd')]
641
+ return out
642
+
643
+ def blocks(self, parent, style='body', sid=None): return self.block_nodes(parent.children, style, sid)
644
+
645
+ def block_nodes(self, nodes, style='body', sid=None):
646
+ out, inline = [], []
647
+ def flush():
648
+ meaningful = [n for n in inline if not isinstance(n, Text) or n.text.strip()]
649
+ if not meaningful:
650
+ inline.clear()
651
+ return
652
+ raw = all(isinstance(n, Element) and _is_raw(n) for n in meaningful)
653
+ if raw:
654
+ for node in meaningful: out.extend(self.block(node, style, sid))
655
+ else:
656
+ extra = self.qindent() if style == 'blockquote' else None
657
+ use = 'firstpara' if self.first and style == 'body' and not sid else style
658
+ self.first = False
659
+ out.append(self.para(self.group_runs(inline, {}), use, extra, sid))
660
+ inline.clear()
661
+ for node in nodes:
662
+ if isinstance(node, Comment): continue
663
+ if isinstance(node, Element) and _tag(node) == 'template':
664
+ flush()
665
+ if runs := self.tmpl_runs(node, {}, 'block'): out.append(self.para(runs))
666
+ continue
667
+ if isinstance(node, Element) and (_tag(node) in INERT_TAGS or _tag(node) == 'script' and not _is_raw(node)): continue
668
+ if isinstance(node, Element) and _tag(node) in BLOCK_TAGS:
669
+ flush()
670
+ out.extend(self.block(node, style, sid))
671
+ elif isinstance(node, (Text, Element)): inline.append(node)
672
+ flush()
673
+ return out
674
+
675
+ # ---- assembly -----------------------------------------------------------
676
+ def document(self, body_blocks):
677
+ "word/document.xml bytes: our blocks + the template's sectPr"
678
+ root = etree.Element(qn('w:document'), nsmap=NS)
679
+ body = etree.SubElement(root, qn('w:body'))
680
+ for b in body_blocks: body.append(b)
681
+ body.append(self.sectpr)
682
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
683
+
684
+ BULLETS = ['•', '◦', '▪']
685
+ NUMFMTS = ['decimal', 'lowerLetter', 'lowerRoman']
686
+
687
+ def numbering_xml(self):
688
+ """word/numbering.xml: the template's part (if any) merged with our list definitions (bullet and
689
+ decimal abstracts, one num per list with startOverrides so each restarts), the heading scheme,
690
+ and .xml contributors' definitions - all abstractNum before all num, as the schema requires"""
691
+ root = self.tmplnum if self.tmplnum is not None else etree.Element(qn('w:numbering'), nsmap=dict(w=W))
692
+ tnums = [e for e in root if etree.QName(e).localname == 'num']
693
+ for e in tnums: root.remove(e)
694
+ if self.nums:
695
+ for aid in (0, 1):
696
+ an = E('w:abstractNum', {'w:abstractNumId': self._absbase + aid},
697
+ E('w:multiLevelType', {'w:val': 'hybridMultilevel'}))
698
+ for i in range(9):
699
+ fmt, txt = ('bullet', self.BULLETS[i % 3]) if aid == 0 else (self.NUMFMTS[i % 3], f'%{i+1}.')
700
+ an.append(E('w:lvl', {'w:ilvl': i}, # chkstyle: ignore-node
701
+ E('w:start', {'w:val': 1}), E('w:numFmt', {'w:val': fmt}),
702
+ E('w:lvlText', {'w:val': txt}), E('w:lvlJc', {'w:val': 'left'}),
703
+ E('w:pPr', E('w:ind', {'w:left': 720 * (i + 1), 'w:hanging': 360}))))
704
+ root.append(an)
705
+ if self.headnum:
706
+ an = E('w:abstractNum', {'w:abstractNumId': self._absbase + 2}, E('w:multiLevelType', {'w:val': 'multilevel'}))
707
+ for i in range(9):
708
+ txt, fmt = self.scheme[i] if i < len(self.scheme) else (f'%{i+1}.', 'decimal')
709
+ an.append(E('w:lvl', {'w:ilvl': i}, # chkstyle: ignore-node
710
+ E('w:start', {'w:val': 1}), E('w:numFmt', {'w:val': fmt}),
711
+ E('w:pStyle', {'w:val': f'Heading{i + 1}'}) if i < 6 else None,
712
+ E('w:lvlText', {'w:val': txt}), E('w:lvlJc', {'w:val': 'left'})))
713
+ root.append(an)
714
+ for e in self.xabs: root.append(e)
715
+ for e in tnums: root.append(e)
716
+ if self.headnum: root.append(E('w:num', {'w:numId': self.headnum}, E('w:abstractNumId', {'w:val': self._absbase + 2})))
717
+ for e in self.xnums: root.append(e)
718
+ for nid, aid, start in self.nums:
719
+ root.append(E('w:num', {'w:numId': nid}, E('w:abstractNumId', {'w:val': self._absbase + aid}), # chkstyle: ignore-node
720
+ *[E('w:lvlOverride', {'w:ilvl': i}, E('w:startOverride', {'w:val': start if i == 0 else 1}))
721
+ for i in range(9)]))
722
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
723
+
724
+ RNS = 'http://schemas.openxmlformats.org/package/2006/relationships'
725
+
726
+ def _add_rels(self, root, rels):
727
+ for rid, typ, target, ext in rels:
728
+ rel = etree.SubElement(root, f'{{{self.RNS}}}Relationship', Id=rid, Type=typ, Target=target)
729
+ if ext: rel.set('TargetMode', 'External')
730
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
731
+
732
+ def relsxml(self):
733
+ "word/_rels/document.xml.rels: template rels plus ours"
734
+ return self._add_rels(etree.fromstring(self.refz.read('word/_rels/document.xml.rels')), self.rels)
735
+
736
+ def fn_relsxml(self):
737
+ "word/_rels/footnotes.xml.rels: relationship ids are per-part, so footnote links/images get their own file"
738
+ return self._add_rels(etree.Element(f'{{{self.RNS}}}Relationships', nsmap={None: self.RNS}), self.fn_rels)
739
+
740
+ DSNS = 'http://schemas.openxmlformats.org/officeDocument/2006/customXml'
741
+
742
+ def bind_item_xml(self):
743
+ "customXml/item1.xml: one empty element per bound variable; every same-name control is a live view of it"
744
+ root = etree.Element(f'{{{self.BIND_NS}}}fields', nsmap=dict(ns0=self.BIND_NS))
745
+ for name in self.bound: etree.SubElement(root, f'{{{self.BIND_NS}}}{name}')
746
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
747
+
748
+ def bind_props_xml(self):
749
+ "customXml/itemProps1.xml: the datastore id that the controls' `storeItemID` points at"
750
+ root = etree.Element(f'{{{self.DSNS}}}datastoreItem', nsmap=dict(ds=self.DSNS))
751
+ root.set(f'{{{self.DSNS}}}itemID', self.BIND_ID)
752
+ srs = etree.SubElement(root, f'{{{self.DSNS}}}schemaRefs')
753
+ etree.SubElement(srs, f'{{{self.DSNS}}}schemaRef').set(f'{{{self.DSNS}}}uri', self.BIND_NS)
754
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
755
+
756
+ def bind_rels_xml(self):
757
+ "customXml/_rels/item1.xml.rels: the item's link to its datastore properties"
758
+ root = etree.Element(f'{{{self.RNS}}}Relationships', nsmap={None: self.RNS})
759
+ return self._add_rels(root, [('rId1', f'{R}/customXmlProps', 'itemProps1.xml', False)])
760
+
761
+
762
+ def footnotes_xml(self):
763
+ "word/footnotes.xml: the two Word-required separator notes plus our harvested ones"
764
+ root = etree.Element(qn('w:footnotes'), nsmap=dict(w=W, r=NS['r']))
765
+ for typ, wid in (('separator', -1), ('continuationSeparator', 0)):
766
+ root.append(E('w:footnote', {'w:type': typ, 'w:id': wid},
767
+ E('w:p', E('w:pPr', E('w:spacing', {'w:after': 0})), E('w:r', E(f'w:{typ}')))))
768
+ for wid, blks in self.fnotes: root.append(E('w:footnote', {'w:id': wid}, *blks))
769
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
770
+
771
+ def content_types(self, extra_parts):
772
+ "[Content_Types].xml with defaults for media extensions and overrides for added parts"
773
+ CT = 'http://schemas.openxmlformats.org/package/2006/content-types'
774
+ root = etree.fromstring(self.refz.read('[Content_Types].xml'))
775
+ have = {d.get('Extension') for d in root if d.tag == f'{{{CT}}}Default'}
776
+ MIME = dict(png='image/png', jpeg='image/jpeg', jpg='image/jpeg', gif='image/gif', tiff='image/tiff')
777
+ for name in self.media:
778
+ ext = posixpath.splitext(name)[1][1:].lower()
779
+ if ext not in have:
780
+ etree.SubElement(root, f'{{{CT}}}Default', Extension=ext, ContentType=MIME.get(ext, 'application/octet-stream'))
781
+ have.add(ext)
782
+ WPML = 'application/vnd.openxmlformats-officedocument.wordprocessingml'
783
+ for part, kind in extra_parts:
784
+ if any(d.get('PartName') == '/' + part for d in root if d.tag == f'{{{CT}}}Override'): continue
785
+ etree.SubElement(root, f'{{{CT}}}Override', PartName='/'+part, ContentType=f'{WPML}.{kind}+xml')
786
+ if self.bound:
787
+ if 'xml' not in have: etree.SubElement(root, f'{{{CT}}}Default', Extension='xml', ContentType='application/xml')
788
+ pn = '/customXml/itemProps1.xml'
789
+ if not any(d.get('PartName') == pn for d in root if d.tag == f'{{{CT}}}Override'):
790
+ etree.SubElement(root, f'{{{CT}}}Override', PartName=pn,
791
+ ContentType='application/vnd.openxmlformats-officedocument.customXmlProperties+xml')
792
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
793
+
794
+ PPR_PRE_NUMPR = {'pStyle', 'keepNext', 'keepLines', 'pageBreakBefore', 'framePr', 'widowControl'}
795
+
796
+ def _number_heading_styles(self):
797
+ "Patch w:numPr into Heading1-6 styles, binding them to the generated heading numbering"
798
+ for i in range(6):
799
+ st = self.sroot.find(f'{{{W}}}style[@{{{W}}}styleId="Heading{i + 1}"]')
800
+ if st is None: continue
801
+ ppr = st.find(qn('w:pPr'))
802
+ if ppr is None:
803
+ ppr = E('w:pPr')
804
+ rpr = st.find(qn('w:rPr'))
805
+ st.insert(list(st).index(rpr) if rpr is not None else len(st), ppr)
806
+ np = E('w:numPr', E('w:ilvl', {'w:val': i}), E('w:numId', {'w:val': self.headnum}))
807
+ pos = next((j for j, c in enumerate(ppr) if etree.QName(c).localname not in self.PPR_PRE_NUMPR), len(ppr))
808
+ ppr.insert(pos, np)
809
+
810
+ def styles_xml(self):
811
+ "word/styles.xml: the merged reference styles, plus stub definitions for undefined custom styles (pandoc's mechanism, plus our warning)"
812
+ if self.headnum: self._number_heading_styles()
813
+ root = self.sroot
814
+ for name, (kind, sid) in self.stubs.items():
815
+ root.append(E('w:style', {'w:type': kind, 'w:customStyle': 1, 'w:styleId': sid}, # chkstyle: ignore-node
816
+ E('w:name', {'w:val': name}),
817
+ E('w:basedOn', {'w:val': 'BodyText' if kind == 'paragraph' else 'DefaultParagraphFont'}),
818
+ E('w:qFormat')))
819
+ if self.has_controls and 'placeholder text' not in self.refstyles:
820
+ root.append(E('w:style', {'w:type': 'character', 'w:styleId': 'PlaceholderText'}, # chkstyle: ignore-node
821
+ E('w:name', {'w:val': 'Placeholder Text'}),
822
+ E('w:basedOn', {'w:val': 'DefaultParagraphFont'}), E('w:semiHidden'),
823
+ E('w:rPr', E('w:color', {'w:val': '808080'}))))
824
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
825
+
826
+ def harvest_footnotes(self, els):
827
+ "Split out footnote endnote sections, indexing their li definitions by id; returns body elements"
828
+ body, fn = [], []
829
+ for el in els:
830
+ is_notes = isinstance(el, Element) and _tag(el) == 'section' and 'footnotes' in _classes(el)
831
+ (fn if is_notes else body).append(el)
832
+ self.fndefs.update({_get(li, 'id'): li for sec in fn for li in _walk(sec) if _tag(li) == 'li' and _get(li, 'id')})
833
+ return body
834
+
835
+ BOOKMARKABLE = {'p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'}
836
+
837
+ def to_docx(self, mdhtml, dest):
838
+ root = parse_frag(mdhtml)
839
+ nodes = self.harvest_footnotes(root.children)
840
+ self.idtext, self.reftarget = {}, {}
841
+ for e in (e for node in nodes if isinstance(node, Element) for e in _walk(node)):
842
+ if not (i := _get(e, 'id')): continue
843
+ self.idtext[i] = ' '.join(e.to_text().split())
844
+ t = _tag(e)
845
+ if t in ('figure', 'table'): self.reftarget[i] = 'caption'
846
+ elif t in self.BOOKMARKABLE: self.reftarget[i] = 'block'
847
+ self.ids = set(self.idtext)
848
+ blocks = self.block_nodes(nodes)
849
+ docxml = self.document(blocks)
850
+ parts = {} # archive name -> (bytes, content-type kind); each also gets a document rel
851
+ if self.nums or self.headnum or self.xnums:
852
+ parts['word/numbering.xml'] = (self.numbering_xml(), 'numbering')
853
+ if self.tmplnum is None: self.rels.append((self.rid(), f'{R}/numbering', 'numbering.xml', False))
854
+ if self.fnotes:
855
+ parts['word/footnotes.xml'] = (self.footnotes_xml(), 'footnotes')
856
+ self.rels.append((self.rid(), f'{R}/footnotes', 'footnotes.xml', False))
857
+ if self.bound: self.rels.append((self.rid(), f'{R}/customXml', '../customXml/item1.xml', False))
858
+ extra = [(name, kind) for name, (data, kind) in parts.items()]
859
+ with zipfile.ZipFile(dest, 'w', zipfile.ZIP_DEFLATED) as zo:
860
+ for i in self.refz.infolist():
861
+ if i.filename == 'word/document.xml': zo.writestr(i.filename, docxml)
862
+ elif i.filename in parts: pass # replaced below (e.g. a template numbering.xml we merged)
863
+ elif i.filename == 'word/_rels/document.xml.rels': zo.writestr(i.filename, self.relsxml())
864
+ elif i.filename == '[Content_Types].xml': zo.writestr(i.filename, self.content_types(extra))
865
+ elif i.filename == 'word/styles.xml': zo.writestr(i.filename, self.styles_xml())
866
+ elif i.filename == 'word/settings.xml' and self.has_fields: zo.writestr(i.filename, self.settings_xml())
867
+ else: zo.writestr(i.filename, self.refz.read(i.filename))
868
+ for name, (data, kind) in parts.items(): zo.writestr(name, data)
869
+ if self.fn_rels: zo.writestr('word/_rels/footnotes.xml.rels', self.fn_relsxml())
870
+ if self.bound:
871
+ zo.writestr('customXml/item1.xml', self.bind_item_xml())
872
+ zo.writestr('customXml/itemProps1.xml', self.bind_props_xml())
873
+ zo.writestr('customXml/_rels/item1.xml.rels', self.bind_rels_xml())
874
+ for name, data in self.media.items(): zo.writestr(name, data)
875
+ return self.warnings
876
+
877
+ SETT_AFTER_UPDATE = {'footnotePr', 'endnotePr', 'compat', 'rsids', 'mathPr', 'attachedSchema', # chkstyle: ignore-node
878
+ 'themeFontLang', 'clrSchemeMapping', 'doNotIncludeSubdocsInStats',
879
+ 'doNotAutoCompressPictures', 'forceUpgrade', 'captions', 'readModeInkLockDown',
880
+ 'smartTagType', 'shapeDefaults', 'doNotEmbedSmartTags', 'decimalSymbol',
881
+ 'listSeparator', 'docId', 'discardImageEditingData', 'defaultImageDpi',
882
+ 'docVars', 'chartTrackingRefBased'}
883
+
884
+ def settings_xml(self):
885
+ "word/settings.xml with w:updateFields added (schema-ordered), so Word refreshes REF fields on open"
886
+ root = etree.fromstring(self.refz.read('word/settings.xml'))
887
+ pos = next((i for i, c in enumerate(root) if etree.QName(c).localname in self.SETT_AFTER_UPDATE), len(root))
888
+ root.insert(pos, E('w:updateFields', {'w:val': 'true'}))
889
+ return etree.tostring(root, xml_declaration=True, encoding='UTF-8', standalone=True)
890
+
891
+ def mustache_fields(body, syntax, form):
892
+ "Mustache variables as live Word `MERGEFIELD`s; section markers (`#`/`/`/`^` sigils) stay literal text"
893
+ if mustache_kind(body) == 'section': return '{{' + body + '}}'
894
+ return 'field', f'MERGEFIELD {body.strip()}'
895
+
896
+
897
+ def jinja_literal(body, syntax, form):
898
+ "Jinja tokens re-spelled canonically (`{{ x }}`/`{% x %}`) as text, for docxtpl-style downstream pipelines"
899
+ o, c = ('{%', '%}') if syntax == 'jinja-stmt' else ('{{', '}}')
900
+ return f'{o} {body.strip()} {c}'
901
+
902
+
903
+ def convert(mdhtml, dest, reference=None, base=None, reftypes=None, number_headings=None, tmpl=None):
904
+ """Convert an MDHTML string or mutable fast5ever DOM to a docx file at `dest`; returns warnings.
905
+ `reference` is a reference docx path, or a list of them: the first supplies the whole archive
906
+ (default, or when None: the built-in template), later entries contribute styles only, later-wins -
907
+ each a .docx path, a raw styles/numbering .xml path, or a fastpylight theme name (whose Hl*/Source
908
+ Code styles are generated; see styles.theme_ref). Default adds 'github_light' when fastpylight is
909
+ installed, so code blocks are colored; pass a bare reference for plain code. Relative image srcs
910
+ resolve against `base` ('.'). Cross-references (`data-ref` anchors from Markdown `[@sec-x]`) become
911
+ live REF fields; `reftypes` maps type tokens to (singular, plural) prefix words beyond the built-in
912
+ `sec`, and `number_headings` (a styles.SCHEMES name such as 'legal', or a {lvlText: numFmt} dict, one entry per heading level)
913
+ numbers the headings via a multilevel list so `\\w` fields resolve. Template tokens are dropped
914
+ unless `tmpl` is given: a callable `(body, syntax, form) -> str` for a literal text run,
915
+ `('field', instr)` for a live field, `('control', name)` for an interactive plain-text content
916
+ control, `('bound', name)` for a content control data-bound to a shared per-variable XML node
917
+ (same-name controls stay in sync as one is filled), or None to drop - `mustache_fields` and
918
+ `jinja_literal` are ready-made recipes."""
919
+ return Converter(reference, base, reftypes, number_headings, tmpl).to_docx(mdhtml, dest)