mdhtml2docx 0.1.2__tar.gz → 0.1.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. mdhtml2docx-0.1.4/PKG-INFO +177 -0
  2. mdhtml2docx-0.1.4/README.md +152 -0
  3. mdhtml2docx-0.1.4/mdhtml2docx/__init__.py +7 -0
  4. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/asvocab.py +6 -3
  5. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/hilite.py +1 -1
  6. mdhtml2docx-0.1.4/mdhtml2docx/mdhtml2docx.py +832 -0
  7. mdhtml2docx-0.1.4/mdhtml2docx/styles.py +64 -0
  8. mdhtml2docx-0.1.4/mdhtml2docx/templates/reference.docx +0 -0
  9. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/validate.py +7 -3
  10. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/wml.py +13 -23
  11. mdhtml2docx-0.1.4/mdhtml2docx.egg-info/PKG-INFO +177 -0
  12. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx.egg-info/SOURCES.txt +2 -5
  13. mdhtml2docx-0.1.4/mdhtml2docx.egg-info/requires.txt +13 -0
  14. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/pyproject.toml +4 -1
  15. mdhtml2docx-0.1.4/tests/test_convert.py +638 -0
  16. mdhtml2docx-0.1.2/CHANGELOG.md +0 -26
  17. mdhtml2docx-0.1.2/MANIFEST.in +0 -3
  18. mdhtml2docx-0.1.2/PKG-INFO +0 -96
  19. mdhtml2docx-0.1.2/README.md +0 -75
  20. mdhtml2docx-0.1.2/mdhtml2docx/__init__.py +0 -5
  21. mdhtml2docx-0.1.2/mdhtml2docx/convert.py +0 -927
  22. mdhtml2docx-0.1.2/mdhtml2docx/styles.py +0 -85
  23. mdhtml2docx-0.1.2/mdhtml2docx/templates/reference.docx +0 -0
  24. mdhtml2docx-0.1.2/mdhtml2docx.egg-info/PKG-INFO +0 -96
  25. mdhtml2docx-0.1.2/mdhtml2docx.egg-info/requires.txt +0 -8
  26. mdhtml2docx-0.1.2/tests/test_convert.py +0 -447
  27. mdhtml2docx-0.1.2/tests/test_word.py +0 -17
  28. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/LICENSE +0 -0
  29. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/SOURCES.md +0 -0
  30. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-chart.xsd +0 -0
  31. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-chartDrawing.xsd +0 -0
  32. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-diagram.xsd +0 -0
  33. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-lockedCanvas.xsd +0 -0
  34. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-main.xsd +0 -0
  35. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-picture.xsd +0 -0
  36. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/dml-wordprocessingDrawing.xsd +0 -0
  37. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/shared-commonSimpleTypes.xsd +0 -0
  38. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/shared-customXmlSchemaProperties.xsd +0 -0
  39. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/shared-math.xsd +0 -0
  40. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/shared-relationshipReference.xsd +0 -0
  41. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/wml.xsd +0 -0
  42. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/schemas/xml.xsd +0 -0
  43. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/word.py +0 -0
  44. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx/word.sdef +0 -0
  45. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx.egg-info/dependency_links.txt +0 -0
  46. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/mdhtml2docx.egg-info/top_level.txt +0 -0
  47. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/setup.cfg +0 -0
  48. {mdhtml2docx-0.1.2 → mdhtml2docx-0.1.4}/tests/test_validate.py +0 -0
@@ -0,0 +1,177 @@
1
+ Metadata-Version: 2.4
2
+ Name: mdhtml2docx
3
+ Version: 0.1.4
4
+ Summary: Convert MDHTML to Word docx files
5
+ Author-email: Jeremy Howard <j@fast.ai>
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/AnswerDotAI/mdhtml2docx
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Programming Language :: Python :: 3 :: Only
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Requires-Dist: mdhtml>=0.1.39
14
+ Requires-Dist: fast5ever>=0.1.5
15
+ Requires-Dist: oxml>=0.1.1
16
+ Provides-Extra: validation
17
+ Requires-Dist: lxml; extra == "validation"
18
+ Provides-Extra: dev
19
+ Requires-Dist: mdhtml2docx[validation]; extra == "dev"
20
+ Requires-Dist: pytest; extra == "dev"
21
+ Requires-Dist: fastship; extra == "dev"
22
+ Requires-Dist: build; extra == "dev"
23
+ Requires-Dist: twine; extra == "dev"
24
+ Dynamic: license-file
25
+
26
+ # mdhtml2docx
27
+
28
+ Convert [MDHTML](https://github.com/AnswerDotAI/mdhtml) to Word docx files.
29
+
30
+ `mdhtml` renders Markdown to an HTML5 document format shared by format-specific exporters. `mdhtml2docx` converts its portable core to docx from scratch. It uses [fast5ever](https://github.com/AnswerDotAI/fast5ever)'s mutable WHATWG DOM for input and `oxml` for WordprocessingML construction, reference editing, DOCX packaging, and atomic saving. XML construction uses namespace-bound factories such as `e.tcW(type='dxa', w=2400)`, backed by fastcore's XML builder. Conversion requires neither lxml nor a .NET or Office runtime.
31
+
32
+ MDHTML accepts the full HTML vocabulary. This exporter supports the elements and annotations listed below. It preserves the text of unknown inline elements and recurses through block containers. It returns a warning when an unsupported block must become a plain paragraph. HTML parsing and repair belong to `mdhtml`.
33
+
34
+ ## Usage
35
+
36
+ ```python
37
+ from mdhtml import md2mdhtml
38
+ from mdhtml2docx import mdhtml2docx
39
+
40
+ html = md2mdhtml(markdown_text)
41
+ warnings = mdhtml2docx(html, 'out.docx')
42
+ ```
43
+
44
+ `mdhtml2docx` takes an MDHTML string, writes a docx, and returns a list of warning strings. It parses strings with `mdhtml.mdhtml2dom`. Normal HTML5 repair applies; the input need not be well-formed XML.
45
+
46
+ You can also pass an existing mutable fast5ever DOM:
47
+
48
+ ```python
49
+ from mdhtml import mdhtml2dom
50
+
51
+ document = mdhtml2dom(html)
52
+ document.children[0].attrs['custom-style'] = 'Contract Title'
53
+ warnings = mdhtml2docx(document, 'out.docx')
54
+ ```
55
+
56
+ Body-position text and phrasing content become implicit Word paragraphs. Whitespace between blocks does not produce content. HTML templates are not rendered.
57
+
58
+ ## Supported content
59
+
60
+ The exporter supports these elements:
61
+
62
+ - Headings, paragraphs, block quotes, fenced and indented code blocks, and horizontal rules.
63
+ - Bold, italic, strikethrough, underline (`u`), highlight, superscript, subscript, inline code, and abbreviations.
64
+ - External links and internal links through bookmarks.
65
+ - Images from local files, embedded with their dimensions. Remote image URLs become links.
66
+ - Bullet and numbered lists, including nesting, `start`, and task-list checkboxes.
67
+ - Pipe and grid tables, including row and column spans and header rows.
68
+ - Definition lists and footnotes.
69
+
70
+ A `details` div becomes its label as a bold line above the body. It does not become a numbered heading or retain its collapsible behavior.
71
+
72
+ Math spans and blocks become native Word math zones (`m:oMath`). These contain the source text unchanged, whether TeX, UnicodeMath, or AsciiMath. Math build-up and rendering are left to downstream tools. Pandoc reads the zones back as math.
73
+
74
+ When [fastpylight](https://github.com/AnswerDotAI/fastpylight) is installed, code blocks with a language use syntax coloring. Tokens receive `Hl*` character styles such as `Hl Keyword`. Their colors come from the reference theme described below. You can restyle them in Word.
75
+
76
+ ## Styling
77
+
78
+ The generated document uses named styles instead of inline formatting. Change the styles to change its appearance.
79
+
80
+ Markdown headings map to Word styles as follows:
81
+
82
+ - h1 uses Title and displays no number. It carries the numbering's invisible level 0, which restarts the count below each title. A file containing several documents can therefore number each from 1.
83
+ - h2 through h6 use Heading 1 through Heading 5.
84
+
85
+ Prose uses Body Text. The first paragraph after a heading or similar block uses First Paragraph, following Pandoc's convention. Other styles include Quote, Source Code, Verbatim Char, Hyperlink, List Paragraph, Compact for table cells, Definition Term, Definition, caption, and footnote styles. Tables default to Table Grid. The built-in template also provides Borderless Table for signature blocks and other layout tables.
86
+
87
+ ### Reference documents and themes
88
+
89
+ Pass `reference='mydoc.docx'` to use your own document's styles, as with Pandoc's `--reference-doc`.
90
+
91
+ Reference parts are located through their relationships, not assumed filenames. Existing reference media and unrelated parts are preserved; new images receive non-colliding part names, and relationship IDs are allocated in each source part's scope.
92
+
93
+ `reference` also accepts a list. The first entry supplies the document, including page setup, fonts, and base styles. Later entries contribute styles and replace earlier styles with the same name. Each can be another `.docx` or a fastpylight theme name such as `'dracula'`.
94
+
95
+ The default is the built-in template plus `'github_light'`. Pass a reference document without a theme for plain, uncolored code. To write a theme's styles to a standalone docx for inspection or editing, use `mdhtml2docx.styles.theme_ref('dracula', 'dracula.docx')`.
96
+
97
+ The built-in template is generated by `tools/createref.py` from a stock Word document. It defines the styles the converter emits and next-paragraph chains for continued editing in Word.
98
+
99
+ ### Style annotations and table widths
100
+
101
+ A `custom-style="Name"` attribute, written as `{custom-style="Name"}` in Markdown, applies that style from your reference document. If the style is missing, the converter inserts a stub and returns a warning.
102
+
103
+ A plain class such as `{.note}` applies a style only when the reference document defines one named `note`. Otherwise it is ignored. Both annotations also work on tables. A named table style replaces Table Grid.
104
+
105
+ Tables can mix fixed and proportional column widths. Add an attribute list after the table: `{: colwidths="10em 2fr 1fr"}`. Lengths fix a column's width. The `fr` values divide the remaining width proportionally, as in CSS grid.
106
+
107
+ ## Raw docx
108
+
109
+ The converter accepts MDHTML raw data with `data-format="docx"`:
110
+
111
+ ```html
112
+ <script type="application/vnd.mdhtml.raw" data-format="docx">…</script>
113
+ ```
114
+
115
+ A ```` ```{=docx} ```` fenced block in Markdown, or inline code followed by `{=docx}`, produces this element. The payload is parsed as WordprocessingML and inserted verbatim. Block payloads contain elements such as `w:p` or `w:tbl`. Inline payloads contain elements such as `w:r`. The prefixes `w`, `r`, `wp`, `a`, `pic`, and `m` are predeclared.
116
+
117
+ Payload encoding follows these rules:
118
+
119
+ - No `data-encoding`: use the literal payload.
120
+ - `data-encoding="html"`: perform one character-reference decoding pass.
121
+ - `data-encoding="base64"`: decode a base64-encoded UTF-8 payload.
122
+
123
+ Malformed payloads and unknown encodings are dropped with a warning. Raw data for other formats is skipped silently.
124
+
125
+ For example, insert a page break with:
126
+
127
+ ```{=docx}
128
+ <w:p><w:r><w:br w:type="page"/></w:r></w:p>
129
+ ```
130
+
131
+ ## Cross-references
132
+
133
+ Markdown references such as `[@sec-payment]` become live Word REF fields. MDHTML represents them as `a` elements with `data-ref`.
134
+
135
+ The default field is `REF <bookmark> \w \h`. It displays the full-context paragraph number, such as "3.(c)(iii)", as a hyperlink. The converter sets `updateFields` in `settings.xml` so Word refreshes fields on open.
136
+
137
+ The `leaf`, `rel`, `text`, and `page` tokens select other fields. The independent `bare` token suppresses the prefix word. Unknown or conflicting tokens are conversion errors.
138
+
139
+ ### Prefixes and groups
140
+
141
+ The reference type is the part of the target id before its first `-`. It determines the word before the number. The built-in `sec` type uses Section or Sections. Add types with `reftypes=dict(exh=('Exhibit', 'Exhibits'))`.
142
+
143
+ Use `[Clause @sec-x]` to override the word for one reference, or `[-@sec-x]` to suppress it.
144
+
145
+ Grouped references use a `span` with `data-refs`. They render as "Sections 3.1 and 4.2", with one field per number. Groups never collapse into static ranges such as "3.1-3.3", whose meaning could change when a clause is inserted.
146
+
147
+ A missing target id raises a conversion error. A reference type with no defined prefix also raises when a prefix is required. A lawyer's document must not open showing "Error! Reference source not found."
148
+
149
+ ### Heading numbering
150
+
151
+ Number fields require numbered headings. If your reference docx already numbers its heading styles, the converter leaves that numbering unchanged.
152
+
153
+ Otherwise, select a scheme with `number_headings`:
154
+
155
+ - `'legal'` uses 1. / (a) / (i) numbering.
156
+ - `'decimal'` uses 1. / 1.1. / 1.1.1. numbering.
157
+ - A `{lvlText: numFmt}` dictionary supplies a custom scheme, with one entry per heading level from h1 down. Level 0 is the h1 title with an empty `lvlText`. `%2` is the h2 counter.
158
+
159
+ The named schemes come from mdhtml's `SCHEMES`. The custom dictionary uses the same format as mdhtml.
160
+
161
+ A reference-list entry ending in `.xml` can also supply styles and numbering. It contains raw `w:style`, `w:abstractNum`, and `w:num` elements. Each contributor's ids and references are remapped before merging, so separate contributors can reuse the same original numbering ids; later styles still win.
162
+
163
+ ### Figures and tables
164
+
165
+ Figures and captioned tables use live SEQ fields. A figure's caption appears below its image as "Figure 1: caption", using the caption style. A table's caption appears above it as "Table 1: caption".
166
+
167
+ When the element has an id, the converter bookmarks its label and number. `[@fig-plot]` inserts "Figure 1" from that bookmark without adding another prefix. A second, number-only bookmark supplies the bare number for references such as `[-@tbl-stages]`.
168
+
169
+ `fig` and `tbl` are built-in reference types alongside `sec`. Their labels come from the same `reftypes` table. Same-type groups pluralize once. Mixed-type groups use each item's singular prefix, as in "Figure 1 and Table 2".
170
+
171
+ Reference targets must be headings, paragraphs, figures, or tables with ids. A reference to anything else raises a conversion error.
172
+
173
+ ## Validation
174
+
175
+ Independent schema validation is optional. Install `mdhtml2docx[validation]` to use `mdhtml2docx.validate.fast_checks(path)`; this extra requires lxml. Normal conversion does not import it.
176
+
177
+ The test suite checks docx containers, CRCs, and XML, validates against the ECMA-376 schemas with lxml, and performs semantic round trips through Pandoc's independent docx reader. Periodic acceptance runs open documents in Microsoft Word through AppleScript.
@@ -0,0 +1,152 @@
1
+ # mdhtml2docx
2
+
3
+ Convert [MDHTML](https://github.com/AnswerDotAI/mdhtml) to Word docx files.
4
+
5
+ `mdhtml` renders Markdown to an HTML5 document format shared by format-specific exporters. `mdhtml2docx` converts its portable core to docx from scratch. It uses [fast5ever](https://github.com/AnswerDotAI/fast5ever)'s mutable WHATWG DOM for input and `oxml` for WordprocessingML construction, reference editing, DOCX packaging, and atomic saving. XML construction uses namespace-bound factories such as `e.tcW(type='dxa', w=2400)`, backed by fastcore's XML builder. Conversion requires neither lxml nor a .NET or Office runtime.
6
+
7
+ MDHTML accepts the full HTML vocabulary. This exporter supports the elements and annotations listed below. It preserves the text of unknown inline elements and recurses through block containers. It returns a warning when an unsupported block must become a plain paragraph. HTML parsing and repair belong to `mdhtml`.
8
+
9
+ ## Usage
10
+
11
+ ```python
12
+ from mdhtml import md2mdhtml
13
+ from mdhtml2docx import mdhtml2docx
14
+
15
+ html = md2mdhtml(markdown_text)
16
+ warnings = mdhtml2docx(html, 'out.docx')
17
+ ```
18
+
19
+ `mdhtml2docx` takes an MDHTML string, writes a docx, and returns a list of warning strings. It parses strings with `mdhtml.mdhtml2dom`. Normal HTML5 repair applies; the input need not be well-formed XML.
20
+
21
+ You can also pass an existing mutable fast5ever DOM:
22
+
23
+ ```python
24
+ from mdhtml import mdhtml2dom
25
+
26
+ document = mdhtml2dom(html)
27
+ document.children[0].attrs['custom-style'] = 'Contract Title'
28
+ warnings = mdhtml2docx(document, 'out.docx')
29
+ ```
30
+
31
+ Body-position text and phrasing content become implicit Word paragraphs. Whitespace between blocks does not produce content. HTML templates are not rendered.
32
+
33
+ ## Supported content
34
+
35
+ The exporter supports these elements:
36
+
37
+ - Headings, paragraphs, block quotes, fenced and indented code blocks, and horizontal rules.
38
+ - Bold, italic, strikethrough, underline (`u`), highlight, superscript, subscript, inline code, and abbreviations.
39
+ - External links and internal links through bookmarks.
40
+ - Images from local files, embedded with their dimensions. Remote image URLs become links.
41
+ - Bullet and numbered lists, including nesting, `start`, and task-list checkboxes.
42
+ - Pipe and grid tables, including row and column spans and header rows.
43
+ - Definition lists and footnotes.
44
+
45
+ A `details` div becomes its label as a bold line above the body. It does not become a numbered heading or retain its collapsible behavior.
46
+
47
+ Math spans and blocks become native Word math zones (`m:oMath`). These contain the source text unchanged, whether TeX, UnicodeMath, or AsciiMath. Math build-up and rendering are left to downstream tools. Pandoc reads the zones back as math.
48
+
49
+ When [fastpylight](https://github.com/AnswerDotAI/fastpylight) is installed, code blocks with a language use syntax coloring. Tokens receive `Hl*` character styles such as `Hl Keyword`. Their colors come from the reference theme described below. You can restyle them in Word.
50
+
51
+ ## Styling
52
+
53
+ The generated document uses named styles instead of inline formatting. Change the styles to change its appearance.
54
+
55
+ Markdown headings map to Word styles as follows:
56
+
57
+ - h1 uses Title and displays no number. It carries the numbering's invisible level 0, which restarts the count below each title. A file containing several documents can therefore number each from 1.
58
+ - h2 through h6 use Heading 1 through Heading 5.
59
+
60
+ Prose uses Body Text. The first paragraph after a heading or similar block uses First Paragraph, following Pandoc's convention. Other styles include Quote, Source Code, Verbatim Char, Hyperlink, List Paragraph, Compact for table cells, Definition Term, Definition, caption, and footnote styles. Tables default to Table Grid. The built-in template also provides Borderless Table for signature blocks and other layout tables.
61
+
62
+ ### Reference documents and themes
63
+
64
+ Pass `reference='mydoc.docx'` to use your own document's styles, as with Pandoc's `--reference-doc`.
65
+
66
+ Reference parts are located through their relationships, not assumed filenames. Existing reference media and unrelated parts are preserved; new images receive non-colliding part names, and relationship IDs are allocated in each source part's scope.
67
+
68
+ `reference` also accepts a list. The first entry supplies the document, including page setup, fonts, and base styles. Later entries contribute styles and replace earlier styles with the same name. Each can be another `.docx` or a fastpylight theme name such as `'dracula'`.
69
+
70
+ The default is the built-in template plus `'github_light'`. Pass a reference document without a theme for plain, uncolored code. To write a theme's styles to a standalone docx for inspection or editing, use `mdhtml2docx.styles.theme_ref('dracula', 'dracula.docx')`.
71
+
72
+ The built-in template is generated by `tools/createref.py` from a stock Word document. It defines the styles the converter emits and next-paragraph chains for continued editing in Word.
73
+
74
+ ### Style annotations and table widths
75
+
76
+ A `custom-style="Name"` attribute, written as `{custom-style="Name"}` in Markdown, applies that style from your reference document. If the style is missing, the converter inserts a stub and returns a warning.
77
+
78
+ A plain class such as `{.note}` applies a style only when the reference document defines one named `note`. Otherwise it is ignored. Both annotations also work on tables. A named table style replaces Table Grid.
79
+
80
+ Tables can mix fixed and proportional column widths. Add an attribute list after the table: `{: colwidths="10em 2fr 1fr"}`. Lengths fix a column's width. The `fr` values divide the remaining width proportionally, as in CSS grid.
81
+
82
+ ## Raw docx
83
+
84
+ The converter accepts MDHTML raw data with `data-format="docx"`:
85
+
86
+ ```html
87
+ <script type="application/vnd.mdhtml.raw" data-format="docx">…</script>
88
+ ```
89
+
90
+ A ```` ```{=docx} ```` fenced block in Markdown, or inline code followed by `{=docx}`, produces this element. The payload is parsed as WordprocessingML and inserted verbatim. Block payloads contain elements such as `w:p` or `w:tbl`. Inline payloads contain elements such as `w:r`. The prefixes `w`, `r`, `wp`, `a`, `pic`, and `m` are predeclared.
91
+
92
+ Payload encoding follows these rules:
93
+
94
+ - No `data-encoding`: use the literal payload.
95
+ - `data-encoding="html"`: perform one character-reference decoding pass.
96
+ - `data-encoding="base64"`: decode a base64-encoded UTF-8 payload.
97
+
98
+ Malformed payloads and unknown encodings are dropped with a warning. Raw data for other formats is skipped silently.
99
+
100
+ For example, insert a page break with:
101
+
102
+ ```{=docx}
103
+ <w:p><w:r><w:br w:type="page"/></w:r></w:p>
104
+ ```
105
+
106
+ ## Cross-references
107
+
108
+ Markdown references such as `[@sec-payment]` become live Word REF fields. MDHTML represents them as `a` elements with `data-ref`.
109
+
110
+ The default field is `REF <bookmark> \w \h`. It displays the full-context paragraph number, such as "3.(c)(iii)", as a hyperlink. The converter sets `updateFields` in `settings.xml` so Word refreshes fields on open.
111
+
112
+ The `leaf`, `rel`, `text`, and `page` tokens select other fields. The independent `bare` token suppresses the prefix word. Unknown or conflicting tokens are conversion errors.
113
+
114
+ ### Prefixes and groups
115
+
116
+ The reference type is the part of the target id before its first `-`. It determines the word before the number. The built-in `sec` type uses Section or Sections. Add types with `reftypes=dict(exh=('Exhibit', 'Exhibits'))`.
117
+
118
+ Use `[Clause @sec-x]` to override the word for one reference, or `[-@sec-x]` to suppress it.
119
+
120
+ Grouped references use a `span` with `data-refs`. They render as "Sections 3.1 and 4.2", with one field per number. Groups never collapse into static ranges such as "3.1-3.3", whose meaning could change when a clause is inserted.
121
+
122
+ A missing target id raises a conversion error. A reference type with no defined prefix also raises when a prefix is required. A lawyer's document must not open showing "Error! Reference source not found."
123
+
124
+ ### Heading numbering
125
+
126
+ Number fields require numbered headings. If your reference docx already numbers its heading styles, the converter leaves that numbering unchanged.
127
+
128
+ Otherwise, select a scheme with `number_headings`:
129
+
130
+ - `'legal'` uses 1. / (a) / (i) numbering.
131
+ - `'decimal'` uses 1. / 1.1. / 1.1.1. numbering.
132
+ - A `{lvlText: numFmt}` dictionary supplies a custom scheme, with one entry per heading level from h1 down. Level 0 is the h1 title with an empty `lvlText`. `%2` is the h2 counter.
133
+
134
+ The named schemes come from mdhtml's `SCHEMES`. The custom dictionary uses the same format as mdhtml.
135
+
136
+ A reference-list entry ending in `.xml` can also supply styles and numbering. It contains raw `w:style`, `w:abstractNum`, and `w:num` elements. Each contributor's ids and references are remapped before merging, so separate contributors can reuse the same original numbering ids; later styles still win.
137
+
138
+ ### Figures and tables
139
+
140
+ Figures and captioned tables use live SEQ fields. A figure's caption appears below its image as "Figure 1: caption", using the caption style. A table's caption appears above it as "Table 1: caption".
141
+
142
+ When the element has an id, the converter bookmarks its label and number. `[@fig-plot]` inserts "Figure 1" from that bookmark without adding another prefix. A second, number-only bookmark supplies the bare number for references such as `[-@tbl-stages]`.
143
+
144
+ `fig` and `tbl` are built-in reference types alongside `sec`. Their labels come from the same `reftypes` table. Same-type groups pluralize once. Mixed-type groups use each item's singular prefix, as in "Figure 1 and Table 2".
145
+
146
+ Reference targets must be headings, paragraphs, figures, or tables with ids. A reference to anything else raises a conversion error.
147
+
148
+ ## Validation
149
+
150
+ Independent schema validation is optional. Install `mdhtml2docx[validation]` to use `mdhtml2docx.validate.fast_checks(path)`; this extra requires lxml. Normal conversion does not import it.
151
+
152
+ The test suite checks docx containers, CRCs, and XML, validates against the ECMA-376 schemas with lxml, and performs semantic round trips through Pandoc's independent docx reader. Periodic acceptance runs open documents in Microsoft Word through AppleScript.
@@ -0,0 +1,7 @@
1
+ __version__ = "0.1.4"
2
+
3
+
4
+
5
+
6
+
7
+ from .mdhtml2docx import mdhtml2docx
@@ -3,7 +3,7 @@ import re
3
3
  from functools import cache
4
4
  from importlib.resources import files
5
5
  from aem import AEEnum
6
- from lxml import etree
6
+ from xml.etree import ElementTree as etree
7
7
 
8
8
  __all__ = ['vocab', 'props', 'sd', 'sdfind']
9
9
 
@@ -80,9 +80,12 @@ def sdfind(pat, path=SDEF, maxlen=80):
80
80
  "Search sdef names and descriptions for regex `pat`, as (kind, name, description) rows; child nodes show as parent.name"
81
81
  r = re.compile(pat, re.I)
82
82
  res = []
83
- for e in _sdef(path).iter(*_fmts, *_children):
83
+ root = _sdef(path)
84
+ parents = {c: p for p in root.iter() for c in p}
85
+ for e in root.iter():
86
+ if e.tag not in _fmts and e.tag not in _children: continue
84
87
  n, d = e.get('name') or '', e.get('description') or ''
85
88
  if not (r.search(n) or r.search(d)): continue
86
- if e.tag in _children: n = f'{e.getparent().get("name")}.{n}'
89
+ if e.tag in _children: n = f'{parents[e].get("name")}.{n}'
87
90
  res.append((e.tag, n, d if len(d)<=maxlen else d[:maxlen]+'...'))
88
91
  return res
@@ -1,6 +1,6 @@
1
1
  """Optional syntax scopes for code blocks, via fastpylight (Jeremy's tree-sitter highlighter).
2
2
  When fastpylight is absent or the language unknown, callers fall back to plain runs, so the
3
- package keeps its lxml-only hard dependency. Colors live in the reference doc's Hl* character
3
+ package does not require fastpylight. Colors live in the reference doc's Hl* character
4
4
  styles (see styles.theme_styles); this module only tokenizes."""
5
5
 
6
6
  # theme_colors is probed too: a pre-theme_colors fastpylight is treated as absent, not half-working