markdown-docx 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_docx/__init__.py +1 -0
- markdown_docx/assets/default.docx +0 -0
- markdown_docx/assets/syntax.json +47 -0
- markdown_docx/assets.py +15 -0
- markdown_docx/cli.py +318 -0
- markdown_docx/errors.py +118 -0
- markdown_docx/images.py +161 -0
- markdown_docx/markdown_body.py +121 -0
- markdown_docx/metadata.py +349 -0
- markdown_docx/models.py +180 -0
- markdown_docx/parser.py +456 -0
- markdown_docx/renderer.py +285 -0
- markdown_docx/skill.py +123 -0
- markdown_docx/styles.py +50 -0
- markdown_docx/template.py +121 -0
- markdown_docx-0.1.0.dist-info/METADATA +372 -0
- markdown_docx-0.1.0.dist-info/RECORD +20 -0
- markdown_docx-0.1.0.dist-info/WHEEL +4 -0
- markdown_docx-0.1.0.dist-info/entry_points.txt +2 -0
- markdown_docx-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import tempfile
|
|
5
|
+
from io import BytesIO
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from docx import Document
|
|
10
|
+
from docx.document import Document as DocumentObject
|
|
11
|
+
from docx.enum.section import WD_ORIENT, WD_SECTION
|
|
12
|
+
from docx.enum.table import WD_TABLE_ALIGNMENT
|
|
13
|
+
from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_BREAK
|
|
14
|
+
from docx.shared import Emu
|
|
15
|
+
from docx.text.paragraph import Paragraph
|
|
16
|
+
|
|
17
|
+
from markdown_docx.errors import MarkdownDocxError, RenderError
|
|
18
|
+
from markdown_docx.images import ImageLoader, rendered_width
|
|
19
|
+
from markdown_docx.models import (
|
|
20
|
+
CodeBlock,
|
|
21
|
+
DocumentModel,
|
|
22
|
+
HeadingBlock,
|
|
23
|
+
ImageBlock,
|
|
24
|
+
InlineFragment,
|
|
25
|
+
ListParagraphBlock,
|
|
26
|
+
PageBreakBlock,
|
|
27
|
+
ParagraphBlock,
|
|
28
|
+
SectionBreakBlock,
|
|
29
|
+
SectionSettings,
|
|
30
|
+
TableBlock,
|
|
31
|
+
)
|
|
32
|
+
from markdown_docx.styles import apply_font_overrides, validate_styles
|
|
33
|
+
from markdown_docx.template import load_template
|
|
34
|
+
|
|
35
|
+
PARAGRAPH_ALIGNMENT = {
|
|
36
|
+
"left": WD_ALIGN_PARAGRAPH.LEFT,
|
|
37
|
+
"center": WD_ALIGN_PARAGRAPH.CENTER,
|
|
38
|
+
"right": WD_ALIGN_PARAGRAPH.RIGHT,
|
|
39
|
+
}
|
|
40
|
+
TABLE_ALIGNMENT = {
|
|
41
|
+
"left": WD_TABLE_ALIGNMENT.LEFT,
|
|
42
|
+
"center": WD_TABLE_ALIGNMENT.CENTER,
|
|
43
|
+
"right": WD_TABLE_ALIGNMENT.RIGHT,
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def render_docx(
|
|
48
|
+
model: DocumentModel,
|
|
49
|
+
output_path: Path,
|
|
50
|
+
*,
|
|
51
|
+
template_path: Path | None,
|
|
52
|
+
base_dir: Path,
|
|
53
|
+
allow_remote_images: bool,
|
|
54
|
+
) -> dict[str, Any]:
|
|
55
|
+
document = load_template(template_path)
|
|
56
|
+
validate_styles(document, model.options)
|
|
57
|
+
apply_font_overrides(document, model.options)
|
|
58
|
+
current_settings = model.options.section
|
|
59
|
+
_apply_section_settings(document.sections[0], current_settings)
|
|
60
|
+
image_loader = ImageLoader(base_dir, allow_remote=allow_remote_images)
|
|
61
|
+
reusable = _reusable_initial_paragraph(document)
|
|
62
|
+
warnings = list(model.warnings)
|
|
63
|
+
image_seen = False
|
|
64
|
+
|
|
65
|
+
try:
|
|
66
|
+
for block in model.blocks:
|
|
67
|
+
if isinstance(block, PageBreakBlock):
|
|
68
|
+
document.add_paragraph().add_run().add_break(WD_BREAK.PAGE)
|
|
69
|
+
elif isinstance(block, SectionBreakBlock):
|
|
70
|
+
current_settings = block.settings
|
|
71
|
+
section = document.add_section(WD_SECTION.NEW_PAGE)
|
|
72
|
+
_apply_section_settings(section, current_settings)
|
|
73
|
+
elif isinstance(block, HeadingBlock):
|
|
74
|
+
paragraph, reusable = _new_paragraph(
|
|
75
|
+
document,
|
|
76
|
+
style=model.options.styles.headings[block.level],
|
|
77
|
+
reusable=reusable,
|
|
78
|
+
)
|
|
79
|
+
image_seen |= _render_fragments(
|
|
80
|
+
paragraph,
|
|
81
|
+
block.fragments,
|
|
82
|
+
image_loader=image_loader,
|
|
83
|
+
settings=current_settings,
|
|
84
|
+
monospace=model.options.fonts.monospace,
|
|
85
|
+
line=block.line,
|
|
86
|
+
input_path=model.source_name,
|
|
87
|
+
)
|
|
88
|
+
elif isinstance(block, ParagraphBlock):
|
|
89
|
+
style = (
|
|
90
|
+
model.options.styles.blockquote if block.role == "blockquote" else model.options.styles.paragraph
|
|
91
|
+
)
|
|
92
|
+
paragraph, reusable = _new_paragraph(document, style=style, reusable=reusable)
|
|
93
|
+
image_seen |= _render_fragments(
|
|
94
|
+
paragraph,
|
|
95
|
+
block.fragments,
|
|
96
|
+
image_loader=image_loader,
|
|
97
|
+
settings=current_settings,
|
|
98
|
+
monospace=model.options.fonts.monospace,
|
|
99
|
+
line=block.line,
|
|
100
|
+
input_path=model.source_name,
|
|
101
|
+
)
|
|
102
|
+
elif isinstance(block, CodeBlock):
|
|
103
|
+
paragraph, reusable = _new_paragraph(
|
|
104
|
+
document,
|
|
105
|
+
style=model.options.styles.code_block,
|
|
106
|
+
reusable=reusable,
|
|
107
|
+
)
|
|
108
|
+
paragraph.add_run(block.text.rstrip("\n"))
|
|
109
|
+
elif isinstance(block, ListParagraphBlock):
|
|
110
|
+
styles = (
|
|
111
|
+
model.options.styles.ordered_list
|
|
112
|
+
if block.list_kind == "ordered"
|
|
113
|
+
else model.options.styles.unordered_list
|
|
114
|
+
)
|
|
115
|
+
paragraph, reusable = _new_paragraph(document, style=styles[block.depth], reusable=reusable)
|
|
116
|
+
image_seen |= _render_fragments(
|
|
117
|
+
paragraph,
|
|
118
|
+
block.fragments,
|
|
119
|
+
image_loader=image_loader,
|
|
120
|
+
settings=current_settings,
|
|
121
|
+
monospace=model.options.fonts.monospace,
|
|
122
|
+
line=block.line,
|
|
123
|
+
input_path=model.source_name,
|
|
124
|
+
)
|
|
125
|
+
elif isinstance(block, TableBlock):
|
|
126
|
+
_render_table(
|
|
127
|
+
document,
|
|
128
|
+
block,
|
|
129
|
+
model=model,
|
|
130
|
+
settings=current_settings,
|
|
131
|
+
image_loader=image_loader,
|
|
132
|
+
)
|
|
133
|
+
elif isinstance(block, ImageBlock):
|
|
134
|
+
paragraph, reusable = _new_paragraph(
|
|
135
|
+
document,
|
|
136
|
+
style=model.options.styles.paragraph,
|
|
137
|
+
reusable=reusable,
|
|
138
|
+
)
|
|
139
|
+
paragraph.alignment = PARAGRAPH_ALIGNMENT[block.options.alignment]
|
|
140
|
+
asset = image_loader.load(block.src, line=block.line, input_path=model.source_name)
|
|
141
|
+
width = rendered_width(
|
|
142
|
+
asset,
|
|
143
|
+
block.options,
|
|
144
|
+
usable_width=current_settings.usable_width,
|
|
145
|
+
line=block.line,
|
|
146
|
+
input_path=model.source_name,
|
|
147
|
+
)
|
|
148
|
+
paragraph.add_run().add_picture(BytesIO(asset.data), width=Emu(width))
|
|
149
|
+
image_seen = True
|
|
150
|
+
else:
|
|
151
|
+
raise AssertionError(f"Unhandled block type: {type(block).__name__}")
|
|
152
|
+
if image_seen:
|
|
153
|
+
warnings.append("image_alt_text_not_embedded")
|
|
154
|
+
_save_atomically(document, output_path)
|
|
155
|
+
except MarkdownDocxError:
|
|
156
|
+
raise
|
|
157
|
+
except Exception as exc:
|
|
158
|
+
raise RenderError("render_failed", f"Could not render DOCX: {exc}") from exc
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
"output": str(output_path),
|
|
162
|
+
"sections": len(document.sections),
|
|
163
|
+
"warnings": sorted(set(warnings)),
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _apply_section_settings(section: Any, settings: SectionSettings) -> None:
|
|
168
|
+
section.orientation = WD_ORIENT.LANDSCAPE if settings.orientation == "landscape" else WD_ORIENT.PORTRAIT
|
|
169
|
+
section.page_width = Emu(settings.effective_width)
|
|
170
|
+
section.page_height = Emu(settings.effective_height)
|
|
171
|
+
section.top_margin = Emu(settings.margins.top)
|
|
172
|
+
section.right_margin = Emu(settings.margins.right)
|
|
173
|
+
section.bottom_margin = Emu(settings.margins.bottom)
|
|
174
|
+
section.left_margin = Emu(settings.margins.left)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _reusable_initial_paragraph(document: DocumentObject) -> Paragraph | None:
|
|
178
|
+
if len(document.paragraphs) == 1 and not document.tables:
|
|
179
|
+
paragraph = document.paragraphs[0]
|
|
180
|
+
if not paragraph.text and not paragraph.runs:
|
|
181
|
+
return paragraph
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _new_paragraph(
|
|
186
|
+
document: DocumentObject,
|
|
187
|
+
*,
|
|
188
|
+
style: str,
|
|
189
|
+
reusable: Paragraph | None,
|
|
190
|
+
) -> tuple[Paragraph, None]:
|
|
191
|
+
if reusable is not None:
|
|
192
|
+
reusable.style = style
|
|
193
|
+
return reusable, None
|
|
194
|
+
return document.add_paragraph(style=style), None
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _render_fragments(
|
|
198
|
+
paragraph: Paragraph,
|
|
199
|
+
fragments: list[InlineFragment],
|
|
200
|
+
*,
|
|
201
|
+
image_loader: ImageLoader,
|
|
202
|
+
settings: SectionSettings,
|
|
203
|
+
monospace: str,
|
|
204
|
+
line: int,
|
|
205
|
+
input_path: str,
|
|
206
|
+
) -> bool:
|
|
207
|
+
image_seen = False
|
|
208
|
+
for fragment in fragments:
|
|
209
|
+
if fragment.kind == "break":
|
|
210
|
+
paragraph.add_run().add_break(WD_BREAK.LINE)
|
|
211
|
+
elif fragment.kind == "image":
|
|
212
|
+
asset = image_loader.load(fragment.src or "", line=line, input_path=input_path)
|
|
213
|
+
width = min(asset.natural_width, settings.usable_width)
|
|
214
|
+
run = paragraph.add_run()
|
|
215
|
+
run.bold = fragment.bold or None
|
|
216
|
+
run.italic = fragment.italic or None
|
|
217
|
+
run.add_picture(BytesIO(asset.data), width=Emu(width))
|
|
218
|
+
image_seen = True
|
|
219
|
+
else:
|
|
220
|
+
run = paragraph.add_run(fragment.text or "")
|
|
221
|
+
run.bold = fragment.bold or None
|
|
222
|
+
run.italic = fragment.italic or None
|
|
223
|
+
if fragment.code:
|
|
224
|
+
run.font.name = monospace
|
|
225
|
+
return image_seen
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _render_table(
|
|
229
|
+
document: DocumentObject,
|
|
230
|
+
block: TableBlock,
|
|
231
|
+
*,
|
|
232
|
+
model: DocumentModel,
|
|
233
|
+
settings: SectionSettings,
|
|
234
|
+
image_loader: ImageLoader,
|
|
235
|
+
) -> None:
|
|
236
|
+
row_data = [block.headers, *block.rows]
|
|
237
|
+
column_count = len(block.headers)
|
|
238
|
+
style_name = block.options.style or model.options.styles.table
|
|
239
|
+
table = document.add_table(rows=len(row_data), cols=column_count, style=style_name)
|
|
240
|
+
table.alignment = TABLE_ALIGNMENT[block.options.alignment]
|
|
241
|
+
set_widths = block.options.width == "page" or block.options.column_widths is not None
|
|
242
|
+
table.autofit = not set_widths
|
|
243
|
+
widths: list[int] = []
|
|
244
|
+
if set_widths:
|
|
245
|
+
ratios = list(block.options.column_widths or (1.0,) * column_count)
|
|
246
|
+
ratio_total = sum(ratios)
|
|
247
|
+
widths = [round(settings.usable_width * ratio / ratio_total) for ratio in ratios]
|
|
248
|
+
widths[-1] += settings.usable_width - sum(widths)
|
|
249
|
+
for column, width in zip(table.columns, widths, strict=True):
|
|
250
|
+
column.width = Emu(width)
|
|
251
|
+
|
|
252
|
+
for row_index, cells in enumerate(row_data):
|
|
253
|
+
for column_index, cell_data in enumerate(cells):
|
|
254
|
+
cell = table.cell(row_index, column_index)
|
|
255
|
+
if widths:
|
|
256
|
+
cell.width = Emu(widths[column_index])
|
|
257
|
+
paragraph = cell.paragraphs[0]
|
|
258
|
+
paragraph.style = model.options.styles.paragraph
|
|
259
|
+
paragraph.alignment = PARAGRAPH_ALIGNMENT[cell_data.alignment]
|
|
260
|
+
_render_fragments(
|
|
261
|
+
paragraph,
|
|
262
|
+
cell_data.fragments,
|
|
263
|
+
image_loader=image_loader,
|
|
264
|
+
settings=settings,
|
|
265
|
+
monospace=model.options.fonts.monospace,
|
|
266
|
+
line=block.line,
|
|
267
|
+
input_path=model.source_name,
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _save_atomically(document: DocumentObject, output_path: Path) -> None:
|
|
272
|
+
output_path = output_path.resolve()
|
|
273
|
+
if not output_path.parent.is_dir():
|
|
274
|
+
raise RenderError("render_failed", f"Output directory does not exist: {output_path.parent}")
|
|
275
|
+
descriptor, temporary_name = tempfile.mkstemp(
|
|
276
|
+
prefix=f".{output_path.stem}-", suffix=".docx", dir=output_path.parent
|
|
277
|
+
)
|
|
278
|
+
os.close(descriptor)
|
|
279
|
+
temporary_path = Path(temporary_name)
|
|
280
|
+
try:
|
|
281
|
+
document.save(str(temporary_path))
|
|
282
|
+
Document(str(temporary_path))
|
|
283
|
+
temporary_path.replace(output_path)
|
|
284
|
+
finally:
|
|
285
|
+
temporary_path.unlink(missing_ok=True)
|
markdown_docx/skill.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import shutil
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from markdown_docx.errors import UsageError
|
|
8
|
+
|
|
9
|
+
SKILL_NAME = "markdown-docx"
|
|
10
|
+
MANAGED_MARKER = "<!-- managed-by: markdown-docx -->"
|
|
11
|
+
|
|
12
|
+
SKILL_MD = f"""---
|
|
13
|
+
name: markdown-docx
|
|
14
|
+
description: Create editable Word documents from strict Markdown using `uvx markdown-docx`. Use for authoring, rendering, validating, or inspecting markdown-docx sources and blank DOCX formatting templates.
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
{MANAGED_MARKER}
|
|
18
|
+
|
|
19
|
+
# Markdown DOCX
|
|
20
|
+
|
|
21
|
+
Use the published CLI through `uvx markdown-docx`. It converts strict Markdown into editable Word documents and keeps Word-specific settings inside invisible HTML comments.
|
|
22
|
+
|
|
23
|
+
Always invoke the tool as `uvx markdown-docx ...`. Do not assume a global install.
|
|
24
|
+
|
|
25
|
+
## Inspect first
|
|
26
|
+
|
|
27
|
+
Inspect the syntax and template before authoring a document:
|
|
28
|
+
|
|
29
|
+
```text
|
|
30
|
+
uvx markdown-docx --syntax
|
|
31
|
+
uvx markdown-docx --inspect-template --template formatting.docx --json
|
|
32
|
+
uvx markdown-docx --list-styles --template formatting.docx
|
|
33
|
+
uvx markdown-docx --list-table-styles --template formatting.docx
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Custom templates must be blank `.docx` files. They may contain styles, themes, fonts, numbering definitions, and page defaults. They may not contain body content, tables, drawings, or nonempty headers and footers.
|
|
37
|
+
|
|
38
|
+
## Render a document
|
|
39
|
+
|
|
40
|
+
```text
|
|
41
|
+
uvx markdown-docx input.md output.docx --json
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
If no output is supplied, the tool writes a `.docx` beside the Markdown source. Add `--force` only when replacing generated output is authorized.
|
|
45
|
+
|
|
46
|
+
## Metadata
|
|
47
|
+
|
|
48
|
+
Document metadata must be the first non-whitespace content:
|
|
49
|
+
|
|
50
|
+
```text
|
|
51
|
+
<!-- markdown-docx
|
|
52
|
+
document:
|
|
53
|
+
page_size: letter
|
|
54
|
+
orientation: portrait
|
|
55
|
+
margins:
|
|
56
|
+
top: 1in
|
|
57
|
+
right: 1in
|
|
58
|
+
bottom: 1in
|
|
59
|
+
left: 1in
|
|
60
|
+
-->
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Start a next-page section with a `section` comment. Insert an explicit page break with `<!-- markdown-docx: page-break -->`. Put `table` metadata immediately before a pipe table and `image` metadata immediately before a standalone image.
|
|
64
|
+
|
|
65
|
+
Run `uvx markdown-docx --syntax` for every accepted key and value.
|
|
66
|
+
|
|
67
|
+
## Supported Markdown
|
|
68
|
+
|
|
69
|
+
Use ATX headings, paragraphs, emphasis, strong text, inline code, hard line breaks, fenced code blocks, blockquotes, lists, pipe tables, and local or remote images. Links are rejected in 0.1.0 because `python-docx` 1.2.0 does not provide public hyperlink creation. Raw HTML, task lists, footnotes, horizontal rules, indented code, multi-paragraph list items, and unsupported nested block content are rejected.
|
|
70
|
+
|
|
71
|
+
Relative image paths resolve from the Markdown file. When reading stdin, provide an output and `--base-dir`. Use `--no-remote-images` for offline or untrusted input.
|
|
72
|
+
|
|
73
|
+
## Handle results
|
|
74
|
+
|
|
75
|
+
Prefer `--json` for automation. Success includes the absolute output path, section count, template identifier, and warnings. Failures include a stable code, message, input path, and line when available. Correct the source before retrying.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def default_skills_dir() -> Path:
|
|
80
|
+
return Path.home() / ".agents" / "skills"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def skill_dir(skills_dir: Path | None = None) -> Path:
|
|
84
|
+
return (skills_dir or default_skills_dir()) / SKILL_NAME
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def install_skill(skills_dir: Path | None = None) -> dict[str, Any]:
|
|
88
|
+
target = skill_dir(skills_dir)
|
|
89
|
+
skill_path = target / "SKILL.md"
|
|
90
|
+
if target.exists() and not skill_path.exists():
|
|
91
|
+
raise UsageError(f"Refusing to install into '{target}' because it contains no managed SKILL.md.")
|
|
92
|
+
if skill_path.exists() and MANAGED_MARKER not in skill_path.read_text(encoding="utf-8"):
|
|
93
|
+
raise UsageError(f"Refusing to overwrite unmanaged skill file '{skill_path}'.")
|
|
94
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
95
|
+
existed = skill_path.exists()
|
|
96
|
+
previous = skill_path.read_text(encoding="utf-8") if existed else ""
|
|
97
|
+
updated = existed and previous != SKILL_MD
|
|
98
|
+
skill_path.write_text(SKILL_MD, encoding="utf-8", newline="\n")
|
|
99
|
+
return {
|
|
100
|
+
"installed": True,
|
|
101
|
+
"created": not existed,
|
|
102
|
+
"updated": updated,
|
|
103
|
+
"skill": SKILL_NAME,
|
|
104
|
+
"path": str(skill_path),
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def remove_skill(skills_dir: Path | None = None, *, force: bool = False) -> dict[str, Any]:
|
|
109
|
+
target = skill_dir(skills_dir)
|
|
110
|
+
skill_path = target / "SKILL.md"
|
|
111
|
+
if not target.exists():
|
|
112
|
+
return {"removed": False, "skill": SKILL_NAME, "path": str(target), "reason": "not_installed"}
|
|
113
|
+
if not skill_path.exists():
|
|
114
|
+
raise UsageError(f"Refusing to remove '{target}' because SKILL.md is missing.")
|
|
115
|
+
content = skill_path.read_text(encoding="utf-8")
|
|
116
|
+
if MANAGED_MARKER not in content and not force:
|
|
117
|
+
raise UsageError(f"Refusing to remove unmanaged skill '{target}'. Use --force to override.")
|
|
118
|
+
extra_paths = [path for path in target.iterdir() if path.name != "SKILL.md"]
|
|
119
|
+
if extra_paths and not force:
|
|
120
|
+
names = ", ".join(sorted(path.name for path in extra_paths))
|
|
121
|
+
raise UsageError(f"Refusing to remove '{target}' because it contains unmanaged entries: {names}. Use --force.")
|
|
122
|
+
shutil.rmtree(target)
|
|
123
|
+
return {"removed": True, "skill": SKILL_NAME, "path": str(target)}
|
markdown_docx/styles.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from docx.document import Document as DocumentObject
|
|
4
|
+
from docx.enum.style import WD_STYLE_TYPE
|
|
5
|
+
|
|
6
|
+
from markdown_docx.errors import TemplateError
|
|
7
|
+
from markdown_docx.models import DocumentOptions
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def validate_styles(document: DocumentObject, options: DocumentOptions) -> None:
|
|
11
|
+
paragraph_names = {
|
|
12
|
+
options.styles.paragraph,
|
|
13
|
+
options.styles.blockquote,
|
|
14
|
+
options.styles.code_block,
|
|
15
|
+
*options.styles.headings.values(),
|
|
16
|
+
*options.styles.ordered_list,
|
|
17
|
+
*options.styles.unordered_list,
|
|
18
|
+
}
|
|
19
|
+
for name in sorted(paragraph_names):
|
|
20
|
+
_require_style(document, name, WD_STYLE_TYPE.PARAGRAPH)
|
|
21
|
+
_require_style(document, options.styles.table, WD_STYLE_TYPE.TABLE)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def apply_font_overrides(document: DocumentObject, options: DocumentOptions) -> None:
|
|
25
|
+
fonts = options.fonts
|
|
26
|
+
if fonts.body:
|
|
27
|
+
body_names = {
|
|
28
|
+
options.styles.paragraph,
|
|
29
|
+
options.styles.blockquote,
|
|
30
|
+
*options.styles.ordered_list,
|
|
31
|
+
*options.styles.unordered_list,
|
|
32
|
+
}
|
|
33
|
+
for name in body_names:
|
|
34
|
+
document.styles[name].font.name = fonts.body
|
|
35
|
+
if fonts.headings:
|
|
36
|
+
for name in options.styles.headings.values():
|
|
37
|
+
document.styles[name].font.name = fonts.headings
|
|
38
|
+
document.styles[options.styles.code_block].font.name = fonts.monospace
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _require_style(document: DocumentObject, name: str, expected_type: WD_STYLE_TYPE) -> None:
|
|
42
|
+
try:
|
|
43
|
+
style = document.styles[name]
|
|
44
|
+
except KeyError as exc:
|
|
45
|
+
raise TemplateError("template_style_missing", f"Template is missing required style: {name}") from exc
|
|
46
|
+
if style.type != expected_type:
|
|
47
|
+
raise TemplateError(
|
|
48
|
+
"template_style_type_mismatch",
|
|
49
|
+
f"Style '{name}' must be a {expected_type.name.lower()} style.",
|
|
50
|
+
)
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from io import BytesIO
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Protocol
|
|
6
|
+
|
|
7
|
+
from docx import Document
|
|
8
|
+
from docx.document import Document as DocumentObject
|
|
9
|
+
from docx.enum.style import WD_STYLE_TYPE
|
|
10
|
+
from docx.opc.exceptions import PackageNotFoundError
|
|
11
|
+
from docx.text.paragraph import Paragraph
|
|
12
|
+
|
|
13
|
+
from markdown_docx.assets import default_template_bytes
|
|
14
|
+
from markdown_docx.errors import TemplateError
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Story(Protocol):
|
|
18
|
+
@property
|
|
19
|
+
def paragraphs(self) -> list[Paragraph]: ...
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def tables(self) -> list[Any]: ...
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def load_template(path: Path | None) -> DocumentObject:
|
|
26
|
+
if path is None:
|
|
27
|
+
document = Document(BytesIO(default_template_bytes()))
|
|
28
|
+
validate_blank_template(document, label="packaged default template")
|
|
29
|
+
return document
|
|
30
|
+
resolved = path.resolve()
|
|
31
|
+
if not resolved.is_file():
|
|
32
|
+
raise TemplateError("template_not_found", f"Template does not exist: {resolved}")
|
|
33
|
+
if resolved.suffix.lower() != ".docx":
|
|
34
|
+
raise TemplateError("unsupported_feature", "Templates must use the .docx format. DOTX is not supported.")
|
|
35
|
+
try:
|
|
36
|
+
document = Document(str(resolved))
|
|
37
|
+
except (PackageNotFoundError, ValueError) as exc:
|
|
38
|
+
raise TemplateError("template_invalid", f"Template is not a readable DOCX file: {resolved}") from exc
|
|
39
|
+
validate_blank_template(document, label=str(resolved))
|
|
40
|
+
return document
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def validate_blank_template(document: DocumentObject, *, label: str) -> None:
|
|
44
|
+
errors = blank_template_errors(document)
|
|
45
|
+
if errors:
|
|
46
|
+
raise TemplateError(
|
|
47
|
+
"template_not_blank",
|
|
48
|
+
f"Template must be blank and formatting-only: {errors[0]}",
|
|
49
|
+
details={"template": label, "errors": errors},
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def blank_template_errors(document: DocumentObject) -> list[str]:
|
|
54
|
+
errors: list[str] = []
|
|
55
|
+
if any(paragraph.text.strip() for paragraph in document.paragraphs):
|
|
56
|
+
errors.append("body contains text")
|
|
57
|
+
if document.tables:
|
|
58
|
+
errors.append("body contains tables")
|
|
59
|
+
if len(document.inline_shapes) or any(_paragraph_has_drawing(p) for p in document.paragraphs):
|
|
60
|
+
errors.append("body contains images or drawings")
|
|
61
|
+
for number, section in enumerate(document.sections, start=1):
|
|
62
|
+
for story_name, story in (("header", section.header), ("footer", section.footer)):
|
|
63
|
+
if _story_has_content(story):
|
|
64
|
+
errors.append(f"section {number} {story_name} is not empty")
|
|
65
|
+
return errors
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def inspect_template(path: Path | None) -> dict[str, Any]:
|
|
69
|
+
if path is None:
|
|
70
|
+
label = "packaged-default"
|
|
71
|
+
document = Document(BytesIO(default_template_bytes()))
|
|
72
|
+
else:
|
|
73
|
+
resolved = path.resolve()
|
|
74
|
+
label = str(resolved)
|
|
75
|
+
if not resolved.is_file():
|
|
76
|
+
raise TemplateError("template_not_found", f"Template does not exist: {resolved}")
|
|
77
|
+
try:
|
|
78
|
+
document = Document(str(resolved))
|
|
79
|
+
except (PackageNotFoundError, ValueError) as exc:
|
|
80
|
+
raise TemplateError("template_invalid", f"Template is not a readable DOCX file: {resolved}") from exc
|
|
81
|
+
errors = blank_template_errors(document)
|
|
82
|
+
groups = {
|
|
83
|
+
"paragraph": sorted(style.name for style in document.styles if style.type == WD_STYLE_TYPE.PARAGRAPH),
|
|
84
|
+
"character": sorted(style.name for style in document.styles if style.type == WD_STYLE_TYPE.CHARACTER),
|
|
85
|
+
"table": sorted(style.name for style in document.styles if style.type == WD_STYLE_TYPE.TABLE),
|
|
86
|
+
}
|
|
87
|
+
sections = [
|
|
88
|
+
{
|
|
89
|
+
"index": index,
|
|
90
|
+
"orientation": section.orientation.name.lower() if section.orientation else None,
|
|
91
|
+
"width_inches": round(section.page_width.inches, 4) if section.page_width else None,
|
|
92
|
+
"height_inches": round(section.page_height.inches, 4) if section.page_height else None,
|
|
93
|
+
"margins_inches": {
|
|
94
|
+
"top": round(section.top_margin.inches, 4) if section.top_margin else None,
|
|
95
|
+
"right": round(section.right_margin.inches, 4) if section.right_margin else None,
|
|
96
|
+
"bottom": round(section.bottom_margin.inches, 4) if section.bottom_margin else None,
|
|
97
|
+
"left": round(section.left_margin.inches, 4) if section.left_margin else None,
|
|
98
|
+
},
|
|
99
|
+
}
|
|
100
|
+
for index, section in enumerate(document.sections, start=1)
|
|
101
|
+
]
|
|
102
|
+
return {
|
|
103
|
+
"template": label,
|
|
104
|
+
"valid": not errors,
|
|
105
|
+
"errors": errors,
|
|
106
|
+
"styles": groups,
|
|
107
|
+
"sections": sections,
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _story_has_content(story: Story) -> bool:
|
|
112
|
+
return bool(story.tables) or any(
|
|
113
|
+
paragraph.text.strip() or _paragraph_has_drawing(paragraph) for paragraph in story.paragraphs
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _paragraph_has_drawing(paragraph: Paragraph) -> bool:
|
|
118
|
+
for run in paragraph.runs:
|
|
119
|
+
if any(not isinstance(item, str) for item in run.iter_inner_content()):
|
|
120
|
+
return True
|
|
121
|
+
return False
|