markdown-docx 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,456 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import Any, NoReturn, cast
7
+
8
+ from markdown_it import MarkdownIt
9
+ from markdown_it.token import Token
10
+
11
+ from markdown_docx.errors import ParseError, UnsupportedFeatureError
12
+ from markdown_docx.markdown_body import is_standalone_image, is_task_item, parse_inline
13
+ from markdown_docx.metadata import (
14
+ default_document_options,
15
+ parse_document_options,
16
+ parse_image_options,
17
+ parse_section_options,
18
+ parse_table_options,
19
+ parse_yaml_payload,
20
+ )
21
+ from markdown_docx.models import (
22
+ Alignment,
23
+ Block,
24
+ CodeBlock,
25
+ DocumentModel,
26
+ DocumentOptions,
27
+ HeadingBlock,
28
+ ImageBlock,
29
+ ImageOptions,
30
+ ListKind,
31
+ ListParagraphBlock,
32
+ PageBreakBlock,
33
+ ParagraphBlock,
34
+ SectionBreakBlock,
35
+ TableBlock,
36
+ TableCell,
37
+ TableOptions,
38
+ )
39
+
40
+ COMMENT_PATTERN = re.compile(r"^<!--\s*markdown-docx(?P<body>.*?)-->\s*$", re.DOTALL)
41
+
42
+
43
+ @dataclass(slots=True)
44
+ class Directive:
45
+ kind: str
46
+ value: Any
47
+ line: int
48
+ end_line_index: int
49
+
50
+
51
+ def parse_document(
52
+ source: str,
53
+ *,
54
+ input_path: Path | None,
55
+ source_name: str,
56
+ ) -> DocumentModel:
57
+ markdown = MarkdownIt("commonmark", {"html": True}).enable("table")
58
+ try:
59
+ tokens = markdown.parse(source)
60
+ except Exception as exc:
61
+ raise ParseError(
62
+ "markdown_parse_error",
63
+ f"Markdown parsing failed: {exc}",
64
+ input_path=str(input_path) if input_path else source_name,
65
+ ) from exc
66
+
67
+ input_label = str(input_path) if input_path else source_name
68
+ source_lines = source.splitlines()
69
+ options = default_document_options()
70
+ blocks: list[Block] = []
71
+ document_seen = False
72
+ visible_seen = False
73
+ pending: Directive | None = None
74
+ index = 0
75
+
76
+ while index < len(tokens):
77
+ token = tokens[index]
78
+ if token.type == "html_block":
79
+ directive = _parse_directive(token, input_path=input_label)
80
+ if directive is None:
81
+ _unsupported("Raw HTML and non-reserved HTML comments are not supported.", token, input_label)
82
+ if pending is not None:
83
+ raise ParseError(
84
+ "metadata_placement_error",
85
+ "More than one metadata block cannot attach to the same content block.",
86
+ line=directive.line,
87
+ input_path=input_label,
88
+ metadata_kind=directive.kind,
89
+ )
90
+ if directive.kind == "document":
91
+ if visible_seen or document_seen or blocks:
92
+ raise ParseError(
93
+ "metadata_placement_error",
94
+ "Document metadata must be the first non-whitespace content and may appear only once.",
95
+ line=directive.line,
96
+ input_path=input_label,
97
+ metadata_kind="document",
98
+ )
99
+ options = parse_document_options(directive.value, line=directive.line, input_path=input_label)
100
+ document_seen = True
101
+ else:
102
+ pending = directive
103
+ index += 1
104
+ continue
105
+
106
+ visible_seen = True
107
+ line = _token_line(token)
108
+ table_options: TableOptions | None = None
109
+ image_options: ImageOptions | None = None
110
+ if pending is not None:
111
+ _validate_adjacency(pending, token, source_lines, input_label)
112
+ if pending.kind == "page-break":
113
+ blocks.append(PageBreakBlock(line=pending.line))
114
+ elif pending.kind == "section":
115
+ settings = parse_section_options(
116
+ pending.value,
117
+ defaults=options.section,
118
+ line=pending.line,
119
+ input_path=input_label,
120
+ )
121
+ blocks.append(SectionBreakBlock(line=pending.line, settings=settings))
122
+ elif pending.kind == "table":
123
+ if token.type != "table_open":
124
+ _attachment_error("table", pending.line, input_label)
125
+ table_options = parse_table_options(pending.value, line=pending.line, input_path=input_label)
126
+ elif pending.kind == "image":
127
+ if token.type != "paragraph_open":
128
+ _attachment_error("image", pending.line, input_label)
129
+ image_options = parse_image_options(pending.value, line=pending.line, input_path=input_label)
130
+ else:
131
+ raise AssertionError(f"Unhandled directive: {pending.kind}")
132
+ pending = None
133
+
134
+ token_type = token.type
135
+ if token_type == "paragraph_open":
136
+ paragraph, index = _consume_paragraph(tokens, index, input_label)
137
+ if is_standalone_image(paragraph.fragments):
138
+ image = next(fragment for fragment in paragraph.fragments if fragment.kind == "image")
139
+ blocks.append(
140
+ ImageBlock(
141
+ line=paragraph.line,
142
+ src=image.src or "",
143
+ alt=image.alt or "",
144
+ options=image_options or ImageOptions(),
145
+ )
146
+ )
147
+ elif image_options is not None:
148
+ _attachment_error("image", line, input_label)
149
+ else:
150
+ blocks.append(paragraph)
151
+ elif token_type == "heading_open":
152
+ if image_options is not None:
153
+ _attachment_error("image", line, input_label)
154
+ heading, index = _consume_heading(tokens, index, input_label)
155
+ blocks.append(heading)
156
+ elif token_type == "fence":
157
+ blocks.append(CodeBlock(line=line, text=token.content))
158
+ index += 1
159
+ elif token_type == "code_block":
160
+ _unsupported("Indented code blocks are not supported. Use a fenced code block.", token, input_label)
161
+ elif token_type == "blockquote_open":
162
+ quote_blocks, index = _consume_blockquote(tokens, index, input_label)
163
+ blocks.extend(quote_blocks)
164
+ elif token_type in {"bullet_list_open", "ordered_list_open"}:
165
+ list_blocks, index = _consume_list(tokens, index, depth=0, options=options, input_path=input_label)
166
+ blocks.extend(list_blocks)
167
+ elif token_type == "table_open":
168
+ table, index = _consume_table(tokens, index, table_options or TableOptions(), input_label)
169
+ blocks.append(table)
170
+ elif token_type == "hr":
171
+ _unsupported("Horizontal rules are not supported.", token, input_label)
172
+ elif token_type == "html_inline":
173
+ _unsupported("Raw inline HTML is not supported.", token, input_label)
174
+ else:
175
+ _unsupported(f"Markdown token '{token_type}' is not supported.", token, input_label)
176
+
177
+ if pending is not None:
178
+ raise ParseError(
179
+ "metadata_placement_error",
180
+ f"{pending.kind} metadata must be followed by a content block.",
181
+ line=pending.line,
182
+ input_path=input_label,
183
+ metadata_kind=pending.kind,
184
+ )
185
+ return DocumentModel(input_path=input_path, source_name=source_name, options=options, blocks=blocks)
186
+
187
+
188
+ def _parse_directive(token: Token, *, input_path: str) -> Directive | None:
189
+ match = COMMENT_PATTERN.fullmatch(token.content)
190
+ if match is None:
191
+ return None
192
+ line = _token_line(token)
193
+ body = match.group("body").strip()
194
+ end_line_index = token.map[1] if token.map else line
195
+ if body.startswith(":"):
196
+ compact = body[1:].strip()
197
+ if compact != "page-break":
198
+ raise ParseError(
199
+ "unknown_metadata_key",
200
+ f"Unknown compact markdown-docx directive: {compact or '<empty>'}",
201
+ line=line,
202
+ input_path=input_path,
203
+ )
204
+ return Directive("page-break", None, line, end_line_index)
205
+ if not body:
206
+ raise ParseError(
207
+ "metadata_parse_error",
208
+ "A markdown-docx metadata comment requires a YAML payload.",
209
+ line=line,
210
+ input_path=input_path,
211
+ )
212
+ payload = parse_yaml_payload(body, line=line, input_path=input_path, metadata_kind="markdown-docx")
213
+ if not isinstance(payload, dict) or len(payload) != 1 or any(not isinstance(key, str) for key in payload):
214
+ raise ParseError(
215
+ "metadata_parse_error",
216
+ "A metadata comment must contain exactly one of document, section, table, or image.",
217
+ line=line,
218
+ input_path=input_path,
219
+ )
220
+ kind, value = next(iter(payload.items()))
221
+ if kind not in {"document", "section", "table", "image"}:
222
+ raise ParseError(
223
+ "unknown_metadata_key",
224
+ f"Unknown metadata directive: {kind}",
225
+ line=line,
226
+ input_path=input_path,
227
+ details={"key": kind},
228
+ )
229
+ return Directive(kind, value, line, end_line_index)
230
+
231
+
232
+ def _validate_adjacency(directive: Directive, token: Token, lines: list[str], input_path: str) -> None:
233
+ if directive.kind not in {"table", "image"} or token.map is None:
234
+ return
235
+ next_start = token.map[0]
236
+ if any(line.strip() for line in lines[directive.end_line_index : next_start]):
237
+ _attachment_error(directive.kind, directive.line, input_path)
238
+
239
+
240
+ def _consume_paragraph(tokens: list[Token], index: int, input_path: str) -> tuple[ParagraphBlock, int]:
241
+ opening = tokens[index]
242
+ if index + 2 >= len(tokens) or tokens[index + 1].type != "inline" or tokens[index + 2].type != "paragraph_close":
243
+ _unsupported("This paragraph structure is not supported.", opening, input_path)
244
+ line = _token_line(opening)
245
+ fragments = parse_inline(tokens[index + 1], line=line, input_path=input_path)
246
+ return ParagraphBlock(line=line, fragments=fragments), index + 3
247
+
248
+
249
+ def _consume_heading(tokens: list[Token], index: int, input_path: str) -> tuple[HeadingBlock, int]:
250
+ opening = tokens[index]
251
+ if not opening.markup or set(opening.markup) != {"#"}:
252
+ _unsupported("Setext headings are not supported. Use ATX headings beginning with #.", opening, input_path)
253
+ if index + 2 >= len(tokens) or tokens[index + 1].type != "inline" or tokens[index + 2].type != "heading_close":
254
+ _unsupported("This heading structure is not supported.", opening, input_path)
255
+ level = int(opening.tag[1:])
256
+ line = _token_line(opening)
257
+ fragments = parse_inline(tokens[index + 1], line=line, input_path=input_path)
258
+ return HeadingBlock(line=line, level=level, fragments=fragments), index + 3
259
+
260
+
261
+ def _consume_blockquote(tokens: list[Token], index: int, input_path: str) -> tuple[list[ParagraphBlock], int]:
262
+ opening = tokens[index]
263
+ blocks: list[ParagraphBlock] = []
264
+ index += 1
265
+ while index < len(tokens) and tokens[index].type != "blockquote_close":
266
+ if tokens[index].type != "paragraph_open":
267
+ _unsupported("Blockquotes may contain paragraphs only in 0.1.0.", tokens[index], input_path)
268
+ paragraph, index = _consume_paragraph(tokens, index, input_path)
269
+ if any(fragment.kind == "image" for fragment in paragraph.fragments):
270
+ _unsupported("Images nested in blockquotes are not supported.", opening, input_path)
271
+ paragraph.role = "blockquote"
272
+ blocks.append(paragraph)
273
+ if index >= len(tokens):
274
+ _structure_error(opening, input_path)
275
+ return blocks, index + 1
276
+
277
+
278
+ def _consume_list(
279
+ tokens: list[Token],
280
+ index: int,
281
+ *,
282
+ depth: int,
283
+ options: DocumentOptions,
284
+ input_path: str,
285
+ ) -> tuple[list[ListParagraphBlock], int]:
286
+ opening = tokens[index]
287
+ ordered = opening.type == "ordered_list_open"
288
+ kind: ListKind = "ordered" if ordered else "unordered"
289
+ configured_styles = options.styles.ordered_list if ordered else options.styles.unordered_list
290
+ if depth >= len(configured_styles):
291
+ raise ParseError(
292
+ "list_depth_unsupported",
293
+ f"{kind} list depth {depth + 1} exceeds the {len(configured_styles)} configured styles.",
294
+ line=_token_line(opening),
295
+ input_path=input_path,
296
+ )
297
+ if ordered:
298
+ start = opening.attrGet("start")
299
+ if start is not None and int(start) != 1:
300
+ raise ParseError(
301
+ "ordered_list_start_unsupported",
302
+ "Ordered lists must begin with 1 in 0.1.0.",
303
+ line=_token_line(opening),
304
+ input_path=input_path,
305
+ )
306
+ closing_type = "ordered_list_close" if ordered else "bullet_list_close"
307
+ blocks: list[ListParagraphBlock] = []
308
+ index += 1
309
+ while index < len(tokens) and tokens[index].type != closing_type:
310
+ item_open = tokens[index]
311
+ if item_open.type != "list_item_open":
312
+ _structure_error(item_open, input_path)
313
+ index += 1
314
+ paragraph_seen = False
315
+ while index < len(tokens) and tokens[index].type != "list_item_close":
316
+ token = tokens[index]
317
+ if token.type == "paragraph_open":
318
+ if paragraph_seen:
319
+ _unsupported(
320
+ "Multi-paragraph list items are not supported by the public list-style API.", token, input_path
321
+ )
322
+ paragraph, index = _consume_paragraph(tokens, index, input_path)
323
+ if is_task_item(paragraph.fragments):
324
+ _unsupported("Task list syntax is not supported.", token, input_path)
325
+ if any(fragment.kind == "image" for fragment in paragraph.fragments):
326
+ _unsupported("Images nested in list items are not supported.", token, input_path)
327
+ blocks.append(
328
+ ListParagraphBlock(
329
+ line=paragraph.line,
330
+ fragments=paragraph.fragments,
331
+ list_kind=kind,
332
+ depth=depth,
333
+ )
334
+ )
335
+ paragraph_seen = True
336
+ elif token.type in {"bullet_list_open", "ordered_list_open"}:
337
+ nested, index = _consume_list(tokens, index, depth=depth + 1, options=options, input_path=input_path)
338
+ blocks.extend(nested)
339
+ else:
340
+ _unsupported("This content type is not supported inside list items.", token, input_path)
341
+ if index >= len(tokens) or not paragraph_seen:
342
+ _structure_error(item_open, input_path)
343
+ index += 1
344
+ if index >= len(tokens):
345
+ _structure_error(opening, input_path)
346
+ return blocks, index + 1
347
+
348
+
349
+ def _consume_table(
350
+ tokens: list[Token],
351
+ index: int,
352
+ options: TableOptions,
353
+ input_path: str,
354
+ ) -> tuple[TableBlock, int]:
355
+ opening = tokens[index]
356
+ line = _token_line(opening)
357
+ rows: list[list[TableCell]] = []
358
+ current_row: list[TableCell] | None = None
359
+ in_header = False
360
+ header_row_count = 0
361
+ index += 1
362
+ while index < len(tokens) and tokens[index].type != "table_close":
363
+ token = tokens[index]
364
+ if token.type == "thead_open":
365
+ in_header = True
366
+ index += 1
367
+ elif token.type == "thead_close":
368
+ in_header = False
369
+ index += 1
370
+ elif token.type in {"tbody_open", "tbody_close"}:
371
+ index += 1
372
+ elif token.type == "tr_open":
373
+ current_row = []
374
+ index += 1
375
+ elif token.type == "tr_close":
376
+ if current_row is None:
377
+ _structure_error(token, input_path)
378
+ rows.append(current_row)
379
+ if in_header:
380
+ header_row_count += 1
381
+ current_row = None
382
+ index += 1
383
+ elif token.type in {"th_open", "td_open"}:
384
+ if current_row is None or index + 2 >= len(tokens) or tokens[index + 1].type != "inline":
385
+ _structure_error(token, input_path)
386
+ raw_style = token.attrGet("style")
387
+ style = raw_style if isinstance(raw_style, str) else "text-align:left"
388
+ alignment = style.removeprefix("text-align:")
389
+ if alignment not in {"left", "center", "right"}:
390
+ alignment = "left"
391
+ fragments = parse_inline(tokens[index + 1], line=line, input_path=input_path)
392
+ if any(fragment.kind == "image" for fragment in fragments):
393
+ _unsupported("Images inside table cells are not supported.", token, input_path)
394
+ current_row.append(TableCell(fragments=fragments, alignment=cast(Alignment, alignment)))
395
+ expected_close = "th_close" if token.type == "th_open" else "td_close"
396
+ if tokens[index + 2].type != expected_close:
397
+ _structure_error(token, input_path)
398
+ index += 3
399
+ else:
400
+ _structure_error(token, input_path)
401
+ if index >= len(tokens) or not rows or header_row_count != 1:
402
+ raise ParseError(
403
+ "table_shape_invalid",
404
+ "A pipe table requires one header row and at least one row.",
405
+ line=line,
406
+ input_path=input_path,
407
+ )
408
+ column_count = len(rows[0])
409
+ if column_count == 0 or any(len(row) != column_count for row in rows):
410
+ raise ParseError(
411
+ "table_shape_invalid",
412
+ "Every table row must have the same number of cells.",
413
+ line=line,
414
+ input_path=input_path,
415
+ )
416
+ if options.column_widths is not None and len(options.column_widths) != column_count:
417
+ raise ParseError(
418
+ "table_shape_invalid",
419
+ "The number of column_widths entries must match the table column count.",
420
+ line=line,
421
+ input_path=input_path,
422
+ metadata_kind="table",
423
+ )
424
+ return TableBlock(line=line, headers=rows[0], rows=rows[1:], options=options), index + 1
425
+
426
+
427
+ def _token_line(token: Token) -> int:
428
+ return token.map[0] + 1 if token.map else 1
429
+
430
+
431
+ def _attachment_error(kind: str, line: int, input_path: str) -> NoReturn:
432
+ raise ParseError(
433
+ "metadata_placement_error",
434
+ f"{kind} metadata must be immediately before a standalone {kind}.",
435
+ line=line,
436
+ input_path=input_path,
437
+ metadata_kind=kind,
438
+ )
439
+
440
+
441
+ def _unsupported(message: str, token: Token, input_path: str) -> NoReturn:
442
+ raise UnsupportedFeatureError(
443
+ message,
444
+ line=_token_line(token),
445
+ input_path=input_path,
446
+ code="unsupported_markdown",
447
+ )
448
+
449
+
450
+ def _structure_error(token: Token, input_path: str) -> NoReturn:
451
+ raise ParseError(
452
+ "markdown_parse_error",
453
+ "The parsed Markdown structure is not supported.",
454
+ line=_token_line(token),
455
+ input_path=input_path,
456
+ )