docxwright 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ Metadata-Version: 2.4
2
+ Name: docxwright
3
+ Version: 0.1.0
4
+ Summary: A pragmatic LaTeX-to-DOCX converter for editable scientific manuscripts.
5
+ Author: Kenan Hanke
6
+ Requires-Python: >=3.10
7
+ Description-Content-Type: text/markdown
8
+ Requires-Dist: python-docx>=1.1.0
9
+
10
+ # docxwright
11
+
12
+ `docxwright` converts a LaTeX manuscript into an editable DOCX document without
13
+ requiring external tools such as Pandoc. It is intentionally pragmatic: it walks
14
+ included subfiles, keeps tables, replaces figures with placeholders and captions,
15
+ normalizes common LaTeX text/math to Unicode, emits author-year citation text
16
+ from Better-BibTeX-style keys, and omits the bibliography.
17
+
18
+ ```bash
19
+ docxwright paper-example/main.tex -o paper-example/out/main.docx
20
+ ```
21
+
22
+ For inspection, a Markdown-like intermediate representation can also be written:
23
+
24
+ ```bash
25
+ docxwright paper-example/main.tex -o paper-example/out/main.docx --markdown paper-example/out/main.md
26
+ ```
@@ -0,0 +1,17 @@
1
+ # docxwright
2
+
3
+ `docxwright` converts a LaTeX manuscript into an editable DOCX document without
4
+ requiring external tools such as Pandoc. It is intentionally pragmatic: it walks
5
+ included subfiles, keeps tables, replaces figures with placeholders and captions,
6
+ normalizes common LaTeX text/math to Unicode, emits author-year citation text
7
+ from Better-BibTeX-style keys, and omits the bibliography.
8
+
9
+ ```bash
10
+ docxwright paper-example/main.tex -o paper-example/out/main.docx
11
+ ```
12
+
13
+ For inspection, a Markdown-like intermediate representation can also be written:
14
+
15
+ ```bash
16
+ docxwright paper-example/main.tex -o paper-example/out/main.docx --markdown paper-example/out/main.md
17
+ ```
@@ -0,0 +1,5 @@
1
+ """docxwright package."""
2
+
3
+ from .converter import convert_file
4
+
5
+ __all__ = ["convert_file"]
@@ -0,0 +1,31 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ from pathlib import Path
5
+
6
+ from .converter import convert_file
7
+
8
+
9
+ def build_parser() -> argparse.ArgumentParser:
10
+ parser = argparse.ArgumentParser(
11
+ prog="docxwright",
12
+ description="Convert a LaTeX .tex manuscript to an editable .docx file.",
13
+ )
14
+ parser.add_argument("tex_file", type=Path, help="Path to the main .tex file.")
15
+ parser.add_argument("-o", "--output", type=Path, required=True, help="Output .docx path.")
16
+ parser.add_argument(
17
+ "--markdown",
18
+ type=Path,
19
+ help="Optional Markdown-like intermediate output for inspection.",
20
+ )
21
+ return parser
22
+
23
+
24
+ def main(argv: list[str] | None = None) -> int:
25
+ args = build_parser().parse_args(argv)
26
+ convert_file(args.tex_file, args.output, markdown_path=args.markdown)
27
+ return 0
28
+
29
+
30
+ if __name__ == "__main__":
31
+ raise SystemExit(main())
@@ -0,0 +1,662 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+ from pathlib import Path
5
+ import re
6
+ from typing import Iterable
7
+
8
+ from docx import Document
9
+ from docx.enum.text import WD_ALIGN_PARAGRAPH
10
+ from docx.shared import Pt
11
+
12
+
13
+ @dataclass
14
+ class Block:
15
+ kind: str
16
+ text: str = ""
17
+ level: int = 0
18
+ rows: list[list[str]] = field(default_factory=list)
19
+ caption: str = ""
20
+
21
+
22
+ @dataclass
23
+ class Metadata:
24
+ title: str = ""
25
+ authors: list[str] = field(default_factory=list)
26
+ date: str = ""
27
+
28
+
29
+ @dataclass
30
+ class ParseContext:
31
+ figure_count: int = 0
32
+ table_count: int = 0
33
+ labels: dict[str, str] = field(default_factory=dict)
34
+
35
+
36
+ GREEK = {
37
+ "alpha": "α",
38
+ "beta": "β",
39
+ "gamma": "γ",
40
+ "delta": "δ",
41
+ "epsilon": "ε",
42
+ "varepsilon": "ε",
43
+ "zeta": "ζ",
44
+ "eta": "η",
45
+ "theta": "θ",
46
+ "vartheta": "ϑ",
47
+ "iota": "ι",
48
+ "kappa": "κ",
49
+ "lambda": "λ",
50
+ "mu": "μ",
51
+ "nu": "ν",
52
+ "xi": "ξ",
53
+ "pi": "π",
54
+ "rho": "ρ",
55
+ "sigma": "σ",
56
+ "tau": "τ",
57
+ "upsilon": "υ",
58
+ "phi": "φ",
59
+ "varphi": "φ",
60
+ "chi": "χ",
61
+ "psi": "ψ",
62
+ "omega": "ω",
63
+ "Gamma": "Γ",
64
+ "Delta": "Δ",
65
+ "Theta": "Θ",
66
+ "Lambda": "Λ",
67
+ "Xi": "Ξ",
68
+ "Pi": "Π",
69
+ "Sigma": "Σ",
70
+ "Phi": "Φ",
71
+ "Psi": "Ψ",
72
+ "Omega": "Ω",
73
+ }
74
+
75
+ SYMBOLS = {
76
+ "times": "×",
77
+ "cdot": "·",
78
+ "pm": "±",
79
+ "le": "≤",
80
+ "leq": "≤",
81
+ "ge": "≥",
82
+ "geq": "≥",
83
+ "neq": "≠",
84
+ "ne": "≠",
85
+ "approx": "≈",
86
+ "sim": "∼",
87
+ "in": "∈",
88
+ "notin": "∉",
89
+ "exists": "∃",
90
+ "forall": "∀",
91
+ "bigvee": "∨",
92
+ "bigcup": "⋃",
93
+ "cup": "∪",
94
+ "cap": "∩",
95
+ "infty": "∞",
96
+ "sum": "∑",
97
+ "sqrt": "√",
98
+ "ln": "ln",
99
+ "bar": "¯",
100
+ "textcopyright": "©",
101
+ "copyright": "©",
102
+ "textregistered": "®",
103
+ "texttrademark": "™",
104
+ "degree": "°",
105
+ "circ": "°",
106
+ "qquad": " ",
107
+ "quad": " ",
108
+ }
109
+
110
+ SUPERSCRIPT = str.maketrans("0123456789+-=()n", "⁰¹²³⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾ⁿ")
111
+ SUBSCRIPT_MAP = {
112
+ **dict(zip("0123456789+-=()", "₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎")),
113
+ **dict(zip("aehijklmnoprstuvxy", "ₐₑₕᵢⱼₖₗₘₙₒₚᵣₛₜᵤᵥₓᵧ")),
114
+ }
115
+
116
+
117
+ def convert_file(tex_path: Path, docx_path: Path, markdown_path: Path | None = None) -> None:
118
+ tex_path = tex_path.resolve()
119
+ docx_path = docx_path.resolve()
120
+ source = flatten_tex(tex_path)
121
+ metadata = extract_metadata(source)
122
+ body = extract_document_body(source)
123
+ context = ParseContext(labels=collect_label_map(body))
124
+ blocks = parse_blocks(body, tex_path.parent, context)
125
+ docx_path.parent.mkdir(parents=True, exist_ok=True)
126
+ write_docx(blocks, metadata, docx_path)
127
+ if markdown_path:
128
+ markdown_path = markdown_path.resolve()
129
+ markdown_path.parent.mkdir(parents=True, exist_ok=True)
130
+ markdown_path.write_text(render_markdown(blocks, metadata), encoding="utf-8")
131
+
132
+
133
+ def flatten_tex(path: Path, seen: set[Path] | None = None) -> str:
134
+ seen = seen or set()
135
+ path = path.resolve()
136
+ if path in seen:
137
+ return ""
138
+ seen.add(path)
139
+ text = path.read_text(encoding="utf-8")
140
+
141
+ def replace_include(match: re.Match[str]) -> str:
142
+ command, target = match.group(1), match.group(2)
143
+ candidate = (path.parent / target)
144
+ if candidate.suffix != ".tex":
145
+ candidate = candidate.with_suffix(".tex")
146
+ if command == "input" and not candidate.exists():
147
+ return f"\n[Figure content omitted: {target}]\n"
148
+ if candidate.exists():
149
+ return flatten_tex(candidate, seen)
150
+ return ""
151
+
152
+ return re.sub(r"\\(subfile|input)\s*\{([^{}]+)\}", replace_include, text)
153
+
154
+
155
+ def extract_metadata(source: str) -> Metadata:
156
+ title = normalize_text(extract_braced_command(source, "title") or "")
157
+ author_text = extract_braced_command(source, "author") or ""
158
+ authors = [normalize_text(part).strip() for part in re.split(r"\\and", author_text) if part.strip()]
159
+ date = extract_braced_command(source, "date") or ""
160
+ if r"\today" in date:
161
+ date = ""
162
+ return Metadata(title=title, authors=authors, date=normalize_text(date))
163
+
164
+
165
+ def extract_document_body(source: str) -> str:
166
+ match = re.search(r"\\begin\{document\}(.*)\\end\{document\}", source, flags=re.S)
167
+ return match.group(1) if match else source
168
+
169
+
170
+ def parse_blocks(text: str, base_dir: Path, context: ParseContext) -> list[Block]:
171
+ text = strip_comments(text)
172
+ text = remove_bibliography(text)
173
+ text = remove_wrappers(text)
174
+ text = re.sub(r"\\maketitle|\\tableofcontents", "\n", text)
175
+ blocks: list[Block] = []
176
+ pos = 0
177
+ env_pattern = re.compile(r"\\begin\{(abstract|figure|table|equation|align|cases)\}(?:\[[^]]*\])?", re.S)
178
+ for match in env_pattern.finditer(text):
179
+ if match.start() < pos:
180
+ continue
181
+ blocks.extend(parse_plain(text[pos : match.start()], context))
182
+ env = match.group(1)
183
+ content_start = match.end()
184
+ end_match = find_environment_end(text, env, content_start)
185
+ if not end_match:
186
+ pos = content_start
187
+ continue
188
+ content = text[content_start : end_match.start()]
189
+ if env == "abstract":
190
+ blocks.append(Block("heading", "Abstract", level=1))
191
+ blocks.extend(parse_plain(content, context))
192
+ elif env == "figure":
193
+ blocks.append(parse_figure(content, context))
194
+ elif env == "table":
195
+ blocks.append(parse_table(content, context))
196
+ else:
197
+ blocks.append(Block("equation", normalize_math(content)))
198
+ pos = end_match.end()
199
+ blocks.extend(parse_plain(text[pos:], context))
200
+ return [block for block in blocks if block.kind != "paragraph" or block.text.strip()]
201
+
202
+
203
+ def parse_plain(text: str, context: ParseContext) -> list[Block]:
204
+ text = re.sub(r"\\end\{document\}|\\begin\{document\}|\\documentclass(?:\[[^]]*\])?\{[^{}]*\}", "\n", text)
205
+ blocks: list[Block] = []
206
+ parts = re.split(r"(\\(?:section|subsection|subsubsection|paragraph)\*?\{[^{}]*\})", text)
207
+ current: list[str] = []
208
+ for part in parts:
209
+ command_match = re.fullmatch(r"\\(section|subsection|subsubsection|paragraph)\*?\{([^{}]*)\}", part, flags=re.S)
210
+ if command_match:
211
+ blocks.extend(paragraph_blocks("".join(current), context))
212
+ current = []
213
+ level = {"section": 1, "subsection": 2, "subsubsection": 3, "paragraph": 4}[command_match.group(1)]
214
+ blocks.append(Block("heading", normalize_text(command_match.group(2), context), level=level))
215
+ else:
216
+ current.append(part)
217
+ blocks.extend(paragraph_blocks("".join(current), context))
218
+ return blocks
219
+
220
+
221
+ def paragraph_blocks(text: str, context: ParseContext) -> list[Block]:
222
+ text = text.replace("\\\\", "\n")
223
+ raw_paragraphs = re.split(r"\n\s*\n+", text)
224
+ blocks = []
225
+ for para in raw_paragraphs:
226
+ para = normalize_text(para, context).strip()
227
+ if para:
228
+ blocks.append(Block("paragraph", para))
229
+ return blocks
230
+
231
+
232
+ def parse_figure(content: str, context: ParseContext) -> Block:
233
+ context.figure_count += 1
234
+ label = extract_braced_command(content, "label")
235
+ if label:
236
+ context.labels[label] = str(context.figure_count)
237
+ caption = normalize_text(extract_braced_command(content, "caption") or "", context)
238
+ text = f"[Figure {context.figure_count} omitted]"
239
+ return Block("figure", text=text, caption=caption)
240
+
241
+
242
+ def parse_table(content: str, context: ParseContext) -> Block:
243
+ context.table_count += 1
244
+ label = extract_braced_command(content, "label")
245
+ if label:
246
+ context.labels[label] = str(context.table_count)
247
+ caption = normalize_text(extract_braced_command(content, "caption") or "", context)
248
+ tabular = extract_environment(content, "tabular")
249
+ rows = parse_tabular(tabular or "", context)
250
+ return Block("table", caption=caption, rows=rows)
251
+
252
+
253
+ def parse_tabular(content: str, context: ParseContext) -> list[list[str]]:
254
+ if not content:
255
+ return []
256
+ content = re.sub(r"^\s*\{[^{}]*\}", "", content.strip(), count=1, flags=re.S)
257
+ content = re.sub(r"\\(?:hline|toprule|midrule|bottomrule)\b", "\n", content)
258
+ content = re.sub(r"\\multirow(?:\[[^]]*\])?\{[^{}]*\}\{[^{}]*\}\{([^{}]*)\}", r"\1", content)
259
+ content = re.sub(r"\\multicolumn\{[^{}]*\}\{[^{}]*\}\{([^{}]*)\}", r"\1", content)
260
+ rows = []
261
+ for row_text in split_latex_rows(content):
262
+ cells = [normalize_text(cell, context).strip() for cell in split_latex_cells(row_text)]
263
+ cells = [cell for cell in cells if cell or len(cells) > 1]
264
+ if cells:
265
+ rows.append(cells)
266
+ return rows
267
+
268
+
269
+ def split_latex_rows(text: str) -> list[str]:
270
+ rows: list[str] = []
271
+ buf: list[str] = []
272
+ depth = 0
273
+ i = 0
274
+ while i < len(text):
275
+ ch = text[i]
276
+ if ch == "{":
277
+ depth += 1
278
+ elif ch == "}":
279
+ depth = max(0, depth - 1)
280
+ if depth == 0 and text.startswith(r"\\", i):
281
+ row = "".join(buf).strip()
282
+ if row:
283
+ rows.append(row)
284
+ buf = []
285
+ i += 2
286
+ continue
287
+ buf.append(ch)
288
+ i += 1
289
+ tail = "".join(buf).strip()
290
+ if tail:
291
+ rows.append(tail)
292
+ return rows
293
+
294
+
295
+ def split_latex_cells(row: str) -> list[str]:
296
+ cells: list[str] = []
297
+ buf: list[str] = []
298
+ depth = 0
299
+ for ch in row:
300
+ if ch == "{":
301
+ depth += 1
302
+ elif ch == "}":
303
+ depth = max(0, depth - 1)
304
+ if ch == "&" and depth == 0:
305
+ cells.append("".join(buf))
306
+ buf = []
307
+ else:
308
+ buf.append(ch)
309
+ cells.append("".join(buf))
310
+ return cells
311
+
312
+
313
+ def normalize_text(text: str, context: ParseContext | None = None) -> str:
314
+ context = context or ParseContext()
315
+ text = strip_comments(text)
316
+ text = text.replace("\u2010", "-")
317
+ text = re.sub(r"~|\\[,;:! ]", " ", text)
318
+ text = re.sub(r"\\(?:noindent|centering|small|normalsize|footnotesize|scriptsize)\b", "", text)
319
+ text = text.replace(r"\-", "")
320
+ text = re.sub(r"\\(?:cprotect|protect)\b", "", text)
321
+ text = replace_citations(text)
322
+ text = re.sub(r"\\ref\{([^{}]+)\}", lambda m: context.labels.get(m.group(1), f"reference {m.group(1)}"), text)
323
+ text = re.sub(r"\\label\{[^{}]*\}", "", text)
324
+ text = re.sub(r"\\href\{([^{}]*)\}\{([^{}]*)\}", lambda m: normalize_text(m.group(2), context), text)
325
+ text = re.sub(r"\\url\{([^{}]*)\}|\\nolinkurl\{([^{}]*)\}", lambda m: m.group(1) or m.group(2), text)
326
+ text = replace_math(text)
327
+ text = apply_inline_formatting(text, context)
328
+ text = unwrap_text_commands(text, context)
329
+ for name, value in {**GREEK, **SYMBOLS}.items():
330
+ text = text.replace(f"\\{name}", value)
331
+ text = text.replace("\\%", "%").replace("\\&", "&").replace("\\$", "$").replace("\\#", "#")
332
+ text = text.replace("\\_", "_").replace("\\{", "{").replace("\\}", "}")
333
+ text = text.replace("---", "—").replace("--", "–")
334
+ text = text.replace("``", "“").replace("''", "”")
335
+ text = re.sub(r"\\[a-zA-Z@]+\*?(?:\[[^]]*\])?", "", text)
336
+ text = text.replace("{", "").replace("}", "")
337
+ text = re.sub(r"[ \t\r\f\v]+", " ", text)
338
+ text = re.sub(r" *\n *", "\n", text)
339
+ return text.strip()
340
+
341
+
342
+ def apply_inline_formatting(text: str, context: ParseContext) -> str:
343
+ replacements = [
344
+ (r"\\textbf\*?\{([^{}]*)\}", "**{}**"),
345
+ (r"\\(?:textit|emph|mkbibemph)\*?\{([^{}]*)\}", "*{}*"),
346
+ (r"\\(?:texttt|verb)\*?(?:\|([^|]*)\||\{([^{}]*)\})", "`{}`"),
347
+ ]
348
+ previous = None
349
+ while previous != text:
350
+ previous = text
351
+ for pattern, template in replacements:
352
+ text = re.sub(
353
+ pattern,
354
+ lambda m, template=template: template.format(
355
+ normalize_text((m.group(1) or (m.group(2) if len(m.groups()) > 1 else "")), context)
356
+ ),
357
+ text,
358
+ )
359
+ return text
360
+
361
+
362
+ def replace_citations(text: str) -> str:
363
+ cite_pattern = re.compile(r"\\(?:parencite|cite|textcite|autocite)(?:\[[^]]*\]){0,2}\{([^{}]+)\}")
364
+ return cite_pattern.sub(lambda m: format_citation(m.group(1)), text)
365
+
366
+
367
+ def format_citation(keys: str) -> str:
368
+ citations = []
369
+ for key in keys.split(","):
370
+ key = key.strip()
371
+ year_match = re.search(r"((?:19|20)\d{2})", key)
372
+ year = year_match.group(1) if year_match else "n.d."
373
+ prefix = key[: year_match.start()] if year_match else key
374
+ surname_bits = re.match(r"[a-z\-]+", prefix)
375
+ surname = surname_bits.group(0) if surname_bits else prefix
376
+ surname = "-".join(bit.capitalize() for bit in surname.split("-") if bit)
377
+ if surname == "Muller":
378
+ surname = "Müller"
379
+ citations.append(f"{surname} {year}")
380
+ return "(" + "; ".join(citations) + ")"
381
+
382
+
383
+ def replace_math(text: str) -> str:
384
+ text = re.sub(r"\$\$(.*?)\$\$", lambda m: normalize_math(m.group(1)), text, flags=re.S)
385
+ text = re.sub(r"\$(.*?)\$", lambda m: normalize_math(m.group(1)), text, flags=re.S)
386
+ text = re.sub(r"\\\[(.*?)\\\]", lambda m: normalize_math(m.group(1)), text, flags=re.S)
387
+ text = re.sub(r"\\\((.*?)\\\)", lambda m: normalize_math(m.group(1)), text, flags=re.S)
388
+ return text
389
+
390
+
391
+ def normalize_math(math: str) -> str:
392
+ math = strip_comments(math).strip()
393
+ math = replace_latex_fractions(math)
394
+ math = re.sub(r"\\mathrm\{([^{}]*)\}", r"\1", math)
395
+ math = re.sub(r"\\mathbb\{([^{}]*)\}", r"\1", math)
396
+ math = re.sub(r"\\mathcal\{([^{}]*)\}", r"\1", math)
397
+ math = re.sub(r"\\text\{([^{}]*)\}", r"\1", math)
398
+ math = re.sub(r"\\boldsymbol\{([^{}]*)\}", r"\1", math)
399
+ math = re.sub(r"\\bar\{([^{}])\}", lambda m: m.group(1) + "\u0304", math)
400
+ math = replace_math_commands(math)
401
+ math = re.sub(r"\\frac\{([^{}]+)\}\{([^{}]+)\}", lambda m: f"{normalize_math(m.group(1))}/{normalize_math(m.group(2))}", math)
402
+ math = re.sub(r"\\sqrt\{([^{}]+)\}", lambda m: f"√({normalize_math(m.group(1))})", math)
403
+ math = re.sub(r"\^\{([^{}]+)\}", lambda m: to_superscript(normalize_math(m.group(1))), math)
404
+ math = re.sub(r"_\{([^{}]+)\}", lambda m: to_subscript(normalize_math(m.group(1))), math)
405
+ math = re.sub(r"\^\\circ\b", "°", math)
406
+ math = re.sub(r"\^([A-Za-z0-9+\-=()])", lambda m: to_superscript(m.group(1)), math)
407
+ math = re.sub(r"_([A-Za-z0-9+\-=()])", lambda m: to_subscript(m.group(1)), math)
408
+ math = re.sub(r"\\begin\{cases\}(.*?)\\end\{cases\}", lambda m: normalize_cases(m.group(1)), math, flags=re.S)
409
+ math = re.sub(r"\\begin\{bmatrix\}(.*?)\\end\{bmatrix\}", lambda m: "[" + normalize_math(m.group(1)) + "]", math, flags=re.S)
410
+ math = math.replace(r"\left", "").replace(r"\right", "")
411
+ math = math.replace(r"\!", "")
412
+ math = re.sub(r"\\(?:,|;|:| )", " ", math)
413
+ math = replace_math_commands(math)
414
+ math = re.sub(r"\\binom\{([^{}]+)\}\{([^{}]+)\}", r"C(\1,\2)", math)
415
+ math = math.replace(r"\mbox{-}", "-")
416
+ math = math.replace(r"\prime", "′")
417
+ math = math.replace(r"\{", "{").replace(r"\}", "}")
418
+ math = math.replace("\\", "")
419
+ math = math.replace("{", "").replace("}", "")
420
+ math = math.replace("---", "—").replace("--", "–")
421
+ math = re.sub(r"\s*&\s*", " ", math)
422
+ math = re.sub(r"\s+", " ", math)
423
+ return math.strip()
424
+
425
+
426
+ def replace_math_commands(math: str) -> str:
427
+ replacements = {**GREEK, **SYMBOLS}
428
+ replacements.update({"vee": "∨", "dots": "…", "ldots": "…", "vdots": "⋮", "mapsto": "↦", "mid": "|"})
429
+ for name, value in sorted(replacements.items(), key=lambda item: -len(item[0])):
430
+ math = re.sub(rf"\\{re.escape(name)}(?=[^A-Za-z]|$)", value, math)
431
+ return math
432
+
433
+
434
+ def normalize_cases(text: str) -> str:
435
+ rows = []
436
+ for row in split_latex_rows(text):
437
+ cells = [normalize_math(cell) for cell in split_latex_cells(row)]
438
+ rows.append(" if ".join(cell for cell in cells if cell))
439
+ return "{ " + "; ".join(rows) + " }"
440
+
441
+
442
+ def replace_latex_fractions(text: str) -> str:
443
+ command = r"\frac"
444
+ start = text.find(command)
445
+ while start != -1:
446
+ first_start = start + len(command)
447
+ first = read_braced_at(text, first_start)
448
+ if not first:
449
+ start = text.find(command, start + len(command))
450
+ continue
451
+ numerator, first_end = first
452
+ second = read_braced_at(text, first_end)
453
+ if not second:
454
+ start = text.find(command, start + len(command))
455
+ continue
456
+ denominator, second_end = second
457
+ replacement = f"({normalize_math(numerator)})/({normalize_math(denominator)})"
458
+ text = text[:start] + replacement + text[second_end:]
459
+ start = text.find(command, start + len(replacement))
460
+ return text
461
+
462
+
463
+ def read_braced_at(text: str, index: int) -> tuple[str, int] | None:
464
+ while index < len(text) and text[index].isspace():
465
+ index += 1
466
+ if index >= len(text) or text[index] != "{":
467
+ return None
468
+ depth = 0
469
+ for pos in range(index, len(text)):
470
+ if text[pos] == "{" and (pos == 0 or text[pos - 1] != "\\"):
471
+ depth += 1
472
+ elif text[pos] == "}" and (pos == 0 or text[pos - 1] != "\\"):
473
+ depth -= 1
474
+ if depth == 0:
475
+ return text[index + 1 : pos], pos + 1
476
+ return None
477
+
478
+
479
+ def to_superscript(value: str) -> str:
480
+ return value.translate(SUPERSCRIPT)
481
+
482
+
483
+ def to_subscript(value: str) -> str:
484
+ return "".join(SUBSCRIPT_MAP.get(char, char) for char in value)
485
+
486
+
487
+ def unwrap_text_commands(text: str, context: ParseContext) -> str:
488
+ command_pattern = re.compile(
489
+ r"\\(?:mathbf|mathit|mathbb|operatorname|underline)\*?(?:\|([^|]*)\||\{([^{}]*)\})"
490
+ )
491
+ previous = None
492
+ while previous != text:
493
+ previous = text
494
+ text = command_pattern.sub(lambda m: normalize_text(m.group(1) or m.group(2) or "", context), text)
495
+ return text
496
+
497
+
498
+ def strip_comments(text: str) -> str:
499
+ return re.sub(r"(?<!\\)%.*", "", text)
500
+
501
+
502
+ def remove_bibliography(text: str) -> str:
503
+ text = re.sub(r"\\printbibliography\b(?:\[[^]]*\])?", "", text)
504
+ text = re.sub(r"\\addbibresource\{[^{}]*\}", "", text)
505
+ text = re.sub(r"\\begin\{thebibliography\}.*?\\end\{thebibliography\}", "", text, flags=re.S)
506
+ return text
507
+
508
+
509
+ def remove_wrappers(text: str) -> str:
510
+ text = re.sub(r"\\begingroup|\\endgroup", "\n", text)
511
+ text = re.sub(r"\\(?:emergencystretch|hfuzz)\s*=\s*[^ \n]+", "", text)
512
+ return text
513
+
514
+
515
+ def collect_label_map(text: str) -> dict[str, str]:
516
+ labels: dict[str, str] = {}
517
+ counts = {"figure": 0, "table": 0}
518
+ pattern = re.compile(r"\\begin\{(figure|table)\}(?:\[[^]]*\])?(.*?)\\end\{\1\}", flags=re.S)
519
+ for match in pattern.finditer(text):
520
+ kind = match.group(1)
521
+ counts[kind] += 1
522
+ label = extract_braced_command(match.group(2), "label")
523
+ if label:
524
+ labels[label] = str(counts[kind])
525
+ return labels
526
+
527
+
528
+ def find_environment_end(text: str, env: str, start: int) -> re.Match[str] | None:
529
+ return re.compile(rf"\\end\{{{re.escape(env)}\}}", flags=re.S).search(text, start)
530
+
531
+
532
+ def extract_environment(text: str, env: str) -> str | None:
533
+ match = re.search(rf"\\begin\{{{re.escape(env)}\}}(.*?)\\end\{{{re.escape(env)}\}}", text, flags=re.S)
534
+ return match.group(1) if match else None
535
+
536
+
537
+ def extract_braced_command(text: str, command: str) -> str | None:
538
+ start = re.search(rf"\\{re.escape(command)}(?:\[[^]]*\])?\{{", text)
539
+ if not start:
540
+ return None
541
+ brace_start = start.end() - 1
542
+ depth = 0
543
+ for idx in range(brace_start, len(text)):
544
+ if text[idx] == "{" and (idx == 0 or text[idx - 1] != "\\"):
545
+ depth += 1
546
+ elif text[idx] == "}" and (idx == 0 or text[idx - 1] != "\\"):
547
+ depth -= 1
548
+ if depth == 0:
549
+ return text[brace_start + 1 : idx]
550
+ return None
551
+
552
+
553
+ def write_docx(blocks: Iterable[Block], metadata: Metadata, path: Path) -> None:
554
+ document = Document()
555
+ styles = document.styles
556
+ styles["Normal"].font.name = "Times New Roman"
557
+ styles["Normal"].font.size = Pt(11)
558
+ if metadata.title:
559
+ paragraph = document.add_heading(metadata.title, level=0)
560
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
561
+ if metadata.authors:
562
+ paragraph = document.add_paragraph(", ".join(metadata.authors))
563
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
564
+ if metadata.date:
565
+ paragraph = document.add_paragraph(metadata.date)
566
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
567
+
568
+ for block in blocks:
569
+ if block.kind == "heading":
570
+ document.add_heading(block.text, level=min(max(block.level, 1), 4))
571
+ elif block.kind == "paragraph":
572
+ add_formatted_paragraph(document, block.text)
573
+ elif block.kind == "equation":
574
+ paragraph = document.add_paragraph(block.text)
575
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
576
+ elif block.kind == "figure":
577
+ paragraph = document.add_paragraph(block.text)
578
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
579
+ if block.caption:
580
+ add_formatted_paragraph(document, f"Caption: {block.caption}")
581
+ elif block.kind == "table":
582
+ add_docx_table(document, block)
583
+ document.save(path)
584
+
585
+
586
+ def add_docx_table(document: Document, block: Block) -> None:
587
+ rows = block.rows
588
+ if not rows:
589
+ if block.caption:
590
+ document.add_paragraph(f"Table: {block.caption}")
591
+ return
592
+ column_count = max(len(row) for row in rows)
593
+ table = document.add_table(rows=len(rows), cols=column_count)
594
+ table.style = "Table Grid"
595
+ for row_index, row in enumerate(rows):
596
+ for col_index in range(column_count):
597
+ table.cell(row_index, col_index).text = row[col_index] if col_index < len(row) else ""
598
+ if block.caption:
599
+ add_formatted_paragraph(document, f"Caption: {block.caption}")
600
+
601
+
602
+ def add_formatted_paragraph(document: Document, text: str):
603
+ paragraph = document.add_paragraph()
604
+ add_formatted_runs(paragraph, text)
605
+ return paragraph
606
+
607
+
608
+ def add_formatted_runs(paragraph, text: str) -> None:
609
+ token_pattern = re.compile(r"(\*\*[^*]+\*\*|\*[^*]+\*|`[^`]+`)")
610
+ pos = 0
611
+ for match in token_pattern.finditer(text):
612
+ if match.start() > pos:
613
+ paragraph.add_run(text[pos : match.start()])
614
+ token = match.group(0)
615
+ if token.startswith("**"):
616
+ run = paragraph.add_run(token[2:-2])
617
+ run.bold = True
618
+ elif token.startswith("*"):
619
+ run = paragraph.add_run(token[1:-1])
620
+ run.italic = True
621
+ else:
622
+ run = paragraph.add_run(token[1:-1])
623
+ run.font.name = "Courier New"
624
+ pos = match.end()
625
+ if pos < len(text):
626
+ paragraph.add_run(text[pos:])
627
+
628
+
629
+ def render_markdown(blocks: Iterable[Block], metadata: Metadata) -> str:
630
+ lines: list[str] = []
631
+ if metadata.title:
632
+ lines.extend([f"# {metadata.title}", ""])
633
+ if metadata.authors:
634
+ lines.extend([", ".join(metadata.authors), ""])
635
+ for block in blocks:
636
+ if block.kind == "heading":
637
+ lines.extend([f"{'#' * min(block.level + 1, 6)} {block.text}", ""])
638
+ elif block.kind == "paragraph":
639
+ lines.extend([block.text, ""])
640
+ elif block.kind == "equation":
641
+ lines.extend([f"```text\n{block.text}\n```", ""])
642
+ elif block.kind == "figure":
643
+ lines.extend([block.text, ""])
644
+ if block.caption:
645
+ lines.extend([f"Caption: {block.caption}", ""])
646
+ elif block.kind == "table":
647
+ lines.extend(render_markdown_table(block.rows))
648
+ if block.caption:
649
+ lines.extend(["", f"Caption: {block.caption}", ""])
650
+ return "\n".join(lines).rstrip() + "\n"
651
+
652
+
653
+ def render_markdown_table(rows: list[list[str]]) -> list[str]:
654
+ if not rows:
655
+ return []
656
+ width = max(len(row) for row in rows)
657
+ padded = [row + [""] * (width - len(row)) for row in rows]
658
+ lines = ["| " + " | ".join(padded[0]) + " |"]
659
+ lines.append("| " + " | ".join("---" for _ in range(width)) + " |")
660
+ for row in padded[1:]:
661
+ lines.append("| " + " | ".join(row) + " |")
662
+ return lines
@@ -0,0 +1,26 @@
1
+ Metadata-Version: 2.4
2
+ Name: docxwright
3
+ Version: 0.1.0
4
+ Summary: A pragmatic LaTeX-to-DOCX converter for editable scientific manuscripts.
5
+ Author: Kenan Hanke
6
+ Requires-Python: >=3.10
7
+ Description-Content-Type: text/markdown
8
+ Requires-Dist: python-docx>=1.1.0
9
+
10
+ # docxwright
11
+
12
+ `docxwright` converts a LaTeX manuscript into an editable DOCX document without
13
+ requiring external tools such as Pandoc. It is intentionally pragmatic: it walks
14
+ included subfiles, keeps tables, replaces figures with placeholders and captions,
15
+ normalizes common LaTeX text/math to Unicode, emits author-year citation text
16
+ from Better-BibTeX-style keys, and omits the bibliography.
17
+
18
+ ```bash
19
+ docxwright paper-example/main.tex -o paper-example/out/main.docx
20
+ ```
21
+
22
+ For inspection, a Markdown-like intermediate representation can also be written:
23
+
24
+ ```bash
25
+ docxwright paper-example/main.tex -o paper-example/out/main.docx --markdown paper-example/out/main.md
26
+ ```
@@ -0,0 +1,11 @@
1
+ README.md
2
+ pyproject.toml
3
+ docxwright/__init__.py
4
+ docxwright/cli.py
5
+ docxwright/converter.py
6
+ docxwright.egg-info/PKG-INFO
7
+ docxwright.egg-info/SOURCES.txt
8
+ docxwright.egg-info/dependency_links.txt
9
+ docxwright.egg-info/entry_points.txt
10
+ docxwright.egg-info/requires.txt
11
+ docxwright.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ docxwright = docxwright.cli:main
@@ -0,0 +1 @@
1
+ python-docx>=1.1.0
@@ -0,0 +1 @@
1
+ docxwright
@@ -0,0 +1,22 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "docxwright"
7
+ version = "0.1.0"
8
+ description = "A pragmatic LaTeX-to-DOCX converter for editable scientific manuscripts."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ dependencies = [
12
+ "python-docx>=1.1.0",
13
+ ]
14
+ authors = [
15
+ { name = "Kenan Hanke" },
16
+ ]
17
+
18
+ [project.scripts]
19
+ docxwright = "docxwright.cli:main"
20
+
21
+ [tool.setuptools.packages.find]
22
+ include = ["docxwright*"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+