docxwright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docxwright/__init__.py +5 -0
- docxwright/cli.py +31 -0
- docxwright/converter.py +662 -0
- docxwright-0.1.0.dist-info/METADATA +26 -0
- docxwright-0.1.0.dist-info/RECORD +8 -0
- docxwright-0.1.0.dist-info/WHEEL +5 -0
- docxwright-0.1.0.dist-info/entry_points.txt +2 -0
- docxwright-0.1.0.dist-info/top_level.txt +1 -0
docxwright/__init__.py
ADDED
docxwright/cli.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from .converter import convert_file
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
10
|
+
parser = argparse.ArgumentParser(
|
|
11
|
+
prog="docxwright",
|
|
12
|
+
description="Convert a LaTeX .tex manuscript to an editable .docx file.",
|
|
13
|
+
)
|
|
14
|
+
parser.add_argument("tex_file", type=Path, help="Path to the main .tex file.")
|
|
15
|
+
parser.add_argument("-o", "--output", type=Path, required=True, help="Output .docx path.")
|
|
16
|
+
parser.add_argument(
|
|
17
|
+
"--markdown",
|
|
18
|
+
type=Path,
|
|
19
|
+
help="Optional Markdown-like intermediate output for inspection.",
|
|
20
|
+
)
|
|
21
|
+
return parser
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def main(argv: list[str] | None = None) -> int:
|
|
25
|
+
args = build_parser().parse_args(argv)
|
|
26
|
+
convert_file(args.tex_file, args.output, markdown_path=args.markdown)
|
|
27
|
+
return 0
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
if __name__ == "__main__":
|
|
31
|
+
raise SystemExit(main())
|
docxwright/converter.py
ADDED
|
@@ -0,0 +1,662 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import re
|
|
6
|
+
from typing import Iterable
|
|
7
|
+
|
|
8
|
+
from docx import Document
|
|
9
|
+
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
|
10
|
+
from docx.shared import Pt
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class Block:
|
|
15
|
+
kind: str
|
|
16
|
+
text: str = ""
|
|
17
|
+
level: int = 0
|
|
18
|
+
rows: list[list[str]] = field(default_factory=list)
|
|
19
|
+
caption: str = ""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class Metadata:
|
|
24
|
+
title: str = ""
|
|
25
|
+
authors: list[str] = field(default_factory=list)
|
|
26
|
+
date: str = ""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class ParseContext:
|
|
31
|
+
figure_count: int = 0
|
|
32
|
+
table_count: int = 0
|
|
33
|
+
labels: dict[str, str] = field(default_factory=dict)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
GREEK = {
|
|
37
|
+
"alpha": "α",
|
|
38
|
+
"beta": "β",
|
|
39
|
+
"gamma": "γ",
|
|
40
|
+
"delta": "δ",
|
|
41
|
+
"epsilon": "ε",
|
|
42
|
+
"varepsilon": "ε",
|
|
43
|
+
"zeta": "ζ",
|
|
44
|
+
"eta": "η",
|
|
45
|
+
"theta": "θ",
|
|
46
|
+
"vartheta": "ϑ",
|
|
47
|
+
"iota": "ι",
|
|
48
|
+
"kappa": "κ",
|
|
49
|
+
"lambda": "λ",
|
|
50
|
+
"mu": "μ",
|
|
51
|
+
"nu": "ν",
|
|
52
|
+
"xi": "ξ",
|
|
53
|
+
"pi": "π",
|
|
54
|
+
"rho": "ρ",
|
|
55
|
+
"sigma": "σ",
|
|
56
|
+
"tau": "τ",
|
|
57
|
+
"upsilon": "υ",
|
|
58
|
+
"phi": "φ",
|
|
59
|
+
"varphi": "φ",
|
|
60
|
+
"chi": "χ",
|
|
61
|
+
"psi": "ψ",
|
|
62
|
+
"omega": "ω",
|
|
63
|
+
"Gamma": "Γ",
|
|
64
|
+
"Delta": "Δ",
|
|
65
|
+
"Theta": "Θ",
|
|
66
|
+
"Lambda": "Λ",
|
|
67
|
+
"Xi": "Ξ",
|
|
68
|
+
"Pi": "Π",
|
|
69
|
+
"Sigma": "Σ",
|
|
70
|
+
"Phi": "Φ",
|
|
71
|
+
"Psi": "Ψ",
|
|
72
|
+
"Omega": "Ω",
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
SYMBOLS = {
|
|
76
|
+
"times": "×",
|
|
77
|
+
"cdot": "·",
|
|
78
|
+
"pm": "±",
|
|
79
|
+
"le": "≤",
|
|
80
|
+
"leq": "≤",
|
|
81
|
+
"ge": "≥",
|
|
82
|
+
"geq": "≥",
|
|
83
|
+
"neq": "≠",
|
|
84
|
+
"ne": "≠",
|
|
85
|
+
"approx": "≈",
|
|
86
|
+
"sim": "∼",
|
|
87
|
+
"in": "∈",
|
|
88
|
+
"notin": "∉",
|
|
89
|
+
"exists": "∃",
|
|
90
|
+
"forall": "∀",
|
|
91
|
+
"bigvee": "∨",
|
|
92
|
+
"bigcup": "⋃",
|
|
93
|
+
"cup": "∪",
|
|
94
|
+
"cap": "∩",
|
|
95
|
+
"infty": "∞",
|
|
96
|
+
"sum": "∑",
|
|
97
|
+
"sqrt": "√",
|
|
98
|
+
"ln": "ln",
|
|
99
|
+
"bar": "¯",
|
|
100
|
+
"textcopyright": "©",
|
|
101
|
+
"copyright": "©",
|
|
102
|
+
"textregistered": "®",
|
|
103
|
+
"texttrademark": "™",
|
|
104
|
+
"degree": "°",
|
|
105
|
+
"circ": "°",
|
|
106
|
+
"qquad": " ",
|
|
107
|
+
"quad": " ",
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
SUPERSCRIPT = str.maketrans("0123456789+-=()n", "⁰¹²³⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾ⁿ")
|
|
111
|
+
SUBSCRIPT_MAP = {
|
|
112
|
+
**dict(zip("0123456789+-=()", "₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎")),
|
|
113
|
+
**dict(zip("aehijklmnoprstuvxy", "ₐₑₕᵢⱼₖₗₘₙₒₚᵣₛₜᵤᵥₓᵧ")),
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def convert_file(tex_path: Path, docx_path: Path, markdown_path: Path | None = None) -> None:
|
|
118
|
+
tex_path = tex_path.resolve()
|
|
119
|
+
docx_path = docx_path.resolve()
|
|
120
|
+
source = flatten_tex(tex_path)
|
|
121
|
+
metadata = extract_metadata(source)
|
|
122
|
+
body = extract_document_body(source)
|
|
123
|
+
context = ParseContext(labels=collect_label_map(body))
|
|
124
|
+
blocks = parse_blocks(body, tex_path.parent, context)
|
|
125
|
+
docx_path.parent.mkdir(parents=True, exist_ok=True)
|
|
126
|
+
write_docx(blocks, metadata, docx_path)
|
|
127
|
+
if markdown_path:
|
|
128
|
+
markdown_path = markdown_path.resolve()
|
|
129
|
+
markdown_path.parent.mkdir(parents=True, exist_ok=True)
|
|
130
|
+
markdown_path.write_text(render_markdown(blocks, metadata), encoding="utf-8")
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def flatten_tex(path: Path, seen: set[Path] | None = None) -> str:
|
|
134
|
+
seen = seen or set()
|
|
135
|
+
path = path.resolve()
|
|
136
|
+
if path in seen:
|
|
137
|
+
return ""
|
|
138
|
+
seen.add(path)
|
|
139
|
+
text = path.read_text(encoding="utf-8")
|
|
140
|
+
|
|
141
|
+
def replace_include(match: re.Match[str]) -> str:
|
|
142
|
+
command, target = match.group(1), match.group(2)
|
|
143
|
+
candidate = (path.parent / target)
|
|
144
|
+
if candidate.suffix != ".tex":
|
|
145
|
+
candidate = candidate.with_suffix(".tex")
|
|
146
|
+
if command == "input" and not candidate.exists():
|
|
147
|
+
return f"\n[Figure content omitted: {target}]\n"
|
|
148
|
+
if candidate.exists():
|
|
149
|
+
return flatten_tex(candidate, seen)
|
|
150
|
+
return ""
|
|
151
|
+
|
|
152
|
+
return re.sub(r"\\(subfile|input)\s*\{([^{}]+)\}", replace_include, text)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def extract_metadata(source: str) -> Metadata:
|
|
156
|
+
title = normalize_text(extract_braced_command(source, "title") or "")
|
|
157
|
+
author_text = extract_braced_command(source, "author") or ""
|
|
158
|
+
authors = [normalize_text(part).strip() for part in re.split(r"\\and", author_text) if part.strip()]
|
|
159
|
+
date = extract_braced_command(source, "date") or ""
|
|
160
|
+
if r"\today" in date:
|
|
161
|
+
date = ""
|
|
162
|
+
return Metadata(title=title, authors=authors, date=normalize_text(date))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def extract_document_body(source: str) -> str:
|
|
166
|
+
match = re.search(r"\\begin\{document\}(.*)\\end\{document\}", source, flags=re.S)
|
|
167
|
+
return match.group(1) if match else source
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def parse_blocks(text: str, base_dir: Path, context: ParseContext) -> list[Block]:
|
|
171
|
+
text = strip_comments(text)
|
|
172
|
+
text = remove_bibliography(text)
|
|
173
|
+
text = remove_wrappers(text)
|
|
174
|
+
text = re.sub(r"\\maketitle|\\tableofcontents", "\n", text)
|
|
175
|
+
blocks: list[Block] = []
|
|
176
|
+
pos = 0
|
|
177
|
+
env_pattern = re.compile(r"\\begin\{(abstract|figure|table|equation|align|cases)\}(?:\[[^]]*\])?", re.S)
|
|
178
|
+
for match in env_pattern.finditer(text):
|
|
179
|
+
if match.start() < pos:
|
|
180
|
+
continue
|
|
181
|
+
blocks.extend(parse_plain(text[pos : match.start()], context))
|
|
182
|
+
env = match.group(1)
|
|
183
|
+
content_start = match.end()
|
|
184
|
+
end_match = find_environment_end(text, env, content_start)
|
|
185
|
+
if not end_match:
|
|
186
|
+
pos = content_start
|
|
187
|
+
continue
|
|
188
|
+
content = text[content_start : end_match.start()]
|
|
189
|
+
if env == "abstract":
|
|
190
|
+
blocks.append(Block("heading", "Abstract", level=1))
|
|
191
|
+
blocks.extend(parse_plain(content, context))
|
|
192
|
+
elif env == "figure":
|
|
193
|
+
blocks.append(parse_figure(content, context))
|
|
194
|
+
elif env == "table":
|
|
195
|
+
blocks.append(parse_table(content, context))
|
|
196
|
+
else:
|
|
197
|
+
blocks.append(Block("equation", normalize_math(content)))
|
|
198
|
+
pos = end_match.end()
|
|
199
|
+
blocks.extend(parse_plain(text[pos:], context))
|
|
200
|
+
return [block for block in blocks if block.kind != "paragraph" or block.text.strip()]
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def parse_plain(text: str, context: ParseContext) -> list[Block]:
|
|
204
|
+
text = re.sub(r"\\end\{document\}|\\begin\{document\}|\\documentclass(?:\[[^]]*\])?\{[^{}]*\}", "\n", text)
|
|
205
|
+
blocks: list[Block] = []
|
|
206
|
+
parts = re.split(r"(\\(?:section|subsection|subsubsection|paragraph)\*?\{[^{}]*\})", text)
|
|
207
|
+
current: list[str] = []
|
|
208
|
+
for part in parts:
|
|
209
|
+
command_match = re.fullmatch(r"\\(section|subsection|subsubsection|paragraph)\*?\{([^{}]*)\}", part, flags=re.S)
|
|
210
|
+
if command_match:
|
|
211
|
+
blocks.extend(paragraph_blocks("".join(current), context))
|
|
212
|
+
current = []
|
|
213
|
+
level = {"section": 1, "subsection": 2, "subsubsection": 3, "paragraph": 4}[command_match.group(1)]
|
|
214
|
+
blocks.append(Block("heading", normalize_text(command_match.group(2), context), level=level))
|
|
215
|
+
else:
|
|
216
|
+
current.append(part)
|
|
217
|
+
blocks.extend(paragraph_blocks("".join(current), context))
|
|
218
|
+
return blocks
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def paragraph_blocks(text: str, context: ParseContext) -> list[Block]:
|
|
222
|
+
text = text.replace("\\\\", "\n")
|
|
223
|
+
raw_paragraphs = re.split(r"\n\s*\n+", text)
|
|
224
|
+
blocks = []
|
|
225
|
+
for para in raw_paragraphs:
|
|
226
|
+
para = normalize_text(para, context).strip()
|
|
227
|
+
if para:
|
|
228
|
+
blocks.append(Block("paragraph", para))
|
|
229
|
+
return blocks
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def parse_figure(content: str, context: ParseContext) -> Block:
|
|
233
|
+
context.figure_count += 1
|
|
234
|
+
label = extract_braced_command(content, "label")
|
|
235
|
+
if label:
|
|
236
|
+
context.labels[label] = str(context.figure_count)
|
|
237
|
+
caption = normalize_text(extract_braced_command(content, "caption") or "", context)
|
|
238
|
+
text = f"[Figure {context.figure_count} omitted]"
|
|
239
|
+
return Block("figure", text=text, caption=caption)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def parse_table(content: str, context: ParseContext) -> Block:
|
|
243
|
+
context.table_count += 1
|
|
244
|
+
label = extract_braced_command(content, "label")
|
|
245
|
+
if label:
|
|
246
|
+
context.labels[label] = str(context.table_count)
|
|
247
|
+
caption = normalize_text(extract_braced_command(content, "caption") or "", context)
|
|
248
|
+
tabular = extract_environment(content, "tabular")
|
|
249
|
+
rows = parse_tabular(tabular or "", context)
|
|
250
|
+
return Block("table", caption=caption, rows=rows)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def parse_tabular(content: str, context: ParseContext) -> list[list[str]]:
|
|
254
|
+
if not content:
|
|
255
|
+
return []
|
|
256
|
+
content = re.sub(r"^\s*\{[^{}]*\}", "", content.strip(), count=1, flags=re.S)
|
|
257
|
+
content = re.sub(r"\\(?:hline|toprule|midrule|bottomrule)\b", "\n", content)
|
|
258
|
+
content = re.sub(r"\\multirow(?:\[[^]]*\])?\{[^{}]*\}\{[^{}]*\}\{([^{}]*)\}", r"\1", content)
|
|
259
|
+
content = re.sub(r"\\multicolumn\{[^{}]*\}\{[^{}]*\}\{([^{}]*)\}", r"\1", content)
|
|
260
|
+
rows = []
|
|
261
|
+
for row_text in split_latex_rows(content):
|
|
262
|
+
cells = [normalize_text(cell, context).strip() for cell in split_latex_cells(row_text)]
|
|
263
|
+
cells = [cell for cell in cells if cell or len(cells) > 1]
|
|
264
|
+
if cells:
|
|
265
|
+
rows.append(cells)
|
|
266
|
+
return rows
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def split_latex_rows(text: str) -> list[str]:
|
|
270
|
+
rows: list[str] = []
|
|
271
|
+
buf: list[str] = []
|
|
272
|
+
depth = 0
|
|
273
|
+
i = 0
|
|
274
|
+
while i < len(text):
|
|
275
|
+
ch = text[i]
|
|
276
|
+
if ch == "{":
|
|
277
|
+
depth += 1
|
|
278
|
+
elif ch == "}":
|
|
279
|
+
depth = max(0, depth - 1)
|
|
280
|
+
if depth == 0 and text.startswith(r"\\", i):
|
|
281
|
+
row = "".join(buf).strip()
|
|
282
|
+
if row:
|
|
283
|
+
rows.append(row)
|
|
284
|
+
buf = []
|
|
285
|
+
i += 2
|
|
286
|
+
continue
|
|
287
|
+
buf.append(ch)
|
|
288
|
+
i += 1
|
|
289
|
+
tail = "".join(buf).strip()
|
|
290
|
+
if tail:
|
|
291
|
+
rows.append(tail)
|
|
292
|
+
return rows
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def split_latex_cells(row: str) -> list[str]:
|
|
296
|
+
cells: list[str] = []
|
|
297
|
+
buf: list[str] = []
|
|
298
|
+
depth = 0
|
|
299
|
+
for ch in row:
|
|
300
|
+
if ch == "{":
|
|
301
|
+
depth += 1
|
|
302
|
+
elif ch == "}":
|
|
303
|
+
depth = max(0, depth - 1)
|
|
304
|
+
if ch == "&" and depth == 0:
|
|
305
|
+
cells.append("".join(buf))
|
|
306
|
+
buf = []
|
|
307
|
+
else:
|
|
308
|
+
buf.append(ch)
|
|
309
|
+
cells.append("".join(buf))
|
|
310
|
+
return cells
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def normalize_text(text: str, context: ParseContext | None = None) -> str:
|
|
314
|
+
context = context or ParseContext()
|
|
315
|
+
text = strip_comments(text)
|
|
316
|
+
text = text.replace("\u2010", "-")
|
|
317
|
+
text = re.sub(r"~|\\[,;:! ]", " ", text)
|
|
318
|
+
text = re.sub(r"\\(?:noindent|centering|small|normalsize|footnotesize|scriptsize)\b", "", text)
|
|
319
|
+
text = text.replace(r"\-", "")
|
|
320
|
+
text = re.sub(r"\\(?:cprotect|protect)\b", "", text)
|
|
321
|
+
text = replace_citations(text)
|
|
322
|
+
text = re.sub(r"\\ref\{([^{}]+)\}", lambda m: context.labels.get(m.group(1), f"reference {m.group(1)}"), text)
|
|
323
|
+
text = re.sub(r"\\label\{[^{}]*\}", "", text)
|
|
324
|
+
text = re.sub(r"\\href\{([^{}]*)\}\{([^{}]*)\}", lambda m: normalize_text(m.group(2), context), text)
|
|
325
|
+
text = re.sub(r"\\url\{([^{}]*)\}|\\nolinkurl\{([^{}]*)\}", lambda m: m.group(1) or m.group(2), text)
|
|
326
|
+
text = replace_math(text)
|
|
327
|
+
text = apply_inline_formatting(text, context)
|
|
328
|
+
text = unwrap_text_commands(text, context)
|
|
329
|
+
for name, value in {**GREEK, **SYMBOLS}.items():
|
|
330
|
+
text = text.replace(f"\\{name}", value)
|
|
331
|
+
text = text.replace("\\%", "%").replace("\\&", "&").replace("\\$", "$").replace("\\#", "#")
|
|
332
|
+
text = text.replace("\\_", "_").replace("\\{", "{").replace("\\}", "}")
|
|
333
|
+
text = text.replace("---", "—").replace("--", "–")
|
|
334
|
+
text = text.replace("``", "“").replace("''", "”")
|
|
335
|
+
text = re.sub(r"\\[a-zA-Z@]+\*?(?:\[[^]]*\])?", "", text)
|
|
336
|
+
text = text.replace("{", "").replace("}", "")
|
|
337
|
+
text = re.sub(r"[ \t\r\f\v]+", " ", text)
|
|
338
|
+
text = re.sub(r" *\n *", "\n", text)
|
|
339
|
+
return text.strip()
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def apply_inline_formatting(text: str, context: ParseContext) -> str:
|
|
343
|
+
replacements = [
|
|
344
|
+
(r"\\textbf\*?\{([^{}]*)\}", "**{}**"),
|
|
345
|
+
(r"\\(?:textit|emph|mkbibemph)\*?\{([^{}]*)\}", "*{}*"),
|
|
346
|
+
(r"\\(?:texttt|verb)\*?(?:\|([^|]*)\||\{([^{}]*)\})", "`{}`"),
|
|
347
|
+
]
|
|
348
|
+
previous = None
|
|
349
|
+
while previous != text:
|
|
350
|
+
previous = text
|
|
351
|
+
for pattern, template in replacements:
|
|
352
|
+
text = re.sub(
|
|
353
|
+
pattern,
|
|
354
|
+
lambda m, template=template: template.format(
|
|
355
|
+
normalize_text((m.group(1) or (m.group(2) if len(m.groups()) > 1 else "")), context)
|
|
356
|
+
),
|
|
357
|
+
text,
|
|
358
|
+
)
|
|
359
|
+
return text
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def replace_citations(text: str) -> str:
|
|
363
|
+
cite_pattern = re.compile(r"\\(?:parencite|cite|textcite|autocite)(?:\[[^]]*\]){0,2}\{([^{}]+)\}")
|
|
364
|
+
return cite_pattern.sub(lambda m: format_citation(m.group(1)), text)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def format_citation(keys: str) -> str:
|
|
368
|
+
citations = []
|
|
369
|
+
for key in keys.split(","):
|
|
370
|
+
key = key.strip()
|
|
371
|
+
year_match = re.search(r"((?:19|20)\d{2})", key)
|
|
372
|
+
year = year_match.group(1) if year_match else "n.d."
|
|
373
|
+
prefix = key[: year_match.start()] if year_match else key
|
|
374
|
+
surname_bits = re.match(r"[a-z\-]+", prefix)
|
|
375
|
+
surname = surname_bits.group(0) if surname_bits else prefix
|
|
376
|
+
surname = "-".join(bit.capitalize() for bit in surname.split("-") if bit)
|
|
377
|
+
if surname == "Muller":
|
|
378
|
+
surname = "Müller"
|
|
379
|
+
citations.append(f"{surname} {year}")
|
|
380
|
+
return "(" + "; ".join(citations) + ")"
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def replace_math(text: str) -> str:
|
|
384
|
+
text = re.sub(r"\$\$(.*?)\$\$", lambda m: normalize_math(m.group(1)), text, flags=re.S)
|
|
385
|
+
text = re.sub(r"\$(.*?)\$", lambda m: normalize_math(m.group(1)), text, flags=re.S)
|
|
386
|
+
text = re.sub(r"\\\[(.*?)\\\]", lambda m: normalize_math(m.group(1)), text, flags=re.S)
|
|
387
|
+
text = re.sub(r"\\\((.*?)\\\)", lambda m: normalize_math(m.group(1)), text, flags=re.S)
|
|
388
|
+
return text
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def normalize_math(math: str) -> str:
|
|
392
|
+
math = strip_comments(math).strip()
|
|
393
|
+
math = replace_latex_fractions(math)
|
|
394
|
+
math = re.sub(r"\\mathrm\{([^{}]*)\}", r"\1", math)
|
|
395
|
+
math = re.sub(r"\\mathbb\{([^{}]*)\}", r"\1", math)
|
|
396
|
+
math = re.sub(r"\\mathcal\{([^{}]*)\}", r"\1", math)
|
|
397
|
+
math = re.sub(r"\\text\{([^{}]*)\}", r"\1", math)
|
|
398
|
+
math = re.sub(r"\\boldsymbol\{([^{}]*)\}", r"\1", math)
|
|
399
|
+
math = re.sub(r"\\bar\{([^{}])\}", lambda m: m.group(1) + "\u0304", math)
|
|
400
|
+
math = replace_math_commands(math)
|
|
401
|
+
math = re.sub(r"\\frac\{([^{}]+)\}\{([^{}]+)\}", lambda m: f"{normalize_math(m.group(1))}/{normalize_math(m.group(2))}", math)
|
|
402
|
+
math = re.sub(r"\\sqrt\{([^{}]+)\}", lambda m: f"√({normalize_math(m.group(1))})", math)
|
|
403
|
+
math = re.sub(r"\^\{([^{}]+)\}", lambda m: to_superscript(normalize_math(m.group(1))), math)
|
|
404
|
+
math = re.sub(r"_\{([^{}]+)\}", lambda m: to_subscript(normalize_math(m.group(1))), math)
|
|
405
|
+
math = re.sub(r"\^\\circ\b", "°", math)
|
|
406
|
+
math = re.sub(r"\^([A-Za-z0-9+\-=()])", lambda m: to_superscript(m.group(1)), math)
|
|
407
|
+
math = re.sub(r"_([A-Za-z0-9+\-=()])", lambda m: to_subscript(m.group(1)), math)
|
|
408
|
+
math = re.sub(r"\\begin\{cases\}(.*?)\\end\{cases\}", lambda m: normalize_cases(m.group(1)), math, flags=re.S)
|
|
409
|
+
math = re.sub(r"\\begin\{bmatrix\}(.*?)\\end\{bmatrix\}", lambda m: "[" + normalize_math(m.group(1)) + "]", math, flags=re.S)
|
|
410
|
+
math = math.replace(r"\left", "").replace(r"\right", "")
|
|
411
|
+
math = math.replace(r"\!", "")
|
|
412
|
+
math = re.sub(r"\\(?:,|;|:| )", " ", math)
|
|
413
|
+
math = replace_math_commands(math)
|
|
414
|
+
math = re.sub(r"\\binom\{([^{}]+)\}\{([^{}]+)\}", r"C(\1,\2)", math)
|
|
415
|
+
math = math.replace(r"\mbox{-}", "-")
|
|
416
|
+
math = math.replace(r"\prime", "′")
|
|
417
|
+
math = math.replace(r"\{", "{").replace(r"\}", "}")
|
|
418
|
+
math = math.replace("\\", "")
|
|
419
|
+
math = math.replace("{", "").replace("}", "")
|
|
420
|
+
math = math.replace("---", "—").replace("--", "–")
|
|
421
|
+
math = re.sub(r"\s*&\s*", " ", math)
|
|
422
|
+
math = re.sub(r"\s+", " ", math)
|
|
423
|
+
return math.strip()
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def replace_math_commands(math: str) -> str:
|
|
427
|
+
replacements = {**GREEK, **SYMBOLS}
|
|
428
|
+
replacements.update({"vee": "∨", "dots": "…", "ldots": "…", "vdots": "⋮", "mapsto": "↦", "mid": "|"})
|
|
429
|
+
for name, value in sorted(replacements.items(), key=lambda item: -len(item[0])):
|
|
430
|
+
math = re.sub(rf"\\{re.escape(name)}(?=[^A-Za-z]|$)", value, math)
|
|
431
|
+
return math
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
def normalize_cases(text: str) -> str:
|
|
435
|
+
rows = []
|
|
436
|
+
for row in split_latex_rows(text):
|
|
437
|
+
cells = [normalize_math(cell) for cell in split_latex_cells(row)]
|
|
438
|
+
rows.append(" if ".join(cell for cell in cells if cell))
|
|
439
|
+
return "{ " + "; ".join(rows) + " }"
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def replace_latex_fractions(text: str) -> str:
|
|
443
|
+
command = r"\frac"
|
|
444
|
+
start = text.find(command)
|
|
445
|
+
while start != -1:
|
|
446
|
+
first_start = start + len(command)
|
|
447
|
+
first = read_braced_at(text, first_start)
|
|
448
|
+
if not first:
|
|
449
|
+
start = text.find(command, start + len(command))
|
|
450
|
+
continue
|
|
451
|
+
numerator, first_end = first
|
|
452
|
+
second = read_braced_at(text, first_end)
|
|
453
|
+
if not second:
|
|
454
|
+
start = text.find(command, start + len(command))
|
|
455
|
+
continue
|
|
456
|
+
denominator, second_end = second
|
|
457
|
+
replacement = f"({normalize_math(numerator)})/({normalize_math(denominator)})"
|
|
458
|
+
text = text[:start] + replacement + text[second_end:]
|
|
459
|
+
start = text.find(command, start + len(replacement))
|
|
460
|
+
return text
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def read_braced_at(text: str, index: int) -> tuple[str, int] | None:
|
|
464
|
+
while index < len(text) and text[index].isspace():
|
|
465
|
+
index += 1
|
|
466
|
+
if index >= len(text) or text[index] != "{":
|
|
467
|
+
return None
|
|
468
|
+
depth = 0
|
|
469
|
+
for pos in range(index, len(text)):
|
|
470
|
+
if text[pos] == "{" and (pos == 0 or text[pos - 1] != "\\"):
|
|
471
|
+
depth += 1
|
|
472
|
+
elif text[pos] == "}" and (pos == 0 or text[pos - 1] != "\\"):
|
|
473
|
+
depth -= 1
|
|
474
|
+
if depth == 0:
|
|
475
|
+
return text[index + 1 : pos], pos + 1
|
|
476
|
+
return None
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def to_superscript(value: str) -> str:
|
|
480
|
+
return value.translate(SUPERSCRIPT)
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def to_subscript(value: str) -> str:
|
|
484
|
+
return "".join(SUBSCRIPT_MAP.get(char, char) for char in value)
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def unwrap_text_commands(text: str, context: ParseContext) -> str:
|
|
488
|
+
command_pattern = re.compile(
|
|
489
|
+
r"\\(?:mathbf|mathit|mathbb|operatorname|underline)\*?(?:\|([^|]*)\||\{([^{}]*)\})"
|
|
490
|
+
)
|
|
491
|
+
previous = None
|
|
492
|
+
while previous != text:
|
|
493
|
+
previous = text
|
|
494
|
+
text = command_pattern.sub(lambda m: normalize_text(m.group(1) or m.group(2) or "", context), text)
|
|
495
|
+
return text
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def strip_comments(text: str) -> str:
|
|
499
|
+
return re.sub(r"(?<!\\)%.*", "", text)
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def remove_bibliography(text: str) -> str:
|
|
503
|
+
text = re.sub(r"\\printbibliography\b(?:\[[^]]*\])?", "", text)
|
|
504
|
+
text = re.sub(r"\\addbibresource\{[^{}]*\}", "", text)
|
|
505
|
+
text = re.sub(r"\\begin\{thebibliography\}.*?\\end\{thebibliography\}", "", text, flags=re.S)
|
|
506
|
+
return text
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def remove_wrappers(text: str) -> str:
|
|
510
|
+
text = re.sub(r"\\begingroup|\\endgroup", "\n", text)
|
|
511
|
+
text = re.sub(r"\\(?:emergencystretch|hfuzz)\s*=\s*[^ \n]+", "", text)
|
|
512
|
+
return text
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def collect_label_map(text: str) -> dict[str, str]:
|
|
516
|
+
labels: dict[str, str] = {}
|
|
517
|
+
counts = {"figure": 0, "table": 0}
|
|
518
|
+
pattern = re.compile(r"\\begin\{(figure|table)\}(?:\[[^]]*\])?(.*?)\\end\{\1\}", flags=re.S)
|
|
519
|
+
for match in pattern.finditer(text):
|
|
520
|
+
kind = match.group(1)
|
|
521
|
+
counts[kind] += 1
|
|
522
|
+
label = extract_braced_command(match.group(2), "label")
|
|
523
|
+
if label:
|
|
524
|
+
labels[label] = str(counts[kind])
|
|
525
|
+
return labels
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def find_environment_end(text: str, env: str, start: int) -> re.Match[str] | None:
|
|
529
|
+
return re.compile(rf"\\end\{{{re.escape(env)}\}}", flags=re.S).search(text, start)
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def extract_environment(text: str, env: str) -> str | None:
|
|
533
|
+
match = re.search(rf"\\begin\{{{re.escape(env)}\}}(.*?)\\end\{{{re.escape(env)}\}}", text, flags=re.S)
|
|
534
|
+
return match.group(1) if match else None
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def extract_braced_command(text: str, command: str) -> str | None:
|
|
538
|
+
start = re.search(rf"\\{re.escape(command)}(?:\[[^]]*\])?\{{", text)
|
|
539
|
+
if not start:
|
|
540
|
+
return None
|
|
541
|
+
brace_start = start.end() - 1
|
|
542
|
+
depth = 0
|
|
543
|
+
for idx in range(brace_start, len(text)):
|
|
544
|
+
if text[idx] == "{" and (idx == 0 or text[idx - 1] != "\\"):
|
|
545
|
+
depth += 1
|
|
546
|
+
elif text[idx] == "}" and (idx == 0 or text[idx - 1] != "\\"):
|
|
547
|
+
depth -= 1
|
|
548
|
+
if depth == 0:
|
|
549
|
+
return text[brace_start + 1 : idx]
|
|
550
|
+
return None
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def write_docx(blocks: Iterable[Block], metadata: Metadata, path: Path) -> None:
|
|
554
|
+
document = Document()
|
|
555
|
+
styles = document.styles
|
|
556
|
+
styles["Normal"].font.name = "Times New Roman"
|
|
557
|
+
styles["Normal"].font.size = Pt(11)
|
|
558
|
+
if metadata.title:
|
|
559
|
+
paragraph = document.add_heading(metadata.title, level=0)
|
|
560
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
561
|
+
if metadata.authors:
|
|
562
|
+
paragraph = document.add_paragraph(", ".join(metadata.authors))
|
|
563
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
564
|
+
if metadata.date:
|
|
565
|
+
paragraph = document.add_paragraph(metadata.date)
|
|
566
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
567
|
+
|
|
568
|
+
for block in blocks:
|
|
569
|
+
if block.kind == "heading":
|
|
570
|
+
document.add_heading(block.text, level=min(max(block.level, 1), 4))
|
|
571
|
+
elif block.kind == "paragraph":
|
|
572
|
+
add_formatted_paragraph(document, block.text)
|
|
573
|
+
elif block.kind == "equation":
|
|
574
|
+
paragraph = document.add_paragraph(block.text)
|
|
575
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
576
|
+
elif block.kind == "figure":
|
|
577
|
+
paragraph = document.add_paragraph(block.text)
|
|
578
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
579
|
+
if block.caption:
|
|
580
|
+
add_formatted_paragraph(document, f"Caption: {block.caption}")
|
|
581
|
+
elif block.kind == "table":
|
|
582
|
+
add_docx_table(document, block)
|
|
583
|
+
document.save(path)
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def add_docx_table(document: Document, block: Block) -> None:
|
|
587
|
+
rows = block.rows
|
|
588
|
+
if not rows:
|
|
589
|
+
if block.caption:
|
|
590
|
+
document.add_paragraph(f"Table: {block.caption}")
|
|
591
|
+
return
|
|
592
|
+
column_count = max(len(row) for row in rows)
|
|
593
|
+
table = document.add_table(rows=len(rows), cols=column_count)
|
|
594
|
+
table.style = "Table Grid"
|
|
595
|
+
for row_index, row in enumerate(rows):
|
|
596
|
+
for col_index in range(column_count):
|
|
597
|
+
table.cell(row_index, col_index).text = row[col_index] if col_index < len(row) else ""
|
|
598
|
+
if block.caption:
|
|
599
|
+
add_formatted_paragraph(document, f"Caption: {block.caption}")
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def add_formatted_paragraph(document: Document, text: str):
|
|
603
|
+
paragraph = document.add_paragraph()
|
|
604
|
+
add_formatted_runs(paragraph, text)
|
|
605
|
+
return paragraph
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def add_formatted_runs(paragraph, text: str) -> None:
|
|
609
|
+
token_pattern = re.compile(r"(\*\*[^*]+\*\*|\*[^*]+\*|`[^`]+`)")
|
|
610
|
+
pos = 0
|
|
611
|
+
for match in token_pattern.finditer(text):
|
|
612
|
+
if match.start() > pos:
|
|
613
|
+
paragraph.add_run(text[pos : match.start()])
|
|
614
|
+
token = match.group(0)
|
|
615
|
+
if token.startswith("**"):
|
|
616
|
+
run = paragraph.add_run(token[2:-2])
|
|
617
|
+
run.bold = True
|
|
618
|
+
elif token.startswith("*"):
|
|
619
|
+
run = paragraph.add_run(token[1:-1])
|
|
620
|
+
run.italic = True
|
|
621
|
+
else:
|
|
622
|
+
run = paragraph.add_run(token[1:-1])
|
|
623
|
+
run.font.name = "Courier New"
|
|
624
|
+
pos = match.end()
|
|
625
|
+
if pos < len(text):
|
|
626
|
+
paragraph.add_run(text[pos:])
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
def render_markdown(blocks: Iterable[Block], metadata: Metadata) -> str:
|
|
630
|
+
lines: list[str] = []
|
|
631
|
+
if metadata.title:
|
|
632
|
+
lines.extend([f"# {metadata.title}", ""])
|
|
633
|
+
if metadata.authors:
|
|
634
|
+
lines.extend([", ".join(metadata.authors), ""])
|
|
635
|
+
for block in blocks:
|
|
636
|
+
if block.kind == "heading":
|
|
637
|
+
lines.extend([f"{'#' * min(block.level + 1, 6)} {block.text}", ""])
|
|
638
|
+
elif block.kind == "paragraph":
|
|
639
|
+
lines.extend([block.text, ""])
|
|
640
|
+
elif block.kind == "equation":
|
|
641
|
+
lines.extend([f"```text\n{block.text}\n```", ""])
|
|
642
|
+
elif block.kind == "figure":
|
|
643
|
+
lines.extend([block.text, ""])
|
|
644
|
+
if block.caption:
|
|
645
|
+
lines.extend([f"Caption: {block.caption}", ""])
|
|
646
|
+
elif block.kind == "table":
|
|
647
|
+
lines.extend(render_markdown_table(block.rows))
|
|
648
|
+
if block.caption:
|
|
649
|
+
lines.extend(["", f"Caption: {block.caption}", ""])
|
|
650
|
+
return "\n".join(lines).rstrip() + "\n"
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
def render_markdown_table(rows: list[list[str]]) -> list[str]:
|
|
654
|
+
if not rows:
|
|
655
|
+
return []
|
|
656
|
+
width = max(len(row) for row in rows)
|
|
657
|
+
padded = [row + [""] * (width - len(row)) for row in rows]
|
|
658
|
+
lines = ["| " + " | ".join(padded[0]) + " |"]
|
|
659
|
+
lines.append("| " + " | ".join("---" for _ in range(width)) + " |")
|
|
660
|
+
for row in padded[1:]:
|
|
661
|
+
lines.append("| " + " | ".join(row) + " |")
|
|
662
|
+
return lines
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: docxwright
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A pragmatic LaTeX-to-DOCX converter for editable scientific manuscripts.
|
|
5
|
+
Author: Kenan Hanke
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: python-docx>=1.1.0
|
|
9
|
+
|
|
10
|
+
# docxwright
|
|
11
|
+
|
|
12
|
+
`docxwright` converts a LaTeX manuscript into an editable DOCX document without
|
|
13
|
+
requiring external tools such as Pandoc. It is intentionally pragmatic: it walks
|
|
14
|
+
included subfiles, keeps tables, replaces figures with placeholders and captions,
|
|
15
|
+
normalizes common LaTeX text/math to Unicode, emits author-year citation text
|
|
16
|
+
from Better-BibTeX-style keys, and omits the bibliography.
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
docxwright paper-example/main.tex -o paper-example/out/main.docx
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
For inspection, a Markdown-like intermediate representation can also be written:
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
docxwright paper-example/main.tex -o paper-example/out/main.docx --markdown paper-example/out/main.md
|
|
26
|
+
```
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
docxwright/__init__.py,sha256=NufKK4E_ozdxW64GuD8__6Ly8zvQfphYCsS4_-ixR1s,91
|
|
2
|
+
docxwright/cli.py,sha256=nZHO5hx4UBh30ndtbhGmDCMMVzfgmoz3f72gh6HgfC4,897
|
|
3
|
+
docxwright/converter.py,sha256=BmzND1V-vzH4g6w2wHujLf-wR9xIWP7pOvrrlTxxJCc,24291
|
|
4
|
+
docxwright-0.1.0.dist-info/METADATA,sha256=u-xp2nsw_zkZkLaVHXGfyOhxHXYeDA9W-OtItUYAs64,924
|
|
5
|
+
docxwright-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
6
|
+
docxwright-0.1.0.dist-info/entry_points.txt,sha256=Fi_UYBSMbljWxyZP6TyqMCcdSfo5OHPiz91ulnxgjkQ,51
|
|
7
|
+
docxwright-0.1.0.dist-info/top_level.txt,sha256=RYuEfMhvxbOKr-MTPrhKT9UPn6s3Qcekpv2E6kzMORM,11
|
|
8
|
+
docxwright-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
docxwright
|