docx2tiptap 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docx2tiptap/__init__.py +12 -0
- docx2tiptap/comments_parser.py +279 -0
- docx2tiptap/docx_exporter.py +511 -0
- docx2tiptap/docx_parser.py +624 -0
- docx2tiptap/revisions_parser.py +266 -0
- docx2tiptap/tiptap_converter.py +339 -0
- docx2tiptap-0.1.0.dist-info/METADATA +63 -0
- docx2tiptap-0.1.0.dist-info/RECORD +10 -0
- docx2tiptap-0.1.0.dist-info/WHEEL +4 -0
- docx2tiptap-0.1.0.dist-info/licenses/LICENSE +661 -0
docx2tiptap/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from .comments_parser import comments_to_dict
|
|
2
|
+
from .docx_exporter import create_docx_from_tiptap
|
|
3
|
+
from .docx_parser import elements_to_dict, parse_docx
|
|
4
|
+
from .tiptap_converter import to_tiptap
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"comments_to_dict",
|
|
8
|
+
"create_docx_from_tiptap",
|
|
9
|
+
"elements_to_dict",
|
|
10
|
+
"parse_docx",
|
|
11
|
+
"to_tiptap",
|
|
12
|
+
]
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Comments Parser - Extracts comments from Word documents.
|
|
3
|
+
|
|
4
|
+
Word stores comments in:
|
|
5
|
+
- word/comments.xml - Contains the actual comment text, author, date
|
|
6
|
+
- document.xml - Contains markers for comment ranges:
|
|
7
|
+
- <w:commentRangeStart w:id="0"/> - Start of commented text
|
|
8
|
+
- <w:commentRangeEnd w:id="0"/> - End of commented text
|
|
9
|
+
- <w:commentReference w:id="0"/> - Reference point (usually at end)
|
|
10
|
+
|
|
11
|
+
Comments can have replies, which are stored as separate comments with
|
|
12
|
+
a w:paraId attribute linking them to the parent.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import zipfile
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Optional
|
|
18
|
+
|
|
19
|
+
from lxml import etree
|
|
20
|
+
|
|
21
|
+
WORD_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
22
|
+
WORD_NS_PREFIX = "{" + WORD_NS + "}"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class Comment:
|
|
27
|
+
"""A document comment."""
|
|
28
|
+
|
|
29
|
+
id: str
|
|
30
|
+
author: str
|
|
31
|
+
date: Optional[str]
|
|
32
|
+
text: str
|
|
33
|
+
initials: Optional[str] = None
|
|
34
|
+
replies: list["Comment"] = field(default_factory=list)
|
|
35
|
+
# Range information (to be filled in when processing document)
|
|
36
|
+
range_start_para: Optional[int] = None
|
|
37
|
+
range_start_offset: Optional[int] = None
|
|
38
|
+
range_end_para: Optional[int] = None
|
|
39
|
+
range_end_offset: Optional[int] = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def extract_comments_from_docx(docx_bytes: bytes) -> dict[str, Comment]:
|
|
43
|
+
"""
|
|
44
|
+
Extract all comments from a DOCX file.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
docx_bytes: Raw bytes of the DOCX file
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
Dict mapping comment ID to Comment object
|
|
51
|
+
"""
|
|
52
|
+
from io import BytesIO
|
|
53
|
+
|
|
54
|
+
comments = {}
|
|
55
|
+
|
|
56
|
+
try:
|
|
57
|
+
with zipfile.ZipFile(BytesIO(docx_bytes)) as zf:
|
|
58
|
+
# Check if comments.xml exists
|
|
59
|
+
if "word/comments.xml" not in zf.namelist():
|
|
60
|
+
return comments
|
|
61
|
+
|
|
62
|
+
comments_xml = zf.read("word/comments.xml")
|
|
63
|
+
tree = etree.fromstring(comments_xml)
|
|
64
|
+
|
|
65
|
+
# Find all comment elements
|
|
66
|
+
for comment_elem in tree.findall(f".//{WORD_NS_PREFIX}comment"):
|
|
67
|
+
comment_id = comment_elem.get(f"{WORD_NS_PREFIX}id")
|
|
68
|
+
author = (
|
|
69
|
+
comment_elem.get(f"{WORD_NS_PREFIX}author") or "Unknown"
|
|
70
|
+
)
|
|
71
|
+
date = comment_elem.get(f"{WORD_NS_PREFIX}date")
|
|
72
|
+
initials = comment_elem.get(f"{WORD_NS_PREFIX}initials")
|
|
73
|
+
|
|
74
|
+
# Extract comment text (can span multiple paragraphs)
|
|
75
|
+
text_parts = []
|
|
76
|
+
for text_elem in comment_elem.iter(f"{WORD_NS_PREFIX}t"):
|
|
77
|
+
if text_elem.text:
|
|
78
|
+
text_parts.append(text_elem.text)
|
|
79
|
+
|
|
80
|
+
text = "".join(text_parts)
|
|
81
|
+
|
|
82
|
+
comments[comment_id] = Comment(
|
|
83
|
+
id=comment_id,
|
|
84
|
+
author=author,
|
|
85
|
+
date=date,
|
|
86
|
+
text=text,
|
|
87
|
+
initials=initials,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# Handle extended comments (replies) if commentsExtended.xml exists
|
|
91
|
+
if "word/commentsExtended.xml" in zf.namelist():
|
|
92
|
+
_process_comment_replies(zf, comments)
|
|
93
|
+
|
|
94
|
+
except (zipfile.BadZipFile, KeyError, etree.XMLSyntaxError):
|
|
95
|
+
# Return empty dict if we can't parse comments
|
|
96
|
+
pass
|
|
97
|
+
|
|
98
|
+
return comments
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _process_comment_replies(zf: zipfile.ZipFile, comments: dict[str, Comment]):
|
|
102
|
+
"""
|
|
103
|
+
Process commentsExtended.xml to link reply comments to their parents.
|
|
104
|
+
|
|
105
|
+
In Word, replies are stored as separate comments with a paraIdParent
|
|
106
|
+
attribute linking them to the parent comment.
|
|
107
|
+
"""
|
|
108
|
+
try:
|
|
109
|
+
extended_xml = zf.read("word/commentsExtended.xml")
|
|
110
|
+
tree = etree.fromstring(extended_xml)
|
|
111
|
+
|
|
112
|
+
# Namespace for commentsExtended
|
|
113
|
+
w15_ns = "http://schemas.microsoft.com/office/word/2012/wordml"
|
|
114
|
+
|
|
115
|
+
# Build parent-child relationships
|
|
116
|
+
for comment_ex in tree.findall(f".//{{{w15_ns}}}commentEx"):
|
|
117
|
+
para_id = comment_ex.get(f"{{{w15_ns}}}paraId")
|
|
118
|
+
parent_para_id = comment_ex.get(f"{{{w15_ns}}}paraIdParent")
|
|
119
|
+
|
|
120
|
+
if parent_para_id:
|
|
121
|
+
# This is a reply - find the comment and its parent
|
|
122
|
+
# Note: This requires mapping paraId to comment ID
|
|
123
|
+
# For now, we'll skip this complexity
|
|
124
|
+
pass
|
|
125
|
+
|
|
126
|
+
except (KeyError, etree.XMLSyntaxError):
|
|
127
|
+
pass
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def find_comment_ranges_in_paragraph(
|
|
131
|
+
para_element, para_index: int
|
|
132
|
+
) -> dict[str, dict]:
|
|
133
|
+
"""
|
|
134
|
+
Find comment range markers in a paragraph.
|
|
135
|
+
|
|
136
|
+
Returns dict mapping comment ID to range info:
|
|
137
|
+
{
|
|
138
|
+
'0': {'start_offset': 5, 'end_offset': 20},
|
|
139
|
+
'1': {'start_offset': 30, 'end_offset': 45}
|
|
140
|
+
}
|
|
141
|
+
"""
|
|
142
|
+
ranges = {}
|
|
143
|
+
current_offset = 0
|
|
144
|
+
active_comments = {} # comment_id -> start_offset
|
|
145
|
+
|
|
146
|
+
def process_element(element):
|
|
147
|
+
nonlocal current_offset
|
|
148
|
+
|
|
149
|
+
tag = element.tag
|
|
150
|
+
|
|
151
|
+
# Comment range start
|
|
152
|
+
if tag == f"{WORD_NS_PREFIX}commentRangeStart":
|
|
153
|
+
comment_id = element.get(f"{WORD_NS_PREFIX}id")
|
|
154
|
+
if comment_id:
|
|
155
|
+
active_comments[comment_id] = current_offset
|
|
156
|
+
|
|
157
|
+
# Comment range end
|
|
158
|
+
elif tag == f"{WORD_NS_PREFIX}commentRangeEnd":
|
|
159
|
+
comment_id = element.get(f"{WORD_NS_PREFIX}id")
|
|
160
|
+
if comment_id and comment_id in active_comments:
|
|
161
|
+
ranges[comment_id] = {
|
|
162
|
+
"para_index": para_index,
|
|
163
|
+
"start_offset": active_comments[comment_id],
|
|
164
|
+
"end_offset": current_offset,
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
# Text - track offset
|
|
168
|
+
elif tag == f"{WORD_NS_PREFIX}t":
|
|
169
|
+
text = element.text or ""
|
|
170
|
+
current_offset += len(text)
|
|
171
|
+
|
|
172
|
+
elif tag == f"{WORD_NS_PREFIX}delText":
|
|
173
|
+
text = element.text or ""
|
|
174
|
+
current_offset += len(text)
|
|
175
|
+
|
|
176
|
+
# Recurse
|
|
177
|
+
for child in element:
|
|
178
|
+
process_element(child)
|
|
179
|
+
|
|
180
|
+
for child in para_element:
|
|
181
|
+
process_element(child)
|
|
182
|
+
|
|
183
|
+
return ranges
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def get_text_with_comments(para_element, para_index: int) -> list[dict]:
|
|
187
|
+
"""
|
|
188
|
+
Extract text runs from a paragraph, tracking comment ranges.
|
|
189
|
+
|
|
190
|
+
Returns a list of text segments with their comment status:
|
|
191
|
+
[
|
|
192
|
+
{'text': 'normal text', 'comments': []},
|
|
193
|
+
{'text': 'commented text', 'comments': ['0', '1']}, # Can have multiple overlapping
|
|
194
|
+
]
|
|
195
|
+
"""
|
|
196
|
+
segments = []
|
|
197
|
+
active_comment_ids = set()
|
|
198
|
+
|
|
199
|
+
def process_element(element):
|
|
200
|
+
tag = element.tag
|
|
201
|
+
|
|
202
|
+
# Comment range start
|
|
203
|
+
if tag == f"{WORD_NS_PREFIX}commentRangeStart":
|
|
204
|
+
comment_id = element.get(f"{WORD_NS_PREFIX}id")
|
|
205
|
+
if comment_id:
|
|
206
|
+
active_comment_ids.add(comment_id)
|
|
207
|
+
return
|
|
208
|
+
|
|
209
|
+
# Comment range end
|
|
210
|
+
if tag == f"{WORD_NS_PREFIX}commentRangeEnd":
|
|
211
|
+
comment_id = element.get(f"{WORD_NS_PREFIX}id")
|
|
212
|
+
if comment_id:
|
|
213
|
+
active_comment_ids.discard(comment_id)
|
|
214
|
+
return
|
|
215
|
+
|
|
216
|
+
# Text element
|
|
217
|
+
if tag == f"{WORD_NS_PREFIX}t":
|
|
218
|
+
text = element.text or ""
|
|
219
|
+
if text:
|
|
220
|
+
segments.append(
|
|
221
|
+
{
|
|
222
|
+
"text": text,
|
|
223
|
+
"comments": (
|
|
224
|
+
list(active_comment_ids)
|
|
225
|
+
if active_comment_ids
|
|
226
|
+
else []
|
|
227
|
+
),
|
|
228
|
+
}
|
|
229
|
+
)
|
|
230
|
+
return
|
|
231
|
+
|
|
232
|
+
# Deleted text
|
|
233
|
+
if tag == f"{WORD_NS_PREFIX}delText":
|
|
234
|
+
text = element.text or ""
|
|
235
|
+
if text:
|
|
236
|
+
segments.append(
|
|
237
|
+
{
|
|
238
|
+
"text": text,
|
|
239
|
+
"comments": (
|
|
240
|
+
list(active_comment_ids)
|
|
241
|
+
if active_comment_ids
|
|
242
|
+
else []
|
|
243
|
+
),
|
|
244
|
+
}
|
|
245
|
+
)
|
|
246
|
+
return
|
|
247
|
+
|
|
248
|
+
# Run element
|
|
249
|
+
if tag == f"{WORD_NS_PREFIX}r":
|
|
250
|
+
for child in element:
|
|
251
|
+
process_element(child)
|
|
252
|
+
return
|
|
253
|
+
|
|
254
|
+
# Recurse into other elements
|
|
255
|
+
for child in element:
|
|
256
|
+
process_element(child)
|
|
257
|
+
|
|
258
|
+
for child in para_element:
|
|
259
|
+
process_element(child)
|
|
260
|
+
|
|
261
|
+
return segments
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def comments_to_dict(comments: dict[str, Comment]) -> list[dict]:
|
|
265
|
+
"""Convert Comment objects to JSON-serializable dictionaries."""
|
|
266
|
+
return [
|
|
267
|
+
{
|
|
268
|
+
"id": c.id,
|
|
269
|
+
"author": c.author,
|
|
270
|
+
"date": c.date,
|
|
271
|
+
"text": c.text,
|
|
272
|
+
"initials": c.initials,
|
|
273
|
+
"replies": [
|
|
274
|
+
{"id": r.id, "author": r.author, "date": r.date, "text": r.text}
|
|
275
|
+
for r in c.replies
|
|
276
|
+
],
|
|
277
|
+
}
|
|
278
|
+
for c in comments.values()
|
|
279
|
+
]
|