docx2tiptap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ from .comments_parser import comments_to_dict
2
+ from .docx_exporter import create_docx_from_tiptap
3
+ from .docx_parser import elements_to_dict, parse_docx
4
+ from .tiptap_converter import to_tiptap
5
+
6
+ __all__ = [
7
+ "comments_to_dict",
8
+ "create_docx_from_tiptap",
9
+ "elements_to_dict",
10
+ "parse_docx",
11
+ "to_tiptap",
12
+ ]
@@ -0,0 +1,279 @@
1
+ """
2
+ Comments Parser - Extracts comments from Word documents.
3
+
4
+ Word stores comments in:
5
+ - word/comments.xml - Contains the actual comment text, author, date
6
+ - document.xml - Contains markers for comment ranges:
7
+ - <w:commentRangeStart w:id="0"/> - Start of commented text
8
+ - <w:commentRangeEnd w:id="0"/> - End of commented text
9
+ - <w:commentReference w:id="0"/> - Reference point (usually at end)
10
+
11
+ Comments can have replies, which are stored as separate comments with
12
+ a w:paraId attribute linking them to the parent.
13
+ """
14
+
15
+ import zipfile
16
+ from dataclasses import dataclass, field
17
+ from typing import Optional
18
+
19
+ from lxml import etree
20
+
21
+ WORD_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
22
+ WORD_NS_PREFIX = "{" + WORD_NS + "}"
23
+
24
+
25
+ @dataclass
26
+ class Comment:
27
+ """A document comment."""
28
+
29
+ id: str
30
+ author: str
31
+ date: Optional[str]
32
+ text: str
33
+ initials: Optional[str] = None
34
+ replies: list["Comment"] = field(default_factory=list)
35
+ # Range information (to be filled in when processing document)
36
+ range_start_para: Optional[int] = None
37
+ range_start_offset: Optional[int] = None
38
+ range_end_para: Optional[int] = None
39
+ range_end_offset: Optional[int] = None
40
+
41
+
42
+ def extract_comments_from_docx(docx_bytes: bytes) -> dict[str, Comment]:
43
+ """
44
+ Extract all comments from a DOCX file.
45
+
46
+ Args:
47
+ docx_bytes: Raw bytes of the DOCX file
48
+
49
+ Returns:
50
+ Dict mapping comment ID to Comment object
51
+ """
52
+ from io import BytesIO
53
+
54
+ comments = {}
55
+
56
+ try:
57
+ with zipfile.ZipFile(BytesIO(docx_bytes)) as zf:
58
+ # Check if comments.xml exists
59
+ if "word/comments.xml" not in zf.namelist():
60
+ return comments
61
+
62
+ comments_xml = zf.read("word/comments.xml")
63
+ tree = etree.fromstring(comments_xml)
64
+
65
+ # Find all comment elements
66
+ for comment_elem in tree.findall(f".//{WORD_NS_PREFIX}comment"):
67
+ comment_id = comment_elem.get(f"{WORD_NS_PREFIX}id")
68
+ author = (
69
+ comment_elem.get(f"{WORD_NS_PREFIX}author") or "Unknown"
70
+ )
71
+ date = comment_elem.get(f"{WORD_NS_PREFIX}date")
72
+ initials = comment_elem.get(f"{WORD_NS_PREFIX}initials")
73
+
74
+ # Extract comment text (can span multiple paragraphs)
75
+ text_parts = []
76
+ for text_elem in comment_elem.iter(f"{WORD_NS_PREFIX}t"):
77
+ if text_elem.text:
78
+ text_parts.append(text_elem.text)
79
+
80
+ text = "".join(text_parts)
81
+
82
+ comments[comment_id] = Comment(
83
+ id=comment_id,
84
+ author=author,
85
+ date=date,
86
+ text=text,
87
+ initials=initials,
88
+ )
89
+
90
+ # Handle extended comments (replies) if commentsExtended.xml exists
91
+ if "word/commentsExtended.xml" in zf.namelist():
92
+ _process_comment_replies(zf, comments)
93
+
94
+ except (zipfile.BadZipFile, KeyError, etree.XMLSyntaxError):
95
+ # Return empty dict if we can't parse comments
96
+ pass
97
+
98
+ return comments
99
+
100
+
101
+ def _process_comment_replies(zf: zipfile.ZipFile, comments: dict[str, Comment]):
102
+ """
103
+ Process commentsExtended.xml to link reply comments to their parents.
104
+
105
+ In Word, replies are stored as separate comments with a paraIdParent
106
+ attribute linking them to the parent comment.
107
+ """
108
+ try:
109
+ extended_xml = zf.read("word/commentsExtended.xml")
110
+ tree = etree.fromstring(extended_xml)
111
+
112
+ # Namespace for commentsExtended
113
+ w15_ns = "http://schemas.microsoft.com/office/word/2012/wordml"
114
+
115
+ # Build parent-child relationships
116
+ for comment_ex in tree.findall(f".//{{{w15_ns}}}commentEx"):
117
+ para_id = comment_ex.get(f"{{{w15_ns}}}paraId")
118
+ parent_para_id = comment_ex.get(f"{{{w15_ns}}}paraIdParent")
119
+
120
+ if parent_para_id:
121
+ # This is a reply - find the comment and its parent
122
+ # Note: This requires mapping paraId to comment ID
123
+ # For now, we'll skip this complexity
124
+ pass
125
+
126
+ except (KeyError, etree.XMLSyntaxError):
127
+ pass
128
+
129
+
130
+ def find_comment_ranges_in_paragraph(
131
+ para_element, para_index: int
132
+ ) -> dict[str, dict]:
133
+ """
134
+ Find comment range markers in a paragraph.
135
+
136
+ Returns dict mapping comment ID to range info:
137
+ {
138
+ '0': {'start_offset': 5, 'end_offset': 20},
139
+ '1': {'start_offset': 30, 'end_offset': 45}
140
+ }
141
+ """
142
+ ranges = {}
143
+ current_offset = 0
144
+ active_comments = {} # comment_id -> start_offset
145
+
146
+ def process_element(element):
147
+ nonlocal current_offset
148
+
149
+ tag = element.tag
150
+
151
+ # Comment range start
152
+ if tag == f"{WORD_NS_PREFIX}commentRangeStart":
153
+ comment_id = element.get(f"{WORD_NS_PREFIX}id")
154
+ if comment_id:
155
+ active_comments[comment_id] = current_offset
156
+
157
+ # Comment range end
158
+ elif tag == f"{WORD_NS_PREFIX}commentRangeEnd":
159
+ comment_id = element.get(f"{WORD_NS_PREFIX}id")
160
+ if comment_id and comment_id in active_comments:
161
+ ranges[comment_id] = {
162
+ "para_index": para_index,
163
+ "start_offset": active_comments[comment_id],
164
+ "end_offset": current_offset,
165
+ }
166
+
167
+ # Text - track offset
168
+ elif tag == f"{WORD_NS_PREFIX}t":
169
+ text = element.text or ""
170
+ current_offset += len(text)
171
+
172
+ elif tag == f"{WORD_NS_PREFIX}delText":
173
+ text = element.text or ""
174
+ current_offset += len(text)
175
+
176
+ # Recurse
177
+ for child in element:
178
+ process_element(child)
179
+
180
+ for child in para_element:
181
+ process_element(child)
182
+
183
+ return ranges
184
+
185
+
186
+ def get_text_with_comments(para_element, para_index: int) -> list[dict]:
187
+ """
188
+ Extract text runs from a paragraph, tracking comment ranges.
189
+
190
+ Returns a list of text segments with their comment status:
191
+ [
192
+ {'text': 'normal text', 'comments': []},
193
+ {'text': 'commented text', 'comments': ['0', '1']}, # Can have multiple overlapping
194
+ ]
195
+ """
196
+ segments = []
197
+ active_comment_ids = set()
198
+
199
+ def process_element(element):
200
+ tag = element.tag
201
+
202
+ # Comment range start
203
+ if tag == f"{WORD_NS_PREFIX}commentRangeStart":
204
+ comment_id = element.get(f"{WORD_NS_PREFIX}id")
205
+ if comment_id:
206
+ active_comment_ids.add(comment_id)
207
+ return
208
+
209
+ # Comment range end
210
+ if tag == f"{WORD_NS_PREFIX}commentRangeEnd":
211
+ comment_id = element.get(f"{WORD_NS_PREFIX}id")
212
+ if comment_id:
213
+ active_comment_ids.discard(comment_id)
214
+ return
215
+
216
+ # Text element
217
+ if tag == f"{WORD_NS_PREFIX}t":
218
+ text = element.text or ""
219
+ if text:
220
+ segments.append(
221
+ {
222
+ "text": text,
223
+ "comments": (
224
+ list(active_comment_ids)
225
+ if active_comment_ids
226
+ else []
227
+ ),
228
+ }
229
+ )
230
+ return
231
+
232
+ # Deleted text
233
+ if tag == f"{WORD_NS_PREFIX}delText":
234
+ text = element.text or ""
235
+ if text:
236
+ segments.append(
237
+ {
238
+ "text": text,
239
+ "comments": (
240
+ list(active_comment_ids)
241
+ if active_comment_ids
242
+ else []
243
+ ),
244
+ }
245
+ )
246
+ return
247
+
248
+ # Run element
249
+ if tag == f"{WORD_NS_PREFIX}r":
250
+ for child in element:
251
+ process_element(child)
252
+ return
253
+
254
+ # Recurse into other elements
255
+ for child in element:
256
+ process_element(child)
257
+
258
+ for child in para_element:
259
+ process_element(child)
260
+
261
+ return segments
262
+
263
+
264
+ def comments_to_dict(comments: dict[str, Comment]) -> list[dict]:
265
+ """Convert Comment objects to JSON-serializable dictionaries."""
266
+ return [
267
+ {
268
+ "id": c.id,
269
+ "author": c.author,
270
+ "date": c.date,
271
+ "text": c.text,
272
+ "initials": c.initials,
273
+ "replies": [
274
+ {"id": r.id, "author": r.author, "date": r.date, "text": r.text}
275
+ for r in c.replies
276
+ ],
277
+ }
278
+ for c in comments.values()
279
+ ]