document-adapter 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. document_adapter-0.4.0/NOTICE +29 -0
  2. {document_adapter-0.3.0 → document_adapter-0.4.0}/PKG-INFO +13 -5
  3. {document_adapter-0.3.0 → document_adapter-0.4.0}/README.md +7 -1
  4. document_adapter-0.4.0/document_adapter/hwpx_adapter.py +348 -0
  5. document_adapter-0.4.0/document_adapter/hwpx_core/__init__.py +67 -0
  6. document_adapter-0.4.0/document_adapter/hwpx_core/constants.py +27 -0
  7. document_adapter-0.4.0/document_adapter/hwpx_core/grid.py +145 -0
  8. document_adapter-0.4.0/document_adapter/hwpx_core/package.py +146 -0
  9. document_adapter-0.4.0/document_adapter/hwpx_core/paragraph.py +106 -0
  10. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/PKG-INFO +13 -5
  11. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/SOURCES.txt +6 -0
  12. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/requires.txt +3 -2
  13. {document_adapter-0.3.0 → document_adapter-0.4.0}/pyproject.toml +7 -5
  14. document_adapter-0.3.0/document_adapter/hwpx_adapter.py +0 -389
  15. {document_adapter-0.3.0 → document_adapter-0.4.0}/LICENSE +0 -0
  16. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/__init__.py +0 -0
  17. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/base.py +0 -0
  18. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/docx_adapter.py +0 -0
  19. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/mcp_server.py +0 -0
  20. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/pptx_adapter.py +0 -0
  21. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/tools.py +0 -0
  22. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/dependency_links.txt +0 -0
  23. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/entry_points.txt +0 -0
  24. {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/top_level.txt +0 -0
  25. {document_adapter-0.3.0 → document_adapter-0.4.0}/setup.cfg +0 -0
  26. {document_adapter-0.3.0 → document_adapter-0.4.0}/tests/test_smoke.py +0 -0
@@ -0,0 +1,29 @@
1
+ document-adapter
2
+ Copyright 2026 Son Seongjun / PlateerLab
3
+
4
+ This product includes software developed at PlateerLab
5
+ (https://github.com/PlateerLab).
6
+
7
+ ===============================================================================
8
+ Third-party code attributions
9
+ ===============================================================================
10
+
11
+ HWPX table grid parsing logic
12
+ -----------------------------
13
+
14
+ Portions of the HWPX table grid construction in
15
+ `document_adapter/hwpx_core/grid.py` (cell position / cell span parsing, and
16
+ the skip-map approach for merged non-anchor cells) are adapted from:
17
+
18
+ PlateerLab/xgen-doc2chunk
19
+ https://github.com/PlateerLab/xgen-doc2chunk
20
+ Licensed under the Apache License, Version 2.0
21
+
22
+ The adapted logic is limited to read-side grid construction. All edit-side
23
+ functionality (cell writing, row insertion, ZIP round-trip preservation) is
24
+ original to this project.
25
+
26
+ You may obtain a copy of the Apache License 2.0 at:
27
+ http://www.apache.org/licenses/LICENSE-2.0
28
+
29
+ ===============================================================================
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: document-adapter
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
5
  Author-email: Son Seongjun <sonsj97@plateer.com>
6
6
  License: MIT
@@ -9,7 +9,7 @@ Project-URL: Repository, https://github.com/PlateerLab/document-adapter
9
9
  Project-URL: Issues, https://github.com/PlateerLab/document-adapter/issues
10
10
  Project-URL: PyPI, https://pypi.org/project/document-adapter/
11
11
  Project-URL: Changelog, https://github.com/PlateerLab/document-adapter/releases
12
- Keywords: mcp,model-context-protocol,llm,claude,anthropic,tool-use,agent,document,docx,pptx,hwpx,hwp,template,template-engine,office,office-automation,word,powerpoint,hancom,korean,docxtpl,python-docx,python-pptx,python-hwpx
12
+ Keywords: mcp,model-context-protocol,llm,claude,anthropic,tool-use,agent,document,docx,pptx,hwpx,hwp,template,template-engine,office,office-automation,word,powerpoint,hancom,korean,docxtpl,python-docx,python-pptx,lxml
13
13
  Classifier: Programming Language :: Python :: 3
14
14
  Classifier: Programming Language :: Python :: 3.10
15
15
  Classifier: Programming Language :: Python :: 3.11
@@ -26,15 +26,17 @@ Classifier: Environment :: Console
26
26
  Requires-Python: >=3.10
27
27
  Description-Content-Type: text/markdown
28
28
  License-File: LICENSE
29
- Requires-Dist: python-docx>=1.1
29
+ License-File: NOTICE
30
+ Requires-Dist: python-docx>=1.2
30
31
  Requires-Dist: docxtpl>=0.20
31
32
  Requires-Dist: python-pptx>=1.0
32
- Requires-Dist: python-hwpx>=2.9
33
+ Requires-Dist: lxml>=5.0
33
34
  Requires-Dist: mcp>=1.0
34
35
  Provides-Extra: claude
35
36
  Requires-Dist: anthropic>=0.40; extra == "claude"
36
37
  Provides-Extra: dev
37
38
  Requires-Dist: pytest>=8; extra == "dev"
39
+ Requires-Dist: python-hwpx>=2.9; extra == "dev"
38
40
  Dynamic: license-file
39
41
 
40
42
  # document-adapter
@@ -57,12 +59,18 @@ Dynamic: license-file
57
59
  |---|---|---|---|---|---|---|---|
58
60
  | `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ | ✅ | ✅ |
59
61
  | `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | — (포맷 미지원) | ✅ | ❌ (미지원) |
60
- | `.hwpx` | `python-hwpx` (Pure Python) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ (v0.3+) |
62
+ | `.hwpx` | 자체 `hwpx_core` (lxml + zipfile) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ |
61
63
 
62
64
  - HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
63
65
  - 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
64
66
  - 병합 셀: 3개 포맷 모두 preview에 `null` 슬롯 + `merges` 메타로 구조 노출. non-anchor 좌표에 쓰기는 `MergedCellWriteError`로 거부.
65
67
 
68
+ ## 라이선스 (상용 사용 가능)
69
+
70
+ - 본 프로젝트: **MIT License**
71
+ - 런타임 의존성(`python-docx`, `docxtpl`, `python-pptx`, `lxml`, `mcp`): 전부 **허용형 OSS** (MIT/BSD/Apache-2.0/LGPL-2.1). 상용·내부 서비스에 그대로 포함 가능.
72
+ - v0.3 이하에서 사용했던 `python-hwpx` (Non-Commercial License) 는 v0.4.0부터 **dev 환경(테스트 fixture 생성) 전용**으로 이동. HWPX 편집은 자체 `hwpx_core` 모듈이 수행합니다.
73
+
66
74
  ## 설치
67
75
 
68
76
  ```bash
@@ -18,12 +18,18 @@
18
18
  |---|---|---|---|---|---|---|---|
19
19
  | `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ | ✅ | ✅ |
20
20
  | `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | — (포맷 미지원) | ✅ | ❌ (미지원) |
21
- | `.hwpx` | `python-hwpx` (Pure Python) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ (v0.3+) |
21
+ | `.hwpx` | 자체 `hwpx_core` (lxml + zipfile) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ |
22
22
 
23
23
  - HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
24
24
  - 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
25
25
  - 병합 셀: 3개 포맷 모두 preview에 `null` 슬롯 + `merges` 메타로 구조 노출. non-anchor 좌표에 쓰기는 `MergedCellWriteError`로 거부.
26
26
 
27
+ ## 라이선스 (상용 사용 가능)
28
+
29
+ - 본 프로젝트: **MIT License**
30
+ - 런타임 의존성(`python-docx`, `docxtpl`, `python-pptx`, `lxml`, `mcp`): 전부 **허용형 OSS** (MIT/BSD/Apache-2.0/LGPL-2.1). 상용·내부 서비스에 그대로 포함 가능.
31
+ - v0.3 이하에서 사용했던 `python-hwpx` (Non-Commercial License) 는 v0.4.0부터 **dev 환경(테스트 fixture 생성) 전용**으로 이동. HWPX 편집은 자체 `hwpx_core` 모듈이 수행합니다.
32
+
27
33
  ## 설치
28
34
 
29
35
  ```bash
@@ -0,0 +1,348 @@
1
+ """HWPX 어댑터: document_adapter.hwpx_core 기반 (python-hwpx 의존 없음).
2
+
3
+ - 패키지 로드/저장은 HwpxPackage가 처리 (bytes-copy 보존, 수정 XML만 재직렬화)
4
+ - 표 순회는 iter_grid 직접 사용 (cellAddr + cellSpan → logical grid)
5
+ - run-level 포맷은 paragraph 헬퍼가 첫 <hp:t>만 갈아끼워 유지
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ import warnings
11
+ from copy import deepcopy
12
+ from pathlib import Path
13
+ from typing import Any, Iterator
14
+
15
+ from lxml import etree
16
+
17
+ from document_adapter.hwpx_core import (
18
+ HP_CELL_ADDR,
19
+ HP_P,
20
+ HP_RUN,
21
+ HP_SUBLIST,
22
+ HP_T,
23
+ HP_TBL,
24
+ HP_TC,
25
+ HP_TR,
26
+ HwpxPackage,
27
+ cell_paragraph_texts,
28
+ cell_paragraphs,
29
+ cell_text,
30
+ iter_grid,
31
+ nested_tables,
32
+ paragraph_text,
33
+ set_paragraph_text,
34
+ table_shape,
35
+ write_cell,
36
+ )
37
+
38
+ from .base import (
39
+ CellContent,
40
+ CellOutOfBoundsError,
41
+ DocumentAdapter,
42
+ MergeInfo,
43
+ MergedCellWriteError,
44
+ NotImplementedForFormat,
45
+ TableIndexError,
46
+ TableSchema,
47
+ )
48
+
49
+ TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
50
+
51
+
52
+ class HwpxAdapter(DocumentAdapter):
53
+ format = "hwpx"
54
+
55
+ def _open(self) -> None:
56
+ self._pkg = HwpxPackage.open(self.path)
57
+
58
+ def save(self, path: Path | str | None = None) -> Path:
59
+ target = Path(path) if path else self.path
60
+ self._pkg.save(target)
61
+ self.path = target
62
+ return target
63
+
64
+ def close(self) -> None:
65
+ self._pkg.close()
66
+
67
+ # ---- 테이블 순회 ----
68
+
69
+ def _iter_tables(
70
+ self,
71
+ ) -> Iterator[tuple[int, etree._Element, str, str]]:
72
+ """(flat_index, tbl_element, parent_path, section_part_name) 순회.
73
+
74
+ 최상위 테이블과 그 안의 중첩 테이블을 DFS 순서로 부여.
75
+ """
76
+ idx_counter = [0]
77
+
78
+ def walk(tbl: etree._Element, parent_path: str, section_name: str):
79
+ current_idx = idx_counter[0]
80
+ idx_counter[0] += 1
81
+ yield current_idx, tbl, parent_path, section_name
82
+ seen_anchors: set[tuple[int, int]] = set()
83
+ for entry in iter_grid(tbl):
84
+ if not entry.is_anchor or entry.anchor in seen_anchors:
85
+ continue
86
+ seen_anchors.add(entry.anchor)
87
+ for child_tbl in nested_tables(entry.cell_element):
88
+ child_parent = (
89
+ f"{parent_path}.tables[{current_idx}].cell"
90
+ f"({entry.anchor[0]},{entry.anchor[1]})"
91
+ )
92
+ yield from walk(child_tbl, child_parent, section_name)
93
+
94
+ for section_name, root in self._pkg.iter_section_roots():
95
+ # 최상위 <hp:tbl> 찾기: root > hp:p > hp:run > hp:tbl
96
+ for p in root.findall(HP_P):
97
+ for run in p.findall(HP_RUN):
98
+ for tbl in run.findall(HP_TBL):
99
+ yield from walk(tbl, "", section_name)
100
+
101
+ def _get_table(self, table_index: int) -> tuple[etree._Element, str]:
102
+ """flat_index로 (tbl_element, section_part_name) 반환."""
103
+ for idx, tbl, _, section_name in self._iter_tables():
104
+ if idx == table_index:
105
+ return tbl, section_name
106
+ raise TableIndexError(f"HWPX table index {table_index} not found")
107
+
108
+ def _find_grid_entry(self, tbl: etree._Element, row: int, col: int):
109
+ rows, cols = table_shape(tbl)
110
+ if row < 0 or col < 0 or row >= rows or col >= cols:
111
+ raise CellOutOfBoundsError(
112
+ f"cell ({row},{col}) out of bounds ({rows}x{cols})"
113
+ )
114
+ for entry in iter_grid(tbl):
115
+ if (entry.row, entry.column) == (row, col):
116
+ return entry
117
+ raise CellOutOfBoundsError(
118
+ f"cell ({row},{col}) does not resolve to any physical cell"
119
+ )
120
+
121
+ def _resolve_anchor_cell(
122
+ self,
123
+ tbl: etree._Element,
124
+ row: int,
125
+ col: int,
126
+ *,
127
+ allow_merge_redirect: bool,
128
+ ):
129
+ entry = self._find_grid_entry(tbl, row, col)
130
+ if not entry.is_anchor:
131
+ anchor_r, anchor_c = entry.anchor
132
+ if not allow_merge_redirect:
133
+ raise MergedCellWriteError(
134
+ f"cell ({row},{col}) is part of a merged region anchored at "
135
+ f"({anchor_r},{anchor_c}) span={entry.span}. "
136
+ f"Write to the anchor coordinate, or pass "
137
+ f"allow_merge_redirect=True."
138
+ )
139
+ warnings.warn(
140
+ f"write to ({row},{col}) redirected to merge anchor "
141
+ f"({anchor_r},{anchor_c})",
142
+ stacklevel=3,
143
+ )
144
+ return entry
145
+
146
+ # ---- 검사 ----
147
+
148
+ def get_placeholders(self) -> list[str]:
149
+ text = self._pkg.export_text()
150
+ return sorted(set(TAG_PATTERN.findall(text)))
151
+
152
+ def get_tables(
153
+ self,
154
+ min_rows: int = 1,
155
+ min_cols: int = 1,
156
+ preview_rows: int = 4,
157
+ max_cell_len: int = 40,
158
+ ) -> list[TableSchema]:
159
+ schemas: list[TableSchema] = []
160
+ for idx, tbl, parent_path, _ in self._iter_tables():
161
+ rows, cols = table_shape(tbl)
162
+ if rows < min_rows or cols < min_cols:
163
+ continue
164
+
165
+ visible_rows = min(rows, preview_rows)
166
+ preview: list[list[str | None]] = [
167
+ [None for _ in range(cols)] for _ in range(visible_rows)
168
+ ]
169
+ merges: list[MergeInfo] = []
170
+ seen_anchors: set[tuple[int, int]] = set()
171
+
172
+ for entry in iter_grid(tbl):
173
+ if entry.anchor in seen_anchors:
174
+ continue
175
+ if entry.is_anchor:
176
+ seen_anchors.add(entry.anchor)
177
+ if entry.row < visible_rows:
178
+ text = cell_text(entry.cell_element).strip()
179
+ preview[entry.row][entry.column] = text[:max_cell_len]
180
+ if entry.span != (1, 1):
181
+ merges.append(MergeInfo(anchor=entry.anchor, span=entry.span))
182
+
183
+ schemas.append(
184
+ TableSchema(
185
+ index=idx,
186
+ rows=rows,
187
+ cols=cols,
188
+ preview=preview,
189
+ merges=merges,
190
+ parent_path=parent_path or None,
191
+ )
192
+ )
193
+ return schemas
194
+
195
+ def get_cell(self, table_index: int, row: int, col: int) -> CellContent:
196
+ tbl, _ = self._get_table(table_index)
197
+ entry = self._find_grid_entry(tbl, row, col)
198
+
199
+ tc = entry.cell_element
200
+ text = cell_text(tc)
201
+ paragraphs = cell_paragraph_texts(tc)
202
+
203
+ nested_indices: list[int] = []
204
+ if entry.is_anchor:
205
+ child_tbls = nested_tables(tc)
206
+ if child_tbls:
207
+ nested_ids = {id(t) for t in child_tbls}
208
+ for child_idx, child_tbl, _, _ in self._iter_tables():
209
+ if id(child_tbl) in nested_ids:
210
+ nested_indices.append(child_idx)
211
+
212
+ return CellContent(
213
+ row=row,
214
+ col=col,
215
+ text=text,
216
+ paragraphs=paragraphs,
217
+ is_anchor=entry.is_anchor,
218
+ anchor=entry.anchor,
219
+ span=entry.span,
220
+ nested_table_indices=nested_indices,
221
+ )
222
+
223
+ # ---- 편집 ----
224
+
225
+ def render_template(self, context: dict[str, Any]) -> None:
226
+ """섹션의 모든 <hp:p> 에서 {{key}} 치환. paragraph 단위로 처리해
227
+ run 포맷은 보존한다 (첫 <hp:t>에 치환 결과를 쓰고 나머지는 비움).
228
+ """
229
+ def substitute(p: etree._Element) -> bool:
230
+ text = paragraph_text(p)
231
+ if not TAG_PATTERN.search(text):
232
+ return False
233
+ new_text = TAG_PATTERN.sub(
234
+ lambda m: str(context.get(m.group(1), m.group(0))), text
235
+ )
236
+ set_paragraph_text(p, new_text)
237
+ return True
238
+
239
+ for section_name, root in self._pkg.iter_section_roots():
240
+ changed = False
241
+ for p in root.iter(HP_P):
242
+ if substitute(p):
243
+ changed = True
244
+ if changed:
245
+ self._pkg.mark_dirty(section_name)
246
+
247
+ def set_cell(
248
+ self,
249
+ table_index: int,
250
+ row: int,
251
+ col: int,
252
+ value: str,
253
+ *,
254
+ allow_merge_redirect: bool = False,
255
+ ) -> str:
256
+ tbl, section_name = self._get_table(table_index)
257
+ entry = self._resolve_anchor_cell(
258
+ tbl, row, col, allow_merge_redirect=allow_merge_redirect
259
+ )
260
+ tc = entry.cell_element
261
+ old = cell_text(tc).strip()
262
+ write_cell(tc, value)
263
+ self._pkg.mark_dirty(section_name)
264
+ return old
265
+
266
+ def append_to_cell(
267
+ self,
268
+ table_index: int,
269
+ row: int,
270
+ col: int,
271
+ value: str,
272
+ separator: str = " ",
273
+ *,
274
+ allow_merge_redirect: bool = False,
275
+ ) -> str:
276
+ tbl, section_name = self._get_table(table_index)
277
+ entry = self._resolve_anchor_cell(
278
+ tbl, row, col, allow_merge_redirect=allow_merge_redirect
279
+ )
280
+ tc = entry.cell_element
281
+ old = cell_text(tc).strip()
282
+ new_value = f"{old}{separator}{value}" if old else value
283
+ write_cell(tc, new_value)
284
+ self._pkg.mark_dirty(section_name)
285
+ return old
286
+
287
+ def append_row(self, table_index: int, values: list[str]) -> None:
288
+ """표 끝에 새 행 추가: 마지막 <hp:tr> deepcopy → 각 셀 비우고
289
+ cellAddr.rowAddr를 새 인덱스로 갱신. 제약은 기존과 동일:
290
+ - 마지막 행이 rowSpan에 걸리면 NotImplementedForFormat
291
+ """
292
+ tbl, section_name = self._get_table(table_index)
293
+ rows_before, _ = table_shape(tbl)
294
+
295
+ trs = tbl.findall(HP_TR)
296
+ if not trs:
297
+ raise NotImplementedForFormat("cannot append row to empty HWPX table")
298
+
299
+ last_row = trs[-1]
300
+ for tc in last_row.findall(HP_TC):
301
+ from document_adapter.hwpx_core.constants import HP_CELL_SPAN
302
+
303
+ span = tc.find(HP_CELL_SPAN)
304
+ addr = tc.find(HP_CELL_ADDR)
305
+ if span is not None and addr is not None:
306
+ try:
307
+ row_span = int(span.get("rowSpan", "1"))
308
+ row_addr = int(addr.get("rowAddr", "0"))
309
+ except (TypeError, ValueError):
310
+ continue
311
+ if row_addr + row_span - 1 != rows_before - 1:
312
+ raise NotImplementedForFormat(
313
+ "last row participates in a cross-row merge; "
314
+ "append_row is not safe for this table."
315
+ )
316
+
317
+ new_row_idx = rows_before
318
+ new_row = deepcopy(last_row)
319
+ for tc in new_row.findall(HP_TC):
320
+ addr = tc.find(HP_CELL_ADDR)
321
+ if addr is not None:
322
+ addr.set("rowAddr", str(new_row_idx))
323
+ # 기존 텍스트만 비우고 run/paragraph 구조는 유지
324
+ sublist = tc.find(HP_SUBLIST)
325
+ if sublist is not None:
326
+ for p in sublist.findall(HP_P):
327
+ for run in p.findall(HP_RUN):
328
+ for t in run.findall(HP_T):
329
+ t.text = ""
330
+
331
+ tbl.append(new_row)
332
+
333
+ # rowCnt 속성 갱신 (있을 때만)
334
+ row_cnt_attr = tbl.get("rowCnt")
335
+ if row_cnt_attr and row_cnt_attr.isdigit():
336
+ tbl.set("rowCnt", str(int(row_cnt_attr) + 1))
337
+
338
+ self._pkg.mark_dirty(section_name)
339
+
340
+ # 값 채우기 (병합된 non-anchor 위치는 스킵)
341
+ for i, value in enumerate(values):
342
+ _, cols = table_shape(tbl)
343
+ if i >= cols:
344
+ break
345
+ try:
346
+ self.set_cell(table_index, new_row_idx, i, value)
347
+ except MergedCellWriteError:
348
+ continue
@@ -0,0 +1,67 @@
1
+ """HWPX 저수준 패키지 — python-hwpx 의존 없이 zipfile+lxml로 HWPX 문서 편집.
2
+
3
+ 이 패키지는 표 편집에 필요한 최소 기능만 제공한다:
4
+ - ZIP 컨테이너 입출력 (수정 안 한 파일은 bytes 그대로 보존)
5
+ - XML 트리의 lazy 파싱과 dirty 추적
6
+ - 병합 셀 인식된 logical grid 순회
7
+
8
+ 구조:
9
+ - constants: HWPX XML 네임스페이스
10
+ - package.HwpxPackage: python-hwpx의 HwpxDocument 대체
11
+ - grid.iter_grid: python-hwpx의 Table.iter_grid() 대체
12
+ """
13
+ from document_adapter.hwpx_core.constants import (
14
+ HC_NS,
15
+ HH_NS,
16
+ HP_NS,
17
+ HS_NS,
18
+ OPF_NS,
19
+ HP_CELL_ADDR,
20
+ HP_CELL_SPAN,
21
+ HP_P,
22
+ HP_RUN,
23
+ HP_SUBLIST,
24
+ HP_T,
25
+ HP_TBL,
26
+ HP_TC,
27
+ HP_TR,
28
+ )
29
+ from document_adapter.hwpx_core.grid import GridEntry, iter_grid, table_shape
30
+ from document_adapter.hwpx_core.package import HwpxPackage
31
+ from document_adapter.hwpx_core.paragraph import (
32
+ cell_paragraph_texts,
33
+ cell_paragraphs,
34
+ cell_text,
35
+ nested_tables,
36
+ paragraph_text,
37
+ set_paragraph_text,
38
+ write_cell,
39
+ )
40
+
41
+ __all__ = [
42
+ "HC_NS",
43
+ "HH_NS",
44
+ "HP_NS",
45
+ "HS_NS",
46
+ "OPF_NS",
47
+ "HP_CELL_ADDR",
48
+ "HP_CELL_SPAN",
49
+ "HP_P",
50
+ "HP_RUN",
51
+ "HP_SUBLIST",
52
+ "HP_T",
53
+ "HP_TBL",
54
+ "HP_TC",
55
+ "HP_TR",
56
+ "GridEntry",
57
+ "HwpxPackage",
58
+ "iter_grid",
59
+ "table_shape",
60
+ "cell_paragraph_texts",
61
+ "cell_paragraphs",
62
+ "cell_text",
63
+ "nested_tables",
64
+ "paragraph_text",
65
+ "set_paragraph_text",
66
+ "write_cell",
67
+ ]
@@ -0,0 +1,27 @@
1
+ """HWPX XML 네임스페이스 상수."""
2
+ from __future__ import annotations
3
+
4
+ HP_NS = "http://www.hancom.co.kr/hwpml/2011/paragraph"
5
+ HC_NS = "http://www.hancom.co.kr/hwpml/2011/core"
6
+ HH_NS = "http://www.hancom.co.kr/hwpml/2011/head"
7
+ HS_NS = "http://www.hancom.co.kr/hwpml/2011/section"
8
+ OPF_NS = "http://www.idpf.org/2007/opf/"
9
+
10
+ HP_P = f"{{{HP_NS}}}p"
11
+ HP_RUN = f"{{{HP_NS}}}run"
12
+ HP_T = f"{{{HP_NS}}}t"
13
+ HP_TBL = f"{{{HP_NS}}}tbl"
14
+ HP_TR = f"{{{HP_NS}}}tr"
15
+ HP_TC = f"{{{HP_NS}}}tc"
16
+ HP_CELL_ADDR = f"{{{HP_NS}}}cellAddr"
17
+ HP_CELL_SPAN = f"{{{HP_NS}}}cellSpan"
18
+ HP_SUBLIST = f"{{{HP_NS}}}subList"
19
+ HS_SEC = f"{{{HS_NS}}}sec"
20
+
21
+ NAMESPACES = {
22
+ "hp": HP_NS,
23
+ "hc": HC_NS,
24
+ "hh": HH_NS,
25
+ "hs": HS_NS,
26
+ "opf": OPF_NS,
27
+ }
@@ -0,0 +1,145 @@
1
+ """<hp:tbl> 에서 병합 셀 인식된 logical grid를 생성.
2
+
3
+ python-hwpx의 ``Table.iter_grid()``를 대체. xgen-doc2chunk의 ``_build_cell_grid``
4
+ 패턴을 차용하되, **읽기만이 아니라 쓰기도 지원**하기 위해 각 ``GridEntry``가
5
+ anchor cell의 lxml ``<hp:tc>`` Element 자체를 노출한다.
6
+
7
+ HWPX 표 구조:
8
+ <hp:tbl rowCnt colCnt>
9
+ <hp:tr>+
10
+ <hp:tc>+
11
+ <hp:cellAddr rowAddr colAddr>
12
+ <hp:cellSpan rowSpan colSpan> (병합 시만 존재 또는 >1)
13
+ <hp:subList> → <hp:p> → <hp:run> → <hp:t>
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+ from typing import Iterator
19
+
20
+ from lxml import etree
21
+
22
+ from document_adapter.hwpx_core.constants import HP_CELL_ADDR, HP_CELL_SPAN, HP_TC, HP_TR
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class GridEntry:
27
+ """Logical grid의 한 슬롯.
28
+
29
+ - ``is_anchor=True``: 이 (row, col)이 셀의 anchor 좌표
30
+ - ``is_anchor=False``: 병합된 셀에 덮여 있는 non-anchor 슬롯. ``anchor``가 앵커 좌표.
31
+ ``cell_element``는 항상 anchor 셀의 ``<hp:tc>``.
32
+ """
33
+ row: int
34
+ column: int
35
+ is_anchor: bool
36
+ anchor: tuple[int, int]
37
+ span: tuple[int, int] # (rowspan, colspan)
38
+ cell_element: etree._Element
39
+
40
+
41
+ def _parse_cell_position(tc: etree._Element) -> tuple[int, int]:
42
+ addr = tc.find(HP_CELL_ADDR)
43
+ if addr is None:
44
+ return 0, 0
45
+ try:
46
+ row = int(addr.get("rowAddr", "0"))
47
+ except (TypeError, ValueError):
48
+ row = 0
49
+ try:
50
+ col = int(addr.get("colAddr", "0"))
51
+ except (TypeError, ValueError):
52
+ col = 0
53
+ return row, col
54
+
55
+
56
+ def _parse_cell_span(tc: etree._Element) -> tuple[int, int]:
57
+ span = tc.find(HP_CELL_SPAN)
58
+ if span is None:
59
+ return 1, 1
60
+ try:
61
+ rs = int(span.get("rowSpan", "1"))
62
+ except (TypeError, ValueError):
63
+ rs = 1
64
+ try:
65
+ cs = int(span.get("colSpan", "1"))
66
+ except (TypeError, ValueError):
67
+ cs = 1
68
+ return max(1, rs), max(1, cs)
69
+
70
+
71
+ def table_shape(tbl: etree._Element) -> tuple[int, int]:
72
+ """표의 (rows, cols). rowCnt/colCnt 속성이 없으면 앵커 좌표로 추정."""
73
+ rows = _safe_int(tbl.get("rowCnt"))
74
+ cols = _safe_int(tbl.get("colCnt"))
75
+ if rows > 0 and cols > 0:
76
+ return rows, cols
77
+
78
+ max_row = -1
79
+ max_col = -1
80
+ for tr in tbl.findall(HP_TR):
81
+ for tc in tr.findall(HP_TC):
82
+ r, c = _parse_cell_position(tc)
83
+ rs, cs = _parse_cell_span(tc)
84
+ max_row = max(max_row, r + rs - 1)
85
+ max_col = max(max_col, c + cs - 1)
86
+ return max(rows, max_row + 1), max(cols, max_col + 1)
87
+
88
+
89
+ def _safe_int(v: str | None) -> int:
90
+ if v is None:
91
+ return 0
92
+ try:
93
+ return int(v)
94
+ except ValueError:
95
+ return 0
96
+
97
+
98
+ def iter_grid(tbl: etree._Element) -> Iterator[GridEntry]:
99
+ """<hp:tbl> 요소를 병합 셀 인식해 logical grid 순서로 순회.
100
+
101
+ row-major: (0,0), (0,1), ..., (0,C-1), (1,0), ...
102
+ """
103
+ rows, cols = table_shape(tbl)
104
+ if rows <= 0 or cols <= 0:
105
+ return
106
+
107
+ # 1) anchor 좌표 → {span, cell_element}
108
+ anchors: dict[tuple[int, int], tuple[tuple[int, int], etree._Element]] = {}
109
+ for tr in tbl.findall(HP_TR):
110
+ for tc in tr.findall(HP_TC):
111
+ r, c = _parse_cell_position(tc)
112
+ span = _parse_cell_span(tc)
113
+ anchors[(r, c)] = (span, tc)
114
+
115
+ # 2) 각 (row, col)이 어느 anchor에 속하는지 역산 매핑
116
+ owner: dict[tuple[int, int], tuple[int, int]] = {}
117
+ for (ar, ac), (span, _tc) in anchors.items():
118
+ rs, cs = span
119
+ for dr in range(rs):
120
+ for dc in range(cs):
121
+ slot = (ar + dr, ac + dc)
122
+ # 여러 anchor가 같은 slot을 주장하면 가장 가까운(= 자기 자신) 우선
123
+ if slot in owner:
124
+ # 이미 자기 자신으로 설정됐다면 유지
125
+ continue
126
+ owner[slot] = (ar, ac)
127
+
128
+ # 3) row-major 순회
129
+ for r in range(rows):
130
+ for c in range(cols):
131
+ slot = (r, c)
132
+ anchor_coord = owner.get(slot)
133
+ if anchor_coord is None:
134
+ # grid에 선언되지 않은 슬롯 (손상 문서). 빈 앵커 취급.
135
+ continue
136
+ span, cell_el = anchors[anchor_coord]
137
+ is_anchor = (anchor_coord == slot)
138
+ yield GridEntry(
139
+ row=r,
140
+ column=c,
141
+ is_anchor=is_anchor,
142
+ anchor=anchor_coord,
143
+ span=span,
144
+ cell_element=cell_el,
145
+ )