document-adapter 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- document_adapter-0.4.0/NOTICE +29 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/PKG-INFO +13 -5
- {document_adapter-0.3.0 → document_adapter-0.4.0}/README.md +7 -1
- document_adapter-0.4.0/document_adapter/hwpx_adapter.py +348 -0
- document_adapter-0.4.0/document_adapter/hwpx_core/__init__.py +67 -0
- document_adapter-0.4.0/document_adapter/hwpx_core/constants.py +27 -0
- document_adapter-0.4.0/document_adapter/hwpx_core/grid.py +145 -0
- document_adapter-0.4.0/document_adapter/hwpx_core/package.py +146 -0
- document_adapter-0.4.0/document_adapter/hwpx_core/paragraph.py +106 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/PKG-INFO +13 -5
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/SOURCES.txt +6 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/requires.txt +3 -2
- {document_adapter-0.3.0 → document_adapter-0.4.0}/pyproject.toml +7 -5
- document_adapter-0.3.0/document_adapter/hwpx_adapter.py +0 -389
- {document_adapter-0.3.0 → document_adapter-0.4.0}/LICENSE +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/__init__.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/base.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/docx_adapter.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/mcp_server.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/pptx_adapter.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter/tools.py +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/dependency_links.txt +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/entry_points.txt +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/document_adapter.egg-info/top_level.txt +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/setup.cfg +0 -0
- {document_adapter-0.3.0 → document_adapter-0.4.0}/tests/test_smoke.py +0 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
document-adapter
|
|
2
|
+
Copyright 2026 Son Seongjun / PlateerLab
|
|
3
|
+
|
|
4
|
+
This product includes software developed at PlateerLab
|
|
5
|
+
(https://github.com/PlateerLab).
|
|
6
|
+
|
|
7
|
+
===============================================================================
|
|
8
|
+
Third-party code attributions
|
|
9
|
+
===============================================================================
|
|
10
|
+
|
|
11
|
+
HWPX table grid parsing logic
|
|
12
|
+
-----------------------------
|
|
13
|
+
|
|
14
|
+
Portions of the HWPX table grid construction in
|
|
15
|
+
`document_adapter/hwpx_core/grid.py` (cell position / cell span parsing, and
|
|
16
|
+
the skip-map approach for merged non-anchor cells) are adapted from:
|
|
17
|
+
|
|
18
|
+
PlateerLab/xgen-doc2chunk
|
|
19
|
+
https://github.com/PlateerLab/xgen-doc2chunk
|
|
20
|
+
Licensed under the Apache License, Version 2.0
|
|
21
|
+
|
|
22
|
+
The adapted logic is limited to read-side grid construction. All edit-side
|
|
23
|
+
functionality (cell writing, row insertion, ZIP round-trip preservation) is
|
|
24
|
+
original to this project.
|
|
25
|
+
|
|
26
|
+
You may obtain a copy of the Apache License 2.0 at:
|
|
27
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
28
|
+
|
|
29
|
+
===============================================================================
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: document-adapter
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
|
|
5
5
|
Author-email: Son Seongjun <sonsj97@plateer.com>
|
|
6
6
|
License: MIT
|
|
@@ -9,7 +9,7 @@ Project-URL: Repository, https://github.com/PlateerLab/document-adapter
|
|
|
9
9
|
Project-URL: Issues, https://github.com/PlateerLab/document-adapter/issues
|
|
10
10
|
Project-URL: PyPI, https://pypi.org/project/document-adapter/
|
|
11
11
|
Project-URL: Changelog, https://github.com/PlateerLab/document-adapter/releases
|
|
12
|
-
Keywords: mcp,model-context-protocol,llm,claude,anthropic,tool-use,agent,document,docx,pptx,hwpx,hwp,template,template-engine,office,office-automation,word,powerpoint,hancom,korean,docxtpl,python-docx,python-pptx,
|
|
12
|
+
Keywords: mcp,model-context-protocol,llm,claude,anthropic,tool-use,agent,document,docx,pptx,hwpx,hwp,template,template-engine,office,office-automation,word,powerpoint,hancom,korean,docxtpl,python-docx,python-pptx,lxml
|
|
13
13
|
Classifier: Programming Language :: Python :: 3
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.10
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -26,15 +26,17 @@ Classifier: Environment :: Console
|
|
|
26
26
|
Requires-Python: >=3.10
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
28
28
|
License-File: LICENSE
|
|
29
|
-
|
|
29
|
+
License-File: NOTICE
|
|
30
|
+
Requires-Dist: python-docx>=1.2
|
|
30
31
|
Requires-Dist: docxtpl>=0.20
|
|
31
32
|
Requires-Dist: python-pptx>=1.0
|
|
32
|
-
Requires-Dist:
|
|
33
|
+
Requires-Dist: lxml>=5.0
|
|
33
34
|
Requires-Dist: mcp>=1.0
|
|
34
35
|
Provides-Extra: claude
|
|
35
36
|
Requires-Dist: anthropic>=0.40; extra == "claude"
|
|
36
37
|
Provides-Extra: dev
|
|
37
38
|
Requires-Dist: pytest>=8; extra == "dev"
|
|
39
|
+
Requires-Dist: python-hwpx>=2.9; extra == "dev"
|
|
38
40
|
Dynamic: license-file
|
|
39
41
|
|
|
40
42
|
# document-adapter
|
|
@@ -57,12 +59,18 @@ Dynamic: license-file
|
|
|
57
59
|
|---|---|---|---|---|---|---|---|
|
|
58
60
|
| `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ | ✅ | ✅ |
|
|
59
61
|
| `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | — (포맷 미지원) | ✅ | ❌ (미지원) |
|
|
60
|
-
| `.hwpx` | `
|
|
62
|
+
| `.hwpx` | 자체 `hwpx_core` (lxml + zipfile) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ |
|
|
61
63
|
|
|
62
64
|
- HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
|
|
63
65
|
- 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
|
|
64
66
|
- 병합 셀: 3개 포맷 모두 preview에 `null` 슬롯 + `merges` 메타로 구조 노출. non-anchor 좌표에 쓰기는 `MergedCellWriteError`로 거부.
|
|
65
67
|
|
|
68
|
+
## 라이선스 (상용 사용 가능)
|
|
69
|
+
|
|
70
|
+
- 본 프로젝트: **MIT License**
|
|
71
|
+
- 런타임 의존성(`python-docx`, `docxtpl`, `python-pptx`, `lxml`, `mcp`): 전부 **허용형 OSS** (MIT/BSD/Apache-2.0/LGPL-2.1). 상용·내부 서비스에 그대로 포함 가능.
|
|
72
|
+
- v0.3 이하에서 사용했던 `python-hwpx` (Non-Commercial License) 는 v0.4.0부터 **dev 환경(테스트 fixture 생성) 전용**으로 이동. HWPX 편집은 자체 `hwpx_core` 모듈이 수행합니다.
|
|
73
|
+
|
|
66
74
|
## 설치
|
|
67
75
|
|
|
68
76
|
```bash
|
|
@@ -18,12 +18,18 @@
|
|
|
18
18
|
|---|---|---|---|---|---|---|---|
|
|
19
19
|
| `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ | ✅ | ✅ |
|
|
20
20
|
| `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | — (포맷 미지원) | ✅ | ❌ (미지원) |
|
|
21
|
-
| `.hwpx` | `
|
|
21
|
+
| `.hwpx` | 자체 `hwpx_core` (lxml + zipfile) | `{{key}}` 치환 | ✅ | ✅ | ✅ | ✅ | ✅ |
|
|
22
22
|
|
|
23
23
|
- HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
|
|
24
24
|
- 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
|
|
25
25
|
- 병합 셀: 3개 포맷 모두 preview에 `null` 슬롯 + `merges` 메타로 구조 노출. non-anchor 좌표에 쓰기는 `MergedCellWriteError`로 거부.
|
|
26
26
|
|
|
27
|
+
## 라이선스 (상용 사용 가능)
|
|
28
|
+
|
|
29
|
+
- 본 프로젝트: **MIT License**
|
|
30
|
+
- 런타임 의존성(`python-docx`, `docxtpl`, `python-pptx`, `lxml`, `mcp`): 전부 **허용형 OSS** (MIT/BSD/Apache-2.0/LGPL-2.1). 상용·내부 서비스에 그대로 포함 가능.
|
|
31
|
+
- v0.3 이하에서 사용했던 `python-hwpx` (Non-Commercial License) 는 v0.4.0부터 **dev 환경(테스트 fixture 생성) 전용**으로 이동. HWPX 편집은 자체 `hwpx_core` 모듈이 수행합니다.
|
|
32
|
+
|
|
27
33
|
## 설치
|
|
28
34
|
|
|
29
35
|
```bash
|
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
"""HWPX 어댑터: document_adapter.hwpx_core 기반 (python-hwpx 의존 없음).
|
|
2
|
+
|
|
3
|
+
- 패키지 로드/저장은 HwpxPackage가 처리 (bytes-copy 보존, 수정 XML만 재직렬화)
|
|
4
|
+
- 표 순회는 iter_grid 직접 사용 (cellAddr + cellSpan → logical grid)
|
|
5
|
+
- run-level 포맷은 paragraph 헬퍼가 첫 <hp:t>만 갈아끼워 유지
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
import warnings
|
|
11
|
+
from copy import deepcopy
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any, Iterator
|
|
14
|
+
|
|
15
|
+
from lxml import etree
|
|
16
|
+
|
|
17
|
+
from document_adapter.hwpx_core import (
|
|
18
|
+
HP_CELL_ADDR,
|
|
19
|
+
HP_P,
|
|
20
|
+
HP_RUN,
|
|
21
|
+
HP_SUBLIST,
|
|
22
|
+
HP_T,
|
|
23
|
+
HP_TBL,
|
|
24
|
+
HP_TC,
|
|
25
|
+
HP_TR,
|
|
26
|
+
HwpxPackage,
|
|
27
|
+
cell_paragraph_texts,
|
|
28
|
+
cell_paragraphs,
|
|
29
|
+
cell_text,
|
|
30
|
+
iter_grid,
|
|
31
|
+
nested_tables,
|
|
32
|
+
paragraph_text,
|
|
33
|
+
set_paragraph_text,
|
|
34
|
+
table_shape,
|
|
35
|
+
write_cell,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
from .base import (
|
|
39
|
+
CellContent,
|
|
40
|
+
CellOutOfBoundsError,
|
|
41
|
+
DocumentAdapter,
|
|
42
|
+
MergeInfo,
|
|
43
|
+
MergedCellWriteError,
|
|
44
|
+
NotImplementedForFormat,
|
|
45
|
+
TableIndexError,
|
|
46
|
+
TableSchema,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class HwpxAdapter(DocumentAdapter):
|
|
53
|
+
format = "hwpx"
|
|
54
|
+
|
|
55
|
+
def _open(self) -> None:
|
|
56
|
+
self._pkg = HwpxPackage.open(self.path)
|
|
57
|
+
|
|
58
|
+
def save(self, path: Path | str | None = None) -> Path:
|
|
59
|
+
target = Path(path) if path else self.path
|
|
60
|
+
self._pkg.save(target)
|
|
61
|
+
self.path = target
|
|
62
|
+
return target
|
|
63
|
+
|
|
64
|
+
def close(self) -> None:
|
|
65
|
+
self._pkg.close()
|
|
66
|
+
|
|
67
|
+
# ---- 테이블 순회 ----
|
|
68
|
+
|
|
69
|
+
def _iter_tables(
|
|
70
|
+
self,
|
|
71
|
+
) -> Iterator[tuple[int, etree._Element, str, str]]:
|
|
72
|
+
"""(flat_index, tbl_element, parent_path, section_part_name) 순회.
|
|
73
|
+
|
|
74
|
+
최상위 테이블과 그 안의 중첩 테이블을 DFS 순서로 부여.
|
|
75
|
+
"""
|
|
76
|
+
idx_counter = [0]
|
|
77
|
+
|
|
78
|
+
def walk(tbl: etree._Element, parent_path: str, section_name: str):
|
|
79
|
+
current_idx = idx_counter[0]
|
|
80
|
+
idx_counter[0] += 1
|
|
81
|
+
yield current_idx, tbl, parent_path, section_name
|
|
82
|
+
seen_anchors: set[tuple[int, int]] = set()
|
|
83
|
+
for entry in iter_grid(tbl):
|
|
84
|
+
if not entry.is_anchor or entry.anchor in seen_anchors:
|
|
85
|
+
continue
|
|
86
|
+
seen_anchors.add(entry.anchor)
|
|
87
|
+
for child_tbl in nested_tables(entry.cell_element):
|
|
88
|
+
child_parent = (
|
|
89
|
+
f"{parent_path}.tables[{current_idx}].cell"
|
|
90
|
+
f"({entry.anchor[0]},{entry.anchor[1]})"
|
|
91
|
+
)
|
|
92
|
+
yield from walk(child_tbl, child_parent, section_name)
|
|
93
|
+
|
|
94
|
+
for section_name, root in self._pkg.iter_section_roots():
|
|
95
|
+
# 최상위 <hp:tbl> 찾기: root > hp:p > hp:run > hp:tbl
|
|
96
|
+
for p in root.findall(HP_P):
|
|
97
|
+
for run in p.findall(HP_RUN):
|
|
98
|
+
for tbl in run.findall(HP_TBL):
|
|
99
|
+
yield from walk(tbl, "", section_name)
|
|
100
|
+
|
|
101
|
+
def _get_table(self, table_index: int) -> tuple[etree._Element, str]:
|
|
102
|
+
"""flat_index로 (tbl_element, section_part_name) 반환."""
|
|
103
|
+
for idx, tbl, _, section_name in self._iter_tables():
|
|
104
|
+
if idx == table_index:
|
|
105
|
+
return tbl, section_name
|
|
106
|
+
raise TableIndexError(f"HWPX table index {table_index} not found")
|
|
107
|
+
|
|
108
|
+
def _find_grid_entry(self, tbl: etree._Element, row: int, col: int):
|
|
109
|
+
rows, cols = table_shape(tbl)
|
|
110
|
+
if row < 0 or col < 0 or row >= rows or col >= cols:
|
|
111
|
+
raise CellOutOfBoundsError(
|
|
112
|
+
f"cell ({row},{col}) out of bounds ({rows}x{cols})"
|
|
113
|
+
)
|
|
114
|
+
for entry in iter_grid(tbl):
|
|
115
|
+
if (entry.row, entry.column) == (row, col):
|
|
116
|
+
return entry
|
|
117
|
+
raise CellOutOfBoundsError(
|
|
118
|
+
f"cell ({row},{col}) does not resolve to any physical cell"
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
def _resolve_anchor_cell(
|
|
122
|
+
self,
|
|
123
|
+
tbl: etree._Element,
|
|
124
|
+
row: int,
|
|
125
|
+
col: int,
|
|
126
|
+
*,
|
|
127
|
+
allow_merge_redirect: bool,
|
|
128
|
+
):
|
|
129
|
+
entry = self._find_grid_entry(tbl, row, col)
|
|
130
|
+
if not entry.is_anchor:
|
|
131
|
+
anchor_r, anchor_c = entry.anchor
|
|
132
|
+
if not allow_merge_redirect:
|
|
133
|
+
raise MergedCellWriteError(
|
|
134
|
+
f"cell ({row},{col}) is part of a merged region anchored at "
|
|
135
|
+
f"({anchor_r},{anchor_c}) span={entry.span}. "
|
|
136
|
+
f"Write to the anchor coordinate, or pass "
|
|
137
|
+
f"allow_merge_redirect=True."
|
|
138
|
+
)
|
|
139
|
+
warnings.warn(
|
|
140
|
+
f"write to ({row},{col}) redirected to merge anchor "
|
|
141
|
+
f"({anchor_r},{anchor_c})",
|
|
142
|
+
stacklevel=3,
|
|
143
|
+
)
|
|
144
|
+
return entry
|
|
145
|
+
|
|
146
|
+
# ---- 검사 ----
|
|
147
|
+
|
|
148
|
+
def get_placeholders(self) -> list[str]:
|
|
149
|
+
text = self._pkg.export_text()
|
|
150
|
+
return sorted(set(TAG_PATTERN.findall(text)))
|
|
151
|
+
|
|
152
|
+
def get_tables(
|
|
153
|
+
self,
|
|
154
|
+
min_rows: int = 1,
|
|
155
|
+
min_cols: int = 1,
|
|
156
|
+
preview_rows: int = 4,
|
|
157
|
+
max_cell_len: int = 40,
|
|
158
|
+
) -> list[TableSchema]:
|
|
159
|
+
schemas: list[TableSchema] = []
|
|
160
|
+
for idx, tbl, parent_path, _ in self._iter_tables():
|
|
161
|
+
rows, cols = table_shape(tbl)
|
|
162
|
+
if rows < min_rows or cols < min_cols:
|
|
163
|
+
continue
|
|
164
|
+
|
|
165
|
+
visible_rows = min(rows, preview_rows)
|
|
166
|
+
preview: list[list[str | None]] = [
|
|
167
|
+
[None for _ in range(cols)] for _ in range(visible_rows)
|
|
168
|
+
]
|
|
169
|
+
merges: list[MergeInfo] = []
|
|
170
|
+
seen_anchors: set[tuple[int, int]] = set()
|
|
171
|
+
|
|
172
|
+
for entry in iter_grid(tbl):
|
|
173
|
+
if entry.anchor in seen_anchors:
|
|
174
|
+
continue
|
|
175
|
+
if entry.is_anchor:
|
|
176
|
+
seen_anchors.add(entry.anchor)
|
|
177
|
+
if entry.row < visible_rows:
|
|
178
|
+
text = cell_text(entry.cell_element).strip()
|
|
179
|
+
preview[entry.row][entry.column] = text[:max_cell_len]
|
|
180
|
+
if entry.span != (1, 1):
|
|
181
|
+
merges.append(MergeInfo(anchor=entry.anchor, span=entry.span))
|
|
182
|
+
|
|
183
|
+
schemas.append(
|
|
184
|
+
TableSchema(
|
|
185
|
+
index=idx,
|
|
186
|
+
rows=rows,
|
|
187
|
+
cols=cols,
|
|
188
|
+
preview=preview,
|
|
189
|
+
merges=merges,
|
|
190
|
+
parent_path=parent_path or None,
|
|
191
|
+
)
|
|
192
|
+
)
|
|
193
|
+
return schemas
|
|
194
|
+
|
|
195
|
+
def get_cell(self, table_index: int, row: int, col: int) -> CellContent:
|
|
196
|
+
tbl, _ = self._get_table(table_index)
|
|
197
|
+
entry = self._find_grid_entry(tbl, row, col)
|
|
198
|
+
|
|
199
|
+
tc = entry.cell_element
|
|
200
|
+
text = cell_text(tc)
|
|
201
|
+
paragraphs = cell_paragraph_texts(tc)
|
|
202
|
+
|
|
203
|
+
nested_indices: list[int] = []
|
|
204
|
+
if entry.is_anchor:
|
|
205
|
+
child_tbls = nested_tables(tc)
|
|
206
|
+
if child_tbls:
|
|
207
|
+
nested_ids = {id(t) for t in child_tbls}
|
|
208
|
+
for child_idx, child_tbl, _, _ in self._iter_tables():
|
|
209
|
+
if id(child_tbl) in nested_ids:
|
|
210
|
+
nested_indices.append(child_idx)
|
|
211
|
+
|
|
212
|
+
return CellContent(
|
|
213
|
+
row=row,
|
|
214
|
+
col=col,
|
|
215
|
+
text=text,
|
|
216
|
+
paragraphs=paragraphs,
|
|
217
|
+
is_anchor=entry.is_anchor,
|
|
218
|
+
anchor=entry.anchor,
|
|
219
|
+
span=entry.span,
|
|
220
|
+
nested_table_indices=nested_indices,
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
# ---- 편집 ----
|
|
224
|
+
|
|
225
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
226
|
+
"""섹션의 모든 <hp:p> 에서 {{key}} 치환. paragraph 단위로 처리해
|
|
227
|
+
run 포맷은 보존한다 (첫 <hp:t>에 치환 결과를 쓰고 나머지는 비움).
|
|
228
|
+
"""
|
|
229
|
+
def substitute(p: etree._Element) -> bool:
|
|
230
|
+
text = paragraph_text(p)
|
|
231
|
+
if not TAG_PATTERN.search(text):
|
|
232
|
+
return False
|
|
233
|
+
new_text = TAG_PATTERN.sub(
|
|
234
|
+
lambda m: str(context.get(m.group(1), m.group(0))), text
|
|
235
|
+
)
|
|
236
|
+
set_paragraph_text(p, new_text)
|
|
237
|
+
return True
|
|
238
|
+
|
|
239
|
+
for section_name, root in self._pkg.iter_section_roots():
|
|
240
|
+
changed = False
|
|
241
|
+
for p in root.iter(HP_P):
|
|
242
|
+
if substitute(p):
|
|
243
|
+
changed = True
|
|
244
|
+
if changed:
|
|
245
|
+
self._pkg.mark_dirty(section_name)
|
|
246
|
+
|
|
247
|
+
def set_cell(
|
|
248
|
+
self,
|
|
249
|
+
table_index: int,
|
|
250
|
+
row: int,
|
|
251
|
+
col: int,
|
|
252
|
+
value: str,
|
|
253
|
+
*,
|
|
254
|
+
allow_merge_redirect: bool = False,
|
|
255
|
+
) -> str:
|
|
256
|
+
tbl, section_name = self._get_table(table_index)
|
|
257
|
+
entry = self._resolve_anchor_cell(
|
|
258
|
+
tbl, row, col, allow_merge_redirect=allow_merge_redirect
|
|
259
|
+
)
|
|
260
|
+
tc = entry.cell_element
|
|
261
|
+
old = cell_text(tc).strip()
|
|
262
|
+
write_cell(tc, value)
|
|
263
|
+
self._pkg.mark_dirty(section_name)
|
|
264
|
+
return old
|
|
265
|
+
|
|
266
|
+
def append_to_cell(
|
|
267
|
+
self,
|
|
268
|
+
table_index: int,
|
|
269
|
+
row: int,
|
|
270
|
+
col: int,
|
|
271
|
+
value: str,
|
|
272
|
+
separator: str = " ",
|
|
273
|
+
*,
|
|
274
|
+
allow_merge_redirect: bool = False,
|
|
275
|
+
) -> str:
|
|
276
|
+
tbl, section_name = self._get_table(table_index)
|
|
277
|
+
entry = self._resolve_anchor_cell(
|
|
278
|
+
tbl, row, col, allow_merge_redirect=allow_merge_redirect
|
|
279
|
+
)
|
|
280
|
+
tc = entry.cell_element
|
|
281
|
+
old = cell_text(tc).strip()
|
|
282
|
+
new_value = f"{old}{separator}{value}" if old else value
|
|
283
|
+
write_cell(tc, new_value)
|
|
284
|
+
self._pkg.mark_dirty(section_name)
|
|
285
|
+
return old
|
|
286
|
+
|
|
287
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
288
|
+
"""표 끝에 새 행 추가: 마지막 <hp:tr> deepcopy → 각 셀 비우고
|
|
289
|
+
cellAddr.rowAddr를 새 인덱스로 갱신. 제약은 기존과 동일:
|
|
290
|
+
- 마지막 행이 rowSpan에 걸리면 NotImplementedForFormat
|
|
291
|
+
"""
|
|
292
|
+
tbl, section_name = self._get_table(table_index)
|
|
293
|
+
rows_before, _ = table_shape(tbl)
|
|
294
|
+
|
|
295
|
+
trs = tbl.findall(HP_TR)
|
|
296
|
+
if not trs:
|
|
297
|
+
raise NotImplementedForFormat("cannot append row to empty HWPX table")
|
|
298
|
+
|
|
299
|
+
last_row = trs[-1]
|
|
300
|
+
for tc in last_row.findall(HP_TC):
|
|
301
|
+
from document_adapter.hwpx_core.constants import HP_CELL_SPAN
|
|
302
|
+
|
|
303
|
+
span = tc.find(HP_CELL_SPAN)
|
|
304
|
+
addr = tc.find(HP_CELL_ADDR)
|
|
305
|
+
if span is not None and addr is not None:
|
|
306
|
+
try:
|
|
307
|
+
row_span = int(span.get("rowSpan", "1"))
|
|
308
|
+
row_addr = int(addr.get("rowAddr", "0"))
|
|
309
|
+
except (TypeError, ValueError):
|
|
310
|
+
continue
|
|
311
|
+
if row_addr + row_span - 1 != rows_before - 1:
|
|
312
|
+
raise NotImplementedForFormat(
|
|
313
|
+
"last row participates in a cross-row merge; "
|
|
314
|
+
"append_row is not safe for this table."
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
new_row_idx = rows_before
|
|
318
|
+
new_row = deepcopy(last_row)
|
|
319
|
+
for tc in new_row.findall(HP_TC):
|
|
320
|
+
addr = tc.find(HP_CELL_ADDR)
|
|
321
|
+
if addr is not None:
|
|
322
|
+
addr.set("rowAddr", str(new_row_idx))
|
|
323
|
+
# 기존 텍스트만 비우고 run/paragraph 구조는 유지
|
|
324
|
+
sublist = tc.find(HP_SUBLIST)
|
|
325
|
+
if sublist is not None:
|
|
326
|
+
for p in sublist.findall(HP_P):
|
|
327
|
+
for run in p.findall(HP_RUN):
|
|
328
|
+
for t in run.findall(HP_T):
|
|
329
|
+
t.text = ""
|
|
330
|
+
|
|
331
|
+
tbl.append(new_row)
|
|
332
|
+
|
|
333
|
+
# rowCnt 속성 갱신 (있을 때만)
|
|
334
|
+
row_cnt_attr = tbl.get("rowCnt")
|
|
335
|
+
if row_cnt_attr and row_cnt_attr.isdigit():
|
|
336
|
+
tbl.set("rowCnt", str(int(row_cnt_attr) + 1))
|
|
337
|
+
|
|
338
|
+
self._pkg.mark_dirty(section_name)
|
|
339
|
+
|
|
340
|
+
# 값 채우기 (병합된 non-anchor 위치는 스킵)
|
|
341
|
+
for i, value in enumerate(values):
|
|
342
|
+
_, cols = table_shape(tbl)
|
|
343
|
+
if i >= cols:
|
|
344
|
+
break
|
|
345
|
+
try:
|
|
346
|
+
self.set_cell(table_index, new_row_idx, i, value)
|
|
347
|
+
except MergedCellWriteError:
|
|
348
|
+
continue
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""HWPX 저수준 패키지 — python-hwpx 의존 없이 zipfile+lxml로 HWPX 문서 편집.
|
|
2
|
+
|
|
3
|
+
이 패키지는 표 편집에 필요한 최소 기능만 제공한다:
|
|
4
|
+
- ZIP 컨테이너 입출력 (수정 안 한 파일은 bytes 그대로 보존)
|
|
5
|
+
- XML 트리의 lazy 파싱과 dirty 추적
|
|
6
|
+
- 병합 셀 인식된 logical grid 순회
|
|
7
|
+
|
|
8
|
+
구조:
|
|
9
|
+
- constants: HWPX XML 네임스페이스
|
|
10
|
+
- package.HwpxPackage: python-hwpx의 HwpxDocument 대체
|
|
11
|
+
- grid.iter_grid: python-hwpx의 Table.iter_grid() 대체
|
|
12
|
+
"""
|
|
13
|
+
from document_adapter.hwpx_core.constants import (
|
|
14
|
+
HC_NS,
|
|
15
|
+
HH_NS,
|
|
16
|
+
HP_NS,
|
|
17
|
+
HS_NS,
|
|
18
|
+
OPF_NS,
|
|
19
|
+
HP_CELL_ADDR,
|
|
20
|
+
HP_CELL_SPAN,
|
|
21
|
+
HP_P,
|
|
22
|
+
HP_RUN,
|
|
23
|
+
HP_SUBLIST,
|
|
24
|
+
HP_T,
|
|
25
|
+
HP_TBL,
|
|
26
|
+
HP_TC,
|
|
27
|
+
HP_TR,
|
|
28
|
+
)
|
|
29
|
+
from document_adapter.hwpx_core.grid import GridEntry, iter_grid, table_shape
|
|
30
|
+
from document_adapter.hwpx_core.package import HwpxPackage
|
|
31
|
+
from document_adapter.hwpx_core.paragraph import (
|
|
32
|
+
cell_paragraph_texts,
|
|
33
|
+
cell_paragraphs,
|
|
34
|
+
cell_text,
|
|
35
|
+
nested_tables,
|
|
36
|
+
paragraph_text,
|
|
37
|
+
set_paragraph_text,
|
|
38
|
+
write_cell,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
__all__ = [
|
|
42
|
+
"HC_NS",
|
|
43
|
+
"HH_NS",
|
|
44
|
+
"HP_NS",
|
|
45
|
+
"HS_NS",
|
|
46
|
+
"OPF_NS",
|
|
47
|
+
"HP_CELL_ADDR",
|
|
48
|
+
"HP_CELL_SPAN",
|
|
49
|
+
"HP_P",
|
|
50
|
+
"HP_RUN",
|
|
51
|
+
"HP_SUBLIST",
|
|
52
|
+
"HP_T",
|
|
53
|
+
"HP_TBL",
|
|
54
|
+
"HP_TC",
|
|
55
|
+
"HP_TR",
|
|
56
|
+
"GridEntry",
|
|
57
|
+
"HwpxPackage",
|
|
58
|
+
"iter_grid",
|
|
59
|
+
"table_shape",
|
|
60
|
+
"cell_paragraph_texts",
|
|
61
|
+
"cell_paragraphs",
|
|
62
|
+
"cell_text",
|
|
63
|
+
"nested_tables",
|
|
64
|
+
"paragraph_text",
|
|
65
|
+
"set_paragraph_text",
|
|
66
|
+
"write_cell",
|
|
67
|
+
]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""HWPX XML 네임스페이스 상수."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
HP_NS = "http://www.hancom.co.kr/hwpml/2011/paragraph"
|
|
5
|
+
HC_NS = "http://www.hancom.co.kr/hwpml/2011/core"
|
|
6
|
+
HH_NS = "http://www.hancom.co.kr/hwpml/2011/head"
|
|
7
|
+
HS_NS = "http://www.hancom.co.kr/hwpml/2011/section"
|
|
8
|
+
OPF_NS = "http://www.idpf.org/2007/opf/"
|
|
9
|
+
|
|
10
|
+
HP_P = f"{{{HP_NS}}}p"
|
|
11
|
+
HP_RUN = f"{{{HP_NS}}}run"
|
|
12
|
+
HP_T = f"{{{HP_NS}}}t"
|
|
13
|
+
HP_TBL = f"{{{HP_NS}}}tbl"
|
|
14
|
+
HP_TR = f"{{{HP_NS}}}tr"
|
|
15
|
+
HP_TC = f"{{{HP_NS}}}tc"
|
|
16
|
+
HP_CELL_ADDR = f"{{{HP_NS}}}cellAddr"
|
|
17
|
+
HP_CELL_SPAN = f"{{{HP_NS}}}cellSpan"
|
|
18
|
+
HP_SUBLIST = f"{{{HP_NS}}}subList"
|
|
19
|
+
HS_SEC = f"{{{HS_NS}}}sec"
|
|
20
|
+
|
|
21
|
+
NAMESPACES = {
|
|
22
|
+
"hp": HP_NS,
|
|
23
|
+
"hc": HC_NS,
|
|
24
|
+
"hh": HH_NS,
|
|
25
|
+
"hs": HS_NS,
|
|
26
|
+
"opf": OPF_NS,
|
|
27
|
+
}
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""<hp:tbl> 에서 병합 셀 인식된 logical grid를 생성.
|
|
2
|
+
|
|
3
|
+
python-hwpx의 ``Table.iter_grid()``를 대체. xgen-doc2chunk의 ``_build_cell_grid``
|
|
4
|
+
패턴을 차용하되, **읽기만이 아니라 쓰기도 지원**하기 위해 각 ``GridEntry``가
|
|
5
|
+
anchor cell의 lxml ``<hp:tc>`` Element 자체를 노출한다.
|
|
6
|
+
|
|
7
|
+
HWPX 표 구조:
|
|
8
|
+
<hp:tbl rowCnt colCnt>
|
|
9
|
+
<hp:tr>+
|
|
10
|
+
<hp:tc>+
|
|
11
|
+
<hp:cellAddr rowAddr colAddr>
|
|
12
|
+
<hp:cellSpan rowSpan colSpan> (병합 시만 존재 또는 >1)
|
|
13
|
+
<hp:subList> → <hp:p> → <hp:run> → <hp:t>
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Iterator
|
|
19
|
+
|
|
20
|
+
from lxml import etree
|
|
21
|
+
|
|
22
|
+
from document_adapter.hwpx_core.constants import HP_CELL_ADDR, HP_CELL_SPAN, HP_TC, HP_TR
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class GridEntry:
|
|
27
|
+
"""Logical grid의 한 슬롯.
|
|
28
|
+
|
|
29
|
+
- ``is_anchor=True``: 이 (row, col)이 셀의 anchor 좌표
|
|
30
|
+
- ``is_anchor=False``: 병합된 셀에 덮여 있는 non-anchor 슬롯. ``anchor``가 앵커 좌표.
|
|
31
|
+
``cell_element``는 항상 anchor 셀의 ``<hp:tc>``.
|
|
32
|
+
"""
|
|
33
|
+
row: int
|
|
34
|
+
column: int
|
|
35
|
+
is_anchor: bool
|
|
36
|
+
anchor: tuple[int, int]
|
|
37
|
+
span: tuple[int, int] # (rowspan, colspan)
|
|
38
|
+
cell_element: etree._Element
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _parse_cell_position(tc: etree._Element) -> tuple[int, int]:
|
|
42
|
+
addr = tc.find(HP_CELL_ADDR)
|
|
43
|
+
if addr is None:
|
|
44
|
+
return 0, 0
|
|
45
|
+
try:
|
|
46
|
+
row = int(addr.get("rowAddr", "0"))
|
|
47
|
+
except (TypeError, ValueError):
|
|
48
|
+
row = 0
|
|
49
|
+
try:
|
|
50
|
+
col = int(addr.get("colAddr", "0"))
|
|
51
|
+
except (TypeError, ValueError):
|
|
52
|
+
col = 0
|
|
53
|
+
return row, col
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _parse_cell_span(tc: etree._Element) -> tuple[int, int]:
|
|
57
|
+
span = tc.find(HP_CELL_SPAN)
|
|
58
|
+
if span is None:
|
|
59
|
+
return 1, 1
|
|
60
|
+
try:
|
|
61
|
+
rs = int(span.get("rowSpan", "1"))
|
|
62
|
+
except (TypeError, ValueError):
|
|
63
|
+
rs = 1
|
|
64
|
+
try:
|
|
65
|
+
cs = int(span.get("colSpan", "1"))
|
|
66
|
+
except (TypeError, ValueError):
|
|
67
|
+
cs = 1
|
|
68
|
+
return max(1, rs), max(1, cs)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def table_shape(tbl: etree._Element) -> tuple[int, int]:
|
|
72
|
+
"""표의 (rows, cols). rowCnt/colCnt 속성이 없으면 앵커 좌표로 추정."""
|
|
73
|
+
rows = _safe_int(tbl.get("rowCnt"))
|
|
74
|
+
cols = _safe_int(tbl.get("colCnt"))
|
|
75
|
+
if rows > 0 and cols > 0:
|
|
76
|
+
return rows, cols
|
|
77
|
+
|
|
78
|
+
max_row = -1
|
|
79
|
+
max_col = -1
|
|
80
|
+
for tr in tbl.findall(HP_TR):
|
|
81
|
+
for tc in tr.findall(HP_TC):
|
|
82
|
+
r, c = _parse_cell_position(tc)
|
|
83
|
+
rs, cs = _parse_cell_span(tc)
|
|
84
|
+
max_row = max(max_row, r + rs - 1)
|
|
85
|
+
max_col = max(max_col, c + cs - 1)
|
|
86
|
+
return max(rows, max_row + 1), max(cols, max_col + 1)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _safe_int(v: str | None) -> int:
|
|
90
|
+
if v is None:
|
|
91
|
+
return 0
|
|
92
|
+
try:
|
|
93
|
+
return int(v)
|
|
94
|
+
except ValueError:
|
|
95
|
+
return 0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def iter_grid(tbl: etree._Element) -> Iterator[GridEntry]:
|
|
99
|
+
"""<hp:tbl> 요소를 병합 셀 인식해 logical grid 순서로 순회.
|
|
100
|
+
|
|
101
|
+
row-major: (0,0), (0,1), ..., (0,C-1), (1,0), ...
|
|
102
|
+
"""
|
|
103
|
+
rows, cols = table_shape(tbl)
|
|
104
|
+
if rows <= 0 or cols <= 0:
|
|
105
|
+
return
|
|
106
|
+
|
|
107
|
+
# 1) anchor 좌표 → {span, cell_element}
|
|
108
|
+
anchors: dict[tuple[int, int], tuple[tuple[int, int], etree._Element]] = {}
|
|
109
|
+
for tr in tbl.findall(HP_TR):
|
|
110
|
+
for tc in tr.findall(HP_TC):
|
|
111
|
+
r, c = _parse_cell_position(tc)
|
|
112
|
+
span = _parse_cell_span(tc)
|
|
113
|
+
anchors[(r, c)] = (span, tc)
|
|
114
|
+
|
|
115
|
+
# 2) 각 (row, col)이 어느 anchor에 속하는지 역산 매핑
|
|
116
|
+
owner: dict[tuple[int, int], tuple[int, int]] = {}
|
|
117
|
+
for (ar, ac), (span, _tc) in anchors.items():
|
|
118
|
+
rs, cs = span
|
|
119
|
+
for dr in range(rs):
|
|
120
|
+
for dc in range(cs):
|
|
121
|
+
slot = (ar + dr, ac + dc)
|
|
122
|
+
# 여러 anchor가 같은 slot을 주장하면 가장 가까운(= 자기 자신) 우선
|
|
123
|
+
if slot in owner:
|
|
124
|
+
# 이미 자기 자신으로 설정됐다면 유지
|
|
125
|
+
continue
|
|
126
|
+
owner[slot] = (ar, ac)
|
|
127
|
+
|
|
128
|
+
# 3) row-major 순회
|
|
129
|
+
for r in range(rows):
|
|
130
|
+
for c in range(cols):
|
|
131
|
+
slot = (r, c)
|
|
132
|
+
anchor_coord = owner.get(slot)
|
|
133
|
+
if anchor_coord is None:
|
|
134
|
+
# grid에 선언되지 않은 슬롯 (손상 문서). 빈 앵커 취급.
|
|
135
|
+
continue
|
|
136
|
+
span, cell_el = anchors[anchor_coord]
|
|
137
|
+
is_anchor = (anchor_coord == slot)
|
|
138
|
+
yield GridEntry(
|
|
139
|
+
row=r,
|
|
140
|
+
column=c,
|
|
141
|
+
is_anchor=is_anchor,
|
|
142
|
+
anchor=anchor_coord,
|
|
143
|
+
span=span,
|
|
144
|
+
cell_element=cell_el,
|
|
145
|
+
)
|