document-adapter 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {document_adapter-0.1.2 → document_adapter-0.2.0}/PKG-INFO +1 -1
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/base.py +21 -2
- document_adapter-0.2.0/document_adapter/hwpx_adapter.py +246 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/tools.py +23 -3
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/PKG-INFO +1 -1
- {document_adapter-0.1.2 → document_adapter-0.2.0}/pyproject.toml +1 -1
- document_adapter-0.2.0/tests/test_smoke.py +687 -0
- document_adapter-0.1.2/document_adapter/hwpx_adapter.py +0 -126
- document_adapter-0.1.2/tests/test_smoke.py +0 -324
- {document_adapter-0.1.2 → document_adapter-0.2.0}/LICENSE +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/README.md +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/__init__.py +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/docx_adapter.py +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/mcp_server.py +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/pptx_adapter.py +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/SOURCES.txt +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/dependency_links.txt +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/entry_points.txt +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/requires.txt +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/top_level.txt +0 -0
- {document_adapter-0.1.2 → document_adapter-0.2.0}/setup.cfg +0 -0
|
@@ -13,14 +13,31 @@ from pathlib import Path
|
|
|
13
13
|
from typing import Any
|
|
14
14
|
|
|
15
15
|
|
|
16
|
+
@dataclass
|
|
17
|
+
class MergeInfo:
|
|
18
|
+
"""병합 셀 정보. anchor=(row,col)에서 span=(rows,cols)만큼 병합."""
|
|
19
|
+
anchor: tuple[int, int]
|
|
20
|
+
span: tuple[int, int]
|
|
21
|
+
|
|
22
|
+
def to_dict(self) -> dict[str, Any]:
|
|
23
|
+
return {"anchor": list(self.anchor), "span": list(self.span)}
|
|
24
|
+
|
|
25
|
+
|
|
16
26
|
@dataclass
|
|
17
27
|
class TableSchema:
|
|
18
|
-
"""표 한 개의 구조 (LLM에게 넘길 형태).
|
|
28
|
+
"""표 한 개의 구조 (LLM에게 넘길 형태).
|
|
29
|
+
|
|
30
|
+
preview는 logical grid(rows × cols) 형태. 병합된 non-anchor 슬롯은 ``None``.
|
|
31
|
+
merges는 span>1x1인 앵커 목록 (LLM이 병합 구조를 재구성할 수 있게).
|
|
32
|
+
parent_path는 중첩 테이블 위치 표시 (예: ``"tables[0].cell(1,2)"``).
|
|
33
|
+
"""
|
|
19
34
|
index: int
|
|
20
35
|
rows: int
|
|
21
36
|
cols: int
|
|
22
|
-
preview: list[list[str]]
|
|
37
|
+
preview: list[list[str | None]]
|
|
23
38
|
location: str | None = None
|
|
39
|
+
merges: list[MergeInfo] = field(default_factory=list)
|
|
40
|
+
parent_path: str | None = None
|
|
24
41
|
|
|
25
42
|
def to_dict(self) -> dict[str, Any]:
|
|
26
43
|
return {
|
|
@@ -28,7 +45,9 @@ class TableSchema:
|
|
|
28
45
|
"rows": self.rows,
|
|
29
46
|
"cols": self.cols,
|
|
30
47
|
"location": self.location,
|
|
48
|
+
"parent_path": self.parent_path,
|
|
31
49
|
"preview": self.preview,
|
|
50
|
+
"merges": [m.to_dict() for m in self.merges],
|
|
32
51
|
}
|
|
33
52
|
|
|
34
53
|
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""HWPX 어댑터: python-hwpx 기반.
|
|
2
|
+
|
|
3
|
+
버그 회피:
|
|
4
|
+
- set_cell_text()는 빈 셀에서 lxml/ElementTree 혼용 에러가 발생 (v2.9.0) →
|
|
5
|
+
cell.paragraphs[0].text 직접 할당으로 우회
|
|
6
|
+
- replace_text_in_runs()는 한글 공백이 run으로 쪼개질 때 매칭 실패 →
|
|
7
|
+
위치 기반 편집을 권장
|
|
8
|
+
|
|
9
|
+
표 구조:
|
|
10
|
+
- iter_grid()로 병합 셀(rowSpan/colSpan)을 인식해 logical grid를 구성
|
|
11
|
+
- 셀 내부 중첩 테이블은 flat DFS로 인덱싱, parent_path로 위치 표시
|
|
12
|
+
|
|
13
|
+
부가:
|
|
14
|
+
- manifest fallback 로그가 기본적으로 매우 시끄러움 → logging 레벨 조정
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import logging
|
|
19
|
+
import re
|
|
20
|
+
import warnings
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, Iterator
|
|
23
|
+
|
|
24
|
+
# 경고성 로그 억제 (manifest fallback 등)
|
|
25
|
+
logging.getLogger("hwpx").setLevel(logging.ERROR)
|
|
26
|
+
|
|
27
|
+
from hwpx.document import HwpxDocument
|
|
28
|
+
|
|
29
|
+
from .base import DocumentAdapter, MergeInfo, TableSchema
|
|
30
|
+
|
|
31
|
+
TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
|
|
32
|
+
|
|
33
|
+
_HP_NS = "http://www.hancom.co.kr/hwpml/2011/paragraph"
|
|
34
|
+
_HP_T = f"{{{_HP_NS}}}t"
|
|
35
|
+
_HP_RUN = f"{{{_HP_NS}}}run"
|
|
36
|
+
_HP_TBL = f"{{{_HP_NS}}}tbl"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class HwpxAdapter(DocumentAdapter):
|
|
40
|
+
format = "hwpx"
|
|
41
|
+
|
|
42
|
+
def _open(self) -> None:
|
|
43
|
+
self._doc = HwpxDocument.open(self.path)
|
|
44
|
+
|
|
45
|
+
def save(self, path: Path | str | None = None) -> Path:
|
|
46
|
+
target = Path(path) if path else self.path
|
|
47
|
+
self._doc.save_to_path(target)
|
|
48
|
+
self.path = target
|
|
49
|
+
return target
|
|
50
|
+
|
|
51
|
+
def close(self) -> None:
|
|
52
|
+
self._doc.close()
|
|
53
|
+
|
|
54
|
+
# ---- helpers ----
|
|
55
|
+
def _iter_tables(self) -> Iterator[tuple[int, Any, str]]:
|
|
56
|
+
"""(flat_index, table, parent_path) 순회. 중첩 테이블까지 DFS."""
|
|
57
|
+
idx_counter = [0]
|
|
58
|
+
|
|
59
|
+
def walk(tbl, parent_path: str) -> Iterator[tuple[int, Any, str]]:
|
|
60
|
+
current_idx = idx_counter[0]
|
|
61
|
+
idx_counter[0] += 1
|
|
62
|
+
yield current_idx, tbl, parent_path
|
|
63
|
+
# 중첩 테이블: 각 앵커 셀의 tables만 내려간다 (같은 물리 셀 중복 방지)
|
|
64
|
+
seen_cell_ids: set[int] = set()
|
|
65
|
+
for entry in tbl.iter_grid():
|
|
66
|
+
if not entry.is_anchor:
|
|
67
|
+
continue
|
|
68
|
+
cell = entry.cell
|
|
69
|
+
cell_key = id(cell.element)
|
|
70
|
+
if cell_key in seen_cell_ids:
|
|
71
|
+
continue
|
|
72
|
+
seen_cell_ids.add(cell_key)
|
|
73
|
+
for child_tbl in cell.tables:
|
|
74
|
+
child_parent = (
|
|
75
|
+
f"{parent_path}.tables[{current_idx}].cell"
|
|
76
|
+
f"({entry.anchor[0]},{entry.anchor[1]})"
|
|
77
|
+
)
|
|
78
|
+
yield from walk(child_tbl, child_parent)
|
|
79
|
+
|
|
80
|
+
for section in self._doc.sections:
|
|
81
|
+
for para in section.paragraphs:
|
|
82
|
+
for tbl in para.tables:
|
|
83
|
+
yield from walk(tbl, "")
|
|
84
|
+
|
|
85
|
+
def _get_table(self, table_index: int):
|
|
86
|
+
for idx, tbl, _ in self._iter_tables():
|
|
87
|
+
if idx == table_index:
|
|
88
|
+
return tbl
|
|
89
|
+
raise IndexError(f"HWPX table index {table_index} not found")
|
|
90
|
+
|
|
91
|
+
@staticmethod
|
|
92
|
+
def _cell_text(cell) -> str:
|
|
93
|
+
"""셀의 직접 텍스트만 추출 (중첩 테이블의 텍스트는 제외).
|
|
94
|
+
|
|
95
|
+
python-hwpx의 ``paragraph.text``는 ``.//hp:t``로 descendant를 훑어
|
|
96
|
+
중첩 테이블 내부 텍스트까지 흡수한다. LLM에게 이게 그대로 노출되면
|
|
97
|
+
외부 셀의 내용이 중첩 테이블 내용과 뒤섞인 것처럼 보인다.
|
|
98
|
+
따라서 run의 직접 자식 ``<hp:t>``만 읽는다 (중첩된 ``<hp:tbl>`` 서브트리는 자연히 제외).
|
|
99
|
+
"""
|
|
100
|
+
parts: list[str] = []
|
|
101
|
+
for para in cell.paragraphs:
|
|
102
|
+
for run in para.element.findall(_HP_RUN):
|
|
103
|
+
for t in run.findall(_HP_T):
|
|
104
|
+
if t.text:
|
|
105
|
+
parts.append(t.text)
|
|
106
|
+
return "".join(parts).strip()
|
|
107
|
+
|
|
108
|
+
# ---- inspection ----
|
|
109
|
+
def get_placeholders(self) -> list[str]:
|
|
110
|
+
text = self._doc.export_text()
|
|
111
|
+
return sorted(set(TAG_PATTERN.findall(text)))
|
|
112
|
+
|
|
113
|
+
def get_tables(self, min_rows: int = 1, min_cols: int = 1,
|
|
114
|
+
preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
|
|
115
|
+
schemas: list[TableSchema] = []
|
|
116
|
+
for idx, tbl, parent_path in self._iter_tables():
|
|
117
|
+
rows, cols = tbl.row_count, tbl.column_count
|
|
118
|
+
if rows < min_rows or cols < min_cols:
|
|
119
|
+
continue
|
|
120
|
+
|
|
121
|
+
visible_rows = min(rows, preview_rows)
|
|
122
|
+
# 기본 프리뷰 grid: None 채운 뒤 앵커 위치에만 텍스트 주입
|
|
123
|
+
preview: list[list[str | None]] = [
|
|
124
|
+
[None for _ in range(cols)] for _ in range(visible_rows)
|
|
125
|
+
]
|
|
126
|
+
merges: list[MergeInfo] = []
|
|
127
|
+
seen_anchors: set[tuple[int, int]] = set()
|
|
128
|
+
|
|
129
|
+
for entry in tbl.iter_grid():
|
|
130
|
+
if entry.anchor in seen_anchors:
|
|
131
|
+
# 같은 앵커는 한 번만
|
|
132
|
+
if entry.row < visible_rows and entry.is_anchor:
|
|
133
|
+
pass # preview는 이미 채웠으므로 skip
|
|
134
|
+
continue
|
|
135
|
+
if entry.is_anchor:
|
|
136
|
+
seen_anchors.add(entry.anchor)
|
|
137
|
+
if entry.row < visible_rows:
|
|
138
|
+
text = self._cell_text(entry.cell)
|
|
139
|
+
preview[entry.row][entry.column] = text[:max_cell_len]
|
|
140
|
+
if entry.span != (1, 1):
|
|
141
|
+
merges.append(MergeInfo(anchor=entry.anchor, span=entry.span))
|
|
142
|
+
|
|
143
|
+
schemas.append(
|
|
144
|
+
TableSchema(
|
|
145
|
+
index=idx,
|
|
146
|
+
rows=rows,
|
|
147
|
+
cols=cols,
|
|
148
|
+
preview=preview,
|
|
149
|
+
merges=merges,
|
|
150
|
+
parent_path=parent_path or None,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
return schemas
|
|
154
|
+
|
|
155
|
+
# ---- editing ----
|
|
156
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
157
|
+
"""본문 + 표 셀의 {{key}}를 paragraph 단위로 치환.
|
|
158
|
+
|
|
159
|
+
병합 셀의 경우 같은 앵커의 paragraph를 여러 logical 좌표에서 참조하게 되므로,
|
|
160
|
+
is_anchor 위치만 방문해 중복 치환을 피한다.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
def substitute(para) -> None:
|
|
164
|
+
text = para.text
|
|
165
|
+
if TAG_PATTERN.search(text):
|
|
166
|
+
para.text = TAG_PATTERN.sub(
|
|
167
|
+
lambda m: str(context.get(m.group(1), m.group(0))), text
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
# 본문
|
|
171
|
+
for section in self._doc.sections:
|
|
172
|
+
for para in section.paragraphs:
|
|
173
|
+
substitute(para)
|
|
174
|
+
# 표 셀 (중첩 테이블 포함; _iter_tables가 DFS)
|
|
175
|
+
for _, tbl, _ in self._iter_tables():
|
|
176
|
+
for entry in tbl.iter_grid():
|
|
177
|
+
if not entry.is_anchor:
|
|
178
|
+
continue
|
|
179
|
+
for para in entry.cell.paragraphs:
|
|
180
|
+
substitute(para)
|
|
181
|
+
|
|
182
|
+
def set_cell(
|
|
183
|
+
self,
|
|
184
|
+
table_index: int,
|
|
185
|
+
row: int,
|
|
186
|
+
col: int,
|
|
187
|
+
value: str,
|
|
188
|
+
*,
|
|
189
|
+
allow_merge_redirect: bool = False,
|
|
190
|
+
) -> str:
|
|
191
|
+
"""셀 값 교체. 원래 값 반환.
|
|
192
|
+
|
|
193
|
+
병합 셀(non-anchor) 좌표로 호출하면 기본적으로 ``ValueError``를 발생시킨다.
|
|
194
|
+
이는 LLM이 병합 구조를 잘못 이해하고 엉뚱한 앵커를 덮어쓰는 것을 방지한다.
|
|
195
|
+
``allow_merge_redirect=True``를 주면 앵커로 자동 리디렉트하고 경고만 남긴다.
|
|
196
|
+
|
|
197
|
+
set_cell_text 버그 우회: paragraph.text 직접 할당.
|
|
198
|
+
"""
|
|
199
|
+
tbl = self._get_table(table_index)
|
|
200
|
+
if row < 0 or col < 0 or row >= tbl.row_count or col >= tbl.column_count:
|
|
201
|
+
raise IndexError(
|
|
202
|
+
f"cell ({row},{col}) out of bounds for table {table_index} "
|
|
203
|
+
f"({tbl.row_count}x{tbl.column_count})"
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
grid_entry = None
|
|
207
|
+
for entry in tbl.iter_grid():
|
|
208
|
+
if (entry.row, entry.column) == (row, col):
|
|
209
|
+
grid_entry = entry
|
|
210
|
+
break
|
|
211
|
+
if grid_entry is None:
|
|
212
|
+
raise IndexError(
|
|
213
|
+
f"cell ({row},{col}) does not resolve to any physical cell"
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
if not grid_entry.is_anchor:
|
|
217
|
+
anchor_r, anchor_c = grid_entry.anchor
|
|
218
|
+
if not allow_merge_redirect:
|
|
219
|
+
raise ValueError(
|
|
220
|
+
f"cell ({row},{col}) is part of a merged region anchored at "
|
|
221
|
+
f"({anchor_r},{anchor_c}) span={grid_entry.span}. "
|
|
222
|
+
f"Write to the anchor coordinate, or pass "
|
|
223
|
+
f"allow_merge_redirect=True."
|
|
224
|
+
)
|
|
225
|
+
warnings.warn(
|
|
226
|
+
f"set_cell({row},{col}) redirected to merge anchor "
|
|
227
|
+
f"({anchor_r},{anchor_c})",
|
|
228
|
+
stacklevel=2,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
cell = grid_entry.cell
|
|
232
|
+
paragraphs = list(cell.paragraphs)
|
|
233
|
+
old = self._cell_text(cell)
|
|
234
|
+
if paragraphs:
|
|
235
|
+
paragraphs[0].text = value
|
|
236
|
+
for p in paragraphs[1:]:
|
|
237
|
+
p.text = ""
|
|
238
|
+
return old
|
|
239
|
+
|
|
240
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
241
|
+
"""python-hwpx에는 표준 add_row API가 없음.
|
|
242
|
+
대안: 템플릿에 충분한 빈 행을 미리 만들고 set_cell로 채우는 전략."""
|
|
243
|
+
raise NotImplementedError(
|
|
244
|
+
"HWPX는 python-hwpx에 동적 행 추가 공식 API가 없음. "
|
|
245
|
+
"템플릿에 여분 행을 두고 set_cell로 채우는 방식을 권장."
|
|
246
|
+
)
|
|
@@ -69,7 +69,10 @@ TOOL_DEFINITIONS: list[dict[str, Any]] = [
|
|
|
69
69
|
"name": "set_cell",
|
|
70
70
|
"description": (
|
|
71
71
|
"특정 표의 셀 값을 교체한다. table_index는 inspect_document의 tables 배열 인덱스. "
|
|
72
|
-
"PPTX는 슬라이드 경계와 무관한 전역 index."
|
|
72
|
+
"PPTX는 슬라이드 경계와 무관한 전역 index. "
|
|
73
|
+
"HWPX 병합 셀 주의: inspect_document의 tables[i].merges에 나온 anchor 좌표로만 "
|
|
74
|
+
"수정 가능. 병합 영역 내부의 non-anchor 좌표로 호출하면 ValueError가 발생하며, "
|
|
75
|
+
"preview의 해당 슬롯은 null로 표시된다."
|
|
73
76
|
),
|
|
74
77
|
"input_schema": {
|
|
75
78
|
"type": "object",
|
|
@@ -83,6 +86,14 @@ TOOL_DEFINITIONS: list[dict[str, Any]] = [
|
|
|
83
86
|
"type": "string",
|
|
84
87
|
"description": "생략 시 원본 덮어쓰기",
|
|
85
88
|
},
|
|
89
|
+
"allow_merge_redirect": {
|
|
90
|
+
"type": "boolean",
|
|
91
|
+
"description": (
|
|
92
|
+
"HWPX 전용. true면 병합 영역 non-anchor 좌표 호출 시 "
|
|
93
|
+
"앵커로 자동 리디렉트(권장 X, 구조 잘못 이해한 호출을 숨김)."
|
|
94
|
+
),
|
|
95
|
+
"default": False,
|
|
96
|
+
},
|
|
86
97
|
},
|
|
87
98
|
"required": ["path", "table_index", "row", "col", "value"],
|
|
88
99
|
},
|
|
@@ -162,14 +173,23 @@ def render_template(path: str, context: dict[str, Any],
|
|
|
162
173
|
|
|
163
174
|
|
|
164
175
|
def set_cell(path: str, table_index: int, row: int, col: int, value: str,
|
|
165
|
-
output_path: str | None = None
|
|
176
|
+
output_path: str | None = None,
|
|
177
|
+
allow_merge_redirect: bool = False) -> dict[str, Any]:
|
|
166
178
|
target = Path(output_path) if output_path else Path(path)
|
|
167
179
|
if output_path and Path(path) != target:
|
|
168
180
|
shutil.copy2(path, target)
|
|
169
181
|
|
|
170
182
|
doc = load(target)
|
|
171
183
|
try:
|
|
172
|
-
|
|
184
|
+
# allow_merge_redirect는 HWPX 어댑터만 지원하므로 키워드 인자로 전달 시도하고
|
|
185
|
+
# 포맷이 지원 안 하면 무시.
|
|
186
|
+
try:
|
|
187
|
+
old = doc.set_cell(
|
|
188
|
+
table_index, row, col, value,
|
|
189
|
+
allow_merge_redirect=allow_merge_redirect,
|
|
190
|
+
)
|
|
191
|
+
except TypeError:
|
|
192
|
+
old = doc.set_cell(table_index, row, col, value)
|
|
173
193
|
doc.save()
|
|
174
194
|
finally:
|
|
175
195
|
doc.close()
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "document-adapter"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|