document-adapter 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. {document_adapter-0.1.2 → document_adapter-0.2.0}/PKG-INFO +1 -1
  2. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/base.py +21 -2
  3. document_adapter-0.2.0/document_adapter/hwpx_adapter.py +246 -0
  4. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/tools.py +23 -3
  5. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/PKG-INFO +1 -1
  6. {document_adapter-0.1.2 → document_adapter-0.2.0}/pyproject.toml +1 -1
  7. document_adapter-0.2.0/tests/test_smoke.py +687 -0
  8. document_adapter-0.1.2/document_adapter/hwpx_adapter.py +0 -126
  9. document_adapter-0.1.2/tests/test_smoke.py +0 -324
  10. {document_adapter-0.1.2 → document_adapter-0.2.0}/LICENSE +0 -0
  11. {document_adapter-0.1.2 → document_adapter-0.2.0}/README.md +0 -0
  12. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/__init__.py +0 -0
  13. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/docx_adapter.py +0 -0
  14. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/mcp_server.py +0 -0
  15. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter/pptx_adapter.py +0 -0
  16. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/SOURCES.txt +0 -0
  17. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/dependency_links.txt +0 -0
  18. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/entry_points.txt +0 -0
  19. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/requires.txt +0 -0
  20. {document_adapter-0.1.2 → document_adapter-0.2.0}/document_adapter.egg-info/top_level.txt +0 -0
  21. {document_adapter-0.1.2 → document_adapter-0.2.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: document-adapter
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
5
  Author-email: Son Seongjun <sonsj97@plateer.com>
6
6
  License: MIT
@@ -13,14 +13,31 @@ from pathlib import Path
13
13
  from typing import Any
14
14
 
15
15
 
16
+ @dataclass
17
+ class MergeInfo:
18
+ """병합 셀 정보. anchor=(row,col)에서 span=(rows,cols)만큼 병합."""
19
+ anchor: tuple[int, int]
20
+ span: tuple[int, int]
21
+
22
+ def to_dict(self) -> dict[str, Any]:
23
+ return {"anchor": list(self.anchor), "span": list(self.span)}
24
+
25
+
16
26
  @dataclass
17
27
  class TableSchema:
18
- """표 한 개의 구조 (LLM에게 넘길 형태)."""
28
+ """표 한 개의 구조 (LLM에게 넘길 형태).
29
+
30
+ preview는 logical grid(rows × cols) 형태. 병합된 non-anchor 슬롯은 ``None``.
31
+ merges는 span>1x1인 앵커 목록 (LLM이 병합 구조를 재구성할 수 있게).
32
+ parent_path는 중첩 테이블 위치 표시 (예: ``"tables[0].cell(1,2)"``).
33
+ """
19
34
  index: int
20
35
  rows: int
21
36
  cols: int
22
- preview: list[list[str]]
37
+ preview: list[list[str | None]]
23
38
  location: str | None = None
39
+ merges: list[MergeInfo] = field(default_factory=list)
40
+ parent_path: str | None = None
24
41
 
25
42
  def to_dict(self) -> dict[str, Any]:
26
43
  return {
@@ -28,7 +45,9 @@ class TableSchema:
28
45
  "rows": self.rows,
29
46
  "cols": self.cols,
30
47
  "location": self.location,
48
+ "parent_path": self.parent_path,
31
49
  "preview": self.preview,
50
+ "merges": [m.to_dict() for m in self.merges],
32
51
  }
33
52
 
34
53
 
@@ -0,0 +1,246 @@
1
+ """HWPX 어댑터: python-hwpx 기반.
2
+
3
+ 버그 회피:
4
+ - set_cell_text()는 빈 셀에서 lxml/ElementTree 혼용 에러가 발생 (v2.9.0) →
5
+ cell.paragraphs[0].text 직접 할당으로 우회
6
+ - replace_text_in_runs()는 한글 공백이 run으로 쪼개질 때 매칭 실패 →
7
+ 위치 기반 편집을 권장
8
+
9
+ 표 구조:
10
+ - iter_grid()로 병합 셀(rowSpan/colSpan)을 인식해 logical grid를 구성
11
+ - 셀 내부 중첩 테이블은 flat DFS로 인덱싱, parent_path로 위치 표시
12
+
13
+ 부가:
14
+ - manifest fallback 로그가 기본적으로 매우 시끄러움 → logging 레벨 조정
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import logging
19
+ import re
20
+ import warnings
21
+ from pathlib import Path
22
+ from typing import Any, Iterator
23
+
24
+ # 경고성 로그 억제 (manifest fallback 등)
25
+ logging.getLogger("hwpx").setLevel(logging.ERROR)
26
+
27
+ from hwpx.document import HwpxDocument
28
+
29
+ from .base import DocumentAdapter, MergeInfo, TableSchema
30
+
31
+ TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
32
+
33
+ _HP_NS = "http://www.hancom.co.kr/hwpml/2011/paragraph"
34
+ _HP_T = f"{{{_HP_NS}}}t"
35
+ _HP_RUN = f"{{{_HP_NS}}}run"
36
+ _HP_TBL = f"{{{_HP_NS}}}tbl"
37
+
38
+
39
+ class HwpxAdapter(DocumentAdapter):
40
+ format = "hwpx"
41
+
42
+ def _open(self) -> None:
43
+ self._doc = HwpxDocument.open(self.path)
44
+
45
+ def save(self, path: Path | str | None = None) -> Path:
46
+ target = Path(path) if path else self.path
47
+ self._doc.save_to_path(target)
48
+ self.path = target
49
+ return target
50
+
51
+ def close(self) -> None:
52
+ self._doc.close()
53
+
54
+ # ---- helpers ----
55
+ def _iter_tables(self) -> Iterator[tuple[int, Any, str]]:
56
+ """(flat_index, table, parent_path) 순회. 중첩 테이블까지 DFS."""
57
+ idx_counter = [0]
58
+
59
+ def walk(tbl, parent_path: str) -> Iterator[tuple[int, Any, str]]:
60
+ current_idx = idx_counter[0]
61
+ idx_counter[0] += 1
62
+ yield current_idx, tbl, parent_path
63
+ # 중첩 테이블: 각 앵커 셀의 tables만 내려간다 (같은 물리 셀 중복 방지)
64
+ seen_cell_ids: set[int] = set()
65
+ for entry in tbl.iter_grid():
66
+ if not entry.is_anchor:
67
+ continue
68
+ cell = entry.cell
69
+ cell_key = id(cell.element)
70
+ if cell_key in seen_cell_ids:
71
+ continue
72
+ seen_cell_ids.add(cell_key)
73
+ for child_tbl in cell.tables:
74
+ child_parent = (
75
+ f"{parent_path}.tables[{current_idx}].cell"
76
+ f"({entry.anchor[0]},{entry.anchor[1]})"
77
+ )
78
+ yield from walk(child_tbl, child_parent)
79
+
80
+ for section in self._doc.sections:
81
+ for para in section.paragraphs:
82
+ for tbl in para.tables:
83
+ yield from walk(tbl, "")
84
+
85
+ def _get_table(self, table_index: int):
86
+ for idx, tbl, _ in self._iter_tables():
87
+ if idx == table_index:
88
+ return tbl
89
+ raise IndexError(f"HWPX table index {table_index} not found")
90
+
91
+ @staticmethod
92
+ def _cell_text(cell) -> str:
93
+ """셀의 직접 텍스트만 추출 (중첩 테이블의 텍스트는 제외).
94
+
95
+ python-hwpx의 ``paragraph.text``는 ``.//hp:t``로 descendant를 훑어
96
+ 중첩 테이블 내부 텍스트까지 흡수한다. LLM에게 이게 그대로 노출되면
97
+ 외부 셀의 내용이 중첩 테이블 내용과 뒤섞인 것처럼 보인다.
98
+ 따라서 run의 직접 자식 ``<hp:t>``만 읽는다 (중첩된 ``<hp:tbl>`` 서브트리는 자연히 제외).
99
+ """
100
+ parts: list[str] = []
101
+ for para in cell.paragraphs:
102
+ for run in para.element.findall(_HP_RUN):
103
+ for t in run.findall(_HP_T):
104
+ if t.text:
105
+ parts.append(t.text)
106
+ return "".join(parts).strip()
107
+
108
+ # ---- inspection ----
109
+ def get_placeholders(self) -> list[str]:
110
+ text = self._doc.export_text()
111
+ return sorted(set(TAG_PATTERN.findall(text)))
112
+
113
+ def get_tables(self, min_rows: int = 1, min_cols: int = 1,
114
+ preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
115
+ schemas: list[TableSchema] = []
116
+ for idx, tbl, parent_path in self._iter_tables():
117
+ rows, cols = tbl.row_count, tbl.column_count
118
+ if rows < min_rows or cols < min_cols:
119
+ continue
120
+
121
+ visible_rows = min(rows, preview_rows)
122
+ # 기본 프리뷰 grid: None 채운 뒤 앵커 위치에만 텍스트 주입
123
+ preview: list[list[str | None]] = [
124
+ [None for _ in range(cols)] for _ in range(visible_rows)
125
+ ]
126
+ merges: list[MergeInfo] = []
127
+ seen_anchors: set[tuple[int, int]] = set()
128
+
129
+ for entry in tbl.iter_grid():
130
+ if entry.anchor in seen_anchors:
131
+ # 같은 앵커는 한 번만
132
+ if entry.row < visible_rows and entry.is_anchor:
133
+ pass # preview는 이미 채웠으므로 skip
134
+ continue
135
+ if entry.is_anchor:
136
+ seen_anchors.add(entry.anchor)
137
+ if entry.row < visible_rows:
138
+ text = self._cell_text(entry.cell)
139
+ preview[entry.row][entry.column] = text[:max_cell_len]
140
+ if entry.span != (1, 1):
141
+ merges.append(MergeInfo(anchor=entry.anchor, span=entry.span))
142
+
143
+ schemas.append(
144
+ TableSchema(
145
+ index=idx,
146
+ rows=rows,
147
+ cols=cols,
148
+ preview=preview,
149
+ merges=merges,
150
+ parent_path=parent_path or None,
151
+ )
152
+ )
153
+ return schemas
154
+
155
+ # ---- editing ----
156
+ def render_template(self, context: dict[str, Any]) -> None:
157
+ """본문 + 표 셀의 {{key}}를 paragraph 단위로 치환.
158
+
159
+ 병합 셀의 경우 같은 앵커의 paragraph를 여러 logical 좌표에서 참조하게 되므로,
160
+ is_anchor 위치만 방문해 중복 치환을 피한다.
161
+ """
162
+
163
+ def substitute(para) -> None:
164
+ text = para.text
165
+ if TAG_PATTERN.search(text):
166
+ para.text = TAG_PATTERN.sub(
167
+ lambda m: str(context.get(m.group(1), m.group(0))), text
168
+ )
169
+
170
+ # 본문
171
+ for section in self._doc.sections:
172
+ for para in section.paragraphs:
173
+ substitute(para)
174
+ # 표 셀 (중첩 테이블 포함; _iter_tables가 DFS)
175
+ for _, tbl, _ in self._iter_tables():
176
+ for entry in tbl.iter_grid():
177
+ if not entry.is_anchor:
178
+ continue
179
+ for para in entry.cell.paragraphs:
180
+ substitute(para)
181
+
182
+ def set_cell(
183
+ self,
184
+ table_index: int,
185
+ row: int,
186
+ col: int,
187
+ value: str,
188
+ *,
189
+ allow_merge_redirect: bool = False,
190
+ ) -> str:
191
+ """셀 값 교체. 원래 값 반환.
192
+
193
+ 병합 셀(non-anchor) 좌표로 호출하면 기본적으로 ``ValueError``를 발생시킨다.
194
+ 이는 LLM이 병합 구조를 잘못 이해하고 엉뚱한 앵커를 덮어쓰는 것을 방지한다.
195
+ ``allow_merge_redirect=True``를 주면 앵커로 자동 리디렉트하고 경고만 남긴다.
196
+
197
+ set_cell_text 버그 우회: paragraph.text 직접 할당.
198
+ """
199
+ tbl = self._get_table(table_index)
200
+ if row < 0 or col < 0 or row >= tbl.row_count or col >= tbl.column_count:
201
+ raise IndexError(
202
+ f"cell ({row},{col}) out of bounds for table {table_index} "
203
+ f"({tbl.row_count}x{tbl.column_count})"
204
+ )
205
+
206
+ grid_entry = None
207
+ for entry in tbl.iter_grid():
208
+ if (entry.row, entry.column) == (row, col):
209
+ grid_entry = entry
210
+ break
211
+ if grid_entry is None:
212
+ raise IndexError(
213
+ f"cell ({row},{col}) does not resolve to any physical cell"
214
+ )
215
+
216
+ if not grid_entry.is_anchor:
217
+ anchor_r, anchor_c = grid_entry.anchor
218
+ if not allow_merge_redirect:
219
+ raise ValueError(
220
+ f"cell ({row},{col}) is part of a merged region anchored at "
221
+ f"({anchor_r},{anchor_c}) span={grid_entry.span}. "
222
+ f"Write to the anchor coordinate, or pass "
223
+ f"allow_merge_redirect=True."
224
+ )
225
+ warnings.warn(
226
+ f"set_cell({row},{col}) redirected to merge anchor "
227
+ f"({anchor_r},{anchor_c})",
228
+ stacklevel=2,
229
+ )
230
+
231
+ cell = grid_entry.cell
232
+ paragraphs = list(cell.paragraphs)
233
+ old = self._cell_text(cell)
234
+ if paragraphs:
235
+ paragraphs[0].text = value
236
+ for p in paragraphs[1:]:
237
+ p.text = ""
238
+ return old
239
+
240
+ def append_row(self, table_index: int, values: list[str]) -> None:
241
+ """python-hwpx에는 표준 add_row API가 없음.
242
+ 대안: 템플릿에 충분한 빈 행을 미리 만들고 set_cell로 채우는 전략."""
243
+ raise NotImplementedError(
244
+ "HWPX는 python-hwpx에 동적 행 추가 공식 API가 없음. "
245
+ "템플릿에 여분 행을 두고 set_cell로 채우는 방식을 권장."
246
+ )
@@ -69,7 +69,10 @@ TOOL_DEFINITIONS: list[dict[str, Any]] = [
69
69
  "name": "set_cell",
70
70
  "description": (
71
71
  "특정 표의 셀 값을 교체한다. table_index는 inspect_document의 tables 배열 인덱스. "
72
- "PPTX는 슬라이드 경계와 무관한 전역 index."
72
+ "PPTX는 슬라이드 경계와 무관한 전역 index. "
73
+ "HWPX 병합 셀 주의: inspect_document의 tables[i].merges에 나온 anchor 좌표로만 "
74
+ "수정 가능. 병합 영역 내부의 non-anchor 좌표로 호출하면 ValueError가 발생하며, "
75
+ "preview의 해당 슬롯은 null로 표시된다."
73
76
  ),
74
77
  "input_schema": {
75
78
  "type": "object",
@@ -83,6 +86,14 @@ TOOL_DEFINITIONS: list[dict[str, Any]] = [
83
86
  "type": "string",
84
87
  "description": "생략 시 원본 덮어쓰기",
85
88
  },
89
+ "allow_merge_redirect": {
90
+ "type": "boolean",
91
+ "description": (
92
+ "HWPX 전용. true면 병합 영역 non-anchor 좌표 호출 시 "
93
+ "앵커로 자동 리디렉트(권장 X, 구조 잘못 이해한 호출을 숨김)."
94
+ ),
95
+ "default": False,
96
+ },
86
97
  },
87
98
  "required": ["path", "table_index", "row", "col", "value"],
88
99
  },
@@ -162,14 +173,23 @@ def render_template(path: str, context: dict[str, Any],
162
173
 
163
174
 
164
175
  def set_cell(path: str, table_index: int, row: int, col: int, value: str,
165
- output_path: str | None = None) -> dict[str, Any]:
176
+ output_path: str | None = None,
177
+ allow_merge_redirect: bool = False) -> dict[str, Any]:
166
178
  target = Path(output_path) if output_path else Path(path)
167
179
  if output_path and Path(path) != target:
168
180
  shutil.copy2(path, target)
169
181
 
170
182
  doc = load(target)
171
183
  try:
172
- old = doc.set_cell(table_index, row, col, value)
184
+ # allow_merge_redirect는 HWPX 어댑터만 지원하므로 키워드 인자로 전달 시도하고
185
+ # 포맷이 지원 안 하면 무시.
186
+ try:
187
+ old = doc.set_cell(
188
+ table_index, row, col, value,
189
+ allow_merge_redirect=allow_merge_redirect,
190
+ )
191
+ except TypeError:
192
+ old = doc.set_cell(table_index, row, col, value)
173
193
  doc.save()
174
194
  finally:
175
195
  doc.close()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: document-adapter
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
5
  Author-email: Son Seongjun <sonsj97@plateer.com>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "document-adapter"
7
- version = "0.1.2"
7
+ version = "0.2.0"
8
8
  description = "LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"