document-adapter 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- document_adapter/__init__.py +45 -0
- document_adapter/base.py +108 -0
- document_adapter/docx_adapter.py +75 -0
- document_adapter/hwpx_adapter.py +126 -0
- document_adapter/mcp_server.py +64 -0
- document_adapter/pptx_adapter.py +109 -0
- document_adapter/tools.py +232 -0
- document_adapter-0.1.0.dist-info/METADATA +252 -0
- document_adapter-0.1.0.dist-info/RECORD +13 -0
- document_adapter-0.1.0.dist-info/WHEEL +5 -0
- document_adapter-0.1.0.dist-info/entry_points.txt +2 -0
- document_adapter-0.1.0.dist-info/licenses/LICENSE +21 -0
- document_adapter-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Document template editing — 통합 어댑터.
|
|
2
|
+
|
|
3
|
+
사용법:
|
|
4
|
+
from document_adapter import load
|
|
5
|
+
doc = load("report.docx")
|
|
6
|
+
schema = doc.get_schema()
|
|
7
|
+
doc.set_cell(0, 1, 1, "홍길동")
|
|
8
|
+
doc.save("report_filled.docx")
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from .base import DocumentAdapter, DocumentSchema, TableSchema
|
|
15
|
+
from .docx_adapter import DocxAdapter
|
|
16
|
+
from .hwpx_adapter import HwpxAdapter
|
|
17
|
+
from .pptx_adapter import PptxAdapter
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"load",
|
|
21
|
+
"DocumentAdapter",
|
|
22
|
+
"DocumentSchema",
|
|
23
|
+
"TableSchema",
|
|
24
|
+
"DocxAdapter",
|
|
25
|
+
"PptxAdapter",
|
|
26
|
+
"HwpxAdapter",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
_ADAPTERS: dict[str, type[DocumentAdapter]] = {
|
|
30
|
+
".docx": DocxAdapter,
|
|
31
|
+
".pptx": PptxAdapter,
|
|
32
|
+
".hwpx": HwpxAdapter,
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load(path: str | Path) -> DocumentAdapter:
|
|
37
|
+
"""확장자로 적절한 어댑터를 선택해 문서를 연다."""
|
|
38
|
+
p = Path(path)
|
|
39
|
+
suffix = p.suffix.lower()
|
|
40
|
+
cls = _ADAPTERS.get(suffix)
|
|
41
|
+
if cls is None:
|
|
42
|
+
raise ValueError(
|
|
43
|
+
f"지원하지 않는 포맷: {suffix}. 지원: {sorted(_ADAPTERS.keys())}"
|
|
44
|
+
)
|
|
45
|
+
return cls(p)
|
document_adapter/base.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""DocumentAdapter 공통 인터페이스.
|
|
2
|
+
|
|
3
|
+
세 포맷(DOCX/PPTX/HWPX)의 공통 작업을 추상화:
|
|
4
|
+
- 템플릿 렌더링 ({{key}} 치환)
|
|
5
|
+
- 표 스키마 추출 (LLM 입력용)
|
|
6
|
+
- 셀 수정 / 행 추가
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from abc import ABC, abstractmethod
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class TableSchema:
|
|
18
|
+
"""표 한 개의 구조 (LLM에게 넘길 형태)."""
|
|
19
|
+
index: int
|
|
20
|
+
rows: int
|
|
21
|
+
cols: int
|
|
22
|
+
preview: list[list[str]]
|
|
23
|
+
location: str | None = None
|
|
24
|
+
|
|
25
|
+
def to_dict(self) -> dict[str, Any]:
|
|
26
|
+
return {
|
|
27
|
+
"index": self.index,
|
|
28
|
+
"rows": self.rows,
|
|
29
|
+
"cols": self.cols,
|
|
30
|
+
"location": self.location,
|
|
31
|
+
"preview": self.preview,
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class DocumentSchema:
|
|
37
|
+
"""문서 전체 스키마."""
|
|
38
|
+
format: str
|
|
39
|
+
source: str
|
|
40
|
+
placeholders: list[str] = field(default_factory=list)
|
|
41
|
+
tables: list[TableSchema] = field(default_factory=list)
|
|
42
|
+
|
|
43
|
+
def to_dict(self) -> dict[str, Any]:
|
|
44
|
+
return {
|
|
45
|
+
"format": self.format,
|
|
46
|
+
"source": self.source,
|
|
47
|
+
"placeholders": self.placeholders,
|
|
48
|
+
"tables": [t.to_dict() for t in self.tables],
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class DocumentAdapter(ABC):
|
|
53
|
+
"""모든 포맷 어댑터의 공통 부모."""
|
|
54
|
+
|
|
55
|
+
format: str = ""
|
|
56
|
+
|
|
57
|
+
def __init__(self, path: Path) -> None:
|
|
58
|
+
self.path = Path(path)
|
|
59
|
+
self._open()
|
|
60
|
+
|
|
61
|
+
# ---- lifecycle ----
|
|
62
|
+
@abstractmethod
|
|
63
|
+
def _open(self) -> None: ...
|
|
64
|
+
|
|
65
|
+
@abstractmethod
|
|
66
|
+
def save(self, path: Path | str | None = None) -> Path: ...
|
|
67
|
+
|
|
68
|
+
def close(self) -> None:
|
|
69
|
+
"""일부 포맷(HWPX)은 명시적 close 필요."""
|
|
70
|
+
pass
|
|
71
|
+
|
|
72
|
+
# ---- inspection ----
|
|
73
|
+
@abstractmethod
|
|
74
|
+
def get_placeholders(self) -> list[str]:
|
|
75
|
+
"""본문에서 사용된 {{key}} 목록 반환."""
|
|
76
|
+
|
|
77
|
+
@abstractmethod
|
|
78
|
+
def get_tables(self, min_rows: int = 1, min_cols: int = 1,
|
|
79
|
+
preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
|
|
80
|
+
"""필터 조건을 만족하는 표 스키마 목록."""
|
|
81
|
+
|
|
82
|
+
def get_schema(self) -> DocumentSchema:
|
|
83
|
+
return DocumentSchema(
|
|
84
|
+
format=self.format,
|
|
85
|
+
source=str(self.path),
|
|
86
|
+
placeholders=self.get_placeholders(),
|
|
87
|
+
tables=self.get_tables(),
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# ---- editing ----
|
|
91
|
+
@abstractmethod
|
|
92
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
93
|
+
"""템플릿의 {{key}}를 context 값으로 치환."""
|
|
94
|
+
|
|
95
|
+
@abstractmethod
|
|
96
|
+
def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
|
|
97
|
+
"""셀 값 교체. 원래 값 반환."""
|
|
98
|
+
|
|
99
|
+
@abstractmethod
|
|
100
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
101
|
+
"""표 끝에 새 행 추가."""
|
|
102
|
+
|
|
103
|
+
# ---- context manager ----
|
|
104
|
+
def __enter__(self) -> "DocumentAdapter":
|
|
105
|
+
return self
|
|
106
|
+
|
|
107
|
+
def __exit__(self, exc_type, exc, tb) -> None:
|
|
108
|
+
self.close()
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""DOCX 어댑터: python-docx (편집) + docxtpl (템플릿 렌더)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from docx import Document
|
|
9
|
+
from docxtpl import DocxTemplate
|
|
10
|
+
|
|
11
|
+
from .base import DocumentAdapter, TableSchema
|
|
12
|
+
|
|
13
|
+
TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class DocxAdapter(DocumentAdapter):
|
|
17
|
+
format = "docx"
|
|
18
|
+
|
|
19
|
+
def _open(self) -> None:
|
|
20
|
+
self._doc = Document(self.path)
|
|
21
|
+
|
|
22
|
+
def save(self, path: Path | str | None = None) -> Path:
|
|
23
|
+
target = Path(path) if path else self.path
|
|
24
|
+
self._doc.save(target)
|
|
25
|
+
self.path = target
|
|
26
|
+
return target
|
|
27
|
+
|
|
28
|
+
# ---- inspection ----
|
|
29
|
+
def get_placeholders(self) -> list[str]:
|
|
30
|
+
keys: set[str] = set()
|
|
31
|
+
for p in self._doc.paragraphs:
|
|
32
|
+
keys.update(TAG_PATTERN.findall(p.text))
|
|
33
|
+
for table in self._doc.tables:
|
|
34
|
+
for row in table.rows:
|
|
35
|
+
for cell in row.cells:
|
|
36
|
+
keys.update(TAG_PATTERN.findall(cell.text))
|
|
37
|
+
return sorted(keys)
|
|
38
|
+
|
|
39
|
+
def get_tables(self, min_rows: int = 1, min_cols: int = 1,
|
|
40
|
+
preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
|
|
41
|
+
schemas: list[TableSchema] = []
|
|
42
|
+
for i, t in enumerate(self._doc.tables):
|
|
43
|
+
rows, cols = len(t.rows), len(t.columns)
|
|
44
|
+
if rows < min_rows or cols < min_cols:
|
|
45
|
+
continue
|
|
46
|
+
preview: list[list[str]] = []
|
|
47
|
+
for row in list(t.rows)[:preview_rows]:
|
|
48
|
+
preview.append([c.text.strip()[:max_cell_len] for c in row.cells])
|
|
49
|
+
schemas.append(TableSchema(index=i, rows=rows, cols=cols, preview=preview))
|
|
50
|
+
return schemas
|
|
51
|
+
|
|
52
|
+
# ---- editing ----
|
|
53
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
54
|
+
"""docxtpl 기반 Jinja2 렌더. 참고:
|
|
55
|
+
- `{%tr for row in rows %}` / `{%tr endfor %}`는 **각각 별도 행**에 두어야 함
|
|
56
|
+
- 같은 행에 두면 `<w:tr>` 전체가 `{% for %}`로 교체되어 endfor 손실
|
|
57
|
+
"""
|
|
58
|
+
tpl = DocxTemplate(self.path)
|
|
59
|
+
tpl.render(context)
|
|
60
|
+
tpl.save(self.path)
|
|
61
|
+
# 렌더 후 _doc 재로드
|
|
62
|
+
self._doc = Document(self.path)
|
|
63
|
+
|
|
64
|
+
def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
|
|
65
|
+
cell = self._doc.tables[table_index].rows[row].cells[col]
|
|
66
|
+
old = cell.text
|
|
67
|
+
cell.text = value
|
|
68
|
+
return old
|
|
69
|
+
|
|
70
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
71
|
+
table = self._doc.tables[table_index]
|
|
72
|
+
new_row = table.add_row()
|
|
73
|
+
for i, v in enumerate(values):
|
|
74
|
+
if i < len(new_row.cells):
|
|
75
|
+
new_row.cells[i].text = v
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""HWPX 어댑터: python-hwpx 기반.
|
|
2
|
+
|
|
3
|
+
버그 회피:
|
|
4
|
+
- set_cell_text()는 빈 셀에서 lxml/ElementTree 혼용 에러가 발생 (v2.9.0) →
|
|
5
|
+
cell.paragraphs[0].text 직접 할당으로 우회
|
|
6
|
+
- replace_text_in_runs()는 한글 공백이 run으로 쪼개질 때 매칭 실패 →
|
|
7
|
+
위치 기반 편집을 권장
|
|
8
|
+
|
|
9
|
+
부가:
|
|
10
|
+
- manifest fallback 로그가 기본적으로 매우 시끄러움 → logging 레벨 조정
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import re
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Iterator
|
|
18
|
+
|
|
19
|
+
# 경고성 로그 억제 (manifest fallback 등)
|
|
20
|
+
logging.getLogger("hwpx").setLevel(logging.ERROR)
|
|
21
|
+
|
|
22
|
+
from hwpx.document import HwpxDocument
|
|
23
|
+
|
|
24
|
+
from .base import DocumentAdapter, TableSchema
|
|
25
|
+
|
|
26
|
+
TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class HwpxAdapter(DocumentAdapter):
|
|
30
|
+
format = "hwpx"
|
|
31
|
+
|
|
32
|
+
def _open(self) -> None:
|
|
33
|
+
self._doc = HwpxDocument.open(self.path)
|
|
34
|
+
|
|
35
|
+
def save(self, path: Path | str | None = None) -> Path:
|
|
36
|
+
target = Path(path) if path else self.path
|
|
37
|
+
self._doc.save_to_path(target)
|
|
38
|
+
self.path = target
|
|
39
|
+
return target
|
|
40
|
+
|
|
41
|
+
def close(self) -> None:
|
|
42
|
+
self._doc.close()
|
|
43
|
+
|
|
44
|
+
# ---- helpers ----
|
|
45
|
+
def _iter_tables(self) -> Iterator[tuple[int, Any]]:
|
|
46
|
+
idx = 0
|
|
47
|
+
for section in self._doc.sections:
|
|
48
|
+
for para in section.paragraphs:
|
|
49
|
+
for tbl in para.tables:
|
|
50
|
+
yield idx, tbl
|
|
51
|
+
idx += 1
|
|
52
|
+
|
|
53
|
+
def _get_table(self, table_index: int):
|
|
54
|
+
for idx, tbl in self._iter_tables():
|
|
55
|
+
if idx == table_index:
|
|
56
|
+
return tbl
|
|
57
|
+
raise IndexError(f"HWPX table index {table_index} not found")
|
|
58
|
+
|
|
59
|
+
@staticmethod
|
|
60
|
+
def _cell_text(cell) -> str:
|
|
61
|
+
return " ".join(p.text for p in cell.paragraphs).strip()
|
|
62
|
+
|
|
63
|
+
# ---- inspection ----
|
|
64
|
+
def get_placeholders(self) -> list[str]:
|
|
65
|
+
text = self._doc.export_text()
|
|
66
|
+
return sorted(set(TAG_PATTERN.findall(text)))
|
|
67
|
+
|
|
68
|
+
def get_tables(self, min_rows: int = 1, min_cols: int = 1,
|
|
69
|
+
preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
|
|
70
|
+
schemas: list[TableSchema] = []
|
|
71
|
+
for idx, tbl in self._iter_tables():
|
|
72
|
+
rows, cols = tbl.row_count, tbl.column_count
|
|
73
|
+
if rows < min_rows or cols < min_cols:
|
|
74
|
+
continue
|
|
75
|
+
preview: list[list[str]] = []
|
|
76
|
+
for r in range(min(rows, preview_rows)):
|
|
77
|
+
row_cells = []
|
|
78
|
+
for c in range(cols):
|
|
79
|
+
text = self._cell_text(tbl.cell(r, c))
|
|
80
|
+
row_cells.append(text[:max_cell_len])
|
|
81
|
+
preview.append(row_cells)
|
|
82
|
+
schemas.append(TableSchema(index=idx, rows=rows, cols=cols, preview=preview))
|
|
83
|
+
return schemas
|
|
84
|
+
|
|
85
|
+
# ---- editing ----
|
|
86
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
87
|
+
"""본문 + 표 셀의 {{key}}를 paragraph 단위로 치환."""
|
|
88
|
+
# 본문
|
|
89
|
+
for section in self._doc.sections:
|
|
90
|
+
for para in section.paragraphs:
|
|
91
|
+
text = para.text
|
|
92
|
+
if TAG_PATTERN.search(text):
|
|
93
|
+
para.text = TAG_PATTERN.sub(
|
|
94
|
+
lambda m: str(context.get(m.group(1), m.group(0))), text
|
|
95
|
+
)
|
|
96
|
+
# 표 셀
|
|
97
|
+
for _, tbl in self._iter_tables():
|
|
98
|
+
for r in range(tbl.row_count):
|
|
99
|
+
for c in range(tbl.column_count):
|
|
100
|
+
cell = tbl.cell(r, c)
|
|
101
|
+
for para in cell.paragraphs:
|
|
102
|
+
text = para.text
|
|
103
|
+
if TAG_PATTERN.search(text):
|
|
104
|
+
para.text = TAG_PATTERN.sub(
|
|
105
|
+
lambda m: str(context.get(m.group(1), m.group(0))), text
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
|
|
109
|
+
"""set_cell_text 버그 우회: paragraph.text 직접 할당."""
|
|
110
|
+
tbl = self._get_table(table_index)
|
|
111
|
+
cell = tbl.cell(row, col)
|
|
112
|
+
paragraphs = list(cell.paragraphs)
|
|
113
|
+
old = self._cell_text(cell)
|
|
114
|
+
if paragraphs:
|
|
115
|
+
paragraphs[0].text = value
|
|
116
|
+
for p in paragraphs[1:]:
|
|
117
|
+
p.text = ""
|
|
118
|
+
return old
|
|
119
|
+
|
|
120
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
121
|
+
"""python-hwpx에는 표준 add_row API가 없음.
|
|
122
|
+
대안: 템플릿에 충분한 빈 행을 미리 만들고 set_cell로 채우는 전략."""
|
|
123
|
+
raise NotImplementedError(
|
|
124
|
+
"HWPX는 python-hwpx에 동적 행 추가 공식 API가 없음. "
|
|
125
|
+
"템플릿에 여분 행을 두고 set_cell로 채우는 방식을 권장."
|
|
126
|
+
)
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""MCP stdio server — Claude Desktop/Code에서 document-adapter tool을 호출하게 해준다.
|
|
2
|
+
|
|
3
|
+
실행:
|
|
4
|
+
python -m document_adapter.mcp_server
|
|
5
|
+
|
|
6
|
+
Claude Desktop 설정 예시 (~/Library/Application Support/Claude/claude_desktop_config.json):
|
|
7
|
+
{
|
|
8
|
+
"mcpServers": {
|
|
9
|
+
"document-adapter": {
|
|
10
|
+
"command": "/path/to/venv/bin/python",
|
|
11
|
+
"args": ["-m", "document_adapter.mcp_server"]
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
import logging
|
|
21
|
+
|
|
22
|
+
from mcp.server import Server
|
|
23
|
+
from mcp.server.stdio import stdio_server
|
|
24
|
+
from mcp.types import TextContent, Tool
|
|
25
|
+
|
|
26
|
+
from .tools import TOOL_DEFINITIONS, call_tool
|
|
27
|
+
|
|
28
|
+
logging.basicConfig(level=logging.INFO)
|
|
29
|
+
log = logging.getLogger("document-adapter-mcp")
|
|
30
|
+
|
|
31
|
+
server: Server = Server("document-adapter")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@server.list_tools()
|
|
35
|
+
async def list_tools() -> list[Tool]:
|
|
36
|
+
return [
|
|
37
|
+
Tool(
|
|
38
|
+
name=t["name"],
|
|
39
|
+
description=t["description"],
|
|
40
|
+
inputSchema=t["input_schema"],
|
|
41
|
+
)
|
|
42
|
+
for t in TOOL_DEFINITIONS
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@server.call_tool()
|
|
47
|
+
async def on_call_tool(name: str, arguments: dict) -> list[TextContent]:
|
|
48
|
+
log.info("tool call: %s %s", name, list(arguments.keys()))
|
|
49
|
+
result = call_tool(name, arguments)
|
|
50
|
+
return [TextContent(type="text", text=json.dumps(result, ensure_ascii=False, indent=2))]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
async def main() -> None:
|
|
54
|
+
async with stdio_server() as (read, write):
|
|
55
|
+
await server.run(read, write, server.create_initialization_options())
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def main_sync() -> None:
|
|
59
|
+
"""Console script entry point."""
|
|
60
|
+
asyncio.run(main())
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
if __name__ == "__main__":
|
|
64
|
+
main_sync()
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""PPTX 어댑터: python-pptx + 자체 {{key}} 치환 엔진."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any, Iterator
|
|
7
|
+
|
|
8
|
+
from pptx import Presentation
|
|
9
|
+
from pptx.slide import Slide
|
|
10
|
+
|
|
11
|
+
from .base import DocumentAdapter, TableSchema
|
|
12
|
+
|
|
13
|
+
TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class PptxAdapter(DocumentAdapter):
|
|
17
|
+
format = "pptx"
|
|
18
|
+
|
|
19
|
+
def _open(self) -> None:
|
|
20
|
+
self._prs = Presentation(self.path)
|
|
21
|
+
|
|
22
|
+
def save(self, path: Path | str | None = None) -> Path:
|
|
23
|
+
target = Path(path) if path else self.path
|
|
24
|
+
self._prs.save(target)
|
|
25
|
+
self.path = target
|
|
26
|
+
return target
|
|
27
|
+
|
|
28
|
+
# ---- helpers ----
|
|
29
|
+
def _iter_tables(self) -> Iterator[tuple[int, int, Any]]:
|
|
30
|
+
"""(global_index, slide_number_1based, table) 순회."""
|
|
31
|
+
g_idx = 0
|
|
32
|
+
for s_idx, slide in enumerate(self._prs.slides, 1):
|
|
33
|
+
for shape in slide.shapes:
|
|
34
|
+
if shape.has_table:
|
|
35
|
+
yield g_idx, s_idx, shape.table
|
|
36
|
+
g_idx += 1
|
|
37
|
+
|
|
38
|
+
def _iter_text_frames(self) -> Iterator[Any]:
|
|
39
|
+
for slide in self._prs.slides:
|
|
40
|
+
for shape in slide.shapes:
|
|
41
|
+
if shape.has_text_frame:
|
|
42
|
+
yield shape.text_frame
|
|
43
|
+
if shape.has_table:
|
|
44
|
+
for row in shape.table.rows:
|
|
45
|
+
for cell in row.cells:
|
|
46
|
+
yield cell.text_frame
|
|
47
|
+
|
|
48
|
+
# ---- inspection ----
|
|
49
|
+
def get_placeholders(self) -> list[str]:
|
|
50
|
+
keys: set[str] = set()
|
|
51
|
+
for tf in self._iter_text_frames():
|
|
52
|
+
keys.update(TAG_PATTERN.findall(tf.text))
|
|
53
|
+
return sorted(keys)
|
|
54
|
+
|
|
55
|
+
def get_tables(self, min_rows: int = 1, min_cols: int = 1,
|
|
56
|
+
preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
|
|
57
|
+
schemas: list[TableSchema] = []
|
|
58
|
+
for g_idx, s_idx, table in self._iter_tables():
|
|
59
|
+
rows = list(table.rows)
|
|
60
|
+
cols = list(table.columns)
|
|
61
|
+
if len(rows) < min_rows or len(cols) < min_cols:
|
|
62
|
+
continue
|
|
63
|
+
preview: list[list[str]] = []
|
|
64
|
+
for row in rows[:preview_rows]:
|
|
65
|
+
preview.append([c.text.strip()[:max_cell_len] for c in row.cells])
|
|
66
|
+
schemas.append(TableSchema(
|
|
67
|
+
index=g_idx, rows=len(rows), cols=len(cols),
|
|
68
|
+
preview=preview, location=f"slide {s_idx}",
|
|
69
|
+
))
|
|
70
|
+
return schemas
|
|
71
|
+
|
|
72
|
+
# ---- editing ----
|
|
73
|
+
def render_template(self, context: dict[str, Any]) -> None:
|
|
74
|
+
"""paragraph 단위로 {{key}}를 치환. run이 쪼개진 경우를 처리하기 위해
|
|
75
|
+
paragraph 전체 텍스트를 재조립 후 첫 run에 담는다 (서식 일부 손실 가능)."""
|
|
76
|
+
for tf in self._iter_text_frames():
|
|
77
|
+
for para in tf.paragraphs:
|
|
78
|
+
full_text = "".join(run.text for run in para.runs)
|
|
79
|
+
if not TAG_PATTERN.search(full_text):
|
|
80
|
+
continue
|
|
81
|
+
rendered = TAG_PATTERN.sub(
|
|
82
|
+
lambda m: str(context.get(m.group(1), m.group(0))),
|
|
83
|
+
full_text,
|
|
84
|
+
)
|
|
85
|
+
if para.runs:
|
|
86
|
+
para.runs[0].text = rendered
|
|
87
|
+
for run in para.runs[1:]:
|
|
88
|
+
run.text = ""
|
|
89
|
+
|
|
90
|
+
def _get_table(self, table_index: int):
|
|
91
|
+
for g_idx, _, table in self._iter_tables():
|
|
92
|
+
if g_idx == table_index:
|
|
93
|
+
return table
|
|
94
|
+
raise IndexError(f"PPTX table index {table_index} not found")
|
|
95
|
+
|
|
96
|
+
def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
|
|
97
|
+
table = self._get_table(table_index)
|
|
98
|
+
cell = table.cell(row, col)
|
|
99
|
+
old = cell.text
|
|
100
|
+
cell.text = value
|
|
101
|
+
return old
|
|
102
|
+
|
|
103
|
+
def append_row(self, table_index: int, values: list[str]) -> None:
|
|
104
|
+
"""python-pptx는 표 행 추가 API를 제공하지 않는다.
|
|
105
|
+
LLM에게는 '지원 안 함'으로 알리는 게 정직한 방식."""
|
|
106
|
+
raise NotImplementedError(
|
|
107
|
+
"PPTX는 python-pptx에 동적 행 추가 API가 없음. "
|
|
108
|
+
"템플릿 단계에서 충분한 빈 행을 만들어 두고 set_cell로 채우는 방식을 권장."
|
|
109
|
+
)
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
"""LLM tool 정의 + 실행 함수.
|
|
2
|
+
|
|
3
|
+
동일 구현을 MCP 서버와 Claude API Tool Use 양쪽에서 재사용한다.
|
|
4
|
+
각 함수는 JSON-serializable dict를 반환.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import shutil
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from . import load
|
|
13
|
+
|
|
14
|
+
# -------- JSON schemas (Claude API tool use와 MCP 공용) --------
|
|
15
|
+
|
|
16
|
+
TOOL_DEFINITIONS: list[dict[str, Any]] = [
|
|
17
|
+
{
|
|
18
|
+
"name": "inspect_document",
|
|
19
|
+
"description": (
|
|
20
|
+
"문서(.docx/.pptx/.hwpx)의 구조를 분석한다. "
|
|
21
|
+
"placeholders({{key}} 태그 목록)와 tables(각 표의 행/열/미리보기)를 반환한다. "
|
|
22
|
+
"LLM이 어떤 필드를 채우거나 수정할지 판단할 때 먼저 호출해야 한다."
|
|
23
|
+
),
|
|
24
|
+
"input_schema": {
|
|
25
|
+
"type": "object",
|
|
26
|
+
"properties": {
|
|
27
|
+
"path": {
|
|
28
|
+
"type": "string",
|
|
29
|
+
"description": "문서 절대경로",
|
|
30
|
+
},
|
|
31
|
+
"min_rows": {
|
|
32
|
+
"type": "integer",
|
|
33
|
+
"description": "표 필터: 최소 행 수 (기본 1)",
|
|
34
|
+
"default": 1,
|
|
35
|
+
},
|
|
36
|
+
"min_cols": {
|
|
37
|
+
"type": "integer",
|
|
38
|
+
"description": "표 필터: 최소 열 수 (기본 1)",
|
|
39
|
+
"default": 1,
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
"required": ["path"],
|
|
43
|
+
},
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"name": "render_template",
|
|
47
|
+
"description": (
|
|
48
|
+
"문서의 {{key}} placeholder를 context의 값으로 치환해 새 파일로 저장한다. "
|
|
49
|
+
"DOCX는 docxtpl(Jinja2 loop/if 지원), PPTX/HWPX는 단순 {{key}} 치환."
|
|
50
|
+
),
|
|
51
|
+
"input_schema": {
|
|
52
|
+
"type": "object",
|
|
53
|
+
"properties": {
|
|
54
|
+
"path": {"type": "string", "description": "템플릿 파일 경로"},
|
|
55
|
+
"context": {
|
|
56
|
+
"type": "object",
|
|
57
|
+
"description": "{{key}}에 주입할 값 dict",
|
|
58
|
+
"additionalProperties": True,
|
|
59
|
+
},
|
|
60
|
+
"output_path": {
|
|
61
|
+
"type": "string",
|
|
62
|
+
"description": "결과 저장 경로 (생략 시 원본 옆에 _rendered 붙여 저장)",
|
|
63
|
+
},
|
|
64
|
+
},
|
|
65
|
+
"required": ["path", "context"],
|
|
66
|
+
},
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"name": "set_cell",
|
|
70
|
+
"description": (
|
|
71
|
+
"특정 표의 셀 값을 교체한다. table_index는 inspect_document의 tables 배열 인덱스. "
|
|
72
|
+
"PPTX는 슬라이드 경계와 무관한 전역 index."
|
|
73
|
+
),
|
|
74
|
+
"input_schema": {
|
|
75
|
+
"type": "object",
|
|
76
|
+
"properties": {
|
|
77
|
+
"path": {"type": "string"},
|
|
78
|
+
"table_index": {"type": "integer"},
|
|
79
|
+
"row": {"type": "integer"},
|
|
80
|
+
"col": {"type": "integer"},
|
|
81
|
+
"value": {"type": "string"},
|
|
82
|
+
"output_path": {
|
|
83
|
+
"type": "string",
|
|
84
|
+
"description": "생략 시 원본 덮어쓰기",
|
|
85
|
+
},
|
|
86
|
+
},
|
|
87
|
+
"required": ["path", "table_index", "row", "col", "value"],
|
|
88
|
+
},
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"name": "append_row",
|
|
92
|
+
"description": (
|
|
93
|
+
"표 끝에 새 행을 추가한다. **DOCX만 지원** — PPTX/HWPX는 API 미지원으로 에러 반환. "
|
|
94
|
+
"그 경우 템플릿 단계에서 충분한 빈 행을 두고 set_cell로 채워야 한다."
|
|
95
|
+
),
|
|
96
|
+
"input_schema": {
|
|
97
|
+
"type": "object",
|
|
98
|
+
"properties": {
|
|
99
|
+
"path": {"type": "string"},
|
|
100
|
+
"table_index": {"type": "integer"},
|
|
101
|
+
"values": {
|
|
102
|
+
"type": "array",
|
|
103
|
+
"items": {"type": "string"},
|
|
104
|
+
"description": "새 행의 각 셀 값. 열 수보다 적으면 나머지는 공백.",
|
|
105
|
+
},
|
|
106
|
+
"output_path": {"type": "string"},
|
|
107
|
+
},
|
|
108
|
+
"required": ["path", "table_index", "values"],
|
|
109
|
+
},
|
|
110
|
+
},
|
|
111
|
+
]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# -------- 실행 함수 --------
|
|
115
|
+
|
|
116
|
+
def _resolve_output(path: str, output_path: str | None, suffix: str = "_out") -> Path:
|
|
117
|
+
if output_path:
|
|
118
|
+
return Path(output_path)
|
|
119
|
+
p = Path(path)
|
|
120
|
+
return p.with_name(f"{p.stem}{suffix}{p.suffix}")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def inspect_document(path: str, min_rows: int = 1, min_cols: int = 1) -> dict[str, Any]:
|
|
124
|
+
doc = load(path)
|
|
125
|
+
try:
|
|
126
|
+
schema = doc.get_schema()
|
|
127
|
+
# min_rows/min_cols 필터 재적용
|
|
128
|
+
filtered = [t for t in doc.get_tables(min_rows=min_rows, min_cols=min_cols)]
|
|
129
|
+
result = schema.to_dict()
|
|
130
|
+
result["tables"] = [t.to_dict() for t in filtered]
|
|
131
|
+
return result
|
|
132
|
+
finally:
|
|
133
|
+
doc.close()
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def render_template(path: str, context: dict[str, Any],
|
|
137
|
+
output_path: str | None = None) -> dict[str, Any]:
|
|
138
|
+
out = _resolve_output(path, output_path, "_rendered")
|
|
139
|
+
shutil.copy2(path, out)
|
|
140
|
+
|
|
141
|
+
doc = load(out)
|
|
142
|
+
try:
|
|
143
|
+
before = doc.get_placeholders()
|
|
144
|
+
doc.render_template(context)
|
|
145
|
+
doc.save()
|
|
146
|
+
finally:
|
|
147
|
+
doc.close()
|
|
148
|
+
|
|
149
|
+
# 검증 재로드
|
|
150
|
+
doc2 = load(out)
|
|
151
|
+
try:
|
|
152
|
+
after = doc2.get_placeholders()
|
|
153
|
+
finally:
|
|
154
|
+
doc2.close()
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
"output_path": str(out),
|
|
158
|
+
"placeholders_before": before,
|
|
159
|
+
"placeholders_after": after,
|
|
160
|
+
"rendered_count": len(before) - len(after),
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def set_cell(path: str, table_index: int, row: int, col: int, value: str,
|
|
165
|
+
output_path: str | None = None) -> dict[str, Any]:
|
|
166
|
+
target = Path(output_path) if output_path else Path(path)
|
|
167
|
+
if output_path and Path(path) != target:
|
|
168
|
+
shutil.copy2(path, target)
|
|
169
|
+
|
|
170
|
+
doc = load(target)
|
|
171
|
+
try:
|
|
172
|
+
old = doc.set_cell(table_index, row, col, value)
|
|
173
|
+
doc.save()
|
|
174
|
+
finally:
|
|
175
|
+
doc.close()
|
|
176
|
+
|
|
177
|
+
return {
|
|
178
|
+
"output_path": str(target),
|
|
179
|
+
"table_index": table_index,
|
|
180
|
+
"row": row,
|
|
181
|
+
"col": col,
|
|
182
|
+
"previous_value": old,
|
|
183
|
+
"new_value": value,
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def append_row(path: str, table_index: int, values: list[str],
|
|
188
|
+
output_path: str | None = None) -> dict[str, Any]:
|
|
189
|
+
target = Path(output_path) if output_path else Path(path)
|
|
190
|
+
if output_path and Path(path) != target:
|
|
191
|
+
shutil.copy2(path, target)
|
|
192
|
+
|
|
193
|
+
doc = load(target)
|
|
194
|
+
try:
|
|
195
|
+
doc.append_row(table_index, values)
|
|
196
|
+
doc.save()
|
|
197
|
+
new_tables = doc.get_tables()
|
|
198
|
+
finally:
|
|
199
|
+
doc.close()
|
|
200
|
+
|
|
201
|
+
target_schema = next((t for t in new_tables if t.index == table_index), None)
|
|
202
|
+
return {
|
|
203
|
+
"output_path": str(target),
|
|
204
|
+
"table_index": table_index,
|
|
205
|
+
"new_row_count": target_schema.rows if target_schema else None,
|
|
206
|
+
"appended_values": values,
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# -------- 이름으로 dispatch --------
|
|
211
|
+
|
|
212
|
+
TOOL_HANDLERS = {
|
|
213
|
+
"inspect_document": inspect_document,
|
|
214
|
+
"render_template": render_template,
|
|
215
|
+
"set_cell": set_cell,
|
|
216
|
+
"append_row": append_row,
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def call_tool(name: str, arguments: dict[str, Any]) -> dict[str, Any]:
|
|
221
|
+
"""이름으로 tool 실행. 예외도 dict로 직렬화."""
|
|
222
|
+
handler = TOOL_HANDLERS.get(name)
|
|
223
|
+
if handler is None:
|
|
224
|
+
return {"error": f"unknown tool: {name}"}
|
|
225
|
+
try:
|
|
226
|
+
return handler(**arguments)
|
|
227
|
+
except NotImplementedError as e:
|
|
228
|
+
return {"error": "not_implemented", "message": str(e)}
|
|
229
|
+
except (IndexError, ValueError, FileNotFoundError) as e:
|
|
230
|
+
return {"error": type(e).__name__, "message": str(e)}
|
|
231
|
+
except Exception as e:
|
|
232
|
+
return {"error": "unexpected", "type": type(e).__name__, "message": str(e)}
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: document-adapter
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
|
|
5
|
+
Author-email: Son Seongjun <sonsj97@plateer.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/PlateerLab/document-adapter
|
|
8
|
+
Project-URL: Repository, https://github.com/PlateerLab/document-adapter
|
|
9
|
+
Keywords: mcp,llm,document,docx,pptx,hwpx,template,claude
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: python-docx>=1.1
|
|
20
|
+
Requires-Dist: docxtpl>=0.20
|
|
21
|
+
Requires-Dist: python-pptx>=1.0
|
|
22
|
+
Requires-Dist: python-hwpx>=2.9
|
|
23
|
+
Requires-Dist: mcp>=1.0
|
|
24
|
+
Provides-Extra: claude
|
|
25
|
+
Requires-Dist: anthropic>=0.40; extra == "claude"
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# document-adapter
|
|
31
|
+
|
|
32
|
+
**LLM이 DOCX / PPTX / HWPX 문서를 직접 편집할 수 있게 해주는 통합 어댑터 + MCP 서버.**
|
|
33
|
+
|
|
34
|
+
세 가지 오피스 포맷을 하나의 파이썬 인터페이스로 추상화하고, Claude Desktop / Claude Code / Anthropic API Tool Use에서 바로 호출할 수 있는 MCP 도구로 노출합니다. 양식 문서의 빈 셀을 자동으로 채우거나, 템플릿의 `{{key}}`를 치환하거나, 기존 표의 내용을 수정하는 작업을 LLM 에이전트가 수행할 수 있습니다.
|
|
35
|
+
|
|
36
|
+
## 지원 포맷
|
|
37
|
+
|
|
38
|
+
| 포맷 | 백엔드 | 템플릿 렌더 | 표 읽기 | 셀 수정 | 행 추가 |
|
|
39
|
+
|---|---|---|---|---|---|
|
|
40
|
+
| `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ |
|
|
41
|
+
| `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | ❌ (미지원) |
|
|
42
|
+
| `.hwpx` | `python-hwpx` (Pure Python) | `{{key}}` 치환 | ✅ | ✅ | ❌ (미지원) |
|
|
43
|
+
|
|
44
|
+
- HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
|
|
45
|
+
- 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
|
|
46
|
+
|
|
47
|
+
## 설치
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install -e .
|
|
51
|
+
|
|
52
|
+
# Claude API 예시 스크립트까지 쓰려면
|
|
53
|
+
pip install -e ".[claude]"
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Python 3.10+ 필요.
|
|
57
|
+
|
|
58
|
+
## 빠른 시작 — 파이썬 API
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from document_adapter import load
|
|
62
|
+
|
|
63
|
+
doc = load("report_template.docx")
|
|
64
|
+
|
|
65
|
+
# 1. 구조 파악
|
|
66
|
+
schema = doc.get_schema()
|
|
67
|
+
print(schema.placeholders) # ['author', 'date', 'title']
|
|
68
|
+
print(schema.tables) # [TableSchema(index=0, rows=7, cols=2, ...), ...]
|
|
69
|
+
|
|
70
|
+
# 2. 템플릿 렌더
|
|
71
|
+
doc.render_template({
|
|
72
|
+
"title": "Q1 운영 리포트",
|
|
73
|
+
"author": "손성준",
|
|
74
|
+
"date": "2026-04-15",
|
|
75
|
+
})
|
|
76
|
+
doc.save("report_filled.docx")
|
|
77
|
+
|
|
78
|
+
# 3. 기존 양식 파일의 표 셀 수정
|
|
79
|
+
doc = load("checklist.docx")
|
|
80
|
+
old = doc.set_cell(table_index=1, row=1, col=1, value="○○전자")
|
|
81
|
+
doc.append_row(1, ["새 항목", "값"]) # DOCX만 지원
|
|
82
|
+
doc.save("checklist_filled.docx")
|
|
83
|
+
doc.close()
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
확장자로 자동 분기되므로 `.pptx` / `.hwpx`도 동일한 API를 사용합니다.
|
|
87
|
+
|
|
88
|
+
## MCP 서버로 사용 — Claude Desktop / Claude Code
|
|
89
|
+
|
|
90
|
+
### 실행
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
python -m document_adapter.mcp_server
|
|
94
|
+
# 또는 설치 후
|
|
95
|
+
document-adapter-mcp
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
### Claude Desktop 설정
|
|
99
|
+
|
|
100
|
+
`~/Library/Application Support/Claude/claude_desktop_config.json`:
|
|
101
|
+
|
|
102
|
+
```json
|
|
103
|
+
{
|
|
104
|
+
"mcpServers": {
|
|
105
|
+
"document-adapter": {
|
|
106
|
+
"command": "/absolute/path/to/venv/bin/python",
|
|
107
|
+
"args": ["-m", "document_adapter.mcp_server"]
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
재시작하면 Claude Desktop에서 아래 4개 도구를 사용할 수 있습니다.
|
|
114
|
+
|
|
115
|
+
### Claude Code 설정
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
claude mcp add document-adapter \
|
|
119
|
+
/absolute/path/to/venv/bin/python -m document_adapter.mcp_server
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Anthropic API Tool Use로 사용
|
|
123
|
+
|
|
124
|
+
`document_adapter.tools`가 Claude API의 tool schema 형식과 그대로 호환됩니다.
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
import anthropic
|
|
128
|
+
from document_adapter.tools import TOOL_DEFINITIONS, call_tool
|
|
129
|
+
|
|
130
|
+
client = anthropic.Anthropic()
|
|
131
|
+
|
|
132
|
+
resp = client.messages.create(
|
|
133
|
+
model="claude-opus-4-6",
|
|
134
|
+
max_tokens=4096,
|
|
135
|
+
tools=[{
|
|
136
|
+
"name": t["name"],
|
|
137
|
+
"description": t["description"],
|
|
138
|
+
"input_schema": t["input_schema"],
|
|
139
|
+
} for t in TOOL_DEFINITIONS],
|
|
140
|
+
messages=[{
|
|
141
|
+
"role": "user",
|
|
142
|
+
"content": "report_template.docx의 표 구조를 확인하고 빈 셀을 적절히 채워줘",
|
|
143
|
+
}],
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
# tool_use 블록을 받으면 call_tool(name, args)로 실행 후 결과 반환
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
전체 agent loop 예시는 [`examples/claude_api_example.py`](examples/claude_api_example.py) 참고.
|
|
150
|
+
|
|
151
|
+
## 노출되는 4개 도구
|
|
152
|
+
|
|
153
|
+
| 도구 | 설명 |
|
|
154
|
+
|---|---|
|
|
155
|
+
| `inspect_document` | 문서 구조(placeholders, tables)를 JSON으로 반환. **항상 첫 호출로 사용** |
|
|
156
|
+
| `render_template` | `{{key}}`를 context dict 값으로 치환해 새 파일 저장 |
|
|
157
|
+
| `set_cell` | 특정 표의 `(row, col)` 셀 값 교체 |
|
|
158
|
+
| `append_row` | 표 끝에 새 행 추가 (DOCX 전용) |
|
|
159
|
+
|
|
160
|
+
### `inspect_document` 반환 예시
|
|
161
|
+
|
|
162
|
+
```json
|
|
163
|
+
{
|
|
164
|
+
"format": "docx",
|
|
165
|
+
"source": "/path/to/checklist.docx",
|
|
166
|
+
"placeholders": [],
|
|
167
|
+
"tables": [
|
|
168
|
+
{
|
|
169
|
+
"index": 1,
|
|
170
|
+
"rows": 7,
|
|
171
|
+
"cols": 2,
|
|
172
|
+
"location": null,
|
|
173
|
+
"preview": [
|
|
174
|
+
{"row": 0, "cells": ["항목", "기입 내용"]},
|
|
175
|
+
{"row": 1, "cells": ["고객사 / 조직", ""]},
|
|
176
|
+
{"row": 2, "cells": ["현업 담당부서 / 책임자", ""]}
|
|
177
|
+
]
|
|
178
|
+
}
|
|
179
|
+
]
|
|
180
|
+
}
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
LLM은 이 preview를 보고 **"빈 셀이 어디 있는지 / 어떤 값을 넣어야 하는지"** 를 판단하여 `set_cell`을 호출합니다.
|
|
184
|
+
|
|
185
|
+
## 템플릿 작성 규칙
|
|
186
|
+
|
|
187
|
+
### DOCX — Jinja2 전체 문법 사용 가능
|
|
188
|
+
|
|
189
|
+
```
|
|
190
|
+
{{ report_title }}
|
|
191
|
+
작성자: {{ author }}
|
|
192
|
+
|
|
193
|
+
{% for item in items %}- {{ item.name }}: {{ item.value }}
|
|
194
|
+
{% endfor %}
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
**표 행 반복은 `{%tr for ... %}` / `{%tr endfor %}`를 각각 별도 행에 두어야 합니다.**
|
|
198
|
+
같은 행에 두 태그를 넣으면 `<w:tr>` 전체가 `{% for %}`로 교체되어 `endfor`가 손실됩니다.
|
|
199
|
+
|
|
200
|
+
```
|
|
201
|
+
┌─────────────────────┬─────┬─────┐
|
|
202
|
+
│ 항목 │ 목표 │ 실적 │ <- 헤더
|
|
203
|
+
├─────────────────────┼─────┼─────┤
|
|
204
|
+
│ {%tr for r in rows %} │ <- for 행
|
|
205
|
+
├─────────────────────┼─────┼─────┤
|
|
206
|
+
│ {{ r.name }} │ {{ r.target }} │ {{ r.actual }} │ <- 반복 본문
|
|
207
|
+
├─────────────────────┼─────┼─────┤
|
|
208
|
+
│ {%tr endfor %} │ <- endfor 행
|
|
209
|
+
└─────────────────────┴─────┴─────┘
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
### PPTX / HWPX — 단순 `{{key}}` 치환만
|
|
213
|
+
|
|
214
|
+
loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `run`으로 쪼개질 수 있어, 어댑터가 paragraph 전체 텍스트를 재조립한 뒤 첫 `run`에 다시 담는 방식으로 처리합니다 (서식 일부 손실 가능).
|
|
215
|
+
|
|
216
|
+
## 내장된 버그 회피
|
|
217
|
+
|
|
218
|
+
| 포맷 | 문제 | 어댑터의 처리 |
|
|
219
|
+
|---|---|---|
|
|
220
|
+
| HWPX | `python-hwpx 2.9.0`의 `set_cell_text()`가 빈 셀에서 lxml/ElementTree 혼용 `TypeError` 발생 | `paragraphs[0].text = value` 직접 할당으로 우회 |
|
|
221
|
+
| HWPX | `replace_text_in_runs()`가 한글 공백이 run으로 쪼개진 경우 매칭 실패 | 위치 기반 API만 사용 |
|
|
222
|
+
| HWPX | `manifest fallback` 경고 로그가 과도하게 출력됨 | `logging.getLogger("hwpx")` 레벨을 `ERROR`로 조정 |
|
|
223
|
+
| PPTX | placeholder가 여러 `run`으로 쪼개져 단순 `run.text` 치환이 실패 | paragraph 전체 재조립 |
|
|
224
|
+
| DOCX | `docxtpl`의 `{%tr%}`를 같은 행에 두면 파싱 에러 | README에 배치 규칙 명시 |
|
|
225
|
+
|
|
226
|
+
## 프로젝트 구조
|
|
227
|
+
|
|
228
|
+
```
|
|
229
|
+
document_adapter/
|
|
230
|
+
├── __init__.py # load() dispatcher
|
|
231
|
+
├── base.py # DocumentAdapter ABC, TableSchema, DocumentSchema
|
|
232
|
+
├── docx_adapter.py # DocxAdapter
|
|
233
|
+
├── pptx_adapter.py # PptxAdapter
|
|
234
|
+
├── hwpx_adapter.py # HwpxAdapter (버그 회피 포함)
|
|
235
|
+
├── tools.py # Tool 정의 + call_tool dispatcher
|
|
236
|
+
└── mcp_server.py # MCP stdio server
|
|
237
|
+
|
|
238
|
+
examples/
|
|
239
|
+
└── claude_api_example.py
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
## 라이선스
|
|
243
|
+
|
|
244
|
+
MIT
|
|
245
|
+
|
|
246
|
+
## Credits
|
|
247
|
+
|
|
248
|
+
- [`python-docx`](https://github.com/python-openxml/python-docx)
|
|
249
|
+
- [`docxtpl`](https://github.com/elapouya/python-docx-template)
|
|
250
|
+
- [`python-pptx`](https://github.com/scanny/python-pptx)
|
|
251
|
+
- [`python-hwpx`](https://github.com/airmang/python-hwpx)
|
|
252
|
+
- [`mcp`](https://github.com/modelcontextprotocol/python-sdk)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
document_adapter/__init__.py,sha256=YqjF6xhkI1X2sbO5XfZE9kQVA4RMzTuwu7TpuMUGJDk,1121
|
|
2
|
+
document_adapter/base.py,sha256=pQlYsSnMJfYa-GaSkzSlwqVhe1udRyv6STTryuEFftc,3001
|
|
3
|
+
document_adapter/docx_adapter.py,sha256=IxpSXKaD_9kspalKcQ3AbsVJodK6pJ_t5XoUy5fv5Kg,2702
|
|
4
|
+
document_adapter/hwpx_adapter.py,sha256=46uIkn8dDHeJkx4u3xU1ZZaXTTovdgzHWN1ZN0HFep0,4683
|
|
5
|
+
document_adapter/mcp_server.py,sha256=veEl7S5EJMP0OLSlQI6pWvFSKJ-0lWm0GHX46G0EfC4,1609
|
|
6
|
+
document_adapter/pptx_adapter.py,sha256=doV9NjgyK5OrrB-TkrvYrwhusqCfxZXjk_iiLkJOy-g,4215
|
|
7
|
+
document_adapter/tools.py,sha256=84jib-rNQUhWFER1R6k7zIzBDGHgnx6MCHCh2keAcp0,7608
|
|
8
|
+
document_adapter-0.1.0.dist-info/licenses/LICENSE,sha256=QjKz-cVDxa_NbNhNYo9ek6yDMnR4YHA2pKg-upo93x4,1068
|
|
9
|
+
document_adapter-0.1.0.dist-info/METADATA,sha256=pGILinIsgp7YVDeNQ7hfPk_UObx8Ro-rqxYS-wPkQ0E,8809
|
|
10
|
+
document_adapter-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
11
|
+
document_adapter-0.1.0.dist-info/entry_points.txt,sha256=krV-Z-Yc1M3qe2I1pxhEJX7wdwbKwfX-8sEN3-3CuQo,79
|
|
12
|
+
document_adapter-0.1.0.dist-info/top_level.txt,sha256=WGApqbHX1xpY13c2OvG83DxDZToMonYK2xkWN_QeT8M,17
|
|
13
|
+
document_adapter-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Plateer Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
document_adapter
|