document-adapter 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,45 @@
1
+ """Document template editing — 통합 어댑터.
2
+
3
+ 사용법:
4
+ from document_adapter import load
5
+ doc = load("report.docx")
6
+ schema = doc.get_schema()
7
+ doc.set_cell(0, 1, 1, "홍길동")
8
+ doc.save("report_filled.docx")
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+
14
+ from .base import DocumentAdapter, DocumentSchema, TableSchema
15
+ from .docx_adapter import DocxAdapter
16
+ from .hwpx_adapter import HwpxAdapter
17
+ from .pptx_adapter import PptxAdapter
18
+
19
+ __all__ = [
20
+ "load",
21
+ "DocumentAdapter",
22
+ "DocumentSchema",
23
+ "TableSchema",
24
+ "DocxAdapter",
25
+ "PptxAdapter",
26
+ "HwpxAdapter",
27
+ ]
28
+
29
+ _ADAPTERS: dict[str, type[DocumentAdapter]] = {
30
+ ".docx": DocxAdapter,
31
+ ".pptx": PptxAdapter,
32
+ ".hwpx": HwpxAdapter,
33
+ }
34
+
35
+
36
+ def load(path: str | Path) -> DocumentAdapter:
37
+ """확장자로 적절한 어댑터를 선택해 문서를 연다."""
38
+ p = Path(path)
39
+ suffix = p.suffix.lower()
40
+ cls = _ADAPTERS.get(suffix)
41
+ if cls is None:
42
+ raise ValueError(
43
+ f"지원하지 않는 포맷: {suffix}. 지원: {sorted(_ADAPTERS.keys())}"
44
+ )
45
+ return cls(p)
@@ -0,0 +1,108 @@
1
+ """DocumentAdapter 공통 인터페이스.
2
+
3
+ 세 포맷(DOCX/PPTX/HWPX)의 공통 작업을 추상화:
4
+ - 템플릿 렌더링 ({{key}} 치환)
5
+ - 표 스키마 추출 (LLM 입력용)
6
+ - 셀 수정 / 행 추가
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from abc import ABC, abstractmethod
11
+ from dataclasses import dataclass, field
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+
16
+ @dataclass
17
+ class TableSchema:
18
+ """표 한 개의 구조 (LLM에게 넘길 형태)."""
19
+ index: int
20
+ rows: int
21
+ cols: int
22
+ preview: list[list[str]]
23
+ location: str | None = None
24
+
25
+ def to_dict(self) -> dict[str, Any]:
26
+ return {
27
+ "index": self.index,
28
+ "rows": self.rows,
29
+ "cols": self.cols,
30
+ "location": self.location,
31
+ "preview": self.preview,
32
+ }
33
+
34
+
35
+ @dataclass
36
+ class DocumentSchema:
37
+ """문서 전체 스키마."""
38
+ format: str
39
+ source: str
40
+ placeholders: list[str] = field(default_factory=list)
41
+ tables: list[TableSchema] = field(default_factory=list)
42
+
43
+ def to_dict(self) -> dict[str, Any]:
44
+ return {
45
+ "format": self.format,
46
+ "source": self.source,
47
+ "placeholders": self.placeholders,
48
+ "tables": [t.to_dict() for t in self.tables],
49
+ }
50
+
51
+
52
+ class DocumentAdapter(ABC):
53
+ """모든 포맷 어댑터의 공통 부모."""
54
+
55
+ format: str = ""
56
+
57
+ def __init__(self, path: Path) -> None:
58
+ self.path = Path(path)
59
+ self._open()
60
+
61
+ # ---- lifecycle ----
62
+ @abstractmethod
63
+ def _open(self) -> None: ...
64
+
65
+ @abstractmethod
66
+ def save(self, path: Path | str | None = None) -> Path: ...
67
+
68
+ def close(self) -> None:
69
+ """일부 포맷(HWPX)은 명시적 close 필요."""
70
+ pass
71
+
72
+ # ---- inspection ----
73
+ @abstractmethod
74
+ def get_placeholders(self) -> list[str]:
75
+ """본문에서 사용된 {{key}} 목록 반환."""
76
+
77
+ @abstractmethod
78
+ def get_tables(self, min_rows: int = 1, min_cols: int = 1,
79
+ preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
80
+ """필터 조건을 만족하는 표 스키마 목록."""
81
+
82
+ def get_schema(self) -> DocumentSchema:
83
+ return DocumentSchema(
84
+ format=self.format,
85
+ source=str(self.path),
86
+ placeholders=self.get_placeholders(),
87
+ tables=self.get_tables(),
88
+ )
89
+
90
+ # ---- editing ----
91
+ @abstractmethod
92
+ def render_template(self, context: dict[str, Any]) -> None:
93
+ """템플릿의 {{key}}를 context 값으로 치환."""
94
+
95
+ @abstractmethod
96
+ def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
97
+ """셀 값 교체. 원래 값 반환."""
98
+
99
+ @abstractmethod
100
+ def append_row(self, table_index: int, values: list[str]) -> None:
101
+ """표 끝에 새 행 추가."""
102
+
103
+ # ---- context manager ----
104
+ def __enter__(self) -> "DocumentAdapter":
105
+ return self
106
+
107
+ def __exit__(self, exc_type, exc, tb) -> None:
108
+ self.close()
@@ -0,0 +1,75 @@
1
+ """DOCX 어댑터: python-docx (편집) + docxtpl (템플릿 렌더)."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from docx import Document
9
+ from docxtpl import DocxTemplate
10
+
11
+ from .base import DocumentAdapter, TableSchema
12
+
13
+ TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
14
+
15
+
16
+ class DocxAdapter(DocumentAdapter):
17
+ format = "docx"
18
+
19
+ def _open(self) -> None:
20
+ self._doc = Document(self.path)
21
+
22
+ def save(self, path: Path | str | None = None) -> Path:
23
+ target = Path(path) if path else self.path
24
+ self._doc.save(target)
25
+ self.path = target
26
+ return target
27
+
28
+ # ---- inspection ----
29
+ def get_placeholders(self) -> list[str]:
30
+ keys: set[str] = set()
31
+ for p in self._doc.paragraphs:
32
+ keys.update(TAG_PATTERN.findall(p.text))
33
+ for table in self._doc.tables:
34
+ for row in table.rows:
35
+ for cell in row.cells:
36
+ keys.update(TAG_PATTERN.findall(cell.text))
37
+ return sorted(keys)
38
+
39
+ def get_tables(self, min_rows: int = 1, min_cols: int = 1,
40
+ preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
41
+ schemas: list[TableSchema] = []
42
+ for i, t in enumerate(self._doc.tables):
43
+ rows, cols = len(t.rows), len(t.columns)
44
+ if rows < min_rows or cols < min_cols:
45
+ continue
46
+ preview: list[list[str]] = []
47
+ for row in list(t.rows)[:preview_rows]:
48
+ preview.append([c.text.strip()[:max_cell_len] for c in row.cells])
49
+ schemas.append(TableSchema(index=i, rows=rows, cols=cols, preview=preview))
50
+ return schemas
51
+
52
+ # ---- editing ----
53
+ def render_template(self, context: dict[str, Any]) -> None:
54
+ """docxtpl 기반 Jinja2 렌더. 참고:
55
+ - `{%tr for row in rows %}` / `{%tr endfor %}`는 **각각 별도 행**에 두어야 함
56
+ - 같은 행에 두면 `<w:tr>` 전체가 `{% for %}`로 교체되어 endfor 손실
57
+ """
58
+ tpl = DocxTemplate(self.path)
59
+ tpl.render(context)
60
+ tpl.save(self.path)
61
+ # 렌더 후 _doc 재로드
62
+ self._doc = Document(self.path)
63
+
64
+ def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
65
+ cell = self._doc.tables[table_index].rows[row].cells[col]
66
+ old = cell.text
67
+ cell.text = value
68
+ return old
69
+
70
+ def append_row(self, table_index: int, values: list[str]) -> None:
71
+ table = self._doc.tables[table_index]
72
+ new_row = table.add_row()
73
+ for i, v in enumerate(values):
74
+ if i < len(new_row.cells):
75
+ new_row.cells[i].text = v
@@ -0,0 +1,126 @@
1
+ """HWPX 어댑터: python-hwpx 기반.
2
+
3
+ 버그 회피:
4
+ - set_cell_text()는 빈 셀에서 lxml/ElementTree 혼용 에러가 발생 (v2.9.0) →
5
+ cell.paragraphs[0].text 직접 할당으로 우회
6
+ - replace_text_in_runs()는 한글 공백이 run으로 쪼개질 때 매칭 실패 →
7
+ 위치 기반 편집을 권장
8
+
9
+ 부가:
10
+ - manifest fallback 로그가 기본적으로 매우 시끄러움 → logging 레벨 조정
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ import re
16
+ from pathlib import Path
17
+ from typing import Any, Iterator
18
+
19
+ # 경고성 로그 억제 (manifest fallback 등)
20
+ logging.getLogger("hwpx").setLevel(logging.ERROR)
21
+
22
+ from hwpx.document import HwpxDocument
23
+
24
+ from .base import DocumentAdapter, TableSchema
25
+
26
+ TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
27
+
28
+
29
+ class HwpxAdapter(DocumentAdapter):
30
+ format = "hwpx"
31
+
32
+ def _open(self) -> None:
33
+ self._doc = HwpxDocument.open(self.path)
34
+
35
+ def save(self, path: Path | str | None = None) -> Path:
36
+ target = Path(path) if path else self.path
37
+ self._doc.save_to_path(target)
38
+ self.path = target
39
+ return target
40
+
41
+ def close(self) -> None:
42
+ self._doc.close()
43
+
44
+ # ---- helpers ----
45
+ def _iter_tables(self) -> Iterator[tuple[int, Any]]:
46
+ idx = 0
47
+ for section in self._doc.sections:
48
+ for para in section.paragraphs:
49
+ for tbl in para.tables:
50
+ yield idx, tbl
51
+ idx += 1
52
+
53
+ def _get_table(self, table_index: int):
54
+ for idx, tbl in self._iter_tables():
55
+ if idx == table_index:
56
+ return tbl
57
+ raise IndexError(f"HWPX table index {table_index} not found")
58
+
59
+ @staticmethod
60
+ def _cell_text(cell) -> str:
61
+ return " ".join(p.text for p in cell.paragraphs).strip()
62
+
63
+ # ---- inspection ----
64
+ def get_placeholders(self) -> list[str]:
65
+ text = self._doc.export_text()
66
+ return sorted(set(TAG_PATTERN.findall(text)))
67
+
68
+ def get_tables(self, min_rows: int = 1, min_cols: int = 1,
69
+ preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
70
+ schemas: list[TableSchema] = []
71
+ for idx, tbl in self._iter_tables():
72
+ rows, cols = tbl.row_count, tbl.column_count
73
+ if rows < min_rows or cols < min_cols:
74
+ continue
75
+ preview: list[list[str]] = []
76
+ for r in range(min(rows, preview_rows)):
77
+ row_cells = []
78
+ for c in range(cols):
79
+ text = self._cell_text(tbl.cell(r, c))
80
+ row_cells.append(text[:max_cell_len])
81
+ preview.append(row_cells)
82
+ schemas.append(TableSchema(index=idx, rows=rows, cols=cols, preview=preview))
83
+ return schemas
84
+
85
+ # ---- editing ----
86
+ def render_template(self, context: dict[str, Any]) -> None:
87
+ """본문 + 표 셀의 {{key}}를 paragraph 단위로 치환."""
88
+ # 본문
89
+ for section in self._doc.sections:
90
+ for para in section.paragraphs:
91
+ text = para.text
92
+ if TAG_PATTERN.search(text):
93
+ para.text = TAG_PATTERN.sub(
94
+ lambda m: str(context.get(m.group(1), m.group(0))), text
95
+ )
96
+ # 표 셀
97
+ for _, tbl in self._iter_tables():
98
+ for r in range(tbl.row_count):
99
+ for c in range(tbl.column_count):
100
+ cell = tbl.cell(r, c)
101
+ for para in cell.paragraphs:
102
+ text = para.text
103
+ if TAG_PATTERN.search(text):
104
+ para.text = TAG_PATTERN.sub(
105
+ lambda m: str(context.get(m.group(1), m.group(0))), text
106
+ )
107
+
108
+ def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
109
+ """set_cell_text 버그 우회: paragraph.text 직접 할당."""
110
+ tbl = self._get_table(table_index)
111
+ cell = tbl.cell(row, col)
112
+ paragraphs = list(cell.paragraphs)
113
+ old = self._cell_text(cell)
114
+ if paragraphs:
115
+ paragraphs[0].text = value
116
+ for p in paragraphs[1:]:
117
+ p.text = ""
118
+ return old
119
+
120
+ def append_row(self, table_index: int, values: list[str]) -> None:
121
+ """python-hwpx에는 표준 add_row API가 없음.
122
+ 대안: 템플릿에 충분한 빈 행을 미리 만들고 set_cell로 채우는 전략."""
123
+ raise NotImplementedError(
124
+ "HWPX는 python-hwpx에 동적 행 추가 공식 API가 없음. "
125
+ "템플릿에 여분 행을 두고 set_cell로 채우는 방식을 권장."
126
+ )
@@ -0,0 +1,64 @@
1
+ """MCP stdio server — Claude Desktop/Code에서 document-adapter tool을 호출하게 해준다.
2
+
3
+ 실행:
4
+ python -m document_adapter.mcp_server
5
+
6
+ Claude Desktop 설정 예시 (~/Library/Application Support/Claude/claude_desktop_config.json):
7
+ {
8
+ "mcpServers": {
9
+ "document-adapter": {
10
+ "command": "/path/to/venv/bin/python",
11
+ "args": ["-m", "document_adapter.mcp_server"]
12
+ }
13
+ }
14
+ }
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import json
20
+ import logging
21
+
22
+ from mcp.server import Server
23
+ from mcp.server.stdio import stdio_server
24
+ from mcp.types import TextContent, Tool
25
+
26
+ from .tools import TOOL_DEFINITIONS, call_tool
27
+
28
+ logging.basicConfig(level=logging.INFO)
29
+ log = logging.getLogger("document-adapter-mcp")
30
+
31
+ server: Server = Server("document-adapter")
32
+
33
+
34
+ @server.list_tools()
35
+ async def list_tools() -> list[Tool]:
36
+ return [
37
+ Tool(
38
+ name=t["name"],
39
+ description=t["description"],
40
+ inputSchema=t["input_schema"],
41
+ )
42
+ for t in TOOL_DEFINITIONS
43
+ ]
44
+
45
+
46
+ @server.call_tool()
47
+ async def on_call_tool(name: str, arguments: dict) -> list[TextContent]:
48
+ log.info("tool call: %s %s", name, list(arguments.keys()))
49
+ result = call_tool(name, arguments)
50
+ return [TextContent(type="text", text=json.dumps(result, ensure_ascii=False, indent=2))]
51
+
52
+
53
+ async def main() -> None:
54
+ async with stdio_server() as (read, write):
55
+ await server.run(read, write, server.create_initialization_options())
56
+
57
+
58
+ def main_sync() -> None:
59
+ """Console script entry point."""
60
+ asyncio.run(main())
61
+
62
+
63
+ if __name__ == "__main__":
64
+ main_sync()
@@ -0,0 +1,109 @@
1
+ """PPTX 어댑터: python-pptx + 자체 {{key}} 치환 엔진."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from pathlib import Path
6
+ from typing import Any, Iterator
7
+
8
+ from pptx import Presentation
9
+ from pptx.slide import Slide
10
+
11
+ from .base import DocumentAdapter, TableSchema
12
+
13
+ TAG_PATTERN = re.compile(r"\{\{\s*(\w+)\s*\}\}")
14
+
15
+
16
+ class PptxAdapter(DocumentAdapter):
17
+ format = "pptx"
18
+
19
+ def _open(self) -> None:
20
+ self._prs = Presentation(self.path)
21
+
22
+ def save(self, path: Path | str | None = None) -> Path:
23
+ target = Path(path) if path else self.path
24
+ self._prs.save(target)
25
+ self.path = target
26
+ return target
27
+
28
+ # ---- helpers ----
29
+ def _iter_tables(self) -> Iterator[tuple[int, int, Any]]:
30
+ """(global_index, slide_number_1based, table) 순회."""
31
+ g_idx = 0
32
+ for s_idx, slide in enumerate(self._prs.slides, 1):
33
+ for shape in slide.shapes:
34
+ if shape.has_table:
35
+ yield g_idx, s_idx, shape.table
36
+ g_idx += 1
37
+
38
+ def _iter_text_frames(self) -> Iterator[Any]:
39
+ for slide in self._prs.slides:
40
+ for shape in slide.shapes:
41
+ if shape.has_text_frame:
42
+ yield shape.text_frame
43
+ if shape.has_table:
44
+ for row in shape.table.rows:
45
+ for cell in row.cells:
46
+ yield cell.text_frame
47
+
48
+ # ---- inspection ----
49
+ def get_placeholders(self) -> list[str]:
50
+ keys: set[str] = set()
51
+ for tf in self._iter_text_frames():
52
+ keys.update(TAG_PATTERN.findall(tf.text))
53
+ return sorted(keys)
54
+
55
+ def get_tables(self, min_rows: int = 1, min_cols: int = 1,
56
+ preview_rows: int = 4, max_cell_len: int = 40) -> list[TableSchema]:
57
+ schemas: list[TableSchema] = []
58
+ for g_idx, s_idx, table in self._iter_tables():
59
+ rows = list(table.rows)
60
+ cols = list(table.columns)
61
+ if len(rows) < min_rows or len(cols) < min_cols:
62
+ continue
63
+ preview: list[list[str]] = []
64
+ for row in rows[:preview_rows]:
65
+ preview.append([c.text.strip()[:max_cell_len] for c in row.cells])
66
+ schemas.append(TableSchema(
67
+ index=g_idx, rows=len(rows), cols=len(cols),
68
+ preview=preview, location=f"slide {s_idx}",
69
+ ))
70
+ return schemas
71
+
72
+ # ---- editing ----
73
+ def render_template(self, context: dict[str, Any]) -> None:
74
+ """paragraph 단위로 {{key}}를 치환. run이 쪼개진 경우를 처리하기 위해
75
+ paragraph 전체 텍스트를 재조립 후 첫 run에 담는다 (서식 일부 손실 가능)."""
76
+ for tf in self._iter_text_frames():
77
+ for para in tf.paragraphs:
78
+ full_text = "".join(run.text for run in para.runs)
79
+ if not TAG_PATTERN.search(full_text):
80
+ continue
81
+ rendered = TAG_PATTERN.sub(
82
+ lambda m: str(context.get(m.group(1), m.group(0))),
83
+ full_text,
84
+ )
85
+ if para.runs:
86
+ para.runs[0].text = rendered
87
+ for run in para.runs[1:]:
88
+ run.text = ""
89
+
90
+ def _get_table(self, table_index: int):
91
+ for g_idx, _, table in self._iter_tables():
92
+ if g_idx == table_index:
93
+ return table
94
+ raise IndexError(f"PPTX table index {table_index} not found")
95
+
96
+ def set_cell(self, table_index: int, row: int, col: int, value: str) -> str:
97
+ table = self._get_table(table_index)
98
+ cell = table.cell(row, col)
99
+ old = cell.text
100
+ cell.text = value
101
+ return old
102
+
103
+ def append_row(self, table_index: int, values: list[str]) -> None:
104
+ """python-pptx는 표 행 추가 API를 제공하지 않는다.
105
+ LLM에게는 '지원 안 함'으로 알리는 게 정직한 방식."""
106
+ raise NotImplementedError(
107
+ "PPTX는 python-pptx에 동적 행 추가 API가 없음. "
108
+ "템플릿 단계에서 충분한 빈 행을 만들어 두고 set_cell로 채우는 방식을 권장."
109
+ )
@@ -0,0 +1,232 @@
1
+ """LLM tool 정의 + 실행 함수.
2
+
3
+ 동일 구현을 MCP 서버와 Claude API Tool Use 양쪽에서 재사용한다.
4
+ 각 함수는 JSON-serializable dict를 반환.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import shutil
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from . import load
13
+
14
+ # -------- JSON schemas (Claude API tool use와 MCP 공용) --------
15
+
16
+ TOOL_DEFINITIONS: list[dict[str, Any]] = [
17
+ {
18
+ "name": "inspect_document",
19
+ "description": (
20
+ "문서(.docx/.pptx/.hwpx)의 구조를 분석한다. "
21
+ "placeholders({{key}} 태그 목록)와 tables(각 표의 행/열/미리보기)를 반환한다. "
22
+ "LLM이 어떤 필드를 채우거나 수정할지 판단할 때 먼저 호출해야 한다."
23
+ ),
24
+ "input_schema": {
25
+ "type": "object",
26
+ "properties": {
27
+ "path": {
28
+ "type": "string",
29
+ "description": "문서 절대경로",
30
+ },
31
+ "min_rows": {
32
+ "type": "integer",
33
+ "description": "표 필터: 최소 행 수 (기본 1)",
34
+ "default": 1,
35
+ },
36
+ "min_cols": {
37
+ "type": "integer",
38
+ "description": "표 필터: 최소 열 수 (기본 1)",
39
+ "default": 1,
40
+ },
41
+ },
42
+ "required": ["path"],
43
+ },
44
+ },
45
+ {
46
+ "name": "render_template",
47
+ "description": (
48
+ "문서의 {{key}} placeholder를 context의 값으로 치환해 새 파일로 저장한다. "
49
+ "DOCX는 docxtpl(Jinja2 loop/if 지원), PPTX/HWPX는 단순 {{key}} 치환."
50
+ ),
51
+ "input_schema": {
52
+ "type": "object",
53
+ "properties": {
54
+ "path": {"type": "string", "description": "템플릿 파일 경로"},
55
+ "context": {
56
+ "type": "object",
57
+ "description": "{{key}}에 주입할 값 dict",
58
+ "additionalProperties": True,
59
+ },
60
+ "output_path": {
61
+ "type": "string",
62
+ "description": "결과 저장 경로 (생략 시 원본 옆에 _rendered 붙여 저장)",
63
+ },
64
+ },
65
+ "required": ["path", "context"],
66
+ },
67
+ },
68
+ {
69
+ "name": "set_cell",
70
+ "description": (
71
+ "특정 표의 셀 값을 교체한다. table_index는 inspect_document의 tables 배열 인덱스. "
72
+ "PPTX는 슬라이드 경계와 무관한 전역 index."
73
+ ),
74
+ "input_schema": {
75
+ "type": "object",
76
+ "properties": {
77
+ "path": {"type": "string"},
78
+ "table_index": {"type": "integer"},
79
+ "row": {"type": "integer"},
80
+ "col": {"type": "integer"},
81
+ "value": {"type": "string"},
82
+ "output_path": {
83
+ "type": "string",
84
+ "description": "생략 시 원본 덮어쓰기",
85
+ },
86
+ },
87
+ "required": ["path", "table_index", "row", "col", "value"],
88
+ },
89
+ },
90
+ {
91
+ "name": "append_row",
92
+ "description": (
93
+ "표 끝에 새 행을 추가한다. **DOCX만 지원** — PPTX/HWPX는 API 미지원으로 에러 반환. "
94
+ "그 경우 템플릿 단계에서 충분한 빈 행을 두고 set_cell로 채워야 한다."
95
+ ),
96
+ "input_schema": {
97
+ "type": "object",
98
+ "properties": {
99
+ "path": {"type": "string"},
100
+ "table_index": {"type": "integer"},
101
+ "values": {
102
+ "type": "array",
103
+ "items": {"type": "string"},
104
+ "description": "새 행의 각 셀 값. 열 수보다 적으면 나머지는 공백.",
105
+ },
106
+ "output_path": {"type": "string"},
107
+ },
108
+ "required": ["path", "table_index", "values"],
109
+ },
110
+ },
111
+ ]
112
+
113
+
114
+ # -------- 실행 함수 --------
115
+
116
+ def _resolve_output(path: str, output_path: str | None, suffix: str = "_out") -> Path:
117
+ if output_path:
118
+ return Path(output_path)
119
+ p = Path(path)
120
+ return p.with_name(f"{p.stem}{suffix}{p.suffix}")
121
+
122
+
123
+ def inspect_document(path: str, min_rows: int = 1, min_cols: int = 1) -> dict[str, Any]:
124
+ doc = load(path)
125
+ try:
126
+ schema = doc.get_schema()
127
+ # min_rows/min_cols 필터 재적용
128
+ filtered = [t for t in doc.get_tables(min_rows=min_rows, min_cols=min_cols)]
129
+ result = schema.to_dict()
130
+ result["tables"] = [t.to_dict() for t in filtered]
131
+ return result
132
+ finally:
133
+ doc.close()
134
+
135
+
136
+ def render_template(path: str, context: dict[str, Any],
137
+ output_path: str | None = None) -> dict[str, Any]:
138
+ out = _resolve_output(path, output_path, "_rendered")
139
+ shutil.copy2(path, out)
140
+
141
+ doc = load(out)
142
+ try:
143
+ before = doc.get_placeholders()
144
+ doc.render_template(context)
145
+ doc.save()
146
+ finally:
147
+ doc.close()
148
+
149
+ # 검증 재로드
150
+ doc2 = load(out)
151
+ try:
152
+ after = doc2.get_placeholders()
153
+ finally:
154
+ doc2.close()
155
+
156
+ return {
157
+ "output_path": str(out),
158
+ "placeholders_before": before,
159
+ "placeholders_after": after,
160
+ "rendered_count": len(before) - len(after),
161
+ }
162
+
163
+
164
+ def set_cell(path: str, table_index: int, row: int, col: int, value: str,
165
+ output_path: str | None = None) -> dict[str, Any]:
166
+ target = Path(output_path) if output_path else Path(path)
167
+ if output_path and Path(path) != target:
168
+ shutil.copy2(path, target)
169
+
170
+ doc = load(target)
171
+ try:
172
+ old = doc.set_cell(table_index, row, col, value)
173
+ doc.save()
174
+ finally:
175
+ doc.close()
176
+
177
+ return {
178
+ "output_path": str(target),
179
+ "table_index": table_index,
180
+ "row": row,
181
+ "col": col,
182
+ "previous_value": old,
183
+ "new_value": value,
184
+ }
185
+
186
+
187
+ def append_row(path: str, table_index: int, values: list[str],
188
+ output_path: str | None = None) -> dict[str, Any]:
189
+ target = Path(output_path) if output_path else Path(path)
190
+ if output_path and Path(path) != target:
191
+ shutil.copy2(path, target)
192
+
193
+ doc = load(target)
194
+ try:
195
+ doc.append_row(table_index, values)
196
+ doc.save()
197
+ new_tables = doc.get_tables()
198
+ finally:
199
+ doc.close()
200
+
201
+ target_schema = next((t for t in new_tables if t.index == table_index), None)
202
+ return {
203
+ "output_path": str(target),
204
+ "table_index": table_index,
205
+ "new_row_count": target_schema.rows if target_schema else None,
206
+ "appended_values": values,
207
+ }
208
+
209
+
210
+ # -------- 이름으로 dispatch --------
211
+
212
+ TOOL_HANDLERS = {
213
+ "inspect_document": inspect_document,
214
+ "render_template": render_template,
215
+ "set_cell": set_cell,
216
+ "append_row": append_row,
217
+ }
218
+
219
+
220
+ def call_tool(name: str, arguments: dict[str, Any]) -> dict[str, Any]:
221
+ """이름으로 tool 실행. 예외도 dict로 직렬화."""
222
+ handler = TOOL_HANDLERS.get(name)
223
+ if handler is None:
224
+ return {"error": f"unknown tool: {name}"}
225
+ try:
226
+ return handler(**arguments)
227
+ except NotImplementedError as e:
228
+ return {"error": "not_implemented", "message": str(e)}
229
+ except (IndexError, ValueError, FileNotFoundError) as e:
230
+ return {"error": type(e).__name__, "message": str(e)}
231
+ except Exception as e:
232
+ return {"error": "unexpected", "type": type(e).__name__, "message": str(e)}
@@ -0,0 +1,252 @@
1
+ Metadata-Version: 2.4
2
+ Name: document-adapter
3
+ Version: 0.1.0
4
+ Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
+ Author-email: Son Seongjun <sonsj97@plateer.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/PlateerLab/document-adapter
8
+ Project-URL: Repository, https://github.com/PlateerLab/document-adapter
9
+ Keywords: mcp,llm,document,docx,pptx,hwpx,template,claude
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.10
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Requires-Python: >=3.10
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Requires-Dist: python-docx>=1.1
20
+ Requires-Dist: docxtpl>=0.20
21
+ Requires-Dist: python-pptx>=1.0
22
+ Requires-Dist: python-hwpx>=2.9
23
+ Requires-Dist: mcp>=1.0
24
+ Provides-Extra: claude
25
+ Requires-Dist: anthropic>=0.40; extra == "claude"
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest>=8; extra == "dev"
28
+ Dynamic: license-file
29
+
30
+ # document-adapter
31
+
32
+ **LLM이 DOCX / PPTX / HWPX 문서를 직접 편집할 수 있게 해주는 통합 어댑터 + MCP 서버.**
33
+
34
+ 세 가지 오피스 포맷을 하나의 파이썬 인터페이스로 추상화하고, Claude Desktop / Claude Code / Anthropic API Tool Use에서 바로 호출할 수 있는 MCP 도구로 노출합니다. 양식 문서의 빈 셀을 자동으로 채우거나, 템플릿의 `{{key}}`를 치환하거나, 기존 표의 내용을 수정하는 작업을 LLM 에이전트가 수행할 수 있습니다.
35
+
36
+ ## 지원 포맷
37
+
38
+ | 포맷 | 백엔드 | 템플릿 렌더 | 표 읽기 | 셀 수정 | 행 추가 |
39
+ |---|---|---|---|---|---|
40
+ | `.docx` | `docxtpl` + `python-docx` | Jinja2 (`{%tr%}` loop 포함) | ✅ | ✅ | ✅ |
41
+ | `.pptx` | `python-pptx` | `{{key}}` 치환 | ✅ (슬라이드 위치 포함) | ✅ | ❌ (미지원) |
42
+ | `.hwpx` | `python-hwpx` (Pure Python) | `{{key}}` 치환 | ✅ | ✅ | ❌ (미지원) |
43
+
44
+ - HWPX는 한컴오피스 설치가 **불필요**합니다 (macOS/Linux 서버에서 그대로 동작).
45
+ - 구버전 `.hwp`(바이너리 포맷)는 지원하지 않습니다 — `.hwpx`로 변환 후 사용하세요.
46
+
47
+ ## 설치
48
+
49
+ ```bash
50
+ pip install -e .
51
+
52
+ # Claude API 예시 스크립트까지 쓰려면
53
+ pip install -e ".[claude]"
54
+ ```
55
+
56
+ Python 3.10+ 필요.
57
+
58
+ ## 빠른 시작 — 파이썬 API
59
+
60
+ ```python
61
+ from document_adapter import load
62
+
63
+ doc = load("report_template.docx")
64
+
65
+ # 1. 구조 파악
66
+ schema = doc.get_schema()
67
+ print(schema.placeholders) # ['author', 'date', 'title']
68
+ print(schema.tables) # [TableSchema(index=0, rows=7, cols=2, ...), ...]
69
+
70
+ # 2. 템플릿 렌더
71
+ doc.render_template({
72
+ "title": "Q1 운영 리포트",
73
+ "author": "손성준",
74
+ "date": "2026-04-15",
75
+ })
76
+ doc.save("report_filled.docx")
77
+
78
+ # 3. 기존 양식 파일의 표 셀 수정
79
+ doc = load("checklist.docx")
80
+ old = doc.set_cell(table_index=1, row=1, col=1, value="○○전자")
81
+ doc.append_row(1, ["새 항목", "값"]) # DOCX만 지원
82
+ doc.save("checklist_filled.docx")
83
+ doc.close()
84
+ ```
85
+
86
+ 확장자로 자동 분기되므로 `.pptx` / `.hwpx`도 동일한 API를 사용합니다.
87
+
88
+ ## MCP 서버로 사용 — Claude Desktop / Claude Code
89
+
90
+ ### 실행
91
+
92
+ ```bash
93
+ python -m document_adapter.mcp_server
94
+ # 또는 설치 후
95
+ document-adapter-mcp
96
+ ```
97
+
98
+ ### Claude Desktop 설정
99
+
100
+ `~/Library/Application Support/Claude/claude_desktop_config.json`:
101
+
102
+ ```json
103
+ {
104
+ "mcpServers": {
105
+ "document-adapter": {
106
+ "command": "/absolute/path/to/venv/bin/python",
107
+ "args": ["-m", "document_adapter.mcp_server"]
108
+ }
109
+ }
110
+ }
111
+ ```
112
+
113
+ 재시작하면 Claude Desktop에서 아래 4개 도구를 사용할 수 있습니다.
114
+
115
+ ### Claude Code 설정
116
+
117
+ ```bash
118
+ claude mcp add document-adapter \
119
+ /absolute/path/to/venv/bin/python -m document_adapter.mcp_server
120
+ ```
121
+
122
+ ## Anthropic API Tool Use로 사용
123
+
124
+ `document_adapter.tools`가 Claude API의 tool schema 형식과 그대로 호환됩니다.
125
+
126
+ ```python
127
+ import anthropic
128
+ from document_adapter.tools import TOOL_DEFINITIONS, call_tool
129
+
130
+ client = anthropic.Anthropic()
131
+
132
+ resp = client.messages.create(
133
+ model="claude-opus-4-6",
134
+ max_tokens=4096,
135
+ tools=[{
136
+ "name": t["name"],
137
+ "description": t["description"],
138
+ "input_schema": t["input_schema"],
139
+ } for t in TOOL_DEFINITIONS],
140
+ messages=[{
141
+ "role": "user",
142
+ "content": "report_template.docx의 표 구조를 확인하고 빈 셀을 적절히 채워줘",
143
+ }],
144
+ )
145
+
146
+ # tool_use 블록을 받으면 call_tool(name, args)로 실행 후 결과 반환
147
+ ```
148
+
149
+ 전체 agent loop 예시는 [`examples/claude_api_example.py`](examples/claude_api_example.py) 참고.
150
+
151
+ ## 노출되는 4개 도구
152
+
153
+ | 도구 | 설명 |
154
+ |---|---|
155
+ | `inspect_document` | 문서 구조(placeholders, tables)를 JSON으로 반환. **항상 첫 호출로 사용** |
156
+ | `render_template` | `{{key}}`를 context dict 값으로 치환해 새 파일 저장 |
157
+ | `set_cell` | 특정 표의 `(row, col)` 셀 값 교체 |
158
+ | `append_row` | 표 끝에 새 행 추가 (DOCX 전용) |
159
+
160
+ ### `inspect_document` 반환 예시
161
+
162
+ ```json
163
+ {
164
+ "format": "docx",
165
+ "source": "/path/to/checklist.docx",
166
+ "placeholders": [],
167
+ "tables": [
168
+ {
169
+ "index": 1,
170
+ "rows": 7,
171
+ "cols": 2,
172
+ "location": null,
173
+ "preview": [
174
+ {"row": 0, "cells": ["항목", "기입 내용"]},
175
+ {"row": 1, "cells": ["고객사 / 조직", ""]},
176
+ {"row": 2, "cells": ["현업 담당부서 / 책임자", ""]}
177
+ ]
178
+ }
179
+ ]
180
+ }
181
+ ```
182
+
183
+ LLM은 이 preview를 보고 **"빈 셀이 어디 있는지 / 어떤 값을 넣어야 하는지"** 를 판단하여 `set_cell`을 호출합니다.
184
+
185
+ ## 템플릿 작성 규칙
186
+
187
+ ### DOCX — Jinja2 전체 문법 사용 가능
188
+
189
+ ```
190
+ {{ report_title }}
191
+ 작성자: {{ author }}
192
+
193
+ {% for item in items %}- {{ item.name }}: {{ item.value }}
194
+ {% endfor %}
195
+ ```
196
+
197
+ **표 행 반복은 `{%tr for ... %}` / `{%tr endfor %}`를 각각 별도 행에 두어야 합니다.**
198
+ 같은 행에 두 태그를 넣으면 `<w:tr>` 전체가 `{% for %}`로 교체되어 `endfor`가 손실됩니다.
199
+
200
+ ```
201
+ ┌─────────────────────┬─────┬─────┐
202
+ │ 항목 │ 목표 │ 실적 │ <- 헤더
203
+ ├─────────────────────┼─────┼─────┤
204
+ │ {%tr for r in rows %} │ <- for 행
205
+ ├─────────────────────┼─────┼─────┤
206
+ │ {{ r.name }} │ {{ r.target }} │ {{ r.actual }} │ <- 반복 본문
207
+ ├─────────────────────┼─────┼─────┤
208
+ │ {%tr endfor %} │ <- endfor 행
209
+ └─────────────────────┴─────┴─────┘
210
+ ```
211
+
212
+ ### PPTX / HWPX — 단순 `{{key}}` 치환만
213
+
214
+ loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `run`으로 쪼개질 수 있어, 어댑터가 paragraph 전체 텍스트를 재조립한 뒤 첫 `run`에 다시 담는 방식으로 처리합니다 (서식 일부 손실 가능).
215
+
216
+ ## 내장된 버그 회피
217
+
218
+ | 포맷 | 문제 | 어댑터의 처리 |
219
+ |---|---|---|
220
+ | HWPX | `python-hwpx 2.9.0`의 `set_cell_text()`가 빈 셀에서 lxml/ElementTree 혼용 `TypeError` 발생 | `paragraphs[0].text = value` 직접 할당으로 우회 |
221
+ | HWPX | `replace_text_in_runs()`가 한글 공백이 run으로 쪼개진 경우 매칭 실패 | 위치 기반 API만 사용 |
222
+ | HWPX | `manifest fallback` 경고 로그가 과도하게 출력됨 | `logging.getLogger("hwpx")` 레벨을 `ERROR`로 조정 |
223
+ | PPTX | placeholder가 여러 `run`으로 쪼개져 단순 `run.text` 치환이 실패 | paragraph 전체 재조립 |
224
+ | DOCX | `docxtpl`의 `{%tr%}`를 같은 행에 두면 파싱 에러 | README에 배치 규칙 명시 |
225
+
226
+ ## 프로젝트 구조
227
+
228
+ ```
229
+ document_adapter/
230
+ ├── __init__.py # load() dispatcher
231
+ ├── base.py # DocumentAdapter ABC, TableSchema, DocumentSchema
232
+ ├── docx_adapter.py # DocxAdapter
233
+ ├── pptx_adapter.py # PptxAdapter
234
+ ├── hwpx_adapter.py # HwpxAdapter (버그 회피 포함)
235
+ ├── tools.py # Tool 정의 + call_tool dispatcher
236
+ └── mcp_server.py # MCP stdio server
237
+
238
+ examples/
239
+ └── claude_api_example.py
240
+ ```
241
+
242
+ ## 라이선스
243
+
244
+ MIT
245
+
246
+ ## Credits
247
+
248
+ - [`python-docx`](https://github.com/python-openxml/python-docx)
249
+ - [`docxtpl`](https://github.com/elapouya/python-docx-template)
250
+ - [`python-pptx`](https://github.com/scanny/python-pptx)
251
+ - [`python-hwpx`](https://github.com/airmang/python-hwpx)
252
+ - [`mcp`](https://github.com/modelcontextprotocol/python-sdk)
@@ -0,0 +1,13 @@
1
+ document_adapter/__init__.py,sha256=YqjF6xhkI1X2sbO5XfZE9kQVA4RMzTuwu7TpuMUGJDk,1121
2
+ document_adapter/base.py,sha256=pQlYsSnMJfYa-GaSkzSlwqVhe1udRyv6STTryuEFftc,3001
3
+ document_adapter/docx_adapter.py,sha256=IxpSXKaD_9kspalKcQ3AbsVJodK6pJ_t5XoUy5fv5Kg,2702
4
+ document_adapter/hwpx_adapter.py,sha256=46uIkn8dDHeJkx4u3xU1ZZaXTTovdgzHWN1ZN0HFep0,4683
5
+ document_adapter/mcp_server.py,sha256=veEl7S5EJMP0OLSlQI6pWvFSKJ-0lWm0GHX46G0EfC4,1609
6
+ document_adapter/pptx_adapter.py,sha256=doV9NjgyK5OrrB-TkrvYrwhusqCfxZXjk_iiLkJOy-g,4215
7
+ document_adapter/tools.py,sha256=84jib-rNQUhWFER1R6k7zIzBDGHgnx6MCHCh2keAcp0,7608
8
+ document_adapter-0.1.0.dist-info/licenses/LICENSE,sha256=QjKz-cVDxa_NbNhNYo9ek6yDMnR4YHA2pKg-upo93x4,1068
9
+ document_adapter-0.1.0.dist-info/METADATA,sha256=pGILinIsgp7YVDeNQ7hfPk_UObx8Ro-rqxYS-wPkQ0E,8809
10
+ document_adapter-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
11
+ document_adapter-0.1.0.dist-info/entry_points.txt,sha256=krV-Z-Yc1M3qe2I1pxhEJX7wdwbKwfX-8sEN3-3CuQo,79
12
+ document_adapter-0.1.0.dist-info/top_level.txt,sha256=WGApqbHX1xpY13c2OvG83DxDZToMonYK2xkWN_QeT8M,17
13
+ document_adapter-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (82.0.1)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ document-adapter-mcp = document_adapter.mcp_server:main_sync
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Plateer Lab
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ document_adapter