document-adapter 0.7.0__tar.gz → 0.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {document_adapter-0.7.0/document_adapter.egg-info → document_adapter-0.7.2}/PKG-INFO +65 -20
  2. {document_adapter-0.7.0 → document_adapter-0.7.2}/README.md +64 -19
  3. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/base.py +142 -11
  4. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/tools.py +19 -5
  5. {document_adapter-0.7.0 → document_adapter-0.7.2/document_adapter.egg-info}/PKG-INFO +65 -20
  6. {document_adapter-0.7.0 → document_adapter-0.7.2}/pyproject.toml +1 -1
  7. {document_adapter-0.7.0 → document_adapter-0.7.2}/tests/test_smoke.py +44 -2
  8. {document_adapter-0.7.0 → document_adapter-0.7.2}/LICENSE +0 -0
  9. {document_adapter-0.7.0 → document_adapter-0.7.2}/NOTICE +0 -0
  10. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/__init__.py +0 -0
  11. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/docx_adapter.py +0 -0
  12. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_adapter.py +0 -0
  13. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_core/__init__.py +0 -0
  14. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_core/constants.py +0 -0
  15. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_core/grid.py +0 -0
  16. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_core/package.py +0 -0
  17. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/hwpx_core/paragraph.py +0 -0
  18. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/mcp_server.py +0 -0
  19. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter/pptx_adapter.py +0 -0
  20. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter.egg-info/SOURCES.txt +0 -0
  21. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter.egg-info/dependency_links.txt +0 -0
  22. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter.egg-info/entry_points.txt +0 -0
  23. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter.egg-info/requires.txt +0 -0
  24. {document_adapter-0.7.0 → document_adapter-0.7.2}/document_adapter.egg-info/top_level.txt +0 -0
  25. {document_adapter-0.7.0 → document_adapter-0.7.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: document-adapter
3
- Version: 0.7.0
3
+ Version: 0.7.2
4
4
  Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
5
  Author-email: Son Seongjun <sonsj97@plateer.com>
6
6
  License: MIT
@@ -127,13 +127,46 @@ doc.append_to_cell(table_index=2, row=0, col=0, value="홍길동")
127
127
  cell = doc.get_cell(table_index=1, row=3, col=2)
128
128
  print(cell.text, cell.is_anchor, cell.span, cell.nested_table_indices)
129
129
 
130
- # DOCX/HWPX는 행 추가 지원
130
+ # DOCX/PPTX/HWPX 전부 행 추가 지원 (v0.5+)
131
131
  doc.append_row(1, ["새 항목", "값"])
132
132
 
133
133
  doc.save("checklist_filled.docx")
134
134
  doc.close()
135
135
  ```
136
136
 
137
+ ### 라벨 기반 일괄 채우기 (v0.7+)
138
+
139
+ LLM 이 좌표 `(table_index, row, col)` 를 직접 계산하지 않고 "접수번호", "성명" 같은 사람이 읽는 라벨 key-value 로 양식을 채울 수 있습니다.
140
+
141
+ ```python
142
+ doc = load("form.hwpx")
143
+ result = doc.fill_form({
144
+ "접수번호": "2026-0001",
145
+ "성 명": "홍길동",
146
+ "주 소": "서울시 강남구",
147
+ "금융회사": "국민은행",
148
+ })
149
+ # → {"filled": [...], "not_found": [...], "ambiguous": [...]}
150
+ doc.save()
151
+ doc.close()
152
+ ```
153
+
154
+ - `auto` (기본): 라벨 셀 오른쪽 → 아래 → 같은 셀 순으로 값 셀 탐색. 보수적이라 기존 값 있는 셀은 다른 라벨로 간주하고 skip.
155
+ - `direction="right"` 명시: 라벨 오른쪽 셀을 **덮어쓰기** (예시값 있는 PPTX 템플릿 등).
156
+ - **Dot-path 섹션 지정**: 동일 라벨이 여러 섹션에 있으면 `"피해자.금액"`, `"지급정지요청계좌.금액"` 처럼 섹션 힌트 부여. `ambiguous` 반환 시 `hint` 필드에 예시 제공.
157
+ - **팁**: 한 양식의 관련 라벨을 한 번에 dict 로 넘기면 라벨끼리 서로 보호되어 오염을 방지합니다.
158
+
159
+ ### 셀 크기 메타 (v0.6+)
160
+
161
+ `get_tables()`가 `column_widths_cm` / `row_heights_cm` 를, `get_cell()`이 `width_cm` / `height_cm` / `char_count` 를 반환해 LLM이 **좁은 셀에 긴 텍스트를 넣어 오버플로 되는 것을 사전에 판단**할 수 있습니다.
162
+
163
+ ```python
164
+ cell = doc.get_cell(table_index=0, row=0, col=0)
165
+ print(cell.width_cm, cell.char_count) # 1.7cm, 4자 — 작은 배지
166
+ ```
167
+
168
+ DOCX/PPTX는 EMU → cm, HWPX는 HU → cm 자동 환산 (1자리 반올림).
169
+
137
170
  확장자로 자동 분기되므로 `.pptx` / `.hwpx`도 동일한 API를 사용합니다.
138
171
 
139
172
  ### 병합 셀 인지 동작 (v0.2+)
@@ -185,7 +218,7 @@ document-adapter-mcp
185
218
  }
186
219
  ```
187
220
 
188
- 재시작하면 Claude Desktop에서 아래 4개 도구를 사용할 수 있습니다.
221
+ 재시작하면 Claude Desktop에서 아래 7개 도구를 사용할 수 있습니다.
189
222
 
190
223
  ### Claude Code 설정
191
224
 
@@ -227,12 +260,13 @@ resp = client.messages.create(
227
260
 
228
261
  | 도구 | 설명 |
229
262
  |---|---|
230
- | `inspect_document` | 문서 구조(placeholders, tables)를 JSON으로 반환. **항상 첫 호출로 사용** |
263
+ | `inspect_document` | 문서 구조(placeholders, tables + `column_widths_cm`/`row_heights_cm`)를 JSON으로 반환. **항상 첫 호출로 사용** |
231
264
  | `render_template` | `{{key}}`를 context dict 값으로 치환해 새 파일 저장 |
232
- | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 반환 (preview의 40자 잘림 없이) |
265
+ | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 + `width_cm`/`height_cm`/`char_count` 반환 |
233
266
  | `set_cell` | 특정 표의 `(row, col)` 셀 값 교체 (병합 anchor만) |
234
267
  | `append_to_cell` | 기존 텍스트 뒤에 값 덧붙임 (라벨 유지용, 예: `"성 명"` → `"성 명 홍길동"`) |
235
- | `append_row` | 표 끝에 새 행 추가 (DOCX/HWPX 지원) |
268
+ | `fill_form` (v0.7+) | **라벨 이름**으로 일괄 채우기. 좌표 계산 없이 `{"접수번호": "...", "성명": "..."}` dict. dot-path 섹션 해소 지원 |
269
+ | `append_row` | 표 끝에 새 행 추가 (DOCX/PPTX/HWPX 전부 지원, v0.5+) |
236
270
 
237
271
  ### `inspect_document` 반환 예시 (v0.2+)
238
272
 
@@ -294,13 +328,12 @@ LLM은 이 preview를 보고 **"빈 셀이 어디 있는지 / 어떤 값을 넣
294
328
 
295
329
  loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `run`으로 쪼개질 수 있어, 어댑터가 paragraph 전체 텍스트를 재조립한 뒤 첫 `run`에 다시 담는 방식으로 처리합니다 (서식 일부 손실 가능).
296
330
 
297
- ## 내장된 버그 회피
331
+ ## 내장된 버그 회피 / 백엔드 선택
298
332
 
299
333
  | 포맷 | 문제 | 어댑터의 처리 |
300
334
  |---|---|---|
301
- | HWPX | `python-hwpx 2.9.0`의 `set_cell_text()`가 빈 셀에서 lxml/ElementTree 혼용 `TypeError` 발생 | `paragraphs[0].text = value` 직접 할당으로 우회 |
302
- | HWPX | `replace_text_in_runs()`가 한글 공백이 run으로 쪼개진 경우 매칭 실패 | 위치 기반 API만 사용 |
303
- | HWPX | `manifest fallback` 경고 로그가 과도하게 출력됨 | `logging.getLogger("hwpx")` 레벨을 `ERROR`로 조정 |
335
+ | HWPX | `python-hwpx` 가 Non-Commercial License → 상용 배포 블로커 | **v0.4.0 부터 자체 `hwpx_core` 모듈** (zipfile + lxml) 로 교체. 런타임에 `python-hwpx` 불필요. 테스트 fixture 생성에만 사용 (dev extras) |
336
+ | PPTX | `python-pptx` 에 공식 `add_row` API 없음 (issue #86, 2014년부터 open) | **v0.5.0 부터 자체 lxml 구현** (`<a:tr>` deepcopy 패턴) |
304
337
  | PPTX | placeholder가 여러 `run`으로 쪼개져 단순 `run.text` 치환이 실패 | paragraph 전체 재조립 |
305
338
  | DOCX | `docxtpl`의 `{%tr%}`를 같은 행에 두면 파싱 에러 | README에 배치 규칙 명시 |
306
339
 
@@ -309,15 +342,20 @@ loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `
309
342
  ```
310
343
  document_adapter/
311
344
  ├── __init__.py # load() dispatcher
312
- ├── base.py # DocumentAdapter ABC, TableSchema, DocumentSchema
345
+ ├── base.py # DocumentAdapter ABC + fill_form + dataclasses
313
346
  ├── docx_adapter.py # DocxAdapter
314
- ├── pptx_adapter.py # PptxAdapter
315
- ├── hwpx_adapter.py # HwpxAdapter (버그 회피 포함)
316
- ├── tools.py # Tool 정의 + call_tool dispatcher
347
+ ├── pptx_adapter.py # PptxAdapter (append_row 자체 구현 포함)
348
+ ├── hwpx_adapter.py # HwpxAdapter (hwpx_core 기반)
349
+ ├── hwpx_core/ # 자체 HWPX 패키지 (v0.4+)
350
+ │ ├── constants.py
351
+ │ ├── package.py # ZIP + dirty XML 관리
352
+ │ ├── grid.py # iter_grid, table_shape
353
+ │ └── paragraph.py # run-level 편집 헬퍼
354
+ ├── tools.py # 7개 MCP 도구 정의 + call_tool dispatcher
317
355
  └── mcp_server.py # MCP stdio server
318
356
 
319
357
  examples/
320
- └── claude_api_example.py
358
+ └── claude_api_example.py # Claude API Tool Use 에이전트 루프
321
359
  ```
322
360
 
323
361
  ## 라이선스
@@ -326,8 +364,15 @@ MIT
326
364
 
327
365
  ## Credits
328
366
 
329
- - [`python-docx`](https://github.com/python-openxml/python-docx)
330
- - [`docxtpl`](https://github.com/elapouya/python-docx-template)
331
- - [`python-pptx`](https://github.com/scanny/python-pptx)
332
- - [`python-hwpx`](https://github.com/airmang/python-hwpx)
333
- - [`mcp`](https://github.com/modelcontextprotocol/python-sdk)
367
+ **런타임 의존성** (전부 허용형 OSS):
368
+ - [`python-docx`](https://github.com/python-openxml/python-docx) — MIT
369
+ - [`docxtpl`](https://github.com/elapouya/python-docx-template) — LGPL-2.1
370
+ - [`python-pptx`](https://github.com/scanny/python-pptx) — MIT
371
+ - [`lxml`](https://lxml.de/) — BSD
372
+ - [`mcp`](https://github.com/modelcontextprotocol/python-sdk) — MIT
373
+
374
+ **코드 참조**:
375
+ - [`xgen-doc2chunk`](https://github.com/PlateerLab/xgen-doc2chunk) (Apache-2.0) — HWPX table grid 파싱 로직 차용 (`NOTICE` 참조)
376
+
377
+ **Dev 전용** (fixture 생성에만 사용):
378
+ - [`python-hwpx`](https://github.com/airmang/python-hwpx) — Non-Commercial License (v0.4.0 부터 런타임 의존성 제거)
@@ -86,13 +86,46 @@ doc.append_to_cell(table_index=2, row=0, col=0, value="홍길동")
86
86
  cell = doc.get_cell(table_index=1, row=3, col=2)
87
87
  print(cell.text, cell.is_anchor, cell.span, cell.nested_table_indices)
88
88
 
89
- # DOCX/HWPX는 행 추가 지원
89
+ # DOCX/PPTX/HWPX 전부 행 추가 지원 (v0.5+)
90
90
  doc.append_row(1, ["새 항목", "값"])
91
91
 
92
92
  doc.save("checklist_filled.docx")
93
93
  doc.close()
94
94
  ```
95
95
 
96
+ ### 라벨 기반 일괄 채우기 (v0.7+)
97
+
98
+ LLM 이 좌표 `(table_index, row, col)` 를 직접 계산하지 않고 "접수번호", "성명" 같은 사람이 읽는 라벨 key-value 로 양식을 채울 수 있습니다.
99
+
100
+ ```python
101
+ doc = load("form.hwpx")
102
+ result = doc.fill_form({
103
+ "접수번호": "2026-0001",
104
+ "성 명": "홍길동",
105
+ "주 소": "서울시 강남구",
106
+ "금융회사": "국민은행",
107
+ })
108
+ # → {"filled": [...], "not_found": [...], "ambiguous": [...]}
109
+ doc.save()
110
+ doc.close()
111
+ ```
112
+
113
+ - `auto` (기본): 라벨 셀 오른쪽 → 아래 → 같은 셀 순으로 값 셀 탐색. 보수적이라 기존 값 있는 셀은 다른 라벨로 간주하고 skip.
114
+ - `direction="right"` 명시: 라벨 오른쪽 셀을 **덮어쓰기** (예시값 있는 PPTX 템플릿 등).
115
+ - **Dot-path 섹션 지정**: 동일 라벨이 여러 섹션에 있으면 `"피해자.금액"`, `"지급정지요청계좌.금액"` 처럼 섹션 힌트 부여. `ambiguous` 반환 시 `hint` 필드에 예시 제공.
116
+ - **팁**: 한 양식의 관련 라벨을 한 번에 dict 로 넘기면 라벨끼리 서로 보호되어 오염을 방지합니다.
117
+
118
+ ### 셀 크기 메타 (v0.6+)
119
+
120
+ `get_tables()`가 `column_widths_cm` / `row_heights_cm` 를, `get_cell()`이 `width_cm` / `height_cm` / `char_count` 를 반환해 LLM이 **좁은 셀에 긴 텍스트를 넣어 오버플로 되는 것을 사전에 판단**할 수 있습니다.
121
+
122
+ ```python
123
+ cell = doc.get_cell(table_index=0, row=0, col=0)
124
+ print(cell.width_cm, cell.char_count) # 1.7cm, 4자 — 작은 배지
125
+ ```
126
+
127
+ DOCX/PPTX는 EMU → cm, HWPX는 HU → cm 자동 환산 (1자리 반올림).
128
+
96
129
  확장자로 자동 분기되므로 `.pptx` / `.hwpx`도 동일한 API를 사용합니다.
97
130
 
98
131
  ### 병합 셀 인지 동작 (v0.2+)
@@ -144,7 +177,7 @@ document-adapter-mcp
144
177
  }
145
178
  ```
146
179
 
147
- 재시작하면 Claude Desktop에서 아래 4개 도구를 사용할 수 있습니다.
180
+ 재시작하면 Claude Desktop에서 아래 7개 도구를 사용할 수 있습니다.
148
181
 
149
182
  ### Claude Code 설정
150
183
 
@@ -186,12 +219,13 @@ resp = client.messages.create(
186
219
 
187
220
  | 도구 | 설명 |
188
221
  |---|---|
189
- | `inspect_document` | 문서 구조(placeholders, tables)를 JSON으로 반환. **항상 첫 호출로 사용** |
222
+ | `inspect_document` | 문서 구조(placeholders, tables + `column_widths_cm`/`row_heights_cm`)를 JSON으로 반환. **항상 첫 호출로 사용** |
190
223
  | `render_template` | `{{key}}`를 context dict 값으로 치환해 새 파일 저장 |
191
- | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 반환 (preview의 40자 잘림 없이) |
224
+ | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 + `width_cm`/`height_cm`/`char_count` 반환 |
192
225
  | `set_cell` | 특정 표의 `(row, col)` 셀 값 교체 (병합 anchor만) |
193
226
  | `append_to_cell` | 기존 텍스트 뒤에 값 덧붙임 (라벨 유지용, 예: `"성 명"` → `"성 명 홍길동"`) |
194
- | `append_row` | 표 끝에 새 행 추가 (DOCX/HWPX 지원) |
227
+ | `fill_form` (v0.7+) | **라벨 이름**으로 일괄 채우기. 좌표 계산 없이 `{"접수번호": "...", "성명": "..."}` dict. dot-path 섹션 해소 지원 |
228
+ | `append_row` | 표 끝에 새 행 추가 (DOCX/PPTX/HWPX 전부 지원, v0.5+) |
195
229
 
196
230
  ### `inspect_document` 반환 예시 (v0.2+)
197
231
 
@@ -253,13 +287,12 @@ LLM은 이 preview를 보고 **"빈 셀이 어디 있는지 / 어떤 값을 넣
253
287
 
254
288
  loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `run`으로 쪼개질 수 있어, 어댑터가 paragraph 전체 텍스트를 재조립한 뒤 첫 `run`에 다시 담는 방식으로 처리합니다 (서식 일부 손실 가능).
255
289
 
256
- ## 내장된 버그 회피
290
+ ## 내장된 버그 회피 / 백엔드 선택
257
291
 
258
292
  | 포맷 | 문제 | 어댑터의 처리 |
259
293
  |---|---|---|
260
- | HWPX | `python-hwpx 2.9.0`의 `set_cell_text()`가 빈 셀에서 lxml/ElementTree 혼용 `TypeError` 발생 | `paragraphs[0].text = value` 직접 할당으로 우회 |
261
- | HWPX | `replace_text_in_runs()`가 한글 공백이 run으로 쪼개진 경우 매칭 실패 | 위치 기반 API만 사용 |
262
- | HWPX | `manifest fallback` 경고 로그가 과도하게 출력됨 | `logging.getLogger("hwpx")` 레벨을 `ERROR`로 조정 |
294
+ | HWPX | `python-hwpx` 가 Non-Commercial License → 상용 배포 블로커 | **v0.4.0 부터 자체 `hwpx_core` 모듈** (zipfile + lxml) 로 교체. 런타임에 `python-hwpx` 불필요. 테스트 fixture 생성에만 사용 (dev extras) |
295
+ | PPTX | `python-pptx` 에 공식 `add_row` API 없음 (issue #86, 2014년부터 open) | **v0.5.0 부터 자체 lxml 구현** (`<a:tr>` deepcopy 패턴) |
263
296
  | PPTX | placeholder가 여러 `run`으로 쪼개져 단순 `run.text` 치환이 실패 | paragraph 전체 재조립 |
264
297
  | DOCX | `docxtpl`의 `{%tr%}`를 같은 행에 두면 파싱 에러 | README에 배치 규칙 명시 |
265
298
 
@@ -268,15 +301,20 @@ loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `
268
301
  ```
269
302
  document_adapter/
270
303
  ├── __init__.py # load() dispatcher
271
- ├── base.py # DocumentAdapter ABC, TableSchema, DocumentSchema
304
+ ├── base.py # DocumentAdapter ABC + fill_form + dataclasses
272
305
  ├── docx_adapter.py # DocxAdapter
273
- ├── pptx_adapter.py # PptxAdapter
274
- ├── hwpx_adapter.py # HwpxAdapter (버그 회피 포함)
275
- ├── tools.py # Tool 정의 + call_tool dispatcher
306
+ ├── pptx_adapter.py # PptxAdapter (append_row 자체 구현 포함)
307
+ ├── hwpx_adapter.py # HwpxAdapter (hwpx_core 기반)
308
+ ├── hwpx_core/ # 자체 HWPX 패키지 (v0.4+)
309
+ │ ├── constants.py
310
+ │ ├── package.py # ZIP + dirty XML 관리
311
+ │ ├── grid.py # iter_grid, table_shape
312
+ │ └── paragraph.py # run-level 편집 헬퍼
313
+ ├── tools.py # 7개 MCP 도구 정의 + call_tool dispatcher
276
314
  └── mcp_server.py # MCP stdio server
277
315
 
278
316
  examples/
279
- └── claude_api_example.py
317
+ └── claude_api_example.py # Claude API Tool Use 에이전트 루프
280
318
  ```
281
319
 
282
320
  ## 라이선스
@@ -285,8 +323,15 @@ MIT
285
323
 
286
324
  ## Credits
287
325
 
288
- - [`python-docx`](https://github.com/python-openxml/python-docx)
289
- - [`docxtpl`](https://github.com/elapouya/python-docx-template)
290
- - [`python-pptx`](https://github.com/scanny/python-pptx)
291
- - [`python-hwpx`](https://github.com/airmang/python-hwpx)
292
- - [`mcp`](https://github.com/modelcontextprotocol/python-sdk)
326
+ **런타임 의존성** (전부 허용형 OSS):
327
+ - [`python-docx`](https://github.com/python-openxml/python-docx) — MIT
328
+ - [`docxtpl`](https://github.com/elapouya/python-docx-template) — LGPL-2.1
329
+ - [`python-pptx`](https://github.com/scanny/python-pptx) — MIT
330
+ - [`lxml`](https://lxml.de/) — BSD
331
+ - [`mcp`](https://github.com/modelcontextprotocol/python-sdk) — MIT
332
+
333
+ **코드 참조**:
334
+ - [`xgen-doc2chunk`](https://github.com/PlateerLab/xgen-doc2chunk) (Apache-2.0) — HWPX table grid 파싱 로직 차용 (`NOTICE` 참조)
335
+
336
+ **Dev 전용** (fixture 생성에만 사용):
337
+ - [`python-hwpx`](https://github.com/airmang/python-hwpx) — Non-Commercial License (v0.4.0 부터 런타임 의존성 제거)
@@ -28,6 +28,91 @@ def _normalize_label(s: str) -> str:
28
28
  return _LABEL_NORMALIZE_RE.sub("", s).lower().strip()
29
29
 
30
30
 
31
+ def _split_dot_path(label: str) -> tuple[str | None, str]:
32
+ """dot-path 분리: '피해자.금액' → ('피해자', '금액'). dot 없으면 (None, label)."""
33
+ if "." in label:
34
+ section, actual = label.rsplit(".", 1)
35
+ return section.strip() or None, actual.strip()
36
+ return None, label
37
+
38
+
39
+ def _candidate_context_labels(
40
+ candidate: tuple[int, int, int, int, int, str],
41
+ tables_by_idx: dict[int, Any],
42
+ ) -> list[str]:
43
+ """candidate 셀 주변에서 section/row 라벨 후보 텍스트 수집.
44
+
45
+ 전형적으로 양식은:
46
+ - 같은 table 의 (r, 0) 또는 (r-1..0, 0) 위치 anchor cell 이 섹션 헤더
47
+ - 또는 (r-1, c), (0, c) 가 header row
48
+ 단순 휴리스틱: 같은 row 의 col=0 anchor text + 그 위로 올라가면서 나오는
49
+ col=0 anchor text 몇 개를 수집.
50
+ """
51
+ t_idx, r, c = candidate[0], candidate[1], candidate[2]
52
+ t = tables_by_idx.get(t_idx)
53
+ if t is None:
54
+ return []
55
+
56
+ preview = t.preview
57
+ labels: list[str] = []
58
+ # candidate row 자신 또는 위쪽으로 올라가며 col=0 의 가장 가까운 anchor 1개.
59
+ # 병합 anchor 의 rowSpan 으로 candidate row 가 덮이므로 그게 진짜 섹션 헤더.
60
+ # candidate 가 col=0 자체이면 자기 자신이 아니라 **위쪽** row 의 col=0 라벨.
61
+ candidate_col = candidate[2]
62
+ start_r = min(r, len(preview) - 1)
63
+ # col=0 후보인 경우 자기 자신을 skip
64
+ if candidate_col == 0:
65
+ start_r = r - 1
66
+ for cur_r in range(start_r, -1, -1):
67
+ if cur_r < 0:
68
+ break
69
+ row = preview[cur_r]
70
+ if not row:
71
+ continue
72
+ val = row[0]
73
+ if val:
74
+ labels.append(val.strip())
75
+ break
76
+ return labels
77
+
78
+
79
+ def _candidate_matches_section(
80
+ candidate: tuple[int, int, int, int, int, str],
81
+ section_hint_norm: str,
82
+ tables_by_idx: dict[int, Any],
83
+ ) -> bool:
84
+ """candidate 의 주변 섹션 라벨에 section_hint_norm 이 포함되면 매칭."""
85
+ if not section_hint_norm:
86
+ return True
87
+ for ctx_label in _candidate_context_labels(candidate, tables_by_idx):
88
+ if section_hint_norm in _normalize_label(ctx_label):
89
+ return True
90
+ return False
91
+
92
+
93
+ def _describe_cell_context(
94
+ candidate: tuple[int, int, int, int, int, str],
95
+ tables_by_idx: dict[int, Any],
96
+ ) -> str:
97
+ """LLM 에게 보여줄 candidate 셀의 사람 읽는 컨텍스트 문자열."""
98
+ labels = _candidate_context_labels(candidate, tables_by_idx)
99
+ if labels:
100
+ return " / ".join(labels)
101
+ t_idx, r, c = candidate[0], candidate[1], candidate[2]
102
+ t = tables_by_idx.get(t_idx)
103
+ loc = getattr(t, "location", None) if t else None
104
+ return loc or f"table[{t_idx}]"
105
+
106
+
107
+ def _suggest_dot_path(
108
+ candidate: tuple[int, int, int, int, int, str],
109
+ tables_by_idx: dict[int, Any],
110
+ ) -> str:
111
+ """ambiguous hint 용 dot-path prefix 후보 (첫 후보의 섹션 라벨)."""
112
+ labels = _candidate_context_labels(candidate, tables_by_idx)
113
+ return labels[-1] if labels else "섹션"
114
+
115
+
31
116
  # ---- custom exceptions -----------------------------------------------------
32
117
  # 표준 예외를 상속해 기존 ``except ValueError/IndexError`` 흐름과 호환.
33
118
 
@@ -233,9 +318,14 @@ class DocumentAdapter(ABC):
233
318
  "성명" 같은 **사람이 읽는 라벨** 로 값을 넣게 하는 API. 양식 문서의
234
319
  전형적인 라벨-값 패턴 (라벨 오른쪽 또는 아래 셀이 값) 을 자동 탐지한다.
235
320
 
321
+ **Dot-path 로 ambiguous 해소**: 한 양식에 같은 라벨이 여러 번 등장하면
322
+ (예: "금액" 이 피해자 섹션, 지급정지계좌 섹션, 피해금이전계좌 섹션 각각)
323
+ ``"피해자.금액"`` 처럼 dot-path 로 section hint 를 지정하면 section 컨텍스트가
324
+ 일치하는 후보만 선택한다.
325
+
236
326
  Args:
237
327
  data: {라벨: 값} dict. 라벨은 셀 텍스트와 whitespace/특수문자 제거
238
- 후 정규화 비교 (예: "성 명" == "성명").
328
+ 후 정규화 비교 (예: "성 명" == "성명"). "섹션힌트.라벨" 형태도 허용.
239
329
  direction: 값 셀 탐색 방향.
240
330
  - "auto" (기본): right → below → same 순서로 빈 셀 우선
241
331
  - "right": 라벨 셀 오른쪽
@@ -247,7 +337,8 @@ class DocumentAdapter(ABC):
247
337
  {
248
338
  "filled": [{"label", "table_index", "row", "col", "action", "old_value", "new_value"}],
249
339
  "not_found": [라벨 목록],
250
- "ambiguous": [{"label", "candidates": [(t,r,c), ...]}],
340
+ "ambiguous": [{"label", "candidates": [{"table_index", "row", "col", "context"}, ...],
341
+ "hint": "dot-path 예시 (예: '피해자.금액')"}],
251
342
  }
252
343
  """
253
344
  if direction not in ("auto", "right", "below", "same"):
@@ -259,6 +350,7 @@ class DocumentAdapter(ABC):
259
350
  # 동일 라벨이 여러 곳에 있으면 ambiguous 로 분류.
260
351
  label_index: dict[str, list[tuple[int, int, int, int, int, str]]] = {}
261
352
  tables = self.get_tables(preview_rows=10_000, max_cell_len=10_000)
353
+ tables_by_idx = {t.index: t for t in tables}
262
354
  for t in tables:
263
355
  merge_map = {m.anchor: m.span for m in t.merges}
264
356
  for r, row in enumerate(t.preview):
@@ -273,31 +365,65 @@ class DocumentAdapter(ABC):
273
365
  (t.index, r, c, rs, cs, val)
274
366
  )
275
367
 
368
+ # user_keys 는 dot-path 분리 후 뒷부분 기준으로 만든다.
369
+ # "피해자.금액" 과 "지급정지.금액" 이 섞여 있어도 "금액" 단일로 보호.
370
+ user_keys = {_normalize_label(_split_dot_path(k)[1]) for k in data.keys()}
371
+
276
372
  filled: list[dict[str, Any]] = []
277
373
  not_found: list[str] = []
278
374
  ambiguous: list[dict[str, Any]] = []
279
375
 
280
376
  for label, value in data.items():
281
- key = _normalize_label(label)
282
- candidates = label_index.get(key, [])
377
+ section_hint, actual_label = _split_dot_path(label)
378
+ key = _normalize_label(actual_label)
379
+ all_candidates = label_index.get(key, [])
380
+
381
+ # dot-path 가 있으면 section_hint 로 candidate 필터
382
+ if section_hint and all_candidates:
383
+ hint_norm = _normalize_label(section_hint)
384
+ filtered = [
385
+ cand for cand in all_candidates
386
+ if _candidate_matches_section(cand, hint_norm, tables_by_idx)
387
+ ]
388
+ if filtered:
389
+ candidates = filtered
390
+ else:
391
+ # hint 와 매칭 안 되면 원본 후보 유지 (ambiguous 또는 single)
392
+ candidates = all_candidates
393
+ else:
394
+ candidates = all_candidates
395
+
283
396
  if not candidates:
284
397
  if strict:
285
398
  raise ValueError(f"label not found: {label!r}")
286
399
  not_found.append(label)
287
400
  continue
288
401
  if len(candidates) > 1:
289
- # 여러 곳에 있으면 채우지 않음 — LLM 이 명확히 지정하게 유도
402
+ # 여러 곳 — 각 후보의 section context 수집해서 LLM 이 구분 가능하게
290
403
  ambiguous.append({
291
404
  "label": label,
292
- "candidates": [(t, r, c) for (t, r, c, *_) in candidates],
405
+ "candidates": [
406
+ {
407
+ "table_index": cand[0],
408
+ "row": cand[1],
409
+ "col": cand[2],
410
+ "context": _describe_cell_context(cand, tables_by_idx),
411
+ }
412
+ for cand in candidates
413
+ ],
414
+ "hint": (
415
+ f"dot-path 로 재호출 예시: "
416
+ f"'{_suggest_dot_path(candidates[0], tables_by_idx)}.{actual_label}'"
417
+ ),
293
418
  })
294
419
  continue
295
420
 
296
421
  t_idx, r, c, rs, cs, current_text = candidates[0]
297
- # "옆 셀이 다른 라벨이면 skip" 판단에 사용자가 요청한 라벨들만 cross-check.
298
- # (label_index 전체를 쓰면 값 셀 텍스트까지 라벨로 오판 → 덮어쓰기 실패)
299
- user_keys = {_normalize_label(k) for k in data.keys()}
300
- other_label_keys = user_keys - {key}
422
+ # auto 모드에서 인접 라벨 오염 방지 — 문서 내 **모든 anchor cell 텍스트**
423
+ # 를 보호 대상으로. (값 셀이 포함되어 덮어쓰기 차단될 수 있지만 라벨 파괴가
424
+ # 더 치명적이라 보수적 default. 예시값이 있는 양식에서 값 셀을 덮어쓰려면
425
+ # direction="right"/"below" 로 명시.)
426
+ other_label_keys = set(label_index.keys()) - {key}
301
427
  action, coord, old = self._fill_one_cell(
302
428
  t_idx, r, c, rs, cs, str(value), direction, other_label_keys
303
429
  )
@@ -372,13 +498,18 @@ class DocumentAdapter(ABC):
372
498
  return "append_to_cell", (t_idx, target_r, target_c), old
373
499
 
374
500
  # right / below
501
+ if direction == "auto" and not cell.is_anchor:
502
+ # target 이 병합의 non-anchor 면 anchor 로 redirect 되어 엉뚱한 셀에
503
+ # 쓰일 위험 (예: 스페이서 행의 병합 anchor). auto 에서는 skip 하고
504
+ # 다음 mode 로.
505
+ continue
375
506
  if direction == "auto" and target_key and target_key in other_label_keys:
376
507
  # 옆 셀이 다른 라벨 → 덮어쓰면 라벨 손상. 다음 mode 시도.
377
508
  continue
378
509
  try:
379
510
  old = self.set_cell(
380
511
  t_idx, target_r, target_c, value,
381
- allow_merge_redirect=not cell.is_anchor,
512
+ allow_merge_redirect=(direction != "auto"),
382
513
  )
383
514
  except MergedCellWriteError:
384
515
  continue
@@ -177,11 +177,25 @@ TOOL_DEFINITIONS: list[dict[str, Any]] = [
177
177
  "description": (
178
178
  "라벨 이름으로 값 셀을 자동 탐지해 **일괄 채우기**. 좌표 (table_index, row, col) "
179
179
  "계산 없이 '접수번호', '성명' 같은 라벨 key-value dict 로 양식 채움. "
180
- "auto 모드: 라벨 셀 오른쪽 → 아래 → 같은 셀 순으로 값 셀 탐색. "
181
- "오른쪽/아래 셀이 사용자 요청 라벨 중 하나이면 (서로 라벨 공간 보호) skip 후 다음 시도. "
182
- "같은 셀로 fallback 시 append_to_cell 로 라벨 뒤에 값 덧붙임. "
183
- "**팁**: 한 양식의 관련 라벨을 함께 넘기면 라벨끼리 서로 보호하여 덮어쓰기 방지. "
184
- "반환: {filled:[...], not_found:[...], ambiguous:[...]}."
180
+ "\n"
181
+ "**direction 선택 기준 (중요)**: "
182
+ "(a) `auto` (기본, 보수적): 기존 값이 있는 셀은 다른 라벨로 간주하고 skip → 같은 "
183
+ "셀에 append_to_cell. **값 셀이 비어있는 양식**에 적합 (HWPX 공공 서식 등). "
184
+ "(b) `right` / `below` (명시): 라벨 오른쪽/아래 셀을 **덮어쓰기**. **기존에 "
185
+ "예시값이 채워져 있는 양식**(PPTX 템플릿 등) 에는 반드시 direction='right' 명시. "
186
+ "\n"
187
+ "**Dot-path 섹션 지정**: 같은 라벨이 여러 섹션에 있어 ambiguous 로 반환되면 "
188
+ "`{'피해자.금액': '...', '지급정지요청계좌.금액': '...'}` 처럼 섹션힌트.라벨 "
189
+ "형태로 재호출. ambiguous 반환의 hint 필드에 예시 제공됨. "
190
+ "\n"
191
+ "**output_path**: 생략 시 **원본 파일에 덮어쓰기** (대부분의 경우 생략 권장). "
192
+ "다른 위치에 저장이 필요할 때만 지정. "
193
+ "\n"
194
+ "**팁**: 한 양식의 관련 라벨을 **한 번에 dict 로** 넘기면 라벨끼리 서로 보호되어 "
195
+ "인접 라벨 오염이 방지됩니다. "
196
+ "\n"
197
+ "반환: `{filled: [...], not_found: [...], ambiguous: [...]}`. "
198
+ "ambiguous candidates 각각에 `context` 필드 포함 (어느 섹션인지 확인)."
185
199
  ),
186
200
  "input_schema": {
187
201
  "type": "object",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: document-adapter
3
- Version: 0.7.0
3
+ Version: 0.7.2
4
4
  Summary: LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support
5
5
  Author-email: Son Seongjun <sonsj97@plateer.com>
6
6
  License: MIT
@@ -127,13 +127,46 @@ doc.append_to_cell(table_index=2, row=0, col=0, value="홍길동")
127
127
  cell = doc.get_cell(table_index=1, row=3, col=2)
128
128
  print(cell.text, cell.is_anchor, cell.span, cell.nested_table_indices)
129
129
 
130
- # DOCX/HWPX는 행 추가 지원
130
+ # DOCX/PPTX/HWPX 전부 행 추가 지원 (v0.5+)
131
131
  doc.append_row(1, ["새 항목", "값"])
132
132
 
133
133
  doc.save("checklist_filled.docx")
134
134
  doc.close()
135
135
  ```
136
136
 
137
+ ### 라벨 기반 일괄 채우기 (v0.7+)
138
+
139
+ LLM 이 좌표 `(table_index, row, col)` 를 직접 계산하지 않고 "접수번호", "성명" 같은 사람이 읽는 라벨 key-value 로 양식을 채울 수 있습니다.
140
+
141
+ ```python
142
+ doc = load("form.hwpx")
143
+ result = doc.fill_form({
144
+ "접수번호": "2026-0001",
145
+ "성 명": "홍길동",
146
+ "주 소": "서울시 강남구",
147
+ "금융회사": "국민은행",
148
+ })
149
+ # → {"filled": [...], "not_found": [...], "ambiguous": [...]}
150
+ doc.save()
151
+ doc.close()
152
+ ```
153
+
154
+ - `auto` (기본): 라벨 셀 오른쪽 → 아래 → 같은 셀 순으로 값 셀 탐색. 보수적이라 기존 값 있는 셀은 다른 라벨로 간주하고 skip.
155
+ - `direction="right"` 명시: 라벨 오른쪽 셀을 **덮어쓰기** (예시값 있는 PPTX 템플릿 등).
156
+ - **Dot-path 섹션 지정**: 동일 라벨이 여러 섹션에 있으면 `"피해자.금액"`, `"지급정지요청계좌.금액"` 처럼 섹션 힌트 부여. `ambiguous` 반환 시 `hint` 필드에 예시 제공.
157
+ - **팁**: 한 양식의 관련 라벨을 한 번에 dict 로 넘기면 라벨끼리 서로 보호되어 오염을 방지합니다.
158
+
159
+ ### 셀 크기 메타 (v0.6+)
160
+
161
+ `get_tables()`가 `column_widths_cm` / `row_heights_cm` 를, `get_cell()`이 `width_cm` / `height_cm` / `char_count` 를 반환해 LLM이 **좁은 셀에 긴 텍스트를 넣어 오버플로 되는 것을 사전에 판단**할 수 있습니다.
162
+
163
+ ```python
164
+ cell = doc.get_cell(table_index=0, row=0, col=0)
165
+ print(cell.width_cm, cell.char_count) # 1.7cm, 4자 — 작은 배지
166
+ ```
167
+
168
+ DOCX/PPTX는 EMU → cm, HWPX는 HU → cm 자동 환산 (1자리 반올림).
169
+
137
170
  확장자로 자동 분기되므로 `.pptx` / `.hwpx`도 동일한 API를 사용합니다.
138
171
 
139
172
  ### 병합 셀 인지 동작 (v0.2+)
@@ -185,7 +218,7 @@ document-adapter-mcp
185
218
  }
186
219
  ```
187
220
 
188
- 재시작하면 Claude Desktop에서 아래 4개 도구를 사용할 수 있습니다.
221
+ 재시작하면 Claude Desktop에서 아래 7개 도구를 사용할 수 있습니다.
189
222
 
190
223
  ### Claude Code 설정
191
224
 
@@ -227,12 +260,13 @@ resp = client.messages.create(
227
260
 
228
261
  | 도구 | 설명 |
229
262
  |---|---|
230
- | `inspect_document` | 문서 구조(placeholders, tables)를 JSON으로 반환. **항상 첫 호출로 사용** |
263
+ | `inspect_document` | 문서 구조(placeholders, tables + `column_widths_cm`/`row_heights_cm`)를 JSON으로 반환. **항상 첫 호출로 사용** |
231
264
  | `render_template` | `{{key}}`를 context dict 값으로 치환해 새 파일 저장 |
232
- | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 반환 (preview의 40자 잘림 없이) |
265
+ | `get_cell` | 셀 전체 텍스트 + 병합/중첩 메타 + `width_cm`/`height_cm`/`char_count` 반환 |
233
266
  | `set_cell` | 특정 표의 `(row, col)` 셀 값 교체 (병합 anchor만) |
234
267
  | `append_to_cell` | 기존 텍스트 뒤에 값 덧붙임 (라벨 유지용, 예: `"성 명"` → `"성 명 홍길동"`) |
235
- | `append_row` | 표 끝에 새 행 추가 (DOCX/HWPX 지원) |
268
+ | `fill_form` (v0.7+) | **라벨 이름**으로 일괄 채우기. 좌표 계산 없이 `{"접수번호": "...", "성명": "..."}` dict. dot-path 섹션 해소 지원 |
269
+ | `append_row` | 표 끝에 새 행 추가 (DOCX/PPTX/HWPX 전부 지원, v0.5+) |
236
270
 
237
271
  ### `inspect_document` 반환 예시 (v0.2+)
238
272
 
@@ -294,13 +328,12 @@ LLM은 이 preview를 보고 **"빈 셀이 어디 있는지 / 어떤 값을 넣
294
328
 
295
329
  loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `run`으로 쪼개질 수 있어, 어댑터가 paragraph 전체 텍스트를 재조립한 뒤 첫 `run`에 다시 담는 방식으로 처리합니다 (서식 일부 손실 가능).
296
330
 
297
- ## 내장된 버그 회피
331
+ ## 내장된 버그 회피 / 백엔드 선택
298
332
 
299
333
  | 포맷 | 문제 | 어댑터의 처리 |
300
334
  |---|---|---|
301
- | HWPX | `python-hwpx 2.9.0`의 `set_cell_text()`가 빈 셀에서 lxml/ElementTree 혼용 `TypeError` 발생 | `paragraphs[0].text = value` 직접 할당으로 우회 |
302
- | HWPX | `replace_text_in_runs()`가 한글 공백이 run으로 쪼개진 경우 매칭 실패 | 위치 기반 API만 사용 |
303
- | HWPX | `manifest fallback` 경고 로그가 과도하게 출력됨 | `logging.getLogger("hwpx")` 레벨을 `ERROR`로 조정 |
335
+ | HWPX | `python-hwpx` 가 Non-Commercial License → 상용 배포 블로커 | **v0.4.0 부터 자체 `hwpx_core` 모듈** (zipfile + lxml) 로 교체. 런타임에 `python-hwpx` 불필요. 테스트 fixture 생성에만 사용 (dev extras) |
336
+ | PPTX | `python-pptx` 에 공식 `add_row` API 없음 (issue #86, 2014년부터 open) | **v0.5.0 부터 자체 lxml 구현** (`<a:tr>` deepcopy 패턴) |
304
337
  | PPTX | placeholder가 여러 `run`으로 쪼개져 단순 `run.text` 치환이 실패 | paragraph 전체 재조립 |
305
338
  | DOCX | `docxtpl`의 `{%tr%}`를 같은 행에 두면 파싱 에러 | README에 배치 규칙 명시 |
306
339
 
@@ -309,15 +342,20 @@ loop / if / filter는 지원하지 않습니다. PPTX는 placeholder가 여러 `
309
342
  ```
310
343
  document_adapter/
311
344
  ├── __init__.py # load() dispatcher
312
- ├── base.py # DocumentAdapter ABC, TableSchema, DocumentSchema
345
+ ├── base.py # DocumentAdapter ABC + fill_form + dataclasses
313
346
  ├── docx_adapter.py # DocxAdapter
314
- ├── pptx_adapter.py # PptxAdapter
315
- ├── hwpx_adapter.py # HwpxAdapter (버그 회피 포함)
316
- ├── tools.py # Tool 정의 + call_tool dispatcher
347
+ ├── pptx_adapter.py # PptxAdapter (append_row 자체 구현 포함)
348
+ ├── hwpx_adapter.py # HwpxAdapter (hwpx_core 기반)
349
+ ├── hwpx_core/ # 자체 HWPX 패키지 (v0.4+)
350
+ │ ├── constants.py
351
+ │ ├── package.py # ZIP + dirty XML 관리
352
+ │ ├── grid.py # iter_grid, table_shape
353
+ │ └── paragraph.py # run-level 편집 헬퍼
354
+ ├── tools.py # 7개 MCP 도구 정의 + call_tool dispatcher
317
355
  └── mcp_server.py # MCP stdio server
318
356
 
319
357
  examples/
320
- └── claude_api_example.py
358
+ └── claude_api_example.py # Claude API Tool Use 에이전트 루프
321
359
  ```
322
360
 
323
361
  ## 라이선스
@@ -326,8 +364,15 @@ MIT
326
364
 
327
365
  ## Credits
328
366
 
329
- - [`python-docx`](https://github.com/python-openxml/python-docx)
330
- - [`docxtpl`](https://github.com/elapouya/python-docx-template)
331
- - [`python-pptx`](https://github.com/scanny/python-pptx)
332
- - [`python-hwpx`](https://github.com/airmang/python-hwpx)
333
- - [`mcp`](https://github.com/modelcontextprotocol/python-sdk)
367
+ **런타임 의존성** (전부 허용형 OSS):
368
+ - [`python-docx`](https://github.com/python-openxml/python-docx) — MIT
369
+ - [`docxtpl`](https://github.com/elapouya/python-docx-template) — LGPL-2.1
370
+ - [`python-pptx`](https://github.com/scanny/python-pptx) — MIT
371
+ - [`lxml`](https://lxml.de/) — BSD
372
+ - [`mcp`](https://github.com/modelcontextprotocol/python-sdk) — MIT
373
+
374
+ **코드 참조**:
375
+ - [`xgen-doc2chunk`](https://github.com/PlateerLab/xgen-doc2chunk) (Apache-2.0) — HWPX table grid 파싱 로직 차용 (`NOTICE` 참조)
376
+
377
+ **Dev 전용** (fixture 생성에만 사용):
378
+ - [`python-hwpx`](https://github.com/airmang/python-hwpx) — Non-Commercial License (v0.4.0 부터 런타임 의존성 제거)
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "document-adapter"
7
- version = "0.7.0"
7
+ version = "0.7.2"
8
8
  description = "LLM-friendly document template editing (DOCX/PPTX/HWPX) with MCP server and Claude API tool-use support"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -1085,7 +1085,11 @@ def test_fill_form_docx_label_right_value(tmp_path: Path) -> None:
1085
1085
 
1086
1086
 
1087
1087
  def test_fill_form_pptx_label_right_value(tmp_path: Path) -> None:
1088
- """PPTX 2x2 라벨|값 패턴 — fill_form auto 가 오른쪽 값 셀에 set_cell."""
1088
+ """PPTX 2x2 라벨|값 패턴 — 예시값 있는 셀 덮어쓰려면 direction='right' 명시.
1089
+
1090
+ auto 는 보수적 기본값이라 기존 값이 있는 셀은 다른 라벨로 간주하고 skip,
1091
+ 최종적으로 same append. 예시값 덮어쓰기가 의도면 direction 명시 필요.
1092
+ """
1089
1093
  src = tmp_path / "form.pptx"
1090
1094
  prs = Presentation()
1091
1095
  prs.slide_width = Inches(10)
@@ -1101,7 +1105,10 @@ def test_fill_form_pptx_label_right_value(tmp_path: Path) -> None:
1101
1105
 
1102
1106
  adapter = load(src)
1103
1107
  try:
1104
- result = adapter.fill_form({"보고일자": "2026.04.16", "작성자": "홍길동"})
1108
+ result = adapter.fill_form(
1109
+ {"보고일자": "2026.04.16", "작성자": "홍길동"},
1110
+ direction="right",
1111
+ )
1105
1112
  adapter.save()
1106
1113
  finally:
1107
1114
  adapter.close()
@@ -1113,6 +1120,41 @@ def test_fill_form_pptx_label_right_value(tmp_path: Path) -> None:
1113
1120
  assert filled_by_label["보고일자"]["old_value"] == "2026.01.01"
1114
1121
 
1115
1122
 
1123
+ def test_fill_form_dot_path_resolves_ambiguous(tmp_path: Path) -> None:
1124
+ """동일 라벨이 여러 섹션에 있으면 dot-path 로 섹션 지정해 해소."""
1125
+ src = tmp_path / "form.docx"
1126
+ doc = Document()
1127
+ t = doc.add_table(rows=4, cols=2)
1128
+ t.cell(0, 0).text = "재무"
1129
+ t.cell(0, 1).text = ""
1130
+ t.cell(1, 0).text = "금액"
1131
+ t.cell(1, 1).text = ""
1132
+ t.cell(2, 0).text = "영업"
1133
+ t.cell(2, 1).text = ""
1134
+ t.cell(3, 0).text = "금액"
1135
+ t.cell(3, 1).text = ""
1136
+ doc.save(src)
1137
+
1138
+ adapter = load(src)
1139
+ try:
1140
+ r_ambig = adapter.fill_form({"금액": "1"})
1141
+ assert len(r_ambig["ambiguous"]) == 1
1142
+ assert r_ambig["ambiguous"][0]["label"] == "금액"
1143
+ assert len(r_ambig["ambiguous"][0]["candidates"]) == 2
1144
+ contexts = [c["context"] for c in r_ambig["ambiguous"][0]["candidates"]]
1145
+ assert any("재무" in ctx for ctx in contexts)
1146
+ assert any("영업" in ctx for ctx in contexts)
1147
+
1148
+ r_resolved = adapter.fill_form(
1149
+ {"재무.금액": "100", "영업.금액": "200"},
1150
+ direction="right",
1151
+ )
1152
+ assert r_resolved["ambiguous"] == []
1153
+ assert len(r_resolved["filled"]) == 2
1154
+ finally:
1155
+ adapter.close()
1156
+
1157
+
1116
1158
  def test_fill_form_not_found_records_missing(tmp_path: Path) -> None:
1117
1159
  """존재하지 않는 라벨은 not_found 에 기록되고, strict=False 면 예외 없음."""
1118
1160
  src = tmp_path / "form.docx"