dochan 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. dochan/__init__.py +9 -0
  2. dochan/batch.py +154 -0
  3. dochan/cli.py +118 -0
  4. dochan/constants.py +54 -0
  5. dochan/control_char.py +51 -0
  6. dochan/fallback/__init__.py +0 -0
  7. dochan/fallback/filter_server.py +93 -0
  8. dochan/hwp/__init__.py +0 -0
  9. dochan/hwp/bin_data.py +96 -0
  10. dochan/hwp/doc_info.py +212 -0
  11. dochan/hwp/header.py +78 -0
  12. dochan/hwp/records/__init__.py +0 -0
  13. dochan/hwp/records/char_shape.py +115 -0
  14. dochan/hwp/records/ctrl_header.py +67 -0
  15. dochan/hwp/records/para_char_shape.py +35 -0
  16. dochan/hwp/records/para_header.py +63 -0
  17. dochan/hwp/records/para_text.py +66 -0
  18. dochan/hwp/records/style.py +68 -0
  19. dochan/hwp/records/table.py +62 -0
  20. dochan/hwp/section.py +472 -0
  21. dochan/hwpx/__init__.py +0 -0
  22. dochan/hwpx/parser.py +461 -0
  23. dochan/model/__init__.py +0 -0
  24. dochan/model/document.py +97 -0
  25. dochan/model/equation.py +58 -0
  26. dochan/model/header_footer.py +25 -0
  27. dochan/model/image.py +25 -0
  28. dochan/model/style.py +37 -0
  29. dochan/model/table.py +37 -0
  30. dochan/output/__init__.py +0 -0
  31. dochan/output/json_out.py +84 -0
  32. dochan/output/markdown.py +159 -0
  33. dochan/output/plain_text.py +47 -0
  34. dochan/quality/__init__.py +0 -0
  35. dochan/quality/batch_validate.py +222 -0
  36. dochan/quality/checker.py +75 -0
  37. dochan/quality/comparator.py +66 -0
  38. dochan/quality/cross_validator.py +410 -0
  39. dochan/reader.py +174 -0
  40. dochan/tests/__init__.py +0 -0
  41. dochan/tests/test_char_shape.py +72 -0
  42. dochan/tests/test_control_char.py +90 -0
  43. dochan/tests/test_quality.py +82 -0
  44. dochan/utils/__init__.py +0 -0
  45. dochan/utils/error_recovery.py +55 -0
  46. dochan/utils/logger.py +38 -0
  47. dochan/utils/ocr.py +117 -0
  48. dochan/utils/safe_decompress.py +27 -0
  49. dochan-0.1.0.dist-info/METADATA +263 -0
  50. dochan-0.1.0.dist-info/RECORD +55 -0
  51. dochan-0.1.0.dist-info/WHEEL +5 -0
  52. dochan-0.1.0.dist-info/entry_points.txt +2 -0
  53. dochan-0.1.0.dist-info/licenses/LICENSE +21 -0
  54. dochan-0.1.0.dist-info/licenses/NOTICE +8 -0
  55. dochan-0.1.0.dist-info/top_level.txt +1 -0
dochan/__init__.py ADDED
@@ -0,0 +1,9 @@
1
+ """dochan — HWP/HWPX 문서 파서, AI/LLM 최적 Markdown 변환"""
2
+ __version__ = "0.1.0"
3
+
4
+ from .reader import Dochan
5
+
6
+ # 하위 호환 별칭
7
+ HWPReader = Dochan
8
+
9
+ __all__ = ['Dochan', 'HWPReader']
dochan/batch.py ADDED
@@ -0,0 +1,154 @@
1
+ """
2
+ batch.py — 배치 처리
3
+ 다수 HWP 파일의 병렬 변환
4
+ """
5
+
6
+ import os
7
+ import logging
8
+ from concurrent.futures import ProcessPoolExecutor, as_completed
9
+ from dataclasses import dataclass, field
10
+ from pathlib import Path
11
+ from typing import List
12
+
13
+ logger = logging.getLogger('dochan')
14
+
15
+
16
+ @dataclass
17
+ class BatchResult:
18
+ """단일 파일 처리 결과"""
19
+ file_path: str = ""
20
+ success: bool = False
21
+ output_path: str = ""
22
+ error_count: int = 0
23
+ errors: List[str] = field(default_factory=list)
24
+
25
+
26
+ @dataclass
27
+ class BatchSummary:
28
+ """배치 처리 전체 요약"""
29
+ total: int = 0
30
+ success: int = 0
31
+ failed: int = 0
32
+ results: List[BatchResult] = field(default_factory=list)
33
+
34
+ @property
35
+ def success_rate(self) -> float:
36
+ return self.success / max(self.total, 1) * 100
37
+
38
+
39
+ def _process_single(file_path: str, output_dir: str, output_format: str) -> BatchResult:
40
+ """단일 파일 처리 (별도 프로세스에서 실행)"""
41
+ from .reader import Dochan
42
+
43
+ result = BatchResult(file_path=file_path)
44
+
45
+ try:
46
+ reader = Dochan(file_path)
47
+
48
+ # 출력 파일명 생성
49
+ base_name = os.path.splitext(os.path.basename(file_path))[0]
50
+ ext_map = {'markdown': '.md', 'json': '.json', 'text': '.txt'}
51
+ ext = ext_map.get(output_format, '.md')
52
+ output_path = os.path.join(output_dir, base_name + ext)
53
+
54
+ # 변환
55
+ if output_format == 'json':
56
+ content = reader.to_json()
57
+ elif output_format == 'text':
58
+ content = reader.to_plain_text()
59
+ else:
60
+ content = reader.to_markdown()
61
+
62
+ # 저장
63
+ with open(output_path, 'w', encoding='utf-8') as f:
64
+ f.write(content)
65
+
66
+ result.success = True
67
+ result.output_path = output_path
68
+ result.errors = reader.errors
69
+ result.error_count = len(reader.errors)
70
+
71
+ except Exception as e:
72
+ result.success = False
73
+ result.errors = [str(e)]
74
+ result.error_count = 1
75
+
76
+ return result
77
+
78
+
79
+ def batch_convert(
80
+ input_dir: str,
81
+ output_dir: str,
82
+ output_format: str = 'markdown',
83
+ max_workers: int = 4,
84
+ extensions: tuple = ('.hwp', '.hwpx'),
85
+ ) -> BatchSummary:
86
+ """
87
+ 디렉토리 내 HWP 파일 일괄 변환
88
+
89
+ Args:
90
+ input_dir: 입력 디렉토리
91
+ output_dir: 출력 디렉토리
92
+ output_format: 'markdown', 'json', 'text'
93
+ max_workers: 병렬 워커 수
94
+ extensions: 처리할 확장자
95
+
96
+ Returns:
97
+ BatchSummary
98
+ """
99
+ # output_dir 경로 검증
100
+ try:
101
+ Path(output_dir).resolve().relative_to(Path(os.getcwd()).resolve())
102
+ except ValueError:
103
+ pass # output_dir가 cwd 밖이어도 허용하되, 아래에서 symlink 공격 방지
104
+
105
+ os.makedirs(output_dir, exist_ok=True)
106
+
107
+ # 파일 수집
108
+ files = []
109
+ resolved_input = Path(input_dir).resolve()
110
+ for root, _, filenames in os.walk(input_dir):
111
+ for fn in filenames:
112
+ if any(fn.lower().endswith(ext) for ext in extensions):
113
+ full_path = os.path.join(root, fn)
114
+ # Path traversal 방지: 실제 경로가 input_dir 내부인지 검증
115
+ try:
116
+ Path(full_path).resolve().relative_to(resolved_input)
117
+ except ValueError:
118
+ logger.warning(f"경로 이탈 감지, 건너뜀: {full_path}")
119
+ continue
120
+ files.append(full_path)
121
+
122
+ summary = BatchSummary(total=len(files))
123
+ logger.info(f"배치 시작: {len(files)}개 파일, {max_workers} 워커")
124
+
125
+ # 병렬 처리
126
+ with ProcessPoolExecutor(max_workers=max_workers) as executor:
127
+ futures = {
128
+ executor.submit(_process_single, f, output_dir, output_format): f
129
+ for f in files
130
+ }
131
+
132
+ for future in as_completed(futures):
133
+ file_path = futures[future]
134
+ try:
135
+ result = future.result()
136
+ summary.results.append(result)
137
+ if result.success:
138
+ summary.success += 1
139
+ logger.info(f"✓ {os.path.basename(file_path)}")
140
+ else:
141
+ summary.failed += 1
142
+ logger.error(f"✗ {os.path.basename(file_path)}: {result.errors}")
143
+ except Exception as e:
144
+ summary.failed += 1
145
+ summary.results.append(BatchResult(
146
+ file_path=file_path,
147
+ success=False,
148
+ errors=[str(e)],
149
+ error_count=1,
150
+ ))
151
+ logger.error(f"✗ {os.path.basename(file_path)}: {e}")
152
+
153
+ logger.info(f"배치 완료: {summary.success}/{summary.total} 성공 ({summary.success_rate:.1f}%)")
154
+ return summary
dochan/cli.py ADDED
@@ -0,0 +1,118 @@
1
+ """dochan CLI — HWP/HWPX 문서를 터미널에서 변환
2
+
3
+ 사용법:
4
+ dochan convert 문서.hwp # → stdout에 Markdown
5
+ dochan convert 문서.hwp -o output.md # → 파일로 저장
6
+ dochan convert 문서.hwp --format json # → JSON 출력
7
+ dochan convert 문서.hwpx --format text # → Plain text
8
+ dochan batch input_dir/ output_dir/ # → 디렉토리 일괄 변환
9
+ dochan info 문서.hwp # → 문서 메타데이터
10
+ """
11
+
12
+ import argparse
13
+ import sys
14
+ import os
15
+
16
+
17
+ def main():
18
+ parser = argparse.ArgumentParser(
19
+ prog='dochan',
20
+ description='dochan — 독한 HWP/HWPX 파서, AI/LLM 최적 Markdown 변환',
21
+ )
22
+ subparsers = parser.add_subparsers(dest='command', help='명령')
23
+
24
+ # convert
25
+ conv = subparsers.add_parser('convert', help='HWP/HWPX → Markdown/JSON/Text 변환')
26
+ conv.add_argument('file', help='HWP 또는 HWPX 파일 경로')
27
+ conv.add_argument('-o', '--output', default=None, help='출력 파일 경로 (기본: stdout)')
28
+ conv.add_argument('-f', '--format', choices=['markdown', 'json', 'text'],
29
+ default='markdown', help='출력 형식 (기본: markdown)')
30
+ conv.add_argument('--ocr', action='store_true', help='이미지 OCR 활성화')
31
+
32
+ # batch
33
+ bat = subparsers.add_parser('batch', help='디렉토리 일괄 변환')
34
+ bat.add_argument('input_dir', help='입력 디렉토리')
35
+ bat.add_argument('output_dir', help='출력 디렉토리')
36
+ bat.add_argument('-f', '--format', choices=['markdown', 'json', 'text'],
37
+ default='markdown', help='출력 형식')
38
+ bat.add_argument('-w', '--workers', type=int, default=4, help='병렬 워커 수')
39
+
40
+ # info
41
+ inf = subparsers.add_parser('info', help='문서 메타데이터 출력')
42
+ inf.add_argument('file', help='HWP 또는 HWPX 파일 경로')
43
+
44
+ args = parser.parse_args()
45
+
46
+ if not args.command:
47
+ parser.print_help()
48
+ sys.exit(0)
49
+
50
+ if args.command == 'convert':
51
+ _cmd_convert(args)
52
+ elif args.command == 'batch':
53
+ _cmd_batch(args)
54
+ elif args.command == 'info':
55
+ _cmd_info(args)
56
+
57
+
58
+ def _cmd_convert(args):
59
+ from .reader import Dochan
60
+
61
+ if not os.path.exists(args.file):
62
+ print(f"에러: 파일을 찾을 수 없습니다: {args.file}", file=sys.stderr)
63
+ sys.exit(1)
64
+
65
+ doc = Dochan(args.file, ocr=args.ocr)
66
+
67
+ if args.format == 'json':
68
+ content = doc.to_json()
69
+ elif args.format == 'text':
70
+ content = doc.to_plain_text()
71
+ else:
72
+ content = doc.to_markdown()
73
+
74
+ if args.output:
75
+ with open(args.output, 'w', encoding='utf-8') as f:
76
+ f.write(content)
77
+ print(f"저장 완료: {args.output}", file=sys.stderr)
78
+ else:
79
+ print(content)
80
+
81
+ if doc.errors:
82
+ for err in doc.errors:
83
+ print(f"경고: {err}", file=sys.stderr)
84
+
85
+
86
+ def _cmd_batch(args):
87
+ from .batch import batch_convert
88
+
89
+ if not os.path.isdir(args.input_dir):
90
+ print(f"에러: 디렉토리를 찾을 수 없습니다: {args.input_dir}", file=sys.stderr)
91
+ sys.exit(1)
92
+
93
+ summary = batch_convert(
94
+ input_dir=args.input_dir,
95
+ output_dir=args.output_dir,
96
+ output_format=args.format,
97
+ max_workers=args.workers,
98
+ )
99
+ print(f"\n완료: {summary.success}/{summary.total} 성공 ({summary.success_rate:.1f}%)")
100
+
101
+
102
+ def _cmd_info(args):
103
+ from .reader import Dochan
104
+ import json
105
+
106
+ if not os.path.exists(args.file):
107
+ print(f"에러: 파일을 찾을 수 없습니다: {args.file}", file=sys.stderr)
108
+ sys.exit(1)
109
+
110
+ doc = Dochan(args.file)
111
+ info = doc.metadata
112
+ info['file'] = args.file
113
+ info['format'] = 'hwpx' if args.file.lower().endswith('.hwpx') else 'hwp'
114
+ print(json.dumps(info, ensure_ascii=False, indent=2))
115
+
116
+
117
+ if __name__ == '__main__':
118
+ main()
dochan/constants.py ADDED
@@ -0,0 +1,54 @@
1
+ """
2
+ constants.py — 모든 TagID의 단일 진실 소스 (Single Source of Truth)
3
+ 기준: 한글문서파일형식_5.0_revision1.3.pdf
4
+ """
5
+
6
+ HWPTAG_BEGIN = 16 # 0x010
7
+
8
+ # ──── DocInfo 레코드 (스펙 표 13) ────
9
+ HWPTAG_DOCUMENT_PROPERTIES = 16 # BEGIN+0
10
+ HWPTAG_ID_MAPPINGS = 17 # BEGIN+1
11
+ HWPTAG_BIN_DATA = 18 # BEGIN+2
12
+ HWPTAG_FACE_NAME = 19 # BEGIN+3
13
+ HWPTAG_BORDER_FILL = 20 # BEGIN+4
14
+ HWPTAG_CHAR_SHAPE = 21 # BEGIN+5
15
+ HWPTAG_TAB_DEF = 22 # BEGIN+6
16
+ HWPTAG_NUMBERING = 23 # BEGIN+7
17
+ HWPTAG_BULLET = 24 # BEGIN+8
18
+ HWPTAG_PARA_SHAPE = 25 # BEGIN+9
19
+ HWPTAG_STYLE = 26 # BEGIN+10
20
+ HWPTAG_DOC_DATA = 27 # BEGIN+11
21
+ HWPTAG_DISTRIBUTE_DOC_DATA = 28 # BEGIN+12
22
+ HWPTAG_COMPATIBLE_DOCUMENT = 30 # BEGIN+14
23
+ HWPTAG_LAYOUT_COMPATIBILITY = 31 # BEGIN+15
24
+
25
+ # ──── BodyText 레코드 (스펙 표 57) ────
26
+ HWPTAG_PARA_HEADER = 66 # BEGIN+50
27
+ HWPTAG_PARA_TEXT = 67 # BEGIN+51
28
+ HWPTAG_PARA_CHAR_SHAPE = 68 # BEGIN+52
29
+ HWPTAG_PARA_LINE_SEG = 69 # BEGIN+53
30
+ HWPTAG_PARA_RANGE_TAG = 70 # BEGIN+54
31
+ HWPTAG_CTRL_HEADER = 71 # BEGIN+55
32
+ HWPTAG_LIST_HEADER = 72 # BEGIN+56
33
+ HWPTAG_PAGE_DEF = 73 # BEGIN+57
34
+ HWPTAG_FOOTNOTE_SHAPE = 74 # BEGIN+58
35
+ HWPTAG_PAGE_BORDER_FILL = 75 # BEGIN+59
36
+ HWPTAG_SHAPE_COMPONENT = 76 # BEGIN+60
37
+ HWPTAG_TABLE = 77 # BEGIN+61
38
+ HWPTAG_SHAPE_COMP_LINE = 78 # BEGIN+62
39
+ HWPTAG_SHAPE_COMP_RECT = 79 # BEGIN+63
40
+ HWPTAG_SHAPE_COMP_ELLIPSE = 80 # BEGIN+64
41
+ HWPTAG_SHAPE_COMP_ARC = 81 # BEGIN+65
42
+ HWPTAG_SHAPE_COMP_POLYGON = 82 # BEGIN+66
43
+ HWPTAG_SHAPE_COMP_CURVE = 83 # BEGIN+67
44
+ HWPTAG_SHAPE_COMP_OLE = 84 # BEGIN+68
45
+ HWPTAG_SHAPE_COMP_PICTURE = 85 # BEGIN+69
46
+ HWPTAG_SHAPE_COMP_CONTAINER = 86 # BEGIN+70
47
+ HWPTAG_CTRL_DATA = 87 # BEGIN+71
48
+ HWPTAG_EQEDIT = 88 # BEGIN+72
49
+ HWPTAG_SHAPE_COMP_TEXTART = 90 # BEGIN+74
50
+ HWPTAG_FORM_OBJECT = 91 # BEGIN+75
51
+ HWPTAG_MEMO_SHAPE = 92 # BEGIN+76
52
+ HWPTAG_MEMO_LIST = 93 # BEGIN+77
53
+ HWPTAG_CHART_DATA = 95 # BEGIN+79
54
+ HWPTAG_VIDEO_DATA = 98 # BEGIN+82
dochan/control_char.py ADDED
@@ -0,0 +1,51 @@
1
+ """
2
+ control_char.py — 제어 문자별 WCHAR 단위 크기
3
+ 1 WCHAR = 2 bytes. 실제 바이트 크기 = 값 × 2
4
+ 스펙 표 6 (p.10-11) 기준
5
+ """
6
+
7
+ CTRL_CHAR_WCHAR_SIZE = {
8
+ # ── char 타입 (1 WCHAR = 2 bytes) ──
9
+ 10: 1, # 줄바꿈 (line break)
10
+ 13: 1, # 문단 끝 (para break)
11
+ 24: 1, # 하이픈
12
+ 25: 1, 26: 1, 27: 1, 28: 1, 29: 1, # 예약 (char)
13
+ 30: 1, # 묶음 빈칸
14
+ 31: 1, # 고정폭 빈칸
15
+
16
+ # ── inline 타입 (8 WCHAR = 16 bytes) ──
17
+ 4: 8, # 필드 끝
18
+ 5: 8, 6: 8, 7: 8, # 예약 (inline)
19
+ 8: 8, # title mark
20
+ 9: 8, # 탭
21
+ 19: 8, 20: 8, # 예약 (inline)
22
+
23
+ # ── extended 타입 (8 WCHAR = 16 bytes) ──
24
+ 1: 8, # 예약
25
+ 2: 8, # 구역/단 정의
26
+ 3: 8, # 필드 시작
27
+ 11: 8, # 그리기 개체/표
28
+ 12: 8, # 예약
29
+ 14: 8, # 예약
30
+ 15: 8, # 숨은 설명
31
+ 16: 8, # 머리말/꼬리말
32
+ 17: 8, # 각주/미주
33
+ 18: 8, # 자동번호
34
+ 21: 8, # 페이지 컨트롤
35
+ 22: 8, # 책갈피/찾아보기
36
+ 23: 8, # 덧말/글자겹침
37
+ }
38
+
39
+ # extended 타입 ctrlId (이 코드의 컨트롤은 별도 오브젝트가 존재)
40
+ EXTENDED_CTRL_CHARS = {1, 2, 3, 11, 12, 14, 15, 16, 17, 18, 21, 22, 23}
41
+
42
+
43
+ def get_advance_bytes(char_code: int) -> int:
44
+ """제어 문자 하나가 차지하는 바이트 수 반환"""
45
+ wchar_count = CTRL_CHAR_WCHAR_SIZE.get(char_code, 1)
46
+ return wchar_count * 2
47
+
48
+
49
+ def is_extended_ctrl(char_code: int) -> bool:
50
+ """별도 오브젝트(표, 그림 등)를 가리키는 확장 컨트롤인지"""
51
+ return char_code in EXTENDED_CTRL_CHARS
File without changes
@@ -0,0 +1,93 @@
1
+ """
2
+ fallback/filter_server.py — 웹한글 기안기 필터 서버 연동
3
+ 파싱 실패 시 폴백으로 사용하거나, GT(Ground Truth) 비교용
4
+
5
+ 필터 서버 API:
6
+ POST /convert — HWP → HTML/PDF 변환 요청
7
+ """
8
+
9
+ import logging
10
+ from dataclasses import dataclass
11
+ from typing import Optional
12
+
13
+ logger = logging.getLogger('dochan')
14
+
15
+
16
+ @dataclass
17
+ class FilterServerConfig:
18
+ """필터 서버 연결 설정"""
19
+ base_url: str = "http://localhost:8080"
20
+ timeout: int = 30
21
+ api_key: str = ""
22
+
23
+
24
+ class FilterServerClient:
25
+ """웹한글 기안기 필터 서버 클라이언트"""
26
+
27
+ def __init__(self, config: Optional[FilterServerConfig] = None):
28
+ self.config = config or FilterServerConfig()
29
+ self._available = None
30
+
31
+ def is_available(self) -> bool:
32
+ """필터 서버 접속 가능 여부"""
33
+ if self._available is not None:
34
+ return self._available
35
+
36
+ try:
37
+ import urllib.request
38
+ req = urllib.request.Request(
39
+ f"{self.config.base_url}/health",
40
+ method='GET',
41
+ )
42
+ with urllib.request.urlopen(req, timeout=5) as resp:
43
+ self._available = resp.status == 200
44
+ except Exception:
45
+ self._available = False
46
+
47
+ return self._available
48
+
49
+ def convert_to_html(self, hwp_path: str) -> Optional[str]:
50
+ """HWP → HTML 변환 (필터 서버 사용)"""
51
+ if not self.is_available():
52
+ logger.warning("필터 서버 사용 불가")
53
+ return None
54
+
55
+ try:
56
+ import urllib.request
57
+
58
+ with open(hwp_path, 'rb') as f:
59
+ file_data = f.read()
60
+
61
+ # multipart/form-data 전송
62
+ boundary = '----HWPParserBoundary'
63
+ filename = hwp_path.split('/')[-1]
64
+
65
+ body = (
66
+ f'--{boundary}\r\n'
67
+ f'Content-Disposition: form-data; name="file"; filename="{filename}"\r\n'
68
+ f'Content-Type: application/octet-stream\r\n\r\n'
69
+ ).encode('utf-8') + file_data + f'\r\n--{boundary}--\r\n'.encode('utf-8')
70
+
71
+ req = urllib.request.Request(
72
+ f"{self.config.base_url}/convert",
73
+ data=body,
74
+ headers={
75
+ 'Content-Type': f'multipart/form-data; boundary={boundary}',
76
+ },
77
+ method='POST',
78
+ )
79
+
80
+ if self.config.api_key:
81
+ req.add_header('Authorization', f'Bearer {self.config.api_key}')
82
+
83
+ with urllib.request.urlopen(req, timeout=self.config.timeout) as resp:
84
+ return resp.read().decode('utf-8')
85
+
86
+ except Exception as e:
87
+ logger.error(f"필터 서버 변환 실패: {e}")
88
+ return None
89
+
90
+ def convert_as_fallback(self, hwp_path: str, original_errors: list) -> Optional[str]:
91
+ """파싱 실패 시 폴백 변환"""
92
+ logger.info(f"자체 파싱 실패 ({len(original_errors)}건 에러) → 필터 서버 폴백 시도")
93
+ return self.convert_to_html(hwp_path)
dochan/hwp/__init__.py ADDED
File without changes
dochan/hwp/bin_data.py ADDED
@@ -0,0 +1,96 @@
1
+ """
2
+ hwp/bin_data.py — BinData 이미지 연결
3
+ OLE 스토리지의 BinData/ 하위에 저장된 바이너리 데이터를 추출
4
+ """
5
+
6
+ import logging
7
+ import struct
8
+ import olefile
9
+ from dataclasses import dataclass
10
+ from typing import Dict, Optional
11
+
12
+ from ..utils.safe_decompress import safe_zlib_decompress
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ @dataclass
18
+ class BinDataItem:
19
+ """추출된 바이너리 데이터"""
20
+ storage_id: int = 0
21
+ data: bytes = b""
22
+ extension: str = ""
23
+
24
+ @property
25
+ def filename(self) -> str:
26
+ return f"BIN{self.storage_id:04X}.{self.extension}" if self.extension else f"BIN{self.storage_id:04X}"
27
+
28
+
29
+ def extract_bin_data(ole: olefile.OleFileIO, is_compressed: bool) -> Dict[int, BinDataItem]:
30
+ """
31
+ OLE 스토리지에서 BinData/ 하위의 모든 바이너리 데이터 추출
32
+
33
+ 반환: {storage_id: BinDataItem} 딕셔너리
34
+ """
35
+ result = {}
36
+
37
+ for entry in ole.listdir():
38
+ if len(entry) >= 2 and entry[0] == 'BinData':
39
+ storage_name = entry[1] # 예: "BIN0001.bmp"
40
+
41
+ try:
42
+ # 스토리지 ID 추출 (BIN 접두어 제거, 16진수)
43
+ base_name = storage_name.split('.')[0]
44
+ if base_name.upper().startswith('BIN'):
45
+ storage_id = int(base_name[3:], 16)
46
+ else:
47
+ continue
48
+
49
+ # 확장자
50
+ ext = storage_name.split('.')[-1] if '.' in storage_name else ""
51
+
52
+ # 데이터 읽기
53
+ raw_data = ole.openstream('/'.join(entry)).read()
54
+
55
+ # 압축 해제 (문서가 압축 설정인 경우)
56
+ if is_compressed:
57
+ try:
58
+ raw_data = safe_zlib_decompress(raw_data)
59
+ except (ValueError, Exception) as e:
60
+ logger.debug("BinData 압축 해제 실패 (비압축 데이터일 수 있음): %s", e)
61
+
62
+ result[storage_id] = BinDataItem(
63
+ storage_id=storage_id,
64
+ data=raw_data,
65
+ extension=ext.lower(),
66
+ )
67
+
68
+ except (ValueError, struct.error, UnicodeDecodeError, OSError) as e:
69
+ logger.warning("BinData 항목 '%s' 파싱 실패: %s", storage_name, e)
70
+ continue
71
+
72
+ return result
73
+
74
+
75
+ def link_images_to_bin_data(doc, bin_data_items: Dict[int, BinDataItem],
76
+ bin_data_entries: list):
77
+ """
78
+ Document 내 Image 객체에 실제 바이너리 데이터 연결
79
+
80
+ bin_data_entries: DocInfo에서 파싱한 BinDataEntry 목록
81
+ bin_data_items: OLE에서 추출한 BinDataItem 딕셔너리
82
+ """
83
+ from ..model.image import Image
84
+
85
+ for section in doc.sections:
86
+ for elem in section.elements:
87
+ if isinstance(elem, Image) and elem.bin_id >= 0:
88
+ # bin_id는 DocInfo의 BinDataEntry 인덱스 (0-based)
89
+ if elem.bin_id < len(bin_data_entries):
90
+ entry = bin_data_entries[elem.bin_id]
91
+ storage_id = entry.bin_data_id
92
+
93
+ if storage_id in bin_data_items:
94
+ item = bin_data_items[storage_id]
95
+ elem.image_data = item.data
96
+ elem.filename = item.filename