dochan 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dochan/__init__.py +9 -0
- dochan/batch.py +154 -0
- dochan/cli.py +118 -0
- dochan/constants.py +54 -0
- dochan/control_char.py +51 -0
- dochan/fallback/__init__.py +0 -0
- dochan/fallback/filter_server.py +93 -0
- dochan/hwp/__init__.py +0 -0
- dochan/hwp/bin_data.py +96 -0
- dochan/hwp/doc_info.py +212 -0
- dochan/hwp/header.py +78 -0
- dochan/hwp/records/__init__.py +0 -0
- dochan/hwp/records/char_shape.py +115 -0
- dochan/hwp/records/ctrl_header.py +67 -0
- dochan/hwp/records/para_char_shape.py +35 -0
- dochan/hwp/records/para_header.py +63 -0
- dochan/hwp/records/para_text.py +66 -0
- dochan/hwp/records/style.py +68 -0
- dochan/hwp/records/table.py +62 -0
- dochan/hwp/section.py +472 -0
- dochan/hwpx/__init__.py +0 -0
- dochan/hwpx/parser.py +461 -0
- dochan/model/__init__.py +0 -0
- dochan/model/document.py +97 -0
- dochan/model/equation.py +58 -0
- dochan/model/header_footer.py +25 -0
- dochan/model/image.py +25 -0
- dochan/model/style.py +37 -0
- dochan/model/table.py +37 -0
- dochan/output/__init__.py +0 -0
- dochan/output/json_out.py +84 -0
- dochan/output/markdown.py +159 -0
- dochan/output/plain_text.py +47 -0
- dochan/quality/__init__.py +0 -0
- dochan/quality/batch_validate.py +222 -0
- dochan/quality/checker.py +75 -0
- dochan/quality/comparator.py +66 -0
- dochan/quality/cross_validator.py +410 -0
- dochan/reader.py +174 -0
- dochan/tests/__init__.py +0 -0
- dochan/tests/test_char_shape.py +72 -0
- dochan/tests/test_control_char.py +90 -0
- dochan/tests/test_quality.py +82 -0
- dochan/utils/__init__.py +0 -0
- dochan/utils/error_recovery.py +55 -0
- dochan/utils/logger.py +38 -0
- dochan/utils/ocr.py +117 -0
- dochan/utils/safe_decompress.py +27 -0
- dochan-0.1.0.dist-info/METADATA +263 -0
- dochan-0.1.0.dist-info/RECORD +55 -0
- dochan-0.1.0.dist-info/WHEEL +5 -0
- dochan-0.1.0.dist-info/entry_points.txt +2 -0
- dochan-0.1.0.dist-info/licenses/LICENSE +21 -0
- dochan-0.1.0.dist-info/licenses/NOTICE +8 -0
- dochan-0.1.0.dist-info/top_level.txt +1 -0
dochan/__init__.py
ADDED
dochan/batch.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""
|
|
2
|
+
batch.py — 배치 처리
|
|
3
|
+
다수 HWP 파일의 병렬 변환
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
import logging
|
|
8
|
+
from concurrent.futures import ProcessPoolExecutor, as_completed
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import List
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger('dochan')
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class BatchResult:
|
|
18
|
+
"""단일 파일 처리 결과"""
|
|
19
|
+
file_path: str = ""
|
|
20
|
+
success: bool = False
|
|
21
|
+
output_path: str = ""
|
|
22
|
+
error_count: int = 0
|
|
23
|
+
errors: List[str] = field(default_factory=list)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class BatchSummary:
|
|
28
|
+
"""배치 처리 전체 요약"""
|
|
29
|
+
total: int = 0
|
|
30
|
+
success: int = 0
|
|
31
|
+
failed: int = 0
|
|
32
|
+
results: List[BatchResult] = field(default_factory=list)
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def success_rate(self) -> float:
|
|
36
|
+
return self.success / max(self.total, 1) * 100
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _process_single(file_path: str, output_dir: str, output_format: str) -> BatchResult:
|
|
40
|
+
"""단일 파일 처리 (별도 프로세스에서 실행)"""
|
|
41
|
+
from .reader import Dochan
|
|
42
|
+
|
|
43
|
+
result = BatchResult(file_path=file_path)
|
|
44
|
+
|
|
45
|
+
try:
|
|
46
|
+
reader = Dochan(file_path)
|
|
47
|
+
|
|
48
|
+
# 출력 파일명 생성
|
|
49
|
+
base_name = os.path.splitext(os.path.basename(file_path))[0]
|
|
50
|
+
ext_map = {'markdown': '.md', 'json': '.json', 'text': '.txt'}
|
|
51
|
+
ext = ext_map.get(output_format, '.md')
|
|
52
|
+
output_path = os.path.join(output_dir, base_name + ext)
|
|
53
|
+
|
|
54
|
+
# 변환
|
|
55
|
+
if output_format == 'json':
|
|
56
|
+
content = reader.to_json()
|
|
57
|
+
elif output_format == 'text':
|
|
58
|
+
content = reader.to_plain_text()
|
|
59
|
+
else:
|
|
60
|
+
content = reader.to_markdown()
|
|
61
|
+
|
|
62
|
+
# 저장
|
|
63
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
64
|
+
f.write(content)
|
|
65
|
+
|
|
66
|
+
result.success = True
|
|
67
|
+
result.output_path = output_path
|
|
68
|
+
result.errors = reader.errors
|
|
69
|
+
result.error_count = len(reader.errors)
|
|
70
|
+
|
|
71
|
+
except Exception as e:
|
|
72
|
+
result.success = False
|
|
73
|
+
result.errors = [str(e)]
|
|
74
|
+
result.error_count = 1
|
|
75
|
+
|
|
76
|
+
return result
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def batch_convert(
|
|
80
|
+
input_dir: str,
|
|
81
|
+
output_dir: str,
|
|
82
|
+
output_format: str = 'markdown',
|
|
83
|
+
max_workers: int = 4,
|
|
84
|
+
extensions: tuple = ('.hwp', '.hwpx'),
|
|
85
|
+
) -> BatchSummary:
|
|
86
|
+
"""
|
|
87
|
+
디렉토리 내 HWP 파일 일괄 변환
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
input_dir: 입력 디렉토리
|
|
91
|
+
output_dir: 출력 디렉토리
|
|
92
|
+
output_format: 'markdown', 'json', 'text'
|
|
93
|
+
max_workers: 병렬 워커 수
|
|
94
|
+
extensions: 처리할 확장자
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
BatchSummary
|
|
98
|
+
"""
|
|
99
|
+
# output_dir 경로 검증
|
|
100
|
+
try:
|
|
101
|
+
Path(output_dir).resolve().relative_to(Path(os.getcwd()).resolve())
|
|
102
|
+
except ValueError:
|
|
103
|
+
pass # output_dir가 cwd 밖이어도 허용하되, 아래에서 symlink 공격 방지
|
|
104
|
+
|
|
105
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
106
|
+
|
|
107
|
+
# 파일 수집
|
|
108
|
+
files = []
|
|
109
|
+
resolved_input = Path(input_dir).resolve()
|
|
110
|
+
for root, _, filenames in os.walk(input_dir):
|
|
111
|
+
for fn in filenames:
|
|
112
|
+
if any(fn.lower().endswith(ext) for ext in extensions):
|
|
113
|
+
full_path = os.path.join(root, fn)
|
|
114
|
+
# Path traversal 방지: 실제 경로가 input_dir 내부인지 검증
|
|
115
|
+
try:
|
|
116
|
+
Path(full_path).resolve().relative_to(resolved_input)
|
|
117
|
+
except ValueError:
|
|
118
|
+
logger.warning(f"경로 이탈 감지, 건너뜀: {full_path}")
|
|
119
|
+
continue
|
|
120
|
+
files.append(full_path)
|
|
121
|
+
|
|
122
|
+
summary = BatchSummary(total=len(files))
|
|
123
|
+
logger.info(f"배치 시작: {len(files)}개 파일, {max_workers} 워커")
|
|
124
|
+
|
|
125
|
+
# 병렬 처리
|
|
126
|
+
with ProcessPoolExecutor(max_workers=max_workers) as executor:
|
|
127
|
+
futures = {
|
|
128
|
+
executor.submit(_process_single, f, output_dir, output_format): f
|
|
129
|
+
for f in files
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
for future in as_completed(futures):
|
|
133
|
+
file_path = futures[future]
|
|
134
|
+
try:
|
|
135
|
+
result = future.result()
|
|
136
|
+
summary.results.append(result)
|
|
137
|
+
if result.success:
|
|
138
|
+
summary.success += 1
|
|
139
|
+
logger.info(f"✓ {os.path.basename(file_path)}")
|
|
140
|
+
else:
|
|
141
|
+
summary.failed += 1
|
|
142
|
+
logger.error(f"✗ {os.path.basename(file_path)}: {result.errors}")
|
|
143
|
+
except Exception as e:
|
|
144
|
+
summary.failed += 1
|
|
145
|
+
summary.results.append(BatchResult(
|
|
146
|
+
file_path=file_path,
|
|
147
|
+
success=False,
|
|
148
|
+
errors=[str(e)],
|
|
149
|
+
error_count=1,
|
|
150
|
+
))
|
|
151
|
+
logger.error(f"✗ {os.path.basename(file_path)}: {e}")
|
|
152
|
+
|
|
153
|
+
logger.info(f"배치 완료: {summary.success}/{summary.total} 성공 ({summary.success_rate:.1f}%)")
|
|
154
|
+
return summary
|
dochan/cli.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""dochan CLI — HWP/HWPX 문서를 터미널에서 변환
|
|
2
|
+
|
|
3
|
+
사용법:
|
|
4
|
+
dochan convert 문서.hwp # → stdout에 Markdown
|
|
5
|
+
dochan convert 문서.hwp -o output.md # → 파일로 저장
|
|
6
|
+
dochan convert 문서.hwp --format json # → JSON 출력
|
|
7
|
+
dochan convert 문서.hwpx --format text # → Plain text
|
|
8
|
+
dochan batch input_dir/ output_dir/ # → 디렉토리 일괄 변환
|
|
9
|
+
dochan info 문서.hwp # → 문서 메타데이터
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import sys
|
|
14
|
+
import os
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def main():
|
|
18
|
+
parser = argparse.ArgumentParser(
|
|
19
|
+
prog='dochan',
|
|
20
|
+
description='dochan — 독한 HWP/HWPX 파서, AI/LLM 최적 Markdown 변환',
|
|
21
|
+
)
|
|
22
|
+
subparsers = parser.add_subparsers(dest='command', help='명령')
|
|
23
|
+
|
|
24
|
+
# convert
|
|
25
|
+
conv = subparsers.add_parser('convert', help='HWP/HWPX → Markdown/JSON/Text 변환')
|
|
26
|
+
conv.add_argument('file', help='HWP 또는 HWPX 파일 경로')
|
|
27
|
+
conv.add_argument('-o', '--output', default=None, help='출력 파일 경로 (기본: stdout)')
|
|
28
|
+
conv.add_argument('-f', '--format', choices=['markdown', 'json', 'text'],
|
|
29
|
+
default='markdown', help='출력 형식 (기본: markdown)')
|
|
30
|
+
conv.add_argument('--ocr', action='store_true', help='이미지 OCR 활성화')
|
|
31
|
+
|
|
32
|
+
# batch
|
|
33
|
+
bat = subparsers.add_parser('batch', help='디렉토리 일괄 변환')
|
|
34
|
+
bat.add_argument('input_dir', help='입력 디렉토리')
|
|
35
|
+
bat.add_argument('output_dir', help='출력 디렉토리')
|
|
36
|
+
bat.add_argument('-f', '--format', choices=['markdown', 'json', 'text'],
|
|
37
|
+
default='markdown', help='출력 형식')
|
|
38
|
+
bat.add_argument('-w', '--workers', type=int, default=4, help='병렬 워커 수')
|
|
39
|
+
|
|
40
|
+
# info
|
|
41
|
+
inf = subparsers.add_parser('info', help='문서 메타데이터 출력')
|
|
42
|
+
inf.add_argument('file', help='HWP 또는 HWPX 파일 경로')
|
|
43
|
+
|
|
44
|
+
args = parser.parse_args()
|
|
45
|
+
|
|
46
|
+
if not args.command:
|
|
47
|
+
parser.print_help()
|
|
48
|
+
sys.exit(0)
|
|
49
|
+
|
|
50
|
+
if args.command == 'convert':
|
|
51
|
+
_cmd_convert(args)
|
|
52
|
+
elif args.command == 'batch':
|
|
53
|
+
_cmd_batch(args)
|
|
54
|
+
elif args.command == 'info':
|
|
55
|
+
_cmd_info(args)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _cmd_convert(args):
|
|
59
|
+
from .reader import Dochan
|
|
60
|
+
|
|
61
|
+
if not os.path.exists(args.file):
|
|
62
|
+
print(f"에러: 파일을 찾을 수 없습니다: {args.file}", file=sys.stderr)
|
|
63
|
+
sys.exit(1)
|
|
64
|
+
|
|
65
|
+
doc = Dochan(args.file, ocr=args.ocr)
|
|
66
|
+
|
|
67
|
+
if args.format == 'json':
|
|
68
|
+
content = doc.to_json()
|
|
69
|
+
elif args.format == 'text':
|
|
70
|
+
content = doc.to_plain_text()
|
|
71
|
+
else:
|
|
72
|
+
content = doc.to_markdown()
|
|
73
|
+
|
|
74
|
+
if args.output:
|
|
75
|
+
with open(args.output, 'w', encoding='utf-8') as f:
|
|
76
|
+
f.write(content)
|
|
77
|
+
print(f"저장 완료: {args.output}", file=sys.stderr)
|
|
78
|
+
else:
|
|
79
|
+
print(content)
|
|
80
|
+
|
|
81
|
+
if doc.errors:
|
|
82
|
+
for err in doc.errors:
|
|
83
|
+
print(f"경고: {err}", file=sys.stderr)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _cmd_batch(args):
|
|
87
|
+
from .batch import batch_convert
|
|
88
|
+
|
|
89
|
+
if not os.path.isdir(args.input_dir):
|
|
90
|
+
print(f"에러: 디렉토리를 찾을 수 없습니다: {args.input_dir}", file=sys.stderr)
|
|
91
|
+
sys.exit(1)
|
|
92
|
+
|
|
93
|
+
summary = batch_convert(
|
|
94
|
+
input_dir=args.input_dir,
|
|
95
|
+
output_dir=args.output_dir,
|
|
96
|
+
output_format=args.format,
|
|
97
|
+
max_workers=args.workers,
|
|
98
|
+
)
|
|
99
|
+
print(f"\n완료: {summary.success}/{summary.total} 성공 ({summary.success_rate:.1f}%)")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _cmd_info(args):
|
|
103
|
+
from .reader import Dochan
|
|
104
|
+
import json
|
|
105
|
+
|
|
106
|
+
if not os.path.exists(args.file):
|
|
107
|
+
print(f"에러: 파일을 찾을 수 없습니다: {args.file}", file=sys.stderr)
|
|
108
|
+
sys.exit(1)
|
|
109
|
+
|
|
110
|
+
doc = Dochan(args.file)
|
|
111
|
+
info = doc.metadata
|
|
112
|
+
info['file'] = args.file
|
|
113
|
+
info['format'] = 'hwpx' if args.file.lower().endswith('.hwpx') else 'hwp'
|
|
114
|
+
print(json.dumps(info, ensure_ascii=False, indent=2))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
if __name__ == '__main__':
|
|
118
|
+
main()
|
dochan/constants.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""
|
|
2
|
+
constants.py — 모든 TagID의 단일 진실 소스 (Single Source of Truth)
|
|
3
|
+
기준: 한글문서파일형식_5.0_revision1.3.pdf
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
HWPTAG_BEGIN = 16 # 0x010
|
|
7
|
+
|
|
8
|
+
# ──── DocInfo 레코드 (스펙 표 13) ────
|
|
9
|
+
HWPTAG_DOCUMENT_PROPERTIES = 16 # BEGIN+0
|
|
10
|
+
HWPTAG_ID_MAPPINGS = 17 # BEGIN+1
|
|
11
|
+
HWPTAG_BIN_DATA = 18 # BEGIN+2
|
|
12
|
+
HWPTAG_FACE_NAME = 19 # BEGIN+3
|
|
13
|
+
HWPTAG_BORDER_FILL = 20 # BEGIN+4
|
|
14
|
+
HWPTAG_CHAR_SHAPE = 21 # BEGIN+5
|
|
15
|
+
HWPTAG_TAB_DEF = 22 # BEGIN+6
|
|
16
|
+
HWPTAG_NUMBERING = 23 # BEGIN+7
|
|
17
|
+
HWPTAG_BULLET = 24 # BEGIN+8
|
|
18
|
+
HWPTAG_PARA_SHAPE = 25 # BEGIN+9
|
|
19
|
+
HWPTAG_STYLE = 26 # BEGIN+10
|
|
20
|
+
HWPTAG_DOC_DATA = 27 # BEGIN+11
|
|
21
|
+
HWPTAG_DISTRIBUTE_DOC_DATA = 28 # BEGIN+12
|
|
22
|
+
HWPTAG_COMPATIBLE_DOCUMENT = 30 # BEGIN+14
|
|
23
|
+
HWPTAG_LAYOUT_COMPATIBILITY = 31 # BEGIN+15
|
|
24
|
+
|
|
25
|
+
# ──── BodyText 레코드 (스펙 표 57) ────
|
|
26
|
+
HWPTAG_PARA_HEADER = 66 # BEGIN+50
|
|
27
|
+
HWPTAG_PARA_TEXT = 67 # BEGIN+51
|
|
28
|
+
HWPTAG_PARA_CHAR_SHAPE = 68 # BEGIN+52
|
|
29
|
+
HWPTAG_PARA_LINE_SEG = 69 # BEGIN+53
|
|
30
|
+
HWPTAG_PARA_RANGE_TAG = 70 # BEGIN+54
|
|
31
|
+
HWPTAG_CTRL_HEADER = 71 # BEGIN+55
|
|
32
|
+
HWPTAG_LIST_HEADER = 72 # BEGIN+56
|
|
33
|
+
HWPTAG_PAGE_DEF = 73 # BEGIN+57
|
|
34
|
+
HWPTAG_FOOTNOTE_SHAPE = 74 # BEGIN+58
|
|
35
|
+
HWPTAG_PAGE_BORDER_FILL = 75 # BEGIN+59
|
|
36
|
+
HWPTAG_SHAPE_COMPONENT = 76 # BEGIN+60
|
|
37
|
+
HWPTAG_TABLE = 77 # BEGIN+61
|
|
38
|
+
HWPTAG_SHAPE_COMP_LINE = 78 # BEGIN+62
|
|
39
|
+
HWPTAG_SHAPE_COMP_RECT = 79 # BEGIN+63
|
|
40
|
+
HWPTAG_SHAPE_COMP_ELLIPSE = 80 # BEGIN+64
|
|
41
|
+
HWPTAG_SHAPE_COMP_ARC = 81 # BEGIN+65
|
|
42
|
+
HWPTAG_SHAPE_COMP_POLYGON = 82 # BEGIN+66
|
|
43
|
+
HWPTAG_SHAPE_COMP_CURVE = 83 # BEGIN+67
|
|
44
|
+
HWPTAG_SHAPE_COMP_OLE = 84 # BEGIN+68
|
|
45
|
+
HWPTAG_SHAPE_COMP_PICTURE = 85 # BEGIN+69
|
|
46
|
+
HWPTAG_SHAPE_COMP_CONTAINER = 86 # BEGIN+70
|
|
47
|
+
HWPTAG_CTRL_DATA = 87 # BEGIN+71
|
|
48
|
+
HWPTAG_EQEDIT = 88 # BEGIN+72
|
|
49
|
+
HWPTAG_SHAPE_COMP_TEXTART = 90 # BEGIN+74
|
|
50
|
+
HWPTAG_FORM_OBJECT = 91 # BEGIN+75
|
|
51
|
+
HWPTAG_MEMO_SHAPE = 92 # BEGIN+76
|
|
52
|
+
HWPTAG_MEMO_LIST = 93 # BEGIN+77
|
|
53
|
+
HWPTAG_CHART_DATA = 95 # BEGIN+79
|
|
54
|
+
HWPTAG_VIDEO_DATA = 98 # BEGIN+82
|
dochan/control_char.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""
|
|
2
|
+
control_char.py — 제어 문자별 WCHAR 단위 크기
|
|
3
|
+
1 WCHAR = 2 bytes. 실제 바이트 크기 = 값 × 2
|
|
4
|
+
스펙 표 6 (p.10-11) 기준
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
CTRL_CHAR_WCHAR_SIZE = {
|
|
8
|
+
# ── char 타입 (1 WCHAR = 2 bytes) ──
|
|
9
|
+
10: 1, # 줄바꿈 (line break)
|
|
10
|
+
13: 1, # 문단 끝 (para break)
|
|
11
|
+
24: 1, # 하이픈
|
|
12
|
+
25: 1, 26: 1, 27: 1, 28: 1, 29: 1, # 예약 (char)
|
|
13
|
+
30: 1, # 묶음 빈칸
|
|
14
|
+
31: 1, # 고정폭 빈칸
|
|
15
|
+
|
|
16
|
+
# ── inline 타입 (8 WCHAR = 16 bytes) ──
|
|
17
|
+
4: 8, # 필드 끝
|
|
18
|
+
5: 8, 6: 8, 7: 8, # 예약 (inline)
|
|
19
|
+
8: 8, # title mark
|
|
20
|
+
9: 8, # 탭
|
|
21
|
+
19: 8, 20: 8, # 예약 (inline)
|
|
22
|
+
|
|
23
|
+
# ── extended 타입 (8 WCHAR = 16 bytes) ──
|
|
24
|
+
1: 8, # 예약
|
|
25
|
+
2: 8, # 구역/단 정의
|
|
26
|
+
3: 8, # 필드 시작
|
|
27
|
+
11: 8, # 그리기 개체/표
|
|
28
|
+
12: 8, # 예약
|
|
29
|
+
14: 8, # 예약
|
|
30
|
+
15: 8, # 숨은 설명
|
|
31
|
+
16: 8, # 머리말/꼬리말
|
|
32
|
+
17: 8, # 각주/미주
|
|
33
|
+
18: 8, # 자동번호
|
|
34
|
+
21: 8, # 페이지 컨트롤
|
|
35
|
+
22: 8, # 책갈피/찾아보기
|
|
36
|
+
23: 8, # 덧말/글자겹침
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
# extended 타입 ctrlId (이 코드의 컨트롤은 별도 오브젝트가 존재)
|
|
40
|
+
EXTENDED_CTRL_CHARS = {1, 2, 3, 11, 12, 14, 15, 16, 17, 18, 21, 22, 23}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def get_advance_bytes(char_code: int) -> int:
|
|
44
|
+
"""제어 문자 하나가 차지하는 바이트 수 반환"""
|
|
45
|
+
wchar_count = CTRL_CHAR_WCHAR_SIZE.get(char_code, 1)
|
|
46
|
+
return wchar_count * 2
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def is_extended_ctrl(char_code: int) -> bool:
|
|
50
|
+
"""별도 오브젝트(표, 그림 등)를 가리키는 확장 컨트롤인지"""
|
|
51
|
+
return char_code in EXTENDED_CTRL_CHARS
|
|
File without changes
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""
|
|
2
|
+
fallback/filter_server.py — 웹한글 기안기 필터 서버 연동
|
|
3
|
+
파싱 실패 시 폴백으로 사용하거나, GT(Ground Truth) 비교용
|
|
4
|
+
|
|
5
|
+
필터 서버 API:
|
|
6
|
+
POST /convert — HWP → HTML/PDF 변환 요청
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Optional
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger('dochan')
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class FilterServerConfig:
|
|
18
|
+
"""필터 서버 연결 설정"""
|
|
19
|
+
base_url: str = "http://localhost:8080"
|
|
20
|
+
timeout: int = 30
|
|
21
|
+
api_key: str = ""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class FilterServerClient:
|
|
25
|
+
"""웹한글 기안기 필터 서버 클라이언트"""
|
|
26
|
+
|
|
27
|
+
def __init__(self, config: Optional[FilterServerConfig] = None):
|
|
28
|
+
self.config = config or FilterServerConfig()
|
|
29
|
+
self._available = None
|
|
30
|
+
|
|
31
|
+
def is_available(self) -> bool:
|
|
32
|
+
"""필터 서버 접속 가능 여부"""
|
|
33
|
+
if self._available is not None:
|
|
34
|
+
return self._available
|
|
35
|
+
|
|
36
|
+
try:
|
|
37
|
+
import urllib.request
|
|
38
|
+
req = urllib.request.Request(
|
|
39
|
+
f"{self.config.base_url}/health",
|
|
40
|
+
method='GET',
|
|
41
|
+
)
|
|
42
|
+
with urllib.request.urlopen(req, timeout=5) as resp:
|
|
43
|
+
self._available = resp.status == 200
|
|
44
|
+
except Exception:
|
|
45
|
+
self._available = False
|
|
46
|
+
|
|
47
|
+
return self._available
|
|
48
|
+
|
|
49
|
+
def convert_to_html(self, hwp_path: str) -> Optional[str]:
|
|
50
|
+
"""HWP → HTML 변환 (필터 서버 사용)"""
|
|
51
|
+
if not self.is_available():
|
|
52
|
+
logger.warning("필터 서버 사용 불가")
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
import urllib.request
|
|
57
|
+
|
|
58
|
+
with open(hwp_path, 'rb') as f:
|
|
59
|
+
file_data = f.read()
|
|
60
|
+
|
|
61
|
+
# multipart/form-data 전송
|
|
62
|
+
boundary = '----HWPParserBoundary'
|
|
63
|
+
filename = hwp_path.split('/')[-1]
|
|
64
|
+
|
|
65
|
+
body = (
|
|
66
|
+
f'--{boundary}\r\n'
|
|
67
|
+
f'Content-Disposition: form-data; name="file"; filename="{filename}"\r\n'
|
|
68
|
+
f'Content-Type: application/octet-stream\r\n\r\n'
|
|
69
|
+
).encode('utf-8') + file_data + f'\r\n--{boundary}--\r\n'.encode('utf-8')
|
|
70
|
+
|
|
71
|
+
req = urllib.request.Request(
|
|
72
|
+
f"{self.config.base_url}/convert",
|
|
73
|
+
data=body,
|
|
74
|
+
headers={
|
|
75
|
+
'Content-Type': f'multipart/form-data; boundary={boundary}',
|
|
76
|
+
},
|
|
77
|
+
method='POST',
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
if self.config.api_key:
|
|
81
|
+
req.add_header('Authorization', f'Bearer {self.config.api_key}')
|
|
82
|
+
|
|
83
|
+
with urllib.request.urlopen(req, timeout=self.config.timeout) as resp:
|
|
84
|
+
return resp.read().decode('utf-8')
|
|
85
|
+
|
|
86
|
+
except Exception as e:
|
|
87
|
+
logger.error(f"필터 서버 변환 실패: {e}")
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
def convert_as_fallback(self, hwp_path: str, original_errors: list) -> Optional[str]:
|
|
91
|
+
"""파싱 실패 시 폴백 변환"""
|
|
92
|
+
logger.info(f"자체 파싱 실패 ({len(original_errors)}건 에러) → 필터 서버 폴백 시도")
|
|
93
|
+
return self.convert_to_html(hwp_path)
|
dochan/hwp/__init__.py
ADDED
|
File without changes
|
dochan/hwp/bin_data.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""
|
|
2
|
+
hwp/bin_data.py — BinData 이미지 연결
|
|
3
|
+
OLE 스토리지의 BinData/ 하위에 저장된 바이너리 데이터를 추출
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import logging
|
|
7
|
+
import struct
|
|
8
|
+
import olefile
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from typing import Dict, Optional
|
|
11
|
+
|
|
12
|
+
from ..utils.safe_decompress import safe_zlib_decompress
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class BinDataItem:
|
|
19
|
+
"""추출된 바이너리 데이터"""
|
|
20
|
+
storage_id: int = 0
|
|
21
|
+
data: bytes = b""
|
|
22
|
+
extension: str = ""
|
|
23
|
+
|
|
24
|
+
@property
|
|
25
|
+
def filename(self) -> str:
|
|
26
|
+
return f"BIN{self.storage_id:04X}.{self.extension}" if self.extension else f"BIN{self.storage_id:04X}"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def extract_bin_data(ole: olefile.OleFileIO, is_compressed: bool) -> Dict[int, BinDataItem]:
|
|
30
|
+
"""
|
|
31
|
+
OLE 스토리지에서 BinData/ 하위의 모든 바이너리 데이터 추출
|
|
32
|
+
|
|
33
|
+
반환: {storage_id: BinDataItem} 딕셔너리
|
|
34
|
+
"""
|
|
35
|
+
result = {}
|
|
36
|
+
|
|
37
|
+
for entry in ole.listdir():
|
|
38
|
+
if len(entry) >= 2 and entry[0] == 'BinData':
|
|
39
|
+
storage_name = entry[1] # 예: "BIN0001.bmp"
|
|
40
|
+
|
|
41
|
+
try:
|
|
42
|
+
# 스토리지 ID 추출 (BIN 접두어 제거, 16진수)
|
|
43
|
+
base_name = storage_name.split('.')[0]
|
|
44
|
+
if base_name.upper().startswith('BIN'):
|
|
45
|
+
storage_id = int(base_name[3:], 16)
|
|
46
|
+
else:
|
|
47
|
+
continue
|
|
48
|
+
|
|
49
|
+
# 확장자
|
|
50
|
+
ext = storage_name.split('.')[-1] if '.' in storage_name else ""
|
|
51
|
+
|
|
52
|
+
# 데이터 읽기
|
|
53
|
+
raw_data = ole.openstream('/'.join(entry)).read()
|
|
54
|
+
|
|
55
|
+
# 압축 해제 (문서가 압축 설정인 경우)
|
|
56
|
+
if is_compressed:
|
|
57
|
+
try:
|
|
58
|
+
raw_data = safe_zlib_decompress(raw_data)
|
|
59
|
+
except (ValueError, Exception) as e:
|
|
60
|
+
logger.debug("BinData 압축 해제 실패 (비압축 데이터일 수 있음): %s", e)
|
|
61
|
+
|
|
62
|
+
result[storage_id] = BinDataItem(
|
|
63
|
+
storage_id=storage_id,
|
|
64
|
+
data=raw_data,
|
|
65
|
+
extension=ext.lower(),
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
except (ValueError, struct.error, UnicodeDecodeError, OSError) as e:
|
|
69
|
+
logger.warning("BinData 항목 '%s' 파싱 실패: %s", storage_name, e)
|
|
70
|
+
continue
|
|
71
|
+
|
|
72
|
+
return result
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def link_images_to_bin_data(doc, bin_data_items: Dict[int, BinDataItem],
|
|
76
|
+
bin_data_entries: list):
|
|
77
|
+
"""
|
|
78
|
+
Document 내 Image 객체에 실제 바이너리 데이터 연결
|
|
79
|
+
|
|
80
|
+
bin_data_entries: DocInfo에서 파싱한 BinDataEntry 목록
|
|
81
|
+
bin_data_items: OLE에서 추출한 BinDataItem 딕셔너리
|
|
82
|
+
"""
|
|
83
|
+
from ..model.image import Image
|
|
84
|
+
|
|
85
|
+
for section in doc.sections:
|
|
86
|
+
for elem in section.elements:
|
|
87
|
+
if isinstance(elem, Image) and elem.bin_id >= 0:
|
|
88
|
+
# bin_id는 DocInfo의 BinDataEntry 인덱스 (0-based)
|
|
89
|
+
if elem.bin_id < len(bin_data_entries):
|
|
90
|
+
entry = bin_data_entries[elem.bin_id]
|
|
91
|
+
storage_id = entry.bin_data_id
|
|
92
|
+
|
|
93
|
+
if storage_id in bin_data_items:
|
|
94
|
+
item = bin_data_items[storage_id]
|
|
95
|
+
elem.image_data = item.data
|
|
96
|
+
elem.filename = item.filename
|