python-hwpx 6.0.2__py3-none-any.whl → 6.0.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hwpx/body_patch.py +10 -6
- hwpx/ingest/hwpx_converter.py +13 -2
- hwpx/layout/lint.py +3 -1
- hwpx/mutation_report.py +3 -1
- hwpx/opc/package.py +2 -2
- hwpx/opc/security.py +110 -4
- hwpx/patch.py +11 -2
- hwpx/quality/save_pipeline.py +3 -1
- hwpx/table_patch.py +11 -5
- hwpx/tools/archive_cli.py +29 -2
- hwpx/tools/exporter.py +2 -2
- hwpx/tools/idempotence.py +3 -1
- hwpx/tools/ir_equality.py +3 -1
- hwpx/tools/layout_preview.py +2 -2
- hwpx/tools/package_validator.py +8 -6
- hwpx/tools/repair.py +10 -1
- hwpx/tools/text_extractor.py +11 -4
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/METADATA +1 -1
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/RECORD +24 -24
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/WHEEL +0 -0
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/entry_points.txt +0 -0
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/licenses/LICENSE +0 -0
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/licenses/NOTICE +0 -0
- {python_hwpx-6.0.2.dist-info → python_hwpx-6.0.3.dist-info}/top_level.txt +0 -0
hwpx/body_patch.py
CHANGED
|
@@ -35,6 +35,7 @@ from dataclasses import dataclass
|
|
|
35
35
|
from pathlib import Path
|
|
36
36
|
from typing import Any, Mapping, Sequence
|
|
37
37
|
|
|
38
|
+
from .opc.security import guard_zip_file, read_member
|
|
38
39
|
from .mutation_report import MutationReport, project_byte_splice
|
|
39
40
|
from .patch import (
|
|
40
41
|
_finalize,
|
|
@@ -440,10 +441,11 @@ def recolor_runs_by_color(
|
|
|
440
441
|
import io, zipfile
|
|
441
442
|
|
|
442
443
|
with zipfile.ZipFile(io.BytesIO(source_bytes)) as z:
|
|
444
|
+
guard_zip_file(z)
|
|
443
445
|
names = z.namelist()
|
|
444
446
|
header_name = next((n for n in names if n.endswith("header.xml")), None)
|
|
445
|
-
header_xml = z
|
|
446
|
-
sections = {n: z
|
|
447
|
+
header_xml = read_member(z, header_name).decode("utf-8") if header_name else ""
|
|
448
|
+
sections = {n: read_member(z, n).decode("utf-8") for n in names if re.search(r"section\d+\.xml$", n)}
|
|
447
449
|
|
|
448
450
|
ids = set()
|
|
449
451
|
for cm in re.finditer(r"<(?:[A-Za-z_][\w.-]*:)?charPr\b[^>]*?>", header_xml):
|
|
@@ -511,10 +513,11 @@ def strip_runs_by_color(
|
|
|
511
513
|
|
|
512
514
|
targets = {h.upper() for h in hex_colors}
|
|
513
515
|
with zipfile.ZipFile(io.BytesIO(source_bytes)) as z:
|
|
516
|
+
guard_zip_file(z)
|
|
514
517
|
names = z.namelist()
|
|
515
518
|
header_name = next((n for n in names if n.endswith("header.xml")), None)
|
|
516
|
-
header_xml = z
|
|
517
|
-
sections = {n: z
|
|
519
|
+
header_xml = read_member(z, header_name).decode("utf-8") if header_name else ""
|
|
520
|
+
sections = {n: read_member(z, n).decode("utf-8") for n in names if re.search(r"section\d+\.xml$", n)}
|
|
518
521
|
|
|
519
522
|
# 계열 매칭(잔존 게이트와 정렬): 대상 색의 _color_family에 드는 모든 charPr.
|
|
520
523
|
from .oxml.color import color_family
|
|
@@ -589,13 +592,14 @@ def apply_body_ops(
|
|
|
589
592
|
|
|
590
593
|
header_part: str | None = None
|
|
591
594
|
with zipfile.ZipFile(io.BytesIO(source_bytes)) as z:
|
|
595
|
+
guard_zip_file(z)
|
|
592
596
|
sections = {
|
|
593
|
-
n: z
|
|
597
|
+
n: read_member(z, n).decode("utf-8")
|
|
594
598
|
for n in z.namelist()
|
|
595
599
|
if re.search(r"section\d+\.xml$", n)
|
|
596
600
|
}
|
|
597
601
|
header_part = next((n for n in z.namelist() if n.endswith("header.xml")), None)
|
|
598
|
-
header_xml = z
|
|
602
|
+
header_xml = read_member(z, header_part).decode("utf-8") if header_part else ""
|
|
599
603
|
|
|
600
604
|
ctx: dict[str, Any] = {"header": header_xml, "header_changed": False, "charpr_cache": {}}
|
|
601
605
|
skipped: list[dict[str, Any]] = []
|
hwpx/ingest/hwpx_converter.py
CHANGED
|
@@ -10,6 +10,12 @@ from zipfile import BadZipFile, ZipFile
|
|
|
10
10
|
from hwpx.document import HwpxDocument
|
|
11
11
|
|
|
12
12
|
from .base import DocumentIngestResult, DocumentSourceInfo
|
|
13
|
+
from ..opc.security import (
|
|
14
|
+
MAX_ZIP_MIMETYPE_BYTES as _MAX_MIMETYPE_BYTES,
|
|
15
|
+
HwpxSecurityError,
|
|
16
|
+
guard_zip_file,
|
|
17
|
+
read_member,
|
|
18
|
+
)
|
|
13
19
|
|
|
14
20
|
|
|
15
21
|
class HwpxMarkdownConverter:
|
|
@@ -70,16 +76,21 @@ def _looks_like_hwpx_package(file_stream: BinaryIO) -> bool:
|
|
|
70
76
|
cur_pos = file_stream.tell()
|
|
71
77
|
try:
|
|
72
78
|
with ZipFile(file_stream) as archive:
|
|
79
|
+
guard_zip_file(archive)
|
|
73
80
|
names = set(archive.namelist())
|
|
74
81
|
if "mimetype" in names:
|
|
75
82
|
try:
|
|
76
|
-
mimetype =
|
|
83
|
+
mimetype = (
|
|
84
|
+
read_member(archive, "mimetype", limit=_MAX_MIMETYPE_BYTES)
|
|
85
|
+
.decode("utf-8", "replace")
|
|
86
|
+
.strip()
|
|
87
|
+
)
|
|
77
88
|
if "hwp" in mimetype.lower() or "hwpx" in mimetype.lower():
|
|
78
89
|
return True
|
|
79
90
|
except Exception:
|
|
80
91
|
pass
|
|
81
92
|
return any(name.startswith("Contents/section") and name.endswith(".xml") for name in names)
|
|
82
|
-
except (BadZipFile, OSError):
|
|
93
|
+
except (BadZipFile, OSError, HwpxSecurityError):
|
|
83
94
|
return False
|
|
84
95
|
finally:
|
|
85
96
|
file_stream.seek(cur_pos)
|
hwpx/layout/lint.py
CHANGED
|
@@ -35,6 +35,7 @@ from hwpx.tools.package_validator import (
|
|
|
35
35
|
)
|
|
36
36
|
|
|
37
37
|
from .report import LayoutFinding, LayoutLintReport
|
|
38
|
+
from ..opc.security import guard_zip_file, read_member
|
|
38
39
|
|
|
39
40
|
if TYPE_CHECKING:
|
|
40
41
|
from hwpx.quality.ledger import DirtyLayoutLedger
|
|
@@ -173,11 +174,12 @@ def _section_roots(data: bytes) -> list[tuple[str, ET.Element]]:
|
|
|
173
174
|
roots: list[tuple[str, ET.Element]] = []
|
|
174
175
|
try:
|
|
175
176
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
177
|
+
guard_zip_file(archive)
|
|
176
178
|
for info in archive.infolist():
|
|
177
179
|
if info.is_dir() or not is_section_part_name(info.filename):
|
|
178
180
|
continue
|
|
179
181
|
try:
|
|
180
|
-
roots.append((info.filename, ET.fromstring(archive
|
|
182
|
+
roots.append((info.filename, ET.fromstring(read_member(archive, info))))
|
|
181
183
|
except ET.ParseError:
|
|
182
184
|
# Malformed XML is the pipeline's well-formedness floor, not ours.
|
|
183
185
|
continue
|
hwpx/mutation_report.py
CHANGED
|
@@ -19,6 +19,7 @@ import zipfile
|
|
|
19
19
|
from pathlib import Path
|
|
20
20
|
from dataclasses import dataclass, field
|
|
21
21
|
from io import BytesIO
|
|
22
|
+
from .opc.security import guard_zip_file, read_member
|
|
22
23
|
from typing import Any, Literal, Mapping, Sequence
|
|
23
24
|
from zipfile import ZipInfo
|
|
24
25
|
|
|
@@ -58,8 +59,9 @@ def read_archive_members(data: bytes) -> dict[str, bytes]:
|
|
|
58
59
|
"""Return the uncompressed content of every non-directory member of *data*."""
|
|
59
60
|
|
|
60
61
|
with zipfile.ZipFile(BytesIO(data)) as archive:
|
|
62
|
+
guard_zip_file(archive)
|
|
61
63
|
return {
|
|
62
|
-
info.filename: archive
|
|
64
|
+
info.filename: read_member(archive, info)
|
|
63
65
|
for info in archive.infolist()
|
|
64
66
|
if not info.is_dir()
|
|
65
67
|
}
|
hwpx/opc/package.py
CHANGED
|
@@ -24,7 +24,7 @@ from .relationships import (
|
|
|
24
24
|
parse_container_rootfiles,
|
|
25
25
|
parse_manifest_relationships,
|
|
26
26
|
)
|
|
27
|
-
from .security import guard_zip_file
|
|
27
|
+
from .security import guard_zip_file, read_zip_members
|
|
28
28
|
from .xml_utils import (
|
|
29
29
|
extract_xml_declaration,
|
|
30
30
|
iter_declared_namespaces,
|
|
@@ -419,7 +419,7 @@ class HwpxPackage:
|
|
|
419
419
|
with ZipFile(stream, "r") as zf:
|
|
420
420
|
guard_zip_file(zf)
|
|
421
421
|
infos = [info for info in zf.infolist() if not info.is_dir()]
|
|
422
|
-
files =
|
|
422
|
+
files = read_zip_members(zf)
|
|
423
423
|
zip_infos = {info.filename: info for info in infos}
|
|
424
424
|
zip_order = [info.filename for info in infos]
|
|
425
425
|
except BadZipFile as exc:
|
hwpx/opc/security.py
CHANGED
|
@@ -4,9 +4,11 @@
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
|
-
from pathlib import PurePosixPath
|
|
7
|
+
from pathlib import PurePosixPath, PureWindowsPath
|
|
8
8
|
from typing import Iterable
|
|
9
9
|
from xml.etree import ElementTree as ET
|
|
10
|
+
from zipfile import ZIP_DEFLATED as _ZIP_DEFLATED
|
|
11
|
+
from zipfile import ZIP_STORED as _ZIP_STORED
|
|
10
12
|
from zipfile import ZipFile, ZipInfo
|
|
11
13
|
|
|
12
14
|
MAX_XML_BYTES = 64 * 1024 * 1024
|
|
@@ -15,6 +17,20 @@ MAX_ZIP_ENTRIES = 4096
|
|
|
15
17
|
MAX_ZIP_MEMBER_BYTES = 128 * 1024 * 1024
|
|
16
18
|
MAX_ZIP_TOTAL_UNCOMPRESSED_BYTES = 512 * 1024 * 1024
|
|
17
19
|
MAX_ZIP_COMPRESSION_RATIO = 1000.0
|
|
20
|
+
MAX_ZIP_READ_CHUNK = 64 * 1024
|
|
21
|
+
# `mimetype` is a short fixed string in every real package; reading it during
|
|
22
|
+
# format sniffing must not be able to cost the full per-member allowance.
|
|
23
|
+
MAX_ZIP_MIMETYPE_BYTES = 4 * 1024
|
|
24
|
+
# Container/manifest/preview parts are small by construction; reading them
|
|
25
|
+
# during discovery should not cost the generic per-member allowance either.
|
|
26
|
+
MAX_ZIP_SMALL_PART_BYTES = 4 * 1024 * 1024
|
|
27
|
+
|
|
28
|
+
# The only methods HWPX uses, and the only ones CPython decompresses under a
|
|
29
|
+
# bound: ``ZipExtFile._read1`` passes ``max_length`` to zlib for ZIP_DEFLATED but
|
|
30
|
+
# calls ``decompress(data)`` with no limit for every other method, so a bzip2 or
|
|
31
|
+
# lzma member inflates in full before the declared size truncates the result.
|
|
32
|
+
# Chunked reading cannot bound those, so they are refused outright.
|
|
33
|
+
_ALLOWED_METHODS = (_ZIP_STORED, _ZIP_DEFLATED)
|
|
18
34
|
|
|
19
35
|
|
|
20
36
|
class HwpxSecurityError(ValueError):
|
|
@@ -35,8 +51,12 @@ def _iter_file_infos(zf: ZipFile) -> list[ZipInfo]:
|
|
|
35
51
|
|
|
36
52
|
def _guard_zip_name(name: str) -> None:
|
|
37
53
|
normalized = name.replace("\\", "/")
|
|
38
|
-
|
|
39
|
-
|
|
54
|
+
if PurePosixPath(normalized).is_absolute() or ".." in PurePosixPath(normalized).parts:
|
|
55
|
+
raise HwpxSecurityError(f"unsafe ZIP member path: {name!r}")
|
|
56
|
+
# ``PurePosixPath`` reads ``C:/x`` as a relative path, but joining it onto an
|
|
57
|
+
# output directory on Windows discards that directory entirely.
|
|
58
|
+
windows = PureWindowsPath(normalized)
|
|
59
|
+
if windows.drive or windows.is_absolute():
|
|
40
60
|
raise HwpxSecurityError(f"unsafe ZIP member path: {name!r}")
|
|
41
61
|
|
|
42
62
|
|
|
@@ -55,9 +75,20 @@ def guard_zip_file(
|
|
|
55
75
|
f"{len(infos)} > {active_limits.max_entries}"
|
|
56
76
|
)
|
|
57
77
|
|
|
78
|
+
# ``is_dir()`` only tests for a trailing slash in the name, so a member can
|
|
79
|
+
# opt out of the size accounting below just by calling itself a directory
|
|
80
|
+
# while still holding a payload that ``ZipFile.open()`` will read. Check the
|
|
81
|
+
# name of every entry, and require the ones excluded here to be empty.
|
|
82
|
+
for info in zf.infolist():
|
|
83
|
+
_guard_zip_name(info.filename)
|
|
84
|
+
if info.is_dir() and (info.file_size or info.compress_size):
|
|
85
|
+
raise HwpxSecurityError(
|
|
86
|
+
"ZIP directory entry carries data: "
|
|
87
|
+
f"{info.filename}={info.file_size}/{info.compress_size}"
|
|
88
|
+
)
|
|
89
|
+
|
|
58
90
|
total = 0
|
|
59
91
|
for info in infos:
|
|
60
|
-
_guard_zip_name(info.filename)
|
|
61
92
|
if info.file_size > active_limits.max_member_bytes:
|
|
62
93
|
raise HwpxSecurityError(
|
|
63
94
|
"ZIP member exceeds uncompressed size limit: "
|
|
@@ -69,6 +100,18 @@ def guard_zip_file(
|
|
|
69
100
|
"ZIP archive exceeds total uncompressed size limit: "
|
|
70
101
|
f"{total} > {active_limits.max_total_uncompressed_bytes}"
|
|
71
102
|
)
|
|
103
|
+
if info.compress_type not in _ALLOWED_METHODS:
|
|
104
|
+
raise HwpxSecurityError(
|
|
105
|
+
"ZIP member uses an unsupported compression method: "
|
|
106
|
+
f"{info.filename}={info.compress_type}"
|
|
107
|
+
)
|
|
108
|
+
# Must run before the ``file_size <= 0`` skip below: a member declaring 0
|
|
109
|
+
# would otherwise bypass every remaining check.
|
|
110
|
+
if info.compress_size > info.file_size + info.file_size // 1000 + 64:
|
|
111
|
+
raise HwpxSecurityError(
|
|
112
|
+
"ZIP member declares less data than it stores: "
|
|
113
|
+
f"{info.filename}={info.file_size} < {info.compress_size}"
|
|
114
|
+
)
|
|
72
115
|
if info.file_size <= 0:
|
|
73
116
|
continue
|
|
74
117
|
if info.compress_size <= 0:
|
|
@@ -81,6 +124,69 @@ def guard_zip_file(
|
|
|
81
124
|
)
|
|
82
125
|
|
|
83
126
|
|
|
127
|
+
def read_member(
|
|
128
|
+
zf: ZipFile,
|
|
129
|
+
member: str | ZipInfo,
|
|
130
|
+
*,
|
|
131
|
+
limit: int = MAX_ZIP_MEMBER_BYTES,
|
|
132
|
+
) -> bytes:
|
|
133
|
+
"""Read one ZIP member without trusting the size it declares.
|
|
134
|
+
|
|
135
|
+
``ZipFile.read()`` calls ``ZipExtFile.read()`` with no argument, which hands
|
|
136
|
+
zlib a 2 GiB ``max_length`` and inflates the whole stream before truncating
|
|
137
|
+
the result to the declared size. Reading in fixed chunks and counting what
|
|
138
|
+
actually arrives keeps the allocation bounded no matter what the central
|
|
139
|
+
directory claims.
|
|
140
|
+
"""
|
|
141
|
+
|
|
142
|
+
name = member.filename if isinstance(member, ZipInfo) else member
|
|
143
|
+
chunks: list[bytes] = []
|
|
144
|
+
total = 0
|
|
145
|
+
with zf.open(member) as handle:
|
|
146
|
+
while True:
|
|
147
|
+
block = handle.read(MAX_ZIP_READ_CHUNK)
|
|
148
|
+
if not block:
|
|
149
|
+
break
|
|
150
|
+
total += len(block)
|
|
151
|
+
if total > limit:
|
|
152
|
+
raise HwpxSecurityError(
|
|
153
|
+
"ZIP member exceeds uncompressed size limit: "
|
|
154
|
+
f"{name} > {limit}"
|
|
155
|
+
)
|
|
156
|
+
chunks.append(block)
|
|
157
|
+
return b"".join(chunks)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def read_zip_members(
|
|
161
|
+
zf: ZipFile,
|
|
162
|
+
*,
|
|
163
|
+
limits: ZipGuardLimits | None = None,
|
|
164
|
+
names: Iterable[str] | None = None,
|
|
165
|
+
) -> dict[str, bytes]:
|
|
166
|
+
"""Read members with per-member and cumulative bounds on what is produced.
|
|
167
|
+
|
|
168
|
+
``names`` restricts the read to those members; the cumulative budget still
|
|
169
|
+
applies across everything read.
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
active_limits = limits or ZipGuardLimits()
|
|
173
|
+
wanted = None if names is None else set(names)
|
|
174
|
+
payloads: dict[str, bytes] = {}
|
|
175
|
+
total = 0
|
|
176
|
+
for info in _iter_file_infos(zf):
|
|
177
|
+
if wanted is not None and info.filename not in wanted:
|
|
178
|
+
continue
|
|
179
|
+
data = read_member(zf, info, limit=active_limits.max_member_bytes)
|
|
180
|
+
total += len(data)
|
|
181
|
+
if total > active_limits.max_total_uncompressed_bytes:
|
|
182
|
+
raise HwpxSecurityError(
|
|
183
|
+
"ZIP archive exceeds total uncompressed size limit: "
|
|
184
|
+
f"{total} > {active_limits.max_total_uncompressed_bytes}"
|
|
185
|
+
)
|
|
186
|
+
payloads[info.filename] = data
|
|
187
|
+
return payloads
|
|
188
|
+
|
|
189
|
+
|
|
84
190
|
def guard_xml_bytes(
|
|
85
191
|
payload: bytes,
|
|
86
192
|
*,
|
hwpx/patch.py
CHANGED
|
@@ -12,6 +12,7 @@ from pathlib import Path
|
|
|
12
12
|
from typing import Any, Mapping, Sequence
|
|
13
13
|
from zipfile import ZIP_DEFLATED, ZIP_STORED, ZipFile
|
|
14
14
|
|
|
15
|
+
from .opc.security import guard_zip_file, read_member, read_zip_members
|
|
15
16
|
from .mutation_report import MutationReport, project_byte_splice, visual_value_from_status
|
|
16
17
|
from .quality import QualityPolicy, SavePipeline
|
|
17
18
|
from .quality.report import VisualCompleteReport
|
|
@@ -199,6 +200,10 @@ def paragraph_patch(
|
|
|
199
200
|
source_bytes = _read_source_bytes(source)
|
|
200
201
|
normalized_patches = tuple(_normalize_patch(item) for item in patches)
|
|
201
202
|
if not normalized_patches:
|
|
203
|
+
# The early return still hands the source to the save pipeline, so it has
|
|
204
|
+
# to clear the same limits as the patching path below.
|
|
205
|
+
with ZipFile(io.BytesIO(source_bytes), "r") as archive:
|
|
206
|
+
guard_zip_file(archive)
|
|
202
207
|
open_safety, visual_complete = _finalize(source_bytes, output_path, source=source)
|
|
203
208
|
return BytePreservingPatchResult(
|
|
204
209
|
data=source_bytes,
|
|
@@ -212,7 +217,8 @@ def paragraph_patch(
|
|
|
212
217
|
)
|
|
213
218
|
|
|
214
219
|
with ZipFile(io.BytesIO(source_bytes), "r") as archive:
|
|
215
|
-
|
|
220
|
+
guard_zip_file(archive)
|
|
221
|
+
parts = read_zip_members(archive)
|
|
216
222
|
|
|
217
223
|
changed_parts: dict[str, bytes] = {}
|
|
218
224
|
applied: list[PatchApplied] = []
|
|
@@ -481,9 +487,12 @@ def _apply_edits(payload: bytes, edits: Sequence[tuple[int, int, bytes]]) -> byt
|
|
|
481
487
|
def _rewrite_zip_entries(source: bytes, replacements: Mapping[str, bytes]) -> bytes:
|
|
482
488
|
buffer = io.BytesIO()
|
|
483
489
|
with ZipFile(io.BytesIO(source), "r") as src:
|
|
490
|
+
# Also reached directly by the public rewrite_package_parts(), so the
|
|
491
|
+
# entry-count, total-size and ratio limits have to be applied here too.
|
|
492
|
+
guard_zip_file(src)
|
|
484
493
|
with ZipFile(buffer, "w") as dst:
|
|
485
494
|
for info in src.infolist():
|
|
486
|
-
payload = replacements.get(info.filename, src
|
|
495
|
+
payload = replacements.get(info.filename, read_member(src, info))
|
|
487
496
|
dst.writestr(info, payload)
|
|
488
497
|
return buffer.getvalue()
|
|
489
498
|
|
hwpx/quality/save_pipeline.py
CHANGED
|
@@ -48,6 +48,7 @@ from .report import (
|
|
|
48
48
|
VisualCompleteReport,
|
|
49
49
|
VisualCompleteStatus,
|
|
50
50
|
)
|
|
51
|
+
from ..opc.security import guard_zip_file, parse_xml_stdlib, read_member
|
|
51
52
|
|
|
52
53
|
PublishMode = Literal["on_pass", "always", "never"]
|
|
53
54
|
|
|
@@ -277,12 +278,13 @@ class SavePipeline:
|
|
|
277
278
|
|
|
278
279
|
try:
|
|
279
280
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
281
|
+
guard_zip_file(archive)
|
|
280
282
|
names = [info.filename for info in archive.infolist() if not info.is_dir()]
|
|
281
283
|
for name in names:
|
|
282
284
|
base = os.path.basename(name)
|
|
283
285
|
if name.endswith(_XML_SUFFIXES) or base in _XML_NAMES:
|
|
284
286
|
try:
|
|
285
|
-
|
|
287
|
+
parse_xml_stdlib(read_member(archive, name), part_name=name)
|
|
286
288
|
except ET.ParseError as exc:
|
|
287
289
|
errors.append(
|
|
288
290
|
QualityError(
|
hwpx/table_patch.py
CHANGED
|
@@ -30,6 +30,7 @@ from pathlib import Path
|
|
|
30
30
|
from typing import Any, Iterable, Mapping, Sequence
|
|
31
31
|
|
|
32
32
|
from .errors import HwpxError
|
|
33
|
+
from .opc.security import guard_zip_file, read_member, read_zip_members
|
|
33
34
|
from .mutation_report import MutationReport, project_byte_splice
|
|
34
35
|
from .patch import (
|
|
35
36
|
_apply_edits,
|
|
@@ -585,8 +586,9 @@ def resolve_cell_target(
|
|
|
585
586
|
import zipfile
|
|
586
587
|
|
|
587
588
|
with zipfile.ZipFile(io.BytesIO(source_bytes), "r") as archive:
|
|
589
|
+
guard_zip_file(archive)
|
|
588
590
|
parts = {
|
|
589
|
-
info.filename: archive
|
|
591
|
+
info.filename: read_member(archive, info)
|
|
590
592
|
for info in archive.infolist()
|
|
591
593
|
if not info.is_dir()
|
|
592
594
|
}
|
|
@@ -661,7 +663,8 @@ def fill_cells(
|
|
|
661
663
|
import io
|
|
662
664
|
import zipfile
|
|
663
665
|
with zipfile.ZipFile(io.BytesIO(source_bytes), "r") as zf:
|
|
664
|
-
|
|
666
|
+
guard_zip_file(zf)
|
|
667
|
+
parts = read_zip_members(zf)
|
|
665
668
|
|
|
666
669
|
# FR-002: resolve table/cell anchors to concrete (table_index,row,col) first.
|
|
667
670
|
resolved_cells, anchor_skips = _resolve_anchor_cells(parts, cells)
|
|
@@ -1376,7 +1379,8 @@ def _apply_cell_line_spacing(
|
|
|
1376
1379
|
import io
|
|
1377
1380
|
import zipfile
|
|
1378
1381
|
with zipfile.ZipFile(io.BytesIO(source_bytes), "r") as zf:
|
|
1379
|
-
|
|
1382
|
+
guard_zip_file(zf)
|
|
1383
|
+
parts = read_zip_members(zf)
|
|
1380
1384
|
header_name = _header_part_name(parts)
|
|
1381
1385
|
header = parts.get(header_name)
|
|
1382
1386
|
transcript: list[dict[str, Any]] = []
|
|
@@ -1482,7 +1486,8 @@ def _sections(data: bytes) -> dict[str, bytes]:
|
|
|
1482
1486
|
import io
|
|
1483
1487
|
import zipfile
|
|
1484
1488
|
with zipfile.ZipFile(io.BytesIO(data)) as z:
|
|
1485
|
-
|
|
1489
|
+
guard_zip_file(z)
|
|
1490
|
+
return {n: read_member(z, n) for n in z.namelist() if re.search(r"section\d+\.xml$", n)}
|
|
1486
1491
|
|
|
1487
1492
|
|
|
1488
1493
|
def _table_dims(table: str | bytes) -> str:
|
|
@@ -1795,11 +1800,12 @@ def strip_trailing_table_captions(
|
|
|
1795
1800
|
import zipfile
|
|
1796
1801
|
|
|
1797
1802
|
with zipfile.ZipFile(io.BytesIO(source_bytes), "r") as archive:
|
|
1803
|
+
guard_zip_file(archive)
|
|
1798
1804
|
section_names = [
|
|
1799
1805
|
info.filename for info in archive.infolist()
|
|
1800
1806
|
if not info.is_dir() and re.search(r"section\d+\.xml$", info.filename)
|
|
1801
1807
|
]
|
|
1802
|
-
sections = {name: archive
|
|
1808
|
+
sections = {name: read_member(archive, name).decode("utf-8") for name in section_names}
|
|
1803
1809
|
|
|
1804
1810
|
applied: list[CellApplied] = []
|
|
1805
1811
|
changed_parts: dict[str, bytes] = {}
|
hwpx/tools/archive_cli.py
CHANGED
|
@@ -17,6 +17,7 @@ from lxml import etree # type: ignore[reportAttributeAccessIssue]
|
|
|
17
17
|
from ..opc.relationships import is_header_part_name, is_section_part_name
|
|
18
18
|
from ..oxml.namespaces import HWPML_COMPAT_ROOT_NAMESPACES
|
|
19
19
|
from .package_validator import validate_editor_open_safety, validate_package
|
|
20
|
+
from ..opc.security import HwpxSecurityError, guard_xml_bytes, guard_xml_depth, guard_zip_file, read_member
|
|
20
21
|
|
|
21
22
|
_XML_SUFFIXES = (".xml", ".hpf")
|
|
22
23
|
_PACK_METADATA_NAME = ".hwpx-pack-metadata.json"
|
|
@@ -81,18 +82,43 @@ def _prepare_output_path(output_path: Path, *, overwrite: bool) -> None:
|
|
|
81
82
|
raise FileExistsError(f"output file already exists: {output_path}")
|
|
82
83
|
|
|
83
84
|
|
|
85
|
+
_MAX_INDENT_GROWTH = 8
|
|
86
|
+
_MIN_INDENT_BUDGET = 64 * 1024
|
|
87
|
+
|
|
88
|
+
|
|
84
89
|
def _format_xml_bytes(payload: bytes) -> bytes:
|
|
90
|
+
"""Re-indent an XML part, falling back to the original bytes.
|
|
91
|
+
|
|
92
|
+
Indentation is an amplifier: at the depth libxml2 accepts, every leaf gains
|
|
93
|
+
two spaces per level, so a 4-byte element can grow past 500 bytes. The
|
|
94
|
+
payload is guarded first, and a result that grew beyond the per-member
|
|
95
|
+
allowance is discarded in favour of the input.
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
try:
|
|
99
|
+
guard_xml_bytes(payload, part_name="XML part")
|
|
100
|
+
except HwpxSecurityError:
|
|
101
|
+
return payload
|
|
85
102
|
try:
|
|
86
103
|
element = etree.fromstring(payload)
|
|
87
104
|
except etree.XMLSyntaxError:
|
|
88
105
|
return payload
|
|
106
|
+
try:
|
|
107
|
+
guard_xml_depth(element, part_name="XML part")
|
|
108
|
+
except HwpxSecurityError:
|
|
109
|
+
return payload
|
|
89
110
|
etree.indent(element, space=" ")
|
|
90
|
-
|
|
111
|
+
formatted = etree.tostring(
|
|
91
112
|
element,
|
|
92
113
|
pretty_print=True,
|
|
93
114
|
xml_declaration=True,
|
|
94
115
|
encoding="UTF-8",
|
|
95
116
|
)
|
|
117
|
+
# Real parts grow at most ~1.7x when indented (measured across the repo's
|
|
118
|
+
# packages); anything past this is the indentation acting as an amplifier.
|
|
119
|
+
if len(formatted) > max(_MAX_INDENT_GROWTH * len(payload), _MIN_INDENT_BUDGET):
|
|
120
|
+
return payload
|
|
121
|
+
return formatted
|
|
96
122
|
|
|
97
123
|
|
|
98
124
|
def _normalize_hwpml_compat_root(rel_path: str, payload: bytes) -> bytes:
|
|
@@ -225,9 +251,10 @@ def unpack_hwpx(
|
|
|
225
251
|
_prepare_output_dir(destination, overwrite=overwrite)
|
|
226
252
|
|
|
227
253
|
with ZipFile(source_path, "r") as archive:
|
|
254
|
+
guard_zip_file(archive)
|
|
228
255
|
entries = _iter_file_entries(archive)
|
|
229
256
|
for entry in entries:
|
|
230
|
-
data = archive
|
|
257
|
+
data = read_member(archive, entry.path)
|
|
231
258
|
if pretty_xml and entry.path.endswith(_XML_SUFFIXES):
|
|
232
259
|
data = _format_xml_bytes(data)
|
|
233
260
|
target = destination / entry.path
|
hwpx/tools/exporter.py
CHANGED
|
@@ -15,7 +15,7 @@ from typing import TYPE_CHECKING
|
|
|
15
15
|
from xml.etree import ElementTree as ET
|
|
16
16
|
from zipfile import ZipFile
|
|
17
17
|
|
|
18
|
-
from ..opc.security import guard_zip_file, parse_xml_stdlib
|
|
18
|
+
from ..opc.security import guard_zip_file, parse_xml_stdlib, read_member
|
|
19
19
|
#: A caller-supplied redaction step. Declared here rather than imported from
|
|
20
20
|
#: mail_merge, which imports export_text — the two would form a cycle.
|
|
21
21
|
TextSanitizer = Callable[[str], str]
|
|
@@ -41,7 +41,7 @@ def _section_xmls(source: HwpxDocument | bytes) -> list[ET.Element]:
|
|
|
41
41
|
with ZipFile(io.BytesIO(source)) as zf:
|
|
42
42
|
guard_zip_file(zf)
|
|
43
43
|
names = sorted(n for n in zf.namelist() if _SECTION_RE.match(n))
|
|
44
|
-
return [parse_xml_stdlib(zf
|
|
44
|
+
return [parse_xml_stdlib(read_member(zf, n), part_name=n) for n in names]
|
|
45
45
|
return [sec.element for sec in source._root.sections]
|
|
46
46
|
|
|
47
47
|
|
hwpx/tools/idempotence.py
CHANGED
|
@@ -28,6 +28,7 @@ import zipfile
|
|
|
28
28
|
from dataclasses import dataclass
|
|
29
29
|
|
|
30
30
|
from hwpx.document import HwpxDocument
|
|
31
|
+
from ..opc.security import guard_zip_file, read_member
|
|
31
32
|
|
|
32
33
|
__all__ = [
|
|
33
34
|
"IdempotenceReport",
|
|
@@ -78,11 +79,12 @@ def _part_contents(data: bytes) -> tuple[dict[str, bytes], list[str]]:
|
|
|
78
79
|
contents: dict[str, bytes] = {}
|
|
79
80
|
duplicates: list[str] = []
|
|
80
81
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
82
|
+
guard_zip_file(archive)
|
|
81
83
|
for info in archive.infolist():
|
|
82
84
|
name = info.filename
|
|
83
85
|
if name in contents:
|
|
84
86
|
duplicates.append(name)
|
|
85
|
-
contents[name] = archive
|
|
87
|
+
contents[name] = read_member(archive, info)
|
|
86
88
|
return contents, duplicates
|
|
87
89
|
|
|
88
90
|
|
hwpx/tools/ir_equality.py
CHANGED
|
@@ -20,6 +20,7 @@ import re
|
|
|
20
20
|
import xml.etree.ElementTree as ET
|
|
21
21
|
import zipfile
|
|
22
22
|
from dataclasses import dataclass
|
|
23
|
+
from ..opc.security import guard_zip_file, read_member
|
|
23
24
|
|
|
24
25
|
__all__ = [
|
|
25
26
|
"IrEqualityReport",
|
|
@@ -91,12 +92,13 @@ def project_document(data: bytes) -> list:
|
|
|
91
92
|
"""Project a whole HWPX byte blob: paragraphs across all sections in order."""
|
|
92
93
|
projection: list = []
|
|
93
94
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
95
|
+
guard_zip_file(archive)
|
|
94
96
|
names = sorted(
|
|
95
97
|
(n for n in archive.namelist() if _SECTION_RE.match(n)),
|
|
96
98
|
key=lambda n: int(_SECTION_RE.match(n).group(1)),
|
|
97
99
|
)
|
|
98
100
|
for name in names:
|
|
99
|
-
projection.extend(project_section_xml(archive
|
|
101
|
+
projection.extend(project_section_xml(read_member(archive, name)))
|
|
100
102
|
return projection
|
|
101
103
|
|
|
102
104
|
|
hwpx/tools/layout_preview.py
CHANGED
|
@@ -17,7 +17,7 @@ from xml.etree import ElementTree as ET
|
|
|
17
17
|
from zipfile import BadZipFile, ZipFile
|
|
18
18
|
|
|
19
19
|
from ..equation import render_equation
|
|
20
|
-
from ..opc.security import guard_zip_file, parse_xml_stdlib
|
|
20
|
+
from ..opc.security import guard_zip_file, parse_xml_stdlib, read_member
|
|
21
21
|
|
|
22
22
|
_HP_NS = "http://www.hancom.co.kr/hwpml/2011/paragraph"
|
|
23
23
|
_HH_NS = "http://www.hancom.co.kr/hwpml/2011/head"
|
|
@@ -191,7 +191,7 @@ def _read_package_parts(source: str | Path | bytes) -> tuple[dict[str, bytes], l
|
|
|
191
191
|
archive = Path(source)
|
|
192
192
|
with ZipFile(archive) as zf:
|
|
193
193
|
guard_zip_file(zf)
|
|
194
|
-
return {name: zf
|
|
194
|
+
return {name: read_member(zf, name) for name in zf.namelist()}, warnings
|
|
195
195
|
except (BadZipFile, FileNotFoundError, OSError) as exc:
|
|
196
196
|
raise ValueError(f"unable to read HWPX package: {exc}") from exc
|
|
197
197
|
|
hwpx/tools/package_validator.py
CHANGED
|
@@ -13,6 +13,7 @@ from zipfile import ZIP_STORED, BadZipFile, ZipFile, ZipInfo
|
|
|
13
13
|
from lxml import etree as LET # type: ignore[reportMissingImports]
|
|
14
14
|
|
|
15
15
|
from ..oxml.namespaces import HWPML_COMPAT_ROOT_NAMESPACES
|
|
16
|
+
from ..opc.security import HwpxSecurityError, MAX_ZIP_MEMBER_BYTES, MAX_ZIP_MIMETYPE_BYTES, MAX_ZIP_SMALL_PART_BYTES, read_member
|
|
16
17
|
from ..opc.relationships import (
|
|
17
18
|
MAIN_ROOTFILE_MEDIA_TYPE,
|
|
18
19
|
ManifestRelationships,
|
|
@@ -24,7 +25,6 @@ from ..opc.relationships import (
|
|
|
24
25
|
select_main_rootfile,
|
|
25
26
|
)
|
|
26
27
|
from ..opc.security import (
|
|
27
|
-
HwpxSecurityError,
|
|
28
28
|
guard_xml_bytes,
|
|
29
29
|
guard_xml_depth,
|
|
30
30
|
guard_zip_file,
|
|
@@ -543,10 +543,12 @@ def _warning(
|
|
|
543
543
|
issues.append(PackageValidationIssue(part_name, message, "warning"))
|
|
544
544
|
|
|
545
545
|
|
|
546
|
-
def _safe_read(
|
|
546
|
+
def _safe_read(
|
|
547
|
+
zf: ZipFile, part_name: str, *, limit: int = MAX_ZIP_MEMBER_BYTES
|
|
548
|
+
) -> bytes | None:
|
|
547
549
|
try:
|
|
548
|
-
return zf
|
|
549
|
-
except (BadZipFile, KeyError, OSError):
|
|
550
|
+
return read_member(zf, part_name, limit=limit)
|
|
551
|
+
except (BadZipFile, KeyError, OSError, HwpxSecurityError):
|
|
550
552
|
return None
|
|
551
553
|
|
|
552
554
|
|
|
@@ -573,7 +575,7 @@ def _check_mimetype(
|
|
|
573
575
|
if MIMETYPE_PATH not in name_set:
|
|
574
576
|
_error(issues, MIMETYPE_PATH, "missing required file")
|
|
575
577
|
return
|
|
576
|
-
mimetype_bytes = _safe_read(zf, MIMETYPE_PATH)
|
|
578
|
+
mimetype_bytes = _safe_read(zf, MIMETYPE_PATH, limit=MAX_ZIP_MIMETYPE_BYTES)
|
|
577
579
|
if mimetype_bytes is None:
|
|
578
580
|
_error(
|
|
579
581
|
issues,
|
|
@@ -620,7 +622,7 @@ def _check_preview_text(
|
|
|
620
622
|
"missing Preview/PrvText.txt; macOS Hancom compatibility may require it",
|
|
621
623
|
)
|
|
622
624
|
return
|
|
623
|
-
preview_bytes = _safe_read(zf, PREVIEW_TEXT_PATH)
|
|
625
|
+
preview_bytes = _safe_read(zf, PREVIEW_TEXT_PATH, limit=MAX_ZIP_SMALL_PART_BYTES)
|
|
624
626
|
if preview_bytes is None:
|
|
625
627
|
_error(issues, PREVIEW_TEXT_PATH, "unable to read preview text entry")
|
|
626
628
|
elif len(preview_bytes) > 1024 * 1024:
|
hwpx/tools/repair.py
CHANGED
|
@@ -15,6 +15,7 @@ from ..opc.relationships import is_header_part_name, is_section_part_name
|
|
|
15
15
|
from ..oxml.namespaces import HWPML_COMPAT_ROOT_NAMESPACES
|
|
16
16
|
from .package_validator import MIMETYPE_PATH, validate_editor_open_safety, validate_package
|
|
17
17
|
from .recover import recover_entries
|
|
18
|
+
from ..opc.security import guard_zip_file, read_member
|
|
18
19
|
|
|
19
20
|
__all__ = [
|
|
20
21
|
"RepairResult",
|
|
@@ -66,6 +67,9 @@ def _read_entries(
|
|
|
66
67
|
entries: list[_BufferedEntry] = []
|
|
67
68
|
total_size = 0
|
|
68
69
|
with ZipFile(source_path, "r") as archive:
|
|
70
|
+
# The per-entry/total limits below sum declared sizes, so the archive
|
|
71
|
+
# still needs the ratio, entry-count and member-name checks.
|
|
72
|
+
guard_zip_file(archive)
|
|
69
73
|
for info in archive.infolist():
|
|
70
74
|
if info.is_dir():
|
|
71
75
|
continue
|
|
@@ -74,7 +78,12 @@ def _read_entries(
|
|
|
74
78
|
total_size += info.file_size
|
|
75
79
|
if total_size > max_total_size:
|
|
76
80
|
raise ValueError(f"archive exceeds max_total_size={max_total_size}")
|
|
77
|
-
entries.append(
|
|
81
|
+
entries.append(
|
|
82
|
+
_BufferedEntry(
|
|
83
|
+
info=info,
|
|
84
|
+
payload=read_member(archive, info, limit=max_entry_size),
|
|
85
|
+
)
|
|
86
|
+
)
|
|
78
87
|
return tuple(entries)
|
|
79
88
|
|
|
80
89
|
|
hwpx/tools/text_extractor.py
CHANGED
|
@@ -16,7 +16,7 @@ from ..opc.relationships import (
|
|
|
16
16
|
parse_manifest_relationships,
|
|
17
17
|
select_main_rootfile,
|
|
18
18
|
)
|
|
19
|
-
from ..opc.security import guard_zip_file, parse_xml_stdlib
|
|
19
|
+
from ..opc.security import MAX_ZIP_SMALL_PART_BYTES, guard_zip_file, parse_xml_stdlib, read_member
|
|
20
20
|
from ..oxml.namespaces import DEFAULT_NAMESPACES as OWPML_DEFAULT_NAMESPACES
|
|
21
21
|
|
|
22
22
|
__all__ = [
|
|
@@ -213,7 +213,7 @@ class TextExtractor:
|
|
|
213
213
|
archive = self.open()
|
|
214
214
|
section_files = list(self._iter_section_files(archive))
|
|
215
215
|
for index, name in enumerate(section_files):
|
|
216
|
-
data = archive
|
|
216
|
+
data = read_member(archive, name)
|
|
217
217
|
element = parse_xml_stdlib(data, part_name=name)
|
|
218
218
|
yield SectionInfo(index=index, name=name, element=element)
|
|
219
219
|
|
|
@@ -584,7 +584,11 @@ class TextExtractor:
|
|
|
584
584
|
manifest_path: str | None = None
|
|
585
585
|
try:
|
|
586
586
|
container_root = parse_xml_stdlib(
|
|
587
|
-
|
|
587
|
+
read_member(
|
|
588
|
+
archive,
|
|
589
|
+
"META-INF/container.xml",
|
|
590
|
+
limit=MAX_ZIP_SMALL_PART_BYTES,
|
|
591
|
+
),
|
|
588
592
|
part_name="META-INF/container.xml",
|
|
589
593
|
)
|
|
590
594
|
except (ValueError, KeyError):
|
|
@@ -598,7 +602,10 @@ class TextExtractor:
|
|
|
598
602
|
|
|
599
603
|
if manifest_path is not None:
|
|
600
604
|
try:
|
|
601
|
-
manifest_root = parse_xml_stdlib(
|
|
605
|
+
manifest_root = parse_xml_stdlib(
|
|
606
|
+
read_member(archive, manifest_path, limit=MAX_ZIP_SMALL_PART_BYTES),
|
|
607
|
+
part_name=manifest_path,
|
|
608
|
+
)
|
|
602
609
|
except (ValueError, KeyError):
|
|
603
610
|
manifest_root = None
|
|
604
611
|
if manifest_root is not None:
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
hwpx/__init__.py,sha256=DqoPNdD-4-oP4b_56P5ADABnyRGcDZiHch__S-urI4s,16642
|
|
2
|
-
hwpx/body_patch.py,sha256=
|
|
2
|
+
hwpx/body_patch.py,sha256=g8RhaFQ9nf3grVI7qVPEwJc06x0JmwXow8hkY84aT18,29165
|
|
3
3
|
hwpx/capabilities.py,sha256=rhSiHTnt8ZqUvhN3VdQVt8FveifQKlEUQ3LhXiAlaZw,27644
|
|
4
4
|
hwpx/document.py,sha256=rG65RrwgmADeFd4TBolKUMCnBZiecfjphCP7v3dS9T0,28370
|
|
5
5
|
hwpx/errors.py,sha256=otyGkuRVzlItKY-AkrXSqTc_zhe2T4C2q3-2kGrRqbo,13570
|
|
6
6
|
hwpx/experimental.py,sha256=Jvc7WAi3eMFRKYubqwtIWi9oU0OjuqNOfottWkaP2RQ,2113
|
|
7
7
|
hwpx/model.py,sha256=UEShdjlj9piS4RadFBFWZgB8PSM14u7CqdFgPO9Mqcc,3129
|
|
8
|
-
hwpx/mutation_report.py,sha256=
|
|
8
|
+
hwpx/mutation_report.py,sha256=Uw9fxft8kioXpQ0feQwSlq4vOj3-418Oywt7qmDmU_8,19494
|
|
9
9
|
hwpx/package.py,sha256=0rKjGCJbPQvrVBIy07Jpjsu3fI7HhbqFCGWTiTDsJpo,1141
|
|
10
|
-
hwpx/patch.py,sha256=
|
|
10
|
+
hwpx/patch.py,sha256=k_LoVxHzIBNs4kg2V4yBOgAxjsrim7pksG0rmRmLApE,27264
|
|
11
11
|
hwpx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
-
hwpx/table_patch.py,sha256=
|
|
12
|
+
hwpx/table_patch.py,sha256=CQHJ5bN_Kfz-Pi0eIyJJZg5fGJRbq1HjP1jDZqnv0d8,83845
|
|
13
13
|
hwpx/templates.py,sha256=28bYqeJVeDb1Cq8G9NZG9Mhnu4K2GamAKC4QhxvUZyA,1187
|
|
14
14
|
hwpx/_document/__init__.py,sha256=REiNqMbuk_TS4NVC27FIo1rUt3Z8nhJrTBGOgNtxBRg,90
|
|
15
15
|
hwpx/_document/_legacy.py,sha256=8jFKZe8qsYV_3nAvmZBg4qNkcuedG-P2QCA0LeKTyjo,58909
|
|
@@ -55,9 +55,9 @@ hwpx/form_fit/policy.py,sha256=iXAatjS7ZycOdKgMCL4Q_kYg0qnogFPdRNlr_ecOvjc,3106
|
|
|
55
55
|
hwpx/form_fit/report.py,sha256=h1NzQfbj1JYvZtKUi0DckJNWGnUHMZtQ3iCzbUxUjHA,3305
|
|
56
56
|
hwpx/ingest/__init__.py,sha256=cjDQwMdaqbq2xiD32zmaFAyzZbENga_6MLbgo0sLGZE,636
|
|
57
57
|
hwpx/ingest/base.py,sha256=95jXJ70Hkay70BgyryGOzI3UVTixUpR-6f71iF3eqn4,8134
|
|
58
|
-
hwpx/ingest/hwpx_converter.py,sha256=
|
|
58
|
+
hwpx/ingest/hwpx_converter.py,sha256=TOCxlQjHwbvxD7rwnoIy9HSNf-NduDyE-YSDRx8Nslg,4599
|
|
59
59
|
hwpx/layout/__init__.py,sha256=qotAVhol_yrPlOHvQfs68nx7ADYxPcwITYNA9gMutUM,1158
|
|
60
|
-
hwpx/layout/lint.py,sha256=
|
|
60
|
+
hwpx/layout/lint.py,sha256=pUkAa-T8UhLhciuCndSDSu8l4IBivbON-yU7yV6VFO8,15840
|
|
61
61
|
hwpx/layout/report.py,sha256=6jipfJ7PUhSGX8-A2gryE3GrAwN_mbfnaI_PdK931ok,4182
|
|
62
62
|
hwpx/objects/__init__.py,sha256=7az6QN0HoDUV4FvM-2C6449f5AsUmdV6_2fJ7FPDBL4,1755
|
|
63
63
|
hwpx/objects/binary_item.py,sha256=cx3QejHBr26VgQgUyd5jfJrxrSZxrFDgRGwuMC1Bcvw,1884
|
|
@@ -65,9 +65,9 @@ hwpx/objects/checkbox.py,sha256=zOrBBxwQSdvIl0HLqhXc0LF2cIOWbPrA1sEPnqVj2Tc,3085
|
|
|
65
65
|
hwpx/objects/form_field.py,sha256=LHudu4EritJa6G0MOiRiR77RfvbEsQpC7HsSr5o7WVg,5689
|
|
66
66
|
hwpx/objects/results.py,sha256=BkKu50lCIin7fKyUOA4GDy2a3LhWwN13_a0h1Fe_rvA,7574
|
|
67
67
|
hwpx/objects/tracked.py,sha256=uTCm-RQ7c37Yg03nEm9Q6mhcYFGFN2_pBRrwToNPiZ4,2169
|
|
68
|
-
hwpx/opc/package.py,sha256=
|
|
68
|
+
hwpx/opc/package.py,sha256=nz1m3L7palQrbWSedpN_06WJaFpwDO2iZqjCknPZH4E,40067
|
|
69
69
|
hwpx/opc/relationships.py,sha256=tPWLHRMlw0Spvtwou2jCDRfHdcm9FEKKLd95YVHLwYI,6971
|
|
70
|
-
hwpx/opc/security.py,sha256=
|
|
70
|
+
hwpx/opc/security.py,sha256=0wZitXK0J55wkbsWhyMyyufq0syBu_ZsBkqozrp-2DA,9062
|
|
71
71
|
hwpx/opc/xml_utils.py,sha256=L_fHY1-D5I_TfdRkDQV-bn55EnXc6AqEDWItfMpawVs,3840
|
|
72
72
|
hwpx/oxml/__init__.py,sha256=pqHqOK2snS74U_mkmkFcgyCUCLiGmbS84TKFBrwxF5I,5507
|
|
73
73
|
hwpx/oxml/_document_impl.py,sha256=41cxktJqXHER37fjA5HIFR8kw7dHMoISTxEv53PRBYo,5367
|
|
@@ -104,43 +104,43 @@ hwpx/quality/ledger.py,sha256=MvrUxd6LlJPD_dWoD5yH9ceFPNcjBR6tr6STqQqAdk0,3534
|
|
|
104
104
|
hwpx/quality/policy.py,sha256=Buc2CkxfPPkbygf9i9o7FCjwkuL2BV39WG4xKjYkbis,3822
|
|
105
105
|
hwpx/quality/rendering.py,sha256=3uFu-nXCgWYsQDgyRH9DwKGyvSvXnN_pPzEmp5g6NNo,4141
|
|
106
106
|
hwpx/quality/report.py,sha256=n8bOP6vIDy7ZLprApUqUa38HM6NXph57PHSkzT8Wgjk,8721
|
|
107
|
-
hwpx/quality/save_pipeline.py,sha256=
|
|
107
|
+
hwpx/quality/save_pipeline.py,sha256=fUlgQWbvq8uaL8W2JpEPHaN8xVc9e_Q-b-J6wf0QOxI,21838
|
|
108
108
|
hwpx/tools/__init__.py,sha256=6PiL7jvyTfyP_-Ygo119h7kFs3X7tml96kI1JmX2ZWA,3049
|
|
109
|
-
hwpx/tools/archive_cli.py,sha256=
|
|
109
|
+
hwpx/tools/archive_cli.py,sha256=eNuTxdQGn56g7R5HPwqMjExYdvv6uuShiJ-vzEGTF0o,13860
|
|
110
110
|
hwpx/tools/doc_diff.py,sha256=PlHeKlNSyDiifPw-kZfaVJwyHl1tIr2bVlrzR2d80zo,10891
|
|
111
111
|
hwpx/tools/document_viewer.py,sha256=cJb1KfWnbHvNgFDuzpsaF8lJmPTQaplCk08OD-Vh93A,8100
|
|
112
|
-
hwpx/tools/exporter.py,sha256=
|
|
112
|
+
hwpx/tools/exporter.py,sha256=hFyovh7W9WW7wfLw0ol9xK5EadXn7hMa_IXelMheayM,8054
|
|
113
113
|
hwpx/tools/generic_inventory.py,sha256=pHVP8-htX_vO02ARdQR37XFxm7fUPK68VtMeeOJ1NZY,4835
|
|
114
114
|
hwpx/tools/id_integrity.py,sha256=TBI2WOwbaMbwTGyMv0hEO-uZqyewZtdvawqqA42Tums,14245
|
|
115
|
-
hwpx/tools/idempotence.py,sha256=
|
|
116
|
-
hwpx/tools/ir_equality.py,sha256=
|
|
117
|
-
hwpx/tools/layout_preview.py,sha256
|
|
115
|
+
hwpx/tools/idempotence.py,sha256=UNIgfItSCoMFBxSfd3vhA-1AMJ5YJ4eGLtdOpWtfZtc,5274
|
|
116
|
+
hwpx/tools/ir_equality.py,sha256=zku8kKVNwrADcwLXWvmQ4U8VlhOgliPWyv9Ca2ao1yg,4860
|
|
117
|
+
hwpx/tools/layout_preview.py,sha256=-gXLXBgLnIxxFpRh--oyKOodsekzoFZZ_4xSZW2jB2U,22723
|
|
118
118
|
hwpx/tools/mail_merge.py,sha256=8kGlSk2GQL7cFxZgg6P_bKisj1zVUWy8qh8AXIZLK_w,18611
|
|
119
119
|
hwpx/tools/markdown_export.py,sha256=_qhOqnYg2tFQa4-EuXyFdJvkc_-9x6ln9lx-MeqtNeY,20495
|
|
120
120
|
hwpx/tools/object_finder.py,sha256=ovXlbuPiQFdGzNSU08SuEzNFoeYcIb7GfF6vB9MRwBE,13800
|
|
121
121
|
hwpx/tools/package_reconcile.py,sha256=y1Hl7hbPh4YaV59LTdDLzQwgn4g1qEnFmSjmajnrEbA,2416
|
|
122
|
-
hwpx/tools/package_validator.py,sha256=
|
|
122
|
+
hwpx/tools/package_validator.py,sha256=pYNeIp59HIIYfb94prcD-si82-3kdbfkoIpPrI9V-aE,33559
|
|
123
123
|
hwpx/tools/page_guard.py,sha256=Qg9vHP46nnq9W5GCyaHbpNY5Ndt7ecJHAdSs8BfRggQ,9622
|
|
124
124
|
hwpx/tools/read_fidelity.py,sha256=f_a3Kftee3MRaCGraC6KjQDZ-iz3hv3b8-GGC5VCgyw,10528
|
|
125
125
|
hwpx/tools/recover.py,sha256=EOVAzMFAqR9YAT3sinZKCdjSkKygo4dKrs6T6SbGA7o,4963
|
|
126
126
|
hwpx/tools/redline.py,sha256=L9kMQ9i0c_fvqH3WMtu12Gk88WbTfZwxYPd0qH2-o58,7940
|
|
127
|
-
hwpx/tools/repair.py,sha256=
|
|
127
|
+
hwpx/tools/repair.py,sha256=JReRCUiPP44FG-1euzV56DC-RU5FnRxVZtHuGmMiVxM,11550
|
|
128
128
|
hwpx/tools/report_utils.py,sha256=6HYEeQc3ZxTpxbwF11s47uZ-KmV4tsHPE1MV4491KDE,4434
|
|
129
129
|
hwpx/tools/roundtrip_diff.py,sha256=ao0AdpDJkq89u5hwcrsxTijvSsia9Jaw1OOnh4WAco4,1365
|
|
130
130
|
hwpx/tools/table_cleanup.py,sha256=0_f6NnvNp3QD4owKd_bRX6FZbeUmoQC7a4_VGzF2SCE,1796
|
|
131
131
|
hwpx/tools/table_navigation.py,sha256=rtbrWFKpJhqC3LD0ZXImyHgjmDR2hjHCFy3_S-qNBwA,16479
|
|
132
132
|
hwpx/tools/template_analyzer.py,sha256=iH7egGRYNb8UesDB1TJ4tvW41-mlmv5Ap3VacqJFSlE,21355
|
|
133
133
|
hwpx/tools/text_extract_cli.py,sha256=BmsDAwNXpDPhEayb9ez2ORtGNzPd_Xxduy4_cLXhnUw,2188
|
|
134
|
-
hwpx/tools/text_extractor.py,sha256=
|
|
134
|
+
hwpx/tools/text_extractor.py,sha256=DPdDHhep_3ykLGmYS5czIBHR5DPut3FHHDuPe4FvvN4,26387
|
|
135
135
|
hwpx/tools/toc_author.py,sha256=eS1Zm5OBoonuD_H92JxcN2o276d5XmcJDqpfscd4wSg,15445
|
|
136
136
|
hwpx/tools/toc_fidelity.py,sha256=rvoKH8QJ65WNVrxS0d-i2AtODvBy-4O8NRudfQ244v4,19890
|
|
137
137
|
hwpx/tools/validator.py,sha256=U856izL9NcJZOiKDYoCpwOSaFNQ2Un8Jn4pzL20A96Q,7100
|
|
138
138
|
hwpx/tools/_schemas/header.xsd,sha256=mJXuFMuHGT1JnFFaluUpYUglwjMCNlfbFCRVM26eHXE,664
|
|
139
139
|
hwpx/tools/_schemas/section.xsd,sha256=MgvavVHG05RDfUnVPxVU10H4FQOja5ON04_m9Uk_m7E,522
|
|
140
|
-
python_hwpx-6.0.
|
|
141
|
-
python_hwpx-6.0.
|
|
142
|
-
python_hwpx-6.0.
|
|
143
|
-
python_hwpx-6.0.
|
|
144
|
-
python_hwpx-6.0.
|
|
145
|
-
python_hwpx-6.0.
|
|
146
|
-
python_hwpx-6.0.
|
|
140
|
+
python_hwpx-6.0.3.dist-info/licenses/LICENSE,sha256=_ubz4wv-BkkT3l3gu-QuH7JGeVjuRYGZoZK95eNsCHU,9688
|
|
141
|
+
python_hwpx-6.0.3.dist-info/licenses/NOTICE,sha256=he3PBLXnIZK0sNsozc-SAQ2jvWjV61goylWUZ8HPi3s,3837
|
|
142
|
+
python_hwpx-6.0.3.dist-info/METADATA,sha256=KP0TsJBAw0r2A0cIAuCASIwCTUmC7Bl_tbzowFTYUsY,13652
|
|
143
|
+
python_hwpx-6.0.3.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
144
|
+
python_hwpx-6.0.3.dist-info/entry_points.txt,sha256=JUKRxbly9UaeHV7YzOea23y8IiqSTcrhUlooP3fS_Zc,405
|
|
145
|
+
python_hwpx-6.0.3.dist-info/top_level.txt,sha256=R1iToqDh80Nf2oQhRjTN0rbN2X6kyDUizIocZjkhuxc,5
|
|
146
|
+
python_hwpx-6.0.3.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|