python-hwpx 6.4.0__py3-none-any.whl → 6.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hwpx/__init__.py +3 -0
- hwpx/_document/_legacy.py +3 -1
- hwpx/_document/fields.py +50 -2
- hwpx/_document/layout.py +289 -62
- hwpx/_document/media.py +217 -38
- hwpx/_document/memos.py +54 -9
- hwpx/_document/ns/fields.py +26 -1
- hwpx/_document/ns/media.py +23 -4
- hwpx/_document/ns/page.py +27 -4
- hwpx/_document/ns/parts.py +71 -2
- hwpx/_document/ns/shapes.py +6 -1
- hwpx/_document/ns/styles.py +157 -8
- hwpx/_document/ns/text.py +157 -20
- hwpx/_document/ns/tracking.py +5 -1
- hwpx/_document/persistence.py +133 -13
- hwpx/_document/shapes.py +81 -1
- hwpx/_document/tracked.py +49 -1
- hwpx/body_patch.py +6 -2
- hwpx/capabilities.py +1 -1
- hwpx/data/contract_docs/known-traps.md +37 -2
- hwpx/data/contract_docs/mutation-semantics.md +54 -6
- hwpx/data/contract_docs/support-matrix.md +7 -8
- hwpx/document.py +65 -5
- hwpx/equation/authoring.py +13 -10
- hwpx/equation/mathml.py +66 -0
- hwpx/equation/measure.py +371 -0
- hwpx/equation/render.py +2 -2
- hwpx/equation/tokens.py +11 -0
- hwpx/errors.py +49 -2
- hwpx/form_fit/__init__.py +4 -0
- hwpx/form_fit/apply.py +1 -1
- hwpx/form_fit/engine.py +10 -11
- hwpx/form_fit/measure.py +280 -22
- hwpx/hwp5/__init__.py +7 -0
- hwpx/hwp5/binary.py +124 -0
- hwpx/hwp5/bodytext.py +201 -0
- hwpx/hwp5/cfb.py +601 -0
- hwpx/hwp5/controls.py +1175 -0
- hwpx/hwp5/docinfo.py +881 -0
- hwpx/hwp5/docinfo_writer.py +647 -0
- hwpx/hwp5/errors.py +75 -0
- hwpx/hwp5/fileheader.py +87 -0
- hwpx/hwp5/header_xml.py +703 -0
- hwpx/hwp5/owpml.py +390 -0
- hwpx/hwp5/package.py +331 -0
- hwpx/hwp5/reader.py +149 -0
- hwpx/hwp5/records.py +214 -0
- hwpx/hwp5/section_common.py +149 -0
- hwpx/hwp5/section_writer.py +1826 -0
- hwpx/hwp5/section_xml.py +1514 -0
- hwpx/hwp5/shape_xml.py +709 -0
- hwpx/hwp5/shapes.py +842 -0
- hwpx/hwp5/summary.py +204 -0
- hwpx/hwp5/writer.py +325 -0
- hwpx/ingest/hwpx_converter.py +1 -1
- hwpx/layout/__init__.py +2 -0
- hwpx/layout/lint.py +147 -2
- hwpx/model.py +2 -0
- hwpx/objects/__init__.py +12 -1
- hwpx/objects/form_field.py +75 -1
- hwpx/objects/results.py +131 -0
- hwpx/opc/package.py +148 -16
- hwpx/opc/relationships.py +16 -3
- hwpx/opc/xml_utils.py +44 -15
- hwpx/oxml/_document_primitives.py +59 -34
- hwpx/oxml/_paragraph_text_edit.py +117 -0
- hwpx/oxml/color.py +36 -1
- hwpx/oxml/document_parts.py +125 -157
- hwpx/oxml/drop_cap.py +45 -0
- hwpx/oxml/header.py +6 -2
- hwpx/oxml/header_fonts.py +291 -0
- hwpx/oxml/header_part.py +23 -18
- hwpx/oxml/hyperlink_form.py +159 -0
- hwpx/oxml/master_page_authoring.py +66 -4
- hwpx/oxml/memo.py +4 -8
- hwpx/oxml/note_authoring.py +2 -3
- hwpx/oxml/numbering_kinds.py +48 -2
- hwpx/oxml/objects.py +71 -42
- hwpx/oxml/paragraph.py +96 -76
- hwpx/oxml/run.py +209 -44
- hwpx/oxml/section.py +283 -3
- hwpx/oxml/section_format.py +199 -13
- hwpx/oxml/section_layout.py +87 -0
- hwpx/oxml/section_story.py +253 -38
- hwpx/oxml/shape_position.py +128 -0
- hwpx/oxml/table.py +226 -75
- hwpx/oxml/table_merge.py +82 -0
- hwpx/oxml/table_sizes.py +334 -0
- hwpx/oxml/utils.py +121 -1
- hwpx/patch.py +16 -3
- hwpx/table_patch.py +233 -73
- hwpx/tools/_schemas/owpml-body.xsd +18 -0
- hwpx/tools/_schemas/owpml-core.xsd +905 -0
- hwpx/tools/_schemas/owpml-header.xsd +1869 -0
- hwpx/tools/_schemas/owpml-paralist.xsd +2813 -0
- hwpx/tools/_schemas/owpml-xml.xsd +15 -0
- hwpx/tools/document_merge.py +92 -15
- hwpx/tools/exporter.py +503 -59
- hwpx/tools/layout_preview.py +44 -2
- hwpx/tools/mail_merge.py +19 -13
- hwpx/tools/markdown_export.py +37 -16
- hwpx/tools/package_validator.py +17 -4
- hwpx/tools/read_fidelity.py +4 -4
- hwpx/tools/redline.py +4 -4
- hwpx/tools/table_navigation.py +3 -2
- hwpx/tools/template_analyzer.py +5 -1
- hwpx/tools/text_extractor.py +54 -24
- hwpx/tools/toc_author.py +110 -9
- hwpx/tools/validator.py +227 -1
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/METADATA +6 -4
- python_hwpx-6.6.0.dist-info/RECORD +195 -0
- python_hwpx-6.4.0.dist-info/RECORD +0 -162
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/WHEEL +0 -0
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/entry_points.txt +0 -0
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/licenses/LICENSE +0 -0
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/licenses/NOTICE +0 -0
- {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/top_level.txt +0 -0
hwpx/__init__.py
CHANGED
|
@@ -376,6 +376,7 @@ from .tools.package_validator import (
|
|
|
376
376
|
)
|
|
377
377
|
from .ingest import HwpxMarkdownConverter
|
|
378
378
|
from .errors import HwpxError
|
|
379
|
+
from .hwp5.errors import Hwp5ConversionWarning, Hwp5Error
|
|
379
380
|
from .mutation_report import (
|
|
380
381
|
MutationReport,
|
|
381
382
|
PreservationDowngradeError,
|
|
@@ -415,6 +416,8 @@ __all__ = [
|
|
|
415
416
|
"PackageValidationReport",
|
|
416
417
|
"BytePreservingPatchResult",
|
|
417
418
|
"HwpxError",
|
|
419
|
+
"Hwp5Error",
|
|
420
|
+
"Hwp5ConversionWarning",
|
|
418
421
|
"MutationReport",
|
|
419
422
|
"PreservationDowngradeError",
|
|
420
423
|
"ParagraphTextPatch",
|
hwpx/_document/_legacy.py
CHANGED
|
@@ -1241,7 +1241,9 @@ class _LegacyFacade:
|
|
|
1241
1241
|
"""Remove an embedded image by its manifest item id.
|
|
1242
1242
|
|
|
1243
1243
|
This removes the binary data from the ZIP, the manifest entry, and
|
|
1244
|
-
the header binItem entry.
|
|
1244
|
+
the header binItem entry. An item the document still points at is
|
|
1245
|
+
refused with ``HwpxValueError`` (code ``media-item-in-use``);
|
|
1246
|
+
``doc.media.remove_image(item_id, force=True)`` removes it anyway.
|
|
1245
1247
|
|
|
1246
1248
|
Returns:
|
|
1247
1249
|
``True`` if any component was removed.
|
hwpx/_document/fields.py
CHANGED
|
@@ -8,10 +8,11 @@ from typing import TYPE_CHECKING, Any, Iterator, Mapping, Sequence, cast
|
|
|
8
8
|
|
|
9
9
|
from ..errors import HwpxStateError, HwpxValueError
|
|
10
10
|
from ..objects.checkbox import CheckBox
|
|
11
|
-
from ..objects.form_field import FieldLocation, FieldParameter, FormField
|
|
11
|
+
from ..objects.form_field import CellField, FieldLocation, FieldParameter, FormField
|
|
12
12
|
from ..objects.results import FieldFillResult
|
|
13
13
|
from ..oxml import HwpxOxmlParagraph
|
|
14
14
|
from ..oxml.namespaces import HP
|
|
15
|
+
from ..oxml.section_story import control_twins
|
|
15
16
|
|
|
16
17
|
if TYPE_CHECKING:
|
|
17
18
|
from hwpx.document import HwpxDocument
|
|
@@ -291,11 +292,13 @@ def _iter_form_field_matches(doc: "HwpxDocument") -> list[dict[str, Any]]:
|
|
|
291
292
|
paragraph.element: index
|
|
292
293
|
for index, paragraph in enumerate(section.paragraphs)
|
|
293
294
|
}
|
|
295
|
+
# a header/footer python-hwpx keeps twice counts once: its hp:secPr copy
|
|
296
|
+
twins = control_twins(section.element)
|
|
294
297
|
|
|
295
298
|
def iter_content_paragraphs(element: Any) -> Iterator[Any]:
|
|
296
299
|
for child in element:
|
|
297
300
|
local = _local_name(child)
|
|
298
|
-
if local == "memogroup":
|
|
301
|
+
if local == "memogroup" or child in twins:
|
|
299
302
|
continue
|
|
300
303
|
if local == "p":
|
|
301
304
|
yield child
|
|
@@ -403,6 +406,42 @@ def list_form_fields(doc: "HwpxDocument") -> tuple[FormField, ...]:
|
|
|
403
406
|
return tuple(_form_field_from_match(doc, match) for match in _iter_form_field_matches(doc))
|
|
404
407
|
|
|
405
408
|
|
|
409
|
+
def _named_cells(paragraphs: Any) -> Iterator[Any]:
|
|
410
|
+
for paragraph in paragraphs:
|
|
411
|
+
for table in paragraph.tables:
|
|
412
|
+
for row in table.rows:
|
|
413
|
+
for cell in row.cells:
|
|
414
|
+
if cell.field_name:
|
|
415
|
+
yield cell
|
|
416
|
+
yield from _named_cells(cell.paragraphs)
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def list_cell_fields(doc: "HwpxDocument") -> tuple[CellField, ...]:
|
|
420
|
+
"""Named table cells (Hancom's cell fields) in document order: body tables and the tables in their cells."""
|
|
421
|
+
|
|
422
|
+
return tuple(CellField(cell) for section in doc.sections for cell in _named_cells(section.paragraphs))
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def fill_cell_fields(doc: "HwpxDocument", value: str, *, name: str, index: int | None = None) -> tuple[CellField, ...]:
|
|
426
|
+
"""Set the text of every cell field called *name*, or of the *index*-th of them only."""
|
|
427
|
+
|
|
428
|
+
wanted = (name or "").strip()
|
|
429
|
+
fields = [field for field in list_cell_fields(doc) if wanted and field.name == wanted]
|
|
430
|
+
if index is not None:
|
|
431
|
+
fields = fields[index : index + 1] if index >= 0 else []
|
|
432
|
+
if not fields:
|
|
433
|
+
where = f"{wanted!r}" if index is None else f"{wanted!r} at index {index}"
|
|
434
|
+
raise HwpxValueError(
|
|
435
|
+
f"no cell field named {where}",
|
|
436
|
+
code="field-cell-not-found",
|
|
437
|
+
context={"name": wanted, "index": index},
|
|
438
|
+
suggestion="List doc.fields.cells to see the cell field names.",
|
|
439
|
+
)
|
|
440
|
+
for field in fields:
|
|
441
|
+
field.text = value
|
|
442
|
+
return tuple(fields)
|
|
443
|
+
|
|
444
|
+
|
|
406
445
|
_PROMPT_TEXT_COLOR = "#FF0000"
|
|
407
446
|
|
|
408
447
|
|
|
@@ -823,7 +862,10 @@ def _measure_form_field_fit(
|
|
|
823
862
|
) -> "FitResult":
|
|
824
863
|
"""Run the FormFit engine for a native field (plan §2 C)."""
|
|
825
864
|
|
|
865
|
+
from dataclasses import replace
|
|
866
|
+
|
|
826
867
|
from hwpx.form_fit import DEFAULT_SAFETY, FitEngine, FitResult, SlotMetrics
|
|
868
|
+
from hwpx.form_fit.measure import text_style_from_refs
|
|
827
869
|
|
|
828
870
|
runs = match["_runs"]
|
|
829
871
|
begin_index = int(match["_begin_run_index"])
|
|
@@ -850,10 +892,16 @@ def _measure_form_field_fit(
|
|
|
850
892
|
field_id=field_id,
|
|
851
893
|
)
|
|
852
894
|
|
|
895
|
+
# The field sits inside its paragraph, so the paragraph's indent does not
|
|
896
|
+
# apply to the box.
|
|
897
|
+
text_style = text_style_from_refs(
|
|
898
|
+
doc._root, match["_paragraph"].para_pr_id_ref, [begin_ref]
|
|
899
|
+
)
|
|
853
900
|
slot = SlotMetrics(
|
|
854
901
|
available_width=float(box_width) * DEFAULT_SAFETY,
|
|
855
902
|
font_pt=resolved_pt,
|
|
856
903
|
max_lines=fit_policy.effective_max_lines,
|
|
904
|
+
text_style=replace(text_style, indent=0),
|
|
857
905
|
)
|
|
858
906
|
return FitEngine().fit(value, slot, fit_policy, field_id=field_id)
|
|
859
907
|
|
hwpx/_document/layout.py
CHANGED
|
@@ -5,7 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
from typing import TYPE_CHECKING, Any, Mapping, Sequence
|
|
7
7
|
|
|
8
|
-
from ..errors import HwpxStateError, HwpxValueError
|
|
8
|
+
from ..errors import HwpxStateError, HwpxTypeError, HwpxValueError
|
|
9
9
|
from ..objects.results import (
|
|
10
10
|
ColumnLayout,
|
|
11
11
|
ListFormatResult,
|
|
@@ -16,13 +16,14 @@ from ..objects.results import (
|
|
|
16
16
|
Units,
|
|
17
17
|
)
|
|
18
18
|
from ..oxml._document_primitives import NEW_NUM_KINDS
|
|
19
|
-
from ..oxml.namespaces import HH
|
|
19
|
+
from ..oxml.namespaces import HH, HP
|
|
20
|
+
from ..oxml.objects import HwpxOxmlInlineObject
|
|
21
|
+
from ..oxml.section_format import _PAGE_LANDSCAPE, _PAGE_PORTRAIT, _page_orientation_value
|
|
20
22
|
from ._units import _mm_to_hwp_units, _pt_to_hwp_units
|
|
21
23
|
|
|
22
24
|
if TYPE_CHECKING:
|
|
23
25
|
from hwpx.document import HwpxDocument
|
|
24
26
|
from ..oxml import (
|
|
25
|
-
HwpxOxmlInlineObject,
|
|
26
27
|
HwpxOxmlParagraph,
|
|
27
28
|
HwpxOxmlSection,
|
|
28
29
|
HwpxOxmlSectionHeaderFooter,
|
|
@@ -45,16 +46,9 @@ _PAPER_SIZES_MM: dict[str, tuple[float, float]] = {
|
|
|
45
46
|
def _normalize_page_orientation(value: str | None) -> str | None:
|
|
46
47
|
if value is None:
|
|
47
48
|
return None
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
"NARROW": "PORTRAIT",
|
|
52
|
-
"NARROWLY": "PORTRAIT",
|
|
53
|
-
"LANDSCAPE": "WIDELY",
|
|
54
|
-
"WIDE": "WIDELY",
|
|
55
|
-
"WIDELY": "WIDELY",
|
|
56
|
-
}
|
|
57
|
-
orientation = aliases.get(normalized)
|
|
49
|
+
# PORTRAIT/NARROW -> WIDELY and LANDSCAPE/WIDE -> NARROWLY; the stored
|
|
50
|
+
# values themselves keep Hancom's meaning (WIDELY is portrait).
|
|
51
|
+
orientation = _page_orientation_value(value)
|
|
58
52
|
if orientation is None:
|
|
59
53
|
raise HwpxValueError(
|
|
60
54
|
f"unsupported page orientation: {value}",
|
|
@@ -65,6 +59,81 @@ def _normalize_page_orientation(value: str | None) -> str | None:
|
|
|
65
59
|
return orientation
|
|
66
60
|
|
|
67
61
|
|
|
62
|
+
#: Keys of ``set_paragraph_format(border=...)``.
|
|
63
|
+
_PARAGRAPH_BORDER_KEYS = frozenset(
|
|
64
|
+
{"sides", "color", "width", "type", "connect", "offset_mm", "ignore_margin"}
|
|
65
|
+
)
|
|
66
|
+
_PARAGRAPH_BORDER_SIDES = ("left", "right", "top", "bottom")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _paragraph_border_problem(
|
|
70
|
+
spec: Mapping[str, Any], sides: tuple[str, ...], offsets: tuple[Any, ...]
|
|
71
|
+
) -> str | None:
|
|
72
|
+
unknown = sorted(set(spec) - _PARAGRAPH_BORDER_KEYS)
|
|
73
|
+
if unknown:
|
|
74
|
+
return f"unknown paragraph border keys: {unknown}"
|
|
75
|
+
if not sides or any(side not in _PARAGRAPH_BORDER_SIDES for side in sides):
|
|
76
|
+
return f"unsupported paragraph border sides: {list(sides)}"
|
|
77
|
+
if len(offsets) != 4 or any(
|
|
78
|
+
isinstance(value, bool) or not isinstance(value, (int, float)) or value < 0 for value in offsets
|
|
79
|
+
):
|
|
80
|
+
return "offset_mm must be a non-negative number or four of them (left, right, top, bottom)"
|
|
81
|
+
return None
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _paragraph_border_attrs(
|
|
85
|
+
header: Any,
|
|
86
|
+
border: Mapping[str, Any] | None,
|
|
87
|
+
*,
|
|
88
|
+
bottom_border: bool,
|
|
89
|
+
border_color: str,
|
|
90
|
+
border_width: str,
|
|
91
|
+
) -> dict[str, str] | None:
|
|
92
|
+
"""``hh:paraPr/hh:border`` attributes for ``border`` (or ``bottom_border``)."""
|
|
93
|
+
|
|
94
|
+
if border is None and not bottom_border:
|
|
95
|
+
return None
|
|
96
|
+
spec: Mapping[str, Any] = (
|
|
97
|
+
border if border is not None
|
|
98
|
+
else {"sides": ("bottom",), "color": border_color, "width": border_width}
|
|
99
|
+
)
|
|
100
|
+
raw_sides = spec.get("sides", _PARAGRAPH_BORDER_SIDES)
|
|
101
|
+
sides = tuple(str(side).strip().lower() for side in ((raw_sides,) if isinstance(raw_sides, str) else raw_sides))
|
|
102
|
+
raw_offsets = spec.get("offset_mm", 0)
|
|
103
|
+
offsets = tuple((raw_offsets,) * 4 if isinstance(raw_offsets, (int, float)) else raw_offsets)
|
|
104
|
+
problem = (
|
|
105
|
+
"pass either bottom_border or border, not both"
|
|
106
|
+
if border is not None and bottom_border
|
|
107
|
+
else _paragraph_border_problem(spec, sides, offsets)
|
|
108
|
+
)
|
|
109
|
+
if problem is not None:
|
|
110
|
+
raise HwpxValueError(
|
|
111
|
+
problem,
|
|
112
|
+
code="paragraph-border-invalid",
|
|
113
|
+
context={"border": {str(key): str(value) for key, value in spec.items()}},
|
|
114
|
+
suggestion=(
|
|
115
|
+
"border keys: sides, color, width, type, connect, offset_mm "
|
|
116
|
+
"(mm, one number or left/right/top/bottom), ignore_margin."
|
|
117
|
+
),
|
|
118
|
+
)
|
|
119
|
+
border_fill_id = header.ensure_border_fill(
|
|
120
|
+
border_color=str(spec.get("color", "#000000")),
|
|
121
|
+
border_width=str(spec.get("width", "0.12 mm")),
|
|
122
|
+
active_borders=sides,
|
|
123
|
+
border_type=str(spec.get("type", "SOLID")),
|
|
124
|
+
)
|
|
125
|
+
left, right, top, bottom = (str(_mm_to_hwp_units(float(value))) for value in offsets)
|
|
126
|
+
return {
|
|
127
|
+
"borderFillIDRef": border_fill_id,
|
|
128
|
+
"offsetLeft": left,
|
|
129
|
+
"offsetRight": right,
|
|
130
|
+
"offsetTop": top,
|
|
131
|
+
"offsetBottom": bottom,
|
|
132
|
+
"connect": "1" if spec.get("connect") else "0",
|
|
133
|
+
"ignoreMargin": "1" if spec.get("ignore_margin") else "0",
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
|
|
68
137
|
def _resolve_paragraph_targets(
|
|
69
138
|
doc: "HwpxDocument",
|
|
70
139
|
*,
|
|
@@ -104,11 +173,67 @@ def _resolve_paragraph_targets(
|
|
|
104
173
|
return targets
|
|
105
174
|
|
|
106
175
|
|
|
176
|
+
def _tree_root(element: Any) -> Any:
|
|
177
|
+
if hasattr(element, "getparent"):
|
|
178
|
+
while element.getparent() is not None:
|
|
179
|
+
element = element.getparent()
|
|
180
|
+
return element
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _resolve_paragraph_objects(
|
|
184
|
+
doc: "HwpxDocument", paragraphs: Sequence[HwpxOxmlParagraph]
|
|
185
|
+
) -> list[HwpxOxmlParagraph]:
|
|
186
|
+
"""Check that every paragraph object belongs to *doc*, before anything changes.
|
|
187
|
+
|
|
188
|
+
Body, table cell (nested too), header and footer paragraphs all live in a
|
|
189
|
+
section's XML tree, so a paragraph belongs to the document when its element
|
|
190
|
+
is inside one of the document's section elements.
|
|
191
|
+
"""
|
|
192
|
+
|
|
193
|
+
from ..oxml import HwpxOxmlParagraph
|
|
194
|
+
|
|
195
|
+
items = list(paragraphs)
|
|
196
|
+
if not items:
|
|
197
|
+
raise HwpxValueError(
|
|
198
|
+
"paragraphs 가 비어 있습니다.",
|
|
199
|
+
code="paragraph-indexes-empty",
|
|
200
|
+
suggestion="서식을 적용할 문단을 하나 이상 지정하세요.",
|
|
201
|
+
)
|
|
202
|
+
roots = [section.element for section in doc.sections]
|
|
203
|
+
resolved: list[HwpxOxmlParagraph] = []
|
|
204
|
+
for position, paragraph in enumerate(items):
|
|
205
|
+
if not isinstance(paragraph, HwpxOxmlParagraph):
|
|
206
|
+
raise HwpxTypeError(
|
|
207
|
+
f"paragraphs[{position}] 는 문단 객체가 아닙니다 — {type(paragraph).__name__}.",
|
|
208
|
+
code="paragraph-invalid-type",
|
|
209
|
+
context={"position": position, "type": type(paragraph).__name__},
|
|
210
|
+
suggestion="doc.paragraphs, cell.paragraphs, header.paragraphs 의 문단을 넘기세요.",
|
|
211
|
+
)
|
|
212
|
+
element = paragraph.element
|
|
213
|
+
root = _tree_root(element)
|
|
214
|
+
inside = (
|
|
215
|
+
any(root is section_root for section_root in roots)
|
|
216
|
+
if hasattr(element, "getparent")
|
|
217
|
+
else any(node is element for section_root in roots for node in section_root.iter())
|
|
218
|
+
)
|
|
219
|
+
if not inside:
|
|
220
|
+
raise HwpxValueError(
|
|
221
|
+
f"paragraphs[{position}] 는 이 문서에 속한 문단이 아닙니다.",
|
|
222
|
+
code="paragraph-not-in-document",
|
|
223
|
+
context={"position": position},
|
|
224
|
+
suggestion="이 문서에서 얻은 문단(본문·셀·머리말·꼬리말)을 넘기세요. 지운 문단이나 다른 문서의 문단은 받지 않습니다.",
|
|
225
|
+
)
|
|
226
|
+
if all(element is not seen.element for seen in resolved):
|
|
227
|
+
resolved.append(paragraph)
|
|
228
|
+
return resolved
|
|
229
|
+
|
|
230
|
+
|
|
107
231
|
def set_paragraph_format(
|
|
108
232
|
doc: "HwpxDocument",
|
|
109
233
|
*,
|
|
110
234
|
paragraph_index: int | None = None,
|
|
111
235
|
paragraph_indexes: Sequence[int] | None = None,
|
|
236
|
+
paragraphs: Sequence[HwpxOxmlParagraph] | None = None,
|
|
112
237
|
alignment: str | None = None,
|
|
113
238
|
line_spacing_percent: int | float | None = None,
|
|
114
239
|
indent_left_mm: float | None = None,
|
|
@@ -127,9 +252,16 @@ def set_paragraph_format(
|
|
|
127
252
|
tab_stops: Sequence[Mapping[str, Any]] | None = None,
|
|
128
253
|
auto_tab_left: bool | None = None,
|
|
129
254
|
auto_tab_right: bool | None = None,
|
|
255
|
+
border: Mapping[str, Any] | None = None,
|
|
130
256
|
) -> ParagraphFormatResult:
|
|
131
257
|
"""Apply paragraph-level formatting using human units.
|
|
132
258
|
|
|
259
|
+
Targets are body paragraphs by index (``paragraph_index`` /
|
|
260
|
+
``paragraph_indexes``; neither means every body paragraph) or paragraph
|
|
261
|
+
objects of this document (``paragraphs``): body, table cell (nested
|
|
262
|
+
tables too), header and footer paragraphs. The result lists body indexes
|
|
263
|
+
only; ``formatted`` counts every target.
|
|
264
|
+
|
|
133
265
|
Millimetre inputs are converted to HWP units; paragraph spacing uses
|
|
134
266
|
points; line spacing is stored as a percent value. ``keep_with_next`` /
|
|
135
267
|
``keep_lines`` / ``page_break_before`` set the paragraph's keep-together
|
|
@@ -145,6 +277,17 @@ def set_paragraph_format(
|
|
|
145
277
|
position-ascending. Passing ``tab_stops``/``auto_tab_left``/
|
|
146
278
|
``auto_tab_right`` mints (or reuses — dedupe) a ``hh:tabPr`` and wires
|
|
147
279
|
the paragraph's ``tabPrIDRef`` to it.
|
|
280
|
+
|
|
281
|
+
``border`` is a mapping for a paragraph border: ``sides`` (default all
|
|
282
|
+
four of ``"left"``/``"right"``/``"top"``/``"bottom"``), ``color``
|
|
283
|
+
(``"#000000"``), ``width`` (``"0.12 mm"``), ``type`` (``"SOLID"``),
|
|
284
|
+
``offset_mm`` (gap to the text in mm, one number or ``(left, right, top,
|
|
285
|
+
bottom)``, default 0), ``connect`` and ``ignore_margin`` (default
|
|
286
|
+
``False``). With
|
|
287
|
+
``connect=True`` Hancom draws consecutive paragraphs that share the
|
|
288
|
+
paragraph shape as one box, across columns and pages; give an empty
|
|
289
|
+
paragraph inside the box the same format so it does not split the box.
|
|
290
|
+
``bottom_border=True`` is the older bottom-only form.
|
|
148
291
|
"""
|
|
149
292
|
|
|
150
293
|
if not doc._root.headers:
|
|
@@ -207,6 +350,7 @@ def set_paragraph_format(
|
|
|
207
350
|
and not margins
|
|
208
351
|
and heading is None
|
|
209
352
|
and not bottom_border
|
|
353
|
+
and border is None
|
|
210
354
|
and not break_setting
|
|
211
355
|
and not wants_tab_definition
|
|
212
356
|
and column_break is None
|
|
@@ -217,6 +361,26 @@ def set_paragraph_format(
|
|
|
217
361
|
suggestion="Pass alignment, line_spacing_percent, or another option to change.",
|
|
218
362
|
)
|
|
219
363
|
|
|
364
|
+
# Resolve every target before the header gains tab or border definitions,
|
|
365
|
+
# so a bad target changes nothing.
|
|
366
|
+
targets: list[tuple[int | None, HwpxOxmlParagraph]]
|
|
367
|
+
if paragraphs is not None:
|
|
368
|
+
if paragraph_index is not None or paragraph_indexes is not None:
|
|
369
|
+
raise HwpxValueError(
|
|
370
|
+
"use either paragraphs or paragraph_index/paragraph_indexes, not both",
|
|
371
|
+
code="paragraph-argument-conflict",
|
|
372
|
+
suggestion="Pass only one.",
|
|
373
|
+
)
|
|
374
|
+
objects = _resolve_paragraph_objects(doc, paragraphs)
|
|
375
|
+
body = doc.paragraphs
|
|
376
|
+
body_index = {paragraph.element: index for index, paragraph in enumerate(body)}
|
|
377
|
+
targets = [(body_index.get(paragraph.element), paragraph) for paragraph in objects]
|
|
378
|
+
else:
|
|
379
|
+
targets = list(_resolve_paragraph_targets(doc,
|
|
380
|
+
paragraph_index=paragraph_index,
|
|
381
|
+
paragraph_indexes=paragraph_indexes,
|
|
382
|
+
))
|
|
383
|
+
|
|
220
384
|
tab_pr_id: str | None = None
|
|
221
385
|
if wants_tab_definition:
|
|
222
386
|
converted_stops: list[dict[str, object]] = []
|
|
@@ -239,22 +403,13 @@ def set_paragraph_format(
|
|
|
239
403
|
auto_tab_right=bool(auto_tab_right),
|
|
240
404
|
)
|
|
241
405
|
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
border = {
|
|
250
|
-
"borderFillIDRef": border_fill_id,
|
|
251
|
-
"offsetLeft": "0",
|
|
252
|
-
"offsetRight": "0",
|
|
253
|
-
"offsetTop": "0",
|
|
254
|
-
"offsetBottom": "0",
|
|
255
|
-
"connect": "0",
|
|
256
|
-
"ignoreMargin": "0",
|
|
257
|
-
}
|
|
406
|
+
border_attrs = _paragraph_border_attrs(
|
|
407
|
+
header,
|
|
408
|
+
border,
|
|
409
|
+
bottom_border=bottom_border,
|
|
410
|
+
border_color=border_color,
|
|
411
|
+
border_width=border_width,
|
|
412
|
+
)
|
|
258
413
|
|
|
259
414
|
# column_break bypasses paraPr entirely (it's hp:p's own attribute, not
|
|
260
415
|
# a shared style) -- only mint a new paraPr when one of the *other*
|
|
@@ -265,17 +420,12 @@ def set_paragraph_format(
|
|
|
265
420
|
or line_spacing_percent is not None
|
|
266
421
|
or bool(margins)
|
|
267
422
|
or heading is not None
|
|
268
|
-
or
|
|
423
|
+
or border_attrs is not None
|
|
269
424
|
or bool(break_setting)
|
|
270
425
|
or wants_tab_definition
|
|
271
426
|
)
|
|
272
427
|
|
|
273
|
-
|
|
274
|
-
paragraph_index=paragraph_index,
|
|
275
|
-
paragraph_indexes=paragraph_indexes,
|
|
276
|
-
)
|
|
277
|
-
formatted: list[int] = []
|
|
278
|
-
for index, paragraph in targets:
|
|
428
|
+
for paragraph in (paragraph for _, paragraph in targets):
|
|
279
429
|
if wants_para_pr_change:
|
|
280
430
|
para_pr_id = header.ensure_paragraph_format(
|
|
281
431
|
base_para_pr_id=paragraph.para_pr_id_ref,
|
|
@@ -283,18 +433,17 @@ def set_paragraph_format(
|
|
|
283
433
|
line_spacing_percent=line_spacing_percent,
|
|
284
434
|
margins=margins,
|
|
285
435
|
heading=heading,
|
|
286
|
-
border=
|
|
436
|
+
border=border_attrs,
|
|
287
437
|
break_setting=break_setting or None,
|
|
288
438
|
tab_pr_id_ref=tab_pr_id,
|
|
289
439
|
)
|
|
290
440
|
paragraph.para_pr_id_ref = para_pr_id
|
|
291
441
|
if column_break is not None:
|
|
292
442
|
paragraph.column_break = column_break
|
|
293
|
-
formatted.append(index)
|
|
294
443
|
|
|
295
444
|
return ParagraphFormatResult(
|
|
296
|
-
formatted=len(
|
|
297
|
-
paragraphs=tuple(
|
|
445
|
+
formatted=len(targets),
|
|
446
|
+
paragraphs=tuple(index for index, _ in targets if index is not None),
|
|
298
447
|
units=Units(indent="mm", paragraph_spacing="pt", line_spacing="%"),
|
|
299
448
|
)
|
|
300
449
|
|
|
@@ -396,7 +545,12 @@ def set_page_setup(
|
|
|
396
545
|
section: HwpxOxmlSection | None = None,
|
|
397
546
|
section_index: int | None = None,
|
|
398
547
|
) -> PageSetup:
|
|
399
|
-
"""Set page size, margins, orientation, and optional columns in human units.
|
|
548
|
+
"""Set page size, margins, orientation, and optional columns in human units.
|
|
549
|
+
|
|
550
|
+
The page is written as Hancom writes it: ``WIDELY`` for portrait and
|
|
551
|
+
``NARROWLY`` for landscape, both with the paper's portrait size. The
|
|
552
|
+
returned ``page_size`` reports the page as drawn (landscape is wider).
|
|
553
|
+
"""
|
|
400
554
|
|
|
401
555
|
normalized_orientation = _normalize_page_orientation(orientation)
|
|
402
556
|
target_width_mm = width_mm
|
|
@@ -415,10 +569,11 @@ def set_page_setup(
|
|
|
415
569
|
target_height_mm = paper_height if target_height_mm is None else target_height_mm
|
|
416
570
|
|
|
417
571
|
if target_width_mm is not None and target_height_mm is not None:
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
572
|
+
short_side, long_side = sorted((target_width_mm, target_height_mm))
|
|
573
|
+
if normalized_orientation == _PAGE_LANDSCAPE:
|
|
574
|
+
target_width_mm, target_height_mm = long_side, short_side
|
|
575
|
+
elif normalized_orientation == _PAGE_PORTRAIT:
|
|
576
|
+
target_width_mm, target_height_mm = short_side, long_side
|
|
422
577
|
|
|
423
578
|
width = _mm_to_hwp_units(float(target_width_mm)) if target_width_mm is not None else None
|
|
424
579
|
height = _mm_to_hwp_units(float(target_height_mm)) if target_height_mm is not None else None
|
|
@@ -510,10 +665,12 @@ def set_columns(
|
|
|
510
665
|
section: HwpxOxmlSection | None = None,
|
|
511
666
|
section_index: int | None = None,
|
|
512
667
|
) -> HwpxOxmlInlineObject:
|
|
513
|
-
"""
|
|
668
|
+
"""Set the columns of a section, or start new columns at a paragraph.
|
|
514
669
|
|
|
515
|
-
|
|
516
|
-
|
|
670
|
+
Without ``paragraph`` this rewrites the section's own column layout (the
|
|
671
|
+
``hp:colPr`` next to ``hp:secPr``) in place, so the whole section is laid
|
|
672
|
+
out in ``col_count`` columns. With ``paragraph`` it adds a column
|
|
673
|
+
definition control there, and the text from that paragraph on uses it.
|
|
517
674
|
|
|
518
675
|
Args:
|
|
519
676
|
col_count: Number of columns (1–255).
|
|
@@ -521,7 +678,28 @@ def set_columns(
|
|
|
521
678
|
same_gap: Gap in HWPUNIT (7200 = 1 inch).
|
|
522
679
|
separator_type: Optional column separator line type (e.g. ``SOLID``).
|
|
523
680
|
"""
|
|
681
|
+
if not 1 <= col_count <= 255:
|
|
682
|
+
raise HwpxValueError(
|
|
683
|
+
"col_count must be between 1 and 255",
|
|
684
|
+
code="page-columns-invalid",
|
|
685
|
+
context={"requested": col_count},
|
|
686
|
+
suggestion="Use columns=1 to remove columns.",
|
|
687
|
+
)
|
|
524
688
|
if paragraph is None:
|
|
689
|
+
target_section = _resolve_section(doc, section=section, section_index=section_index)
|
|
690
|
+
ctrl = target_section.properties.set_columns(
|
|
691
|
+
col_count,
|
|
692
|
+
col_type=col_type,
|
|
693
|
+
layout=layout,
|
|
694
|
+
same_size=same_size,
|
|
695
|
+
same_gap=same_gap,
|
|
696
|
+
column_widths=column_widths,
|
|
697
|
+
separator_type=separator_type,
|
|
698
|
+
separator_width=separator_width,
|
|
699
|
+
separator_color=separator_color,
|
|
700
|
+
)
|
|
701
|
+
if ctrl is not None:
|
|
702
|
+
return HwpxOxmlInlineObject(ctrl, target_section.paragraphs[0])
|
|
525
703
|
paragraph = doc.add_paragraph(
|
|
526
704
|
"", section=section, section_index=section_index,
|
|
527
705
|
include_run=False,
|
|
@@ -573,21 +751,12 @@ def add_hyperlink(
|
|
|
573
751
|
|
|
574
752
|
The display text follows the Hancom convention (blue ``#0000FF`` text
|
|
575
753
|
with a blue bottom underline — dominant styling across real-corpus
|
|
576
|
-
hyperlinks)
|
|
754
|
+
hyperlinks) on the character look of the paragraph it goes into, unless
|
|
755
|
+
``char_pr_id_ref`` overrides it. ``paragraph.add_hyperlink`` picks that
|
|
756
|
+
style, so a link looks the same whichever way it was added.
|
|
577
757
|
|
|
578
758
|
Returns the ``<hp:ctrl>`` wrapper containing the ``<hp:fieldBegin>``.
|
|
579
759
|
"""
|
|
580
|
-
if char_pr_id_ref is None:
|
|
581
|
-
# `doc._root.ensure_run_style` rather than `doc.ensure_run_style` —
|
|
582
|
-
# that facade name moved in 6.0 (design table row 52) and is a pure
|
|
583
|
-
# passthrough to `_root`, so this is byte-identical minus the
|
|
584
|
-
# DeprecationWarning it would otherwise fire on every hyperlink even
|
|
585
|
-
# when reached via the new `doc.refs.add_hyperlink` namespace path.
|
|
586
|
-
char_pr_id_ref = doc._root.ensure_run_style(
|
|
587
|
-
underline=True,
|
|
588
|
-
color="#0000FF",
|
|
589
|
-
underline_color="#0000FF",
|
|
590
|
-
)
|
|
591
760
|
if paragraph is None:
|
|
592
761
|
paragraph = doc.add_paragraph(
|
|
593
762
|
"", section=section, section_index=section_index,
|
|
@@ -631,7 +800,7 @@ def set_page_size(
|
|
|
631
800
|
target_section.properties.set_page_size(
|
|
632
801
|
width=width,
|
|
633
802
|
height=height,
|
|
634
|
-
orientation=orientation,
|
|
803
|
+
orientation=_normalize_page_orientation(orientation),
|
|
635
804
|
gutter_type=gutter_type,
|
|
636
805
|
)
|
|
637
806
|
|
|
@@ -868,10 +1037,14 @@ def hide_page_elements(
|
|
|
868
1037
|
fill: bool = False,
|
|
869
1038
|
page_num: bool = False,
|
|
870
1039
|
) -> "HwpxOxmlInlineObject":
|
|
871
|
-
"""Hide the named page elements
|
|
1040
|
+
"""Hide the named page elements on *paragraph*'s page only.
|
|
872
1041
|
|
|
873
1042
|
Inserts ``<hp:ctrl><hp:pageHiding .../></hp:ctrl>`` (``ParaList XML
|
|
874
1043
|
schema.xml:148-163`` — six independent booleans, all default unhidden).
|
|
1044
|
+
Hancom applies it to that page alone (its "hide on the current page
|
|
1045
|
+
only"); the next page shows the elements again. *page_num* hides
|
|
1046
|
+
Hancom's page-number control, not the header/footer number that
|
|
1047
|
+
``set_page_number`` writes -- hide that one with *footer* (or *header*).
|
|
875
1048
|
"""
|
|
876
1049
|
|
|
877
1050
|
return paragraph.add_page_hiding(
|
|
@@ -920,3 +1093,57 @@ def remove_footer(
|
|
|
920
1093
|
return
|
|
921
1094
|
target_section = doc._root.sections[-1]
|
|
922
1095
|
target_section.properties.remove_footer(page_type=page_type)
|
|
1096
|
+
|
|
1097
|
+
|
|
1098
|
+
def flow_table_taller_than_page(doc: "HwpxDocument", table: Any) -> None:
|
|
1099
|
+
"""Let a new body *table* flow across pages when its rows alone outgrow a page.
|
|
1100
|
+
|
|
1101
|
+
Hancom never breaks a table laid out as a character (``treatAsChar``, the
|
|
1102
|
+
``add_table`` default) across pages: one taller than the page body is cut
|
|
1103
|
+
off at the paper's edge. Such a table becomes a flowing one instead
|
|
1104
|
+
(``Table.set_treat_as_char(False)``), which Hancom breaks between rows.
|
|
1105
|
+
"""
|
|
1106
|
+
|
|
1107
|
+
properties = table.paragraph.section.properties
|
|
1108
|
+
size, margins = properties.page_size, properties.page_margins
|
|
1109
|
+
body = size.drawn_height - margins.top - margins.bottom - margins.header - margins.footer
|
|
1110
|
+
if body > 0 and _table_min_height(doc, table.element) > body:
|
|
1111
|
+
table.set_treat_as_char(False)
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
def _table_min_height(doc: "HwpxDocument", table: Any) -> int:
|
|
1115
|
+
"""A lower bound of the drawn height: every row is at least its tallest
|
|
1116
|
+
single-row cell, and a cell at least one line of its text plus its top and
|
|
1117
|
+
bottom margins."""
|
|
1118
|
+
|
|
1119
|
+
total = 0
|
|
1120
|
+
for row in table.findall(f"{HP}tr"):
|
|
1121
|
+
tallest = 0
|
|
1122
|
+
for cell in row.findall(f"{HP}tc"):
|
|
1123
|
+
span = cell.find(f"{HP}cellSpan")
|
|
1124
|
+
if span is not None and span.get("rowSpan", "1") != "1":
|
|
1125
|
+
continue
|
|
1126
|
+
run = cell.find(f".//{HP}run")
|
|
1127
|
+
line = _char_height(doc, run.get("charPrIDRef") if run is not None else None)
|
|
1128
|
+
margin = cell.find(f"{HP}cellMargin")
|
|
1129
|
+
padding = _int_attr(margin, "top") + _int_attr(margin, "bottom")
|
|
1130
|
+
tallest = max(tallest, _int_attr(cell.find(f"{HP}cellSz"), "height"), line + padding)
|
|
1131
|
+
total += tallest
|
|
1132
|
+
return total
|
|
1133
|
+
|
|
1134
|
+
|
|
1135
|
+
def _char_height(doc: "HwpxDocument", char_pr_id_ref: str | None) -> int:
|
|
1136
|
+
style = doc._root.char_property(char_pr_id_ref if char_pr_id_ref is not None else "0")
|
|
1137
|
+
try:
|
|
1138
|
+
return int(style.attributes.get("height", "1000")) if style is not None else 1000
|
|
1139
|
+
except ValueError:
|
|
1140
|
+
return 1000
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
def _int_attr(element: Any, name: str) -> int:
|
|
1144
|
+
if element is None:
|
|
1145
|
+
return 0
|
|
1146
|
+
try:
|
|
1147
|
+
return int(element.get(name, "0"))
|
|
1148
|
+
except ValueError:
|
|
1149
|
+
return 0
|