python-hwpx 6.4.0__py3-none-any.whl → 6.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. hwpx/__init__.py +3 -0
  2. hwpx/_document/_legacy.py +3 -1
  3. hwpx/_document/fields.py +50 -2
  4. hwpx/_document/layout.py +289 -62
  5. hwpx/_document/media.py +217 -38
  6. hwpx/_document/memos.py +54 -9
  7. hwpx/_document/ns/fields.py +26 -1
  8. hwpx/_document/ns/media.py +23 -4
  9. hwpx/_document/ns/page.py +27 -4
  10. hwpx/_document/ns/parts.py +71 -2
  11. hwpx/_document/ns/shapes.py +6 -1
  12. hwpx/_document/ns/styles.py +157 -8
  13. hwpx/_document/ns/text.py +157 -20
  14. hwpx/_document/ns/tracking.py +5 -1
  15. hwpx/_document/persistence.py +133 -13
  16. hwpx/_document/shapes.py +81 -1
  17. hwpx/_document/tracked.py +49 -1
  18. hwpx/body_patch.py +6 -2
  19. hwpx/capabilities.py +1 -1
  20. hwpx/data/contract_docs/known-traps.md +37 -2
  21. hwpx/data/contract_docs/mutation-semantics.md +54 -6
  22. hwpx/data/contract_docs/support-matrix.md +7 -8
  23. hwpx/document.py +65 -5
  24. hwpx/equation/authoring.py +13 -10
  25. hwpx/equation/mathml.py +66 -0
  26. hwpx/equation/measure.py +371 -0
  27. hwpx/equation/render.py +2 -2
  28. hwpx/equation/tokens.py +11 -0
  29. hwpx/errors.py +49 -2
  30. hwpx/form_fit/__init__.py +4 -0
  31. hwpx/form_fit/apply.py +1 -1
  32. hwpx/form_fit/engine.py +10 -11
  33. hwpx/form_fit/measure.py +280 -22
  34. hwpx/hwp5/__init__.py +7 -0
  35. hwpx/hwp5/binary.py +124 -0
  36. hwpx/hwp5/bodytext.py +201 -0
  37. hwpx/hwp5/cfb.py +601 -0
  38. hwpx/hwp5/controls.py +1175 -0
  39. hwpx/hwp5/docinfo.py +881 -0
  40. hwpx/hwp5/docinfo_writer.py +647 -0
  41. hwpx/hwp5/errors.py +75 -0
  42. hwpx/hwp5/fileheader.py +87 -0
  43. hwpx/hwp5/header_xml.py +703 -0
  44. hwpx/hwp5/owpml.py +390 -0
  45. hwpx/hwp5/package.py +331 -0
  46. hwpx/hwp5/reader.py +149 -0
  47. hwpx/hwp5/records.py +214 -0
  48. hwpx/hwp5/section_common.py +149 -0
  49. hwpx/hwp5/section_writer.py +1826 -0
  50. hwpx/hwp5/section_xml.py +1514 -0
  51. hwpx/hwp5/shape_xml.py +709 -0
  52. hwpx/hwp5/shapes.py +842 -0
  53. hwpx/hwp5/summary.py +204 -0
  54. hwpx/hwp5/writer.py +325 -0
  55. hwpx/ingest/hwpx_converter.py +1 -1
  56. hwpx/layout/__init__.py +2 -0
  57. hwpx/layout/lint.py +147 -2
  58. hwpx/model.py +2 -0
  59. hwpx/objects/__init__.py +12 -1
  60. hwpx/objects/form_field.py +75 -1
  61. hwpx/objects/results.py +131 -0
  62. hwpx/opc/package.py +148 -16
  63. hwpx/opc/relationships.py +16 -3
  64. hwpx/opc/xml_utils.py +44 -15
  65. hwpx/oxml/_document_primitives.py +59 -34
  66. hwpx/oxml/_paragraph_text_edit.py +117 -0
  67. hwpx/oxml/color.py +36 -1
  68. hwpx/oxml/document_parts.py +125 -157
  69. hwpx/oxml/drop_cap.py +45 -0
  70. hwpx/oxml/header.py +6 -2
  71. hwpx/oxml/header_fonts.py +291 -0
  72. hwpx/oxml/header_part.py +23 -18
  73. hwpx/oxml/hyperlink_form.py +159 -0
  74. hwpx/oxml/master_page_authoring.py +66 -4
  75. hwpx/oxml/memo.py +4 -8
  76. hwpx/oxml/note_authoring.py +2 -3
  77. hwpx/oxml/numbering_kinds.py +48 -2
  78. hwpx/oxml/objects.py +71 -42
  79. hwpx/oxml/paragraph.py +96 -76
  80. hwpx/oxml/run.py +209 -44
  81. hwpx/oxml/section.py +283 -3
  82. hwpx/oxml/section_format.py +199 -13
  83. hwpx/oxml/section_layout.py +87 -0
  84. hwpx/oxml/section_story.py +253 -38
  85. hwpx/oxml/shape_position.py +128 -0
  86. hwpx/oxml/table.py +226 -75
  87. hwpx/oxml/table_merge.py +82 -0
  88. hwpx/oxml/table_sizes.py +334 -0
  89. hwpx/oxml/utils.py +121 -1
  90. hwpx/patch.py +16 -3
  91. hwpx/table_patch.py +233 -73
  92. hwpx/tools/_schemas/owpml-body.xsd +18 -0
  93. hwpx/tools/_schemas/owpml-core.xsd +905 -0
  94. hwpx/tools/_schemas/owpml-header.xsd +1869 -0
  95. hwpx/tools/_schemas/owpml-paralist.xsd +2813 -0
  96. hwpx/tools/_schemas/owpml-xml.xsd +15 -0
  97. hwpx/tools/document_merge.py +92 -15
  98. hwpx/tools/exporter.py +503 -59
  99. hwpx/tools/layout_preview.py +44 -2
  100. hwpx/tools/mail_merge.py +19 -13
  101. hwpx/tools/markdown_export.py +37 -16
  102. hwpx/tools/package_validator.py +17 -4
  103. hwpx/tools/read_fidelity.py +4 -4
  104. hwpx/tools/redline.py +4 -4
  105. hwpx/tools/table_navigation.py +3 -2
  106. hwpx/tools/template_analyzer.py +5 -1
  107. hwpx/tools/text_extractor.py +54 -24
  108. hwpx/tools/toc_author.py +110 -9
  109. hwpx/tools/validator.py +227 -1
  110. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/METADATA +6 -4
  111. python_hwpx-6.6.0.dist-info/RECORD +195 -0
  112. python_hwpx-6.4.0.dist-info/RECORD +0 -162
  113. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/WHEEL +0 -0
  114. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/entry_points.txt +0 -0
  115. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/licenses/LICENSE +0 -0
  116. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/licenses/NOTICE +0 -0
  117. {python_hwpx-6.4.0.dist-info → python_hwpx-6.6.0.dist-info}/top_level.txt +0 -0
hwpx/__init__.py CHANGED
@@ -376,6 +376,7 @@ from .tools.package_validator import (
376
376
  )
377
377
  from .ingest import HwpxMarkdownConverter
378
378
  from .errors import HwpxError
379
+ from .hwp5.errors import Hwp5ConversionWarning, Hwp5Error
379
380
  from .mutation_report import (
380
381
  MutationReport,
381
382
  PreservationDowngradeError,
@@ -415,6 +416,8 @@ __all__ = [
415
416
  "PackageValidationReport",
416
417
  "BytePreservingPatchResult",
417
418
  "HwpxError",
419
+ "Hwp5Error",
420
+ "Hwp5ConversionWarning",
418
421
  "MutationReport",
419
422
  "PreservationDowngradeError",
420
423
  "ParagraphTextPatch",
hwpx/_document/_legacy.py CHANGED
@@ -1241,7 +1241,9 @@ class _LegacyFacade:
1241
1241
  """Remove an embedded image by its manifest item id.
1242
1242
 
1243
1243
  This removes the binary data from the ZIP, the manifest entry, and
1244
- the header binItem entry.
1244
+ the header binItem entry. An item the document still points at is
1245
+ refused with ``HwpxValueError`` (code ``media-item-in-use``);
1246
+ ``doc.media.remove_image(item_id, force=True)`` removes it anyway.
1245
1247
 
1246
1248
  Returns:
1247
1249
  ``True`` if any component was removed.
hwpx/_document/fields.py CHANGED
@@ -8,10 +8,11 @@ from typing import TYPE_CHECKING, Any, Iterator, Mapping, Sequence, cast
8
8
 
9
9
  from ..errors import HwpxStateError, HwpxValueError
10
10
  from ..objects.checkbox import CheckBox
11
- from ..objects.form_field import FieldLocation, FieldParameter, FormField
11
+ from ..objects.form_field import CellField, FieldLocation, FieldParameter, FormField
12
12
  from ..objects.results import FieldFillResult
13
13
  from ..oxml import HwpxOxmlParagraph
14
14
  from ..oxml.namespaces import HP
15
+ from ..oxml.section_story import control_twins
15
16
 
16
17
  if TYPE_CHECKING:
17
18
  from hwpx.document import HwpxDocument
@@ -291,11 +292,13 @@ def _iter_form_field_matches(doc: "HwpxDocument") -> list[dict[str, Any]]:
291
292
  paragraph.element: index
292
293
  for index, paragraph in enumerate(section.paragraphs)
293
294
  }
295
+ # a header/footer python-hwpx keeps twice counts once: its hp:secPr copy
296
+ twins = control_twins(section.element)
294
297
 
295
298
  def iter_content_paragraphs(element: Any) -> Iterator[Any]:
296
299
  for child in element:
297
300
  local = _local_name(child)
298
- if local == "memogroup":
301
+ if local == "memogroup" or child in twins:
299
302
  continue
300
303
  if local == "p":
301
304
  yield child
@@ -403,6 +406,42 @@ def list_form_fields(doc: "HwpxDocument") -> tuple[FormField, ...]:
403
406
  return tuple(_form_field_from_match(doc, match) for match in _iter_form_field_matches(doc))
404
407
 
405
408
 
409
+ def _named_cells(paragraphs: Any) -> Iterator[Any]:
410
+ for paragraph in paragraphs:
411
+ for table in paragraph.tables:
412
+ for row in table.rows:
413
+ for cell in row.cells:
414
+ if cell.field_name:
415
+ yield cell
416
+ yield from _named_cells(cell.paragraphs)
417
+
418
+
419
+ def list_cell_fields(doc: "HwpxDocument") -> tuple[CellField, ...]:
420
+ """Named table cells (Hancom's cell fields) in document order: body tables and the tables in their cells."""
421
+
422
+ return tuple(CellField(cell) for section in doc.sections for cell in _named_cells(section.paragraphs))
423
+
424
+
425
+ def fill_cell_fields(doc: "HwpxDocument", value: str, *, name: str, index: int | None = None) -> tuple[CellField, ...]:
426
+ """Set the text of every cell field called *name*, or of the *index*-th of them only."""
427
+
428
+ wanted = (name or "").strip()
429
+ fields = [field for field in list_cell_fields(doc) if wanted and field.name == wanted]
430
+ if index is not None:
431
+ fields = fields[index : index + 1] if index >= 0 else []
432
+ if not fields:
433
+ where = f"{wanted!r}" if index is None else f"{wanted!r} at index {index}"
434
+ raise HwpxValueError(
435
+ f"no cell field named {where}",
436
+ code="field-cell-not-found",
437
+ context={"name": wanted, "index": index},
438
+ suggestion="List doc.fields.cells to see the cell field names.",
439
+ )
440
+ for field in fields:
441
+ field.text = value
442
+ return tuple(fields)
443
+
444
+
406
445
  _PROMPT_TEXT_COLOR = "#FF0000"
407
446
 
408
447
 
@@ -823,7 +862,10 @@ def _measure_form_field_fit(
823
862
  ) -> "FitResult":
824
863
  """Run the FormFit engine for a native field (plan §2 C)."""
825
864
 
865
+ from dataclasses import replace
866
+
826
867
  from hwpx.form_fit import DEFAULT_SAFETY, FitEngine, FitResult, SlotMetrics
868
+ from hwpx.form_fit.measure import text_style_from_refs
827
869
 
828
870
  runs = match["_runs"]
829
871
  begin_index = int(match["_begin_run_index"])
@@ -850,10 +892,16 @@ def _measure_form_field_fit(
850
892
  field_id=field_id,
851
893
  )
852
894
 
895
+ # The field sits inside its paragraph, so the paragraph's indent does not
896
+ # apply to the box.
897
+ text_style = text_style_from_refs(
898
+ doc._root, match["_paragraph"].para_pr_id_ref, [begin_ref]
899
+ )
853
900
  slot = SlotMetrics(
854
901
  available_width=float(box_width) * DEFAULT_SAFETY,
855
902
  font_pt=resolved_pt,
856
903
  max_lines=fit_policy.effective_max_lines,
904
+ text_style=replace(text_style, indent=0),
857
905
  )
858
906
  return FitEngine().fit(value, slot, fit_policy, field_id=field_id)
859
907
 
hwpx/_document/layout.py CHANGED
@@ -5,7 +5,7 @@ from __future__ import annotations
5
5
 
6
6
  from typing import TYPE_CHECKING, Any, Mapping, Sequence
7
7
 
8
- from ..errors import HwpxStateError, HwpxValueError
8
+ from ..errors import HwpxStateError, HwpxTypeError, HwpxValueError
9
9
  from ..objects.results import (
10
10
  ColumnLayout,
11
11
  ListFormatResult,
@@ -16,13 +16,14 @@ from ..objects.results import (
16
16
  Units,
17
17
  )
18
18
  from ..oxml._document_primitives import NEW_NUM_KINDS
19
- from ..oxml.namespaces import HH
19
+ from ..oxml.namespaces import HH, HP
20
+ from ..oxml.objects import HwpxOxmlInlineObject
21
+ from ..oxml.section_format import _PAGE_LANDSCAPE, _PAGE_PORTRAIT, _page_orientation_value
20
22
  from ._units import _mm_to_hwp_units, _pt_to_hwp_units
21
23
 
22
24
  if TYPE_CHECKING:
23
25
  from hwpx.document import HwpxDocument
24
26
  from ..oxml import (
25
- HwpxOxmlInlineObject,
26
27
  HwpxOxmlParagraph,
27
28
  HwpxOxmlSection,
28
29
  HwpxOxmlSectionHeaderFooter,
@@ -45,16 +46,9 @@ _PAPER_SIZES_MM: dict[str, tuple[float, float]] = {
45
46
  def _normalize_page_orientation(value: str | None) -> str | None:
46
47
  if value is None:
47
48
  return None
48
- normalized = value.strip().upper()
49
- aliases = {
50
- "PORTRAIT": "PORTRAIT",
51
- "NARROW": "PORTRAIT",
52
- "NARROWLY": "PORTRAIT",
53
- "LANDSCAPE": "WIDELY",
54
- "WIDE": "WIDELY",
55
- "WIDELY": "WIDELY",
56
- }
57
- orientation = aliases.get(normalized)
49
+ # PORTRAIT/NARROW -> WIDELY and LANDSCAPE/WIDE -> NARROWLY; the stored
50
+ # values themselves keep Hancom's meaning (WIDELY is portrait).
51
+ orientation = _page_orientation_value(value)
58
52
  if orientation is None:
59
53
  raise HwpxValueError(
60
54
  f"unsupported page orientation: {value}",
@@ -65,6 +59,81 @@ def _normalize_page_orientation(value: str | None) -> str | None:
65
59
  return orientation
66
60
 
67
61
 
62
+ #: Keys of ``set_paragraph_format(border=...)``.
63
+ _PARAGRAPH_BORDER_KEYS = frozenset(
64
+ {"sides", "color", "width", "type", "connect", "offset_mm", "ignore_margin"}
65
+ )
66
+ _PARAGRAPH_BORDER_SIDES = ("left", "right", "top", "bottom")
67
+
68
+
69
+ def _paragraph_border_problem(
70
+ spec: Mapping[str, Any], sides: tuple[str, ...], offsets: tuple[Any, ...]
71
+ ) -> str | None:
72
+ unknown = sorted(set(spec) - _PARAGRAPH_BORDER_KEYS)
73
+ if unknown:
74
+ return f"unknown paragraph border keys: {unknown}"
75
+ if not sides or any(side not in _PARAGRAPH_BORDER_SIDES for side in sides):
76
+ return f"unsupported paragraph border sides: {list(sides)}"
77
+ if len(offsets) != 4 or any(
78
+ isinstance(value, bool) or not isinstance(value, (int, float)) or value < 0 for value in offsets
79
+ ):
80
+ return "offset_mm must be a non-negative number or four of them (left, right, top, bottom)"
81
+ return None
82
+
83
+
84
+ def _paragraph_border_attrs(
85
+ header: Any,
86
+ border: Mapping[str, Any] | None,
87
+ *,
88
+ bottom_border: bool,
89
+ border_color: str,
90
+ border_width: str,
91
+ ) -> dict[str, str] | None:
92
+ """``hh:paraPr/hh:border`` attributes for ``border`` (or ``bottom_border``)."""
93
+
94
+ if border is None and not bottom_border:
95
+ return None
96
+ spec: Mapping[str, Any] = (
97
+ border if border is not None
98
+ else {"sides": ("bottom",), "color": border_color, "width": border_width}
99
+ )
100
+ raw_sides = spec.get("sides", _PARAGRAPH_BORDER_SIDES)
101
+ sides = tuple(str(side).strip().lower() for side in ((raw_sides,) if isinstance(raw_sides, str) else raw_sides))
102
+ raw_offsets = spec.get("offset_mm", 0)
103
+ offsets = tuple((raw_offsets,) * 4 if isinstance(raw_offsets, (int, float)) else raw_offsets)
104
+ problem = (
105
+ "pass either bottom_border or border, not both"
106
+ if border is not None and bottom_border
107
+ else _paragraph_border_problem(spec, sides, offsets)
108
+ )
109
+ if problem is not None:
110
+ raise HwpxValueError(
111
+ problem,
112
+ code="paragraph-border-invalid",
113
+ context={"border": {str(key): str(value) for key, value in spec.items()}},
114
+ suggestion=(
115
+ "border keys: sides, color, width, type, connect, offset_mm "
116
+ "(mm, one number or left/right/top/bottom), ignore_margin."
117
+ ),
118
+ )
119
+ border_fill_id = header.ensure_border_fill(
120
+ border_color=str(spec.get("color", "#000000")),
121
+ border_width=str(spec.get("width", "0.12 mm")),
122
+ active_borders=sides,
123
+ border_type=str(spec.get("type", "SOLID")),
124
+ )
125
+ left, right, top, bottom = (str(_mm_to_hwp_units(float(value))) for value in offsets)
126
+ return {
127
+ "borderFillIDRef": border_fill_id,
128
+ "offsetLeft": left,
129
+ "offsetRight": right,
130
+ "offsetTop": top,
131
+ "offsetBottom": bottom,
132
+ "connect": "1" if spec.get("connect") else "0",
133
+ "ignoreMargin": "1" if spec.get("ignore_margin") else "0",
134
+ }
135
+
136
+
68
137
  def _resolve_paragraph_targets(
69
138
  doc: "HwpxDocument",
70
139
  *,
@@ -104,11 +173,67 @@ def _resolve_paragraph_targets(
104
173
  return targets
105
174
 
106
175
 
176
+ def _tree_root(element: Any) -> Any:
177
+ if hasattr(element, "getparent"):
178
+ while element.getparent() is not None:
179
+ element = element.getparent()
180
+ return element
181
+
182
+
183
+ def _resolve_paragraph_objects(
184
+ doc: "HwpxDocument", paragraphs: Sequence[HwpxOxmlParagraph]
185
+ ) -> list[HwpxOxmlParagraph]:
186
+ """Check that every paragraph object belongs to *doc*, before anything changes.
187
+
188
+ Body, table cell (nested too), header and footer paragraphs all live in a
189
+ section's XML tree, so a paragraph belongs to the document when its element
190
+ is inside one of the document's section elements.
191
+ """
192
+
193
+ from ..oxml import HwpxOxmlParagraph
194
+
195
+ items = list(paragraphs)
196
+ if not items:
197
+ raise HwpxValueError(
198
+ "paragraphs 가 비어 있습니다.",
199
+ code="paragraph-indexes-empty",
200
+ suggestion="서식을 적용할 문단을 하나 이상 지정하세요.",
201
+ )
202
+ roots = [section.element for section in doc.sections]
203
+ resolved: list[HwpxOxmlParagraph] = []
204
+ for position, paragraph in enumerate(items):
205
+ if not isinstance(paragraph, HwpxOxmlParagraph):
206
+ raise HwpxTypeError(
207
+ f"paragraphs[{position}] 는 문단 객체가 아닙니다 — {type(paragraph).__name__}.",
208
+ code="paragraph-invalid-type",
209
+ context={"position": position, "type": type(paragraph).__name__},
210
+ suggestion="doc.paragraphs, cell.paragraphs, header.paragraphs 의 문단을 넘기세요.",
211
+ )
212
+ element = paragraph.element
213
+ root = _tree_root(element)
214
+ inside = (
215
+ any(root is section_root for section_root in roots)
216
+ if hasattr(element, "getparent")
217
+ else any(node is element for section_root in roots for node in section_root.iter())
218
+ )
219
+ if not inside:
220
+ raise HwpxValueError(
221
+ f"paragraphs[{position}] 는 이 문서에 속한 문단이 아닙니다.",
222
+ code="paragraph-not-in-document",
223
+ context={"position": position},
224
+ suggestion="이 문서에서 얻은 문단(본문·셀·머리말·꼬리말)을 넘기세요. 지운 문단이나 다른 문서의 문단은 받지 않습니다.",
225
+ )
226
+ if all(element is not seen.element for seen in resolved):
227
+ resolved.append(paragraph)
228
+ return resolved
229
+
230
+
107
231
  def set_paragraph_format(
108
232
  doc: "HwpxDocument",
109
233
  *,
110
234
  paragraph_index: int | None = None,
111
235
  paragraph_indexes: Sequence[int] | None = None,
236
+ paragraphs: Sequence[HwpxOxmlParagraph] | None = None,
112
237
  alignment: str | None = None,
113
238
  line_spacing_percent: int | float | None = None,
114
239
  indent_left_mm: float | None = None,
@@ -127,9 +252,16 @@ def set_paragraph_format(
127
252
  tab_stops: Sequence[Mapping[str, Any]] | None = None,
128
253
  auto_tab_left: bool | None = None,
129
254
  auto_tab_right: bool | None = None,
255
+ border: Mapping[str, Any] | None = None,
130
256
  ) -> ParagraphFormatResult:
131
257
  """Apply paragraph-level formatting using human units.
132
258
 
259
+ Targets are body paragraphs by index (``paragraph_index`` /
260
+ ``paragraph_indexes``; neither means every body paragraph) or paragraph
261
+ objects of this document (``paragraphs``): body, table cell (nested
262
+ tables too), header and footer paragraphs. The result lists body indexes
263
+ only; ``formatted`` counts every target.
264
+
133
265
  Millimetre inputs are converted to HWP units; paragraph spacing uses
134
266
  points; line spacing is stored as a percent value. ``keep_with_next`` /
135
267
  ``keep_lines`` / ``page_break_before`` set the paragraph's keep-together
@@ -145,6 +277,17 @@ def set_paragraph_format(
145
277
  position-ascending. Passing ``tab_stops``/``auto_tab_left``/
146
278
  ``auto_tab_right`` mints (or reuses — dedupe) a ``hh:tabPr`` and wires
147
279
  the paragraph's ``tabPrIDRef`` to it.
280
+
281
+ ``border`` is a mapping for a paragraph border: ``sides`` (default all
282
+ four of ``"left"``/``"right"``/``"top"``/``"bottom"``), ``color``
283
+ (``"#000000"``), ``width`` (``"0.12 mm"``), ``type`` (``"SOLID"``),
284
+ ``offset_mm`` (gap to the text in mm, one number or ``(left, right, top,
285
+ bottom)``, default 0), ``connect`` and ``ignore_margin`` (default
286
+ ``False``). With
287
+ ``connect=True`` Hancom draws consecutive paragraphs that share the
288
+ paragraph shape as one box, across columns and pages; give an empty
289
+ paragraph inside the box the same format so it does not split the box.
290
+ ``bottom_border=True`` is the older bottom-only form.
148
291
  """
149
292
 
150
293
  if not doc._root.headers:
@@ -207,6 +350,7 @@ def set_paragraph_format(
207
350
  and not margins
208
351
  and heading is None
209
352
  and not bottom_border
353
+ and border is None
210
354
  and not break_setting
211
355
  and not wants_tab_definition
212
356
  and column_break is None
@@ -217,6 +361,26 @@ def set_paragraph_format(
217
361
  suggestion="Pass alignment, line_spacing_percent, or another option to change.",
218
362
  )
219
363
 
364
+ # Resolve every target before the header gains tab or border definitions,
365
+ # so a bad target changes nothing.
366
+ targets: list[tuple[int | None, HwpxOxmlParagraph]]
367
+ if paragraphs is not None:
368
+ if paragraph_index is not None or paragraph_indexes is not None:
369
+ raise HwpxValueError(
370
+ "use either paragraphs or paragraph_index/paragraph_indexes, not both",
371
+ code="paragraph-argument-conflict",
372
+ suggestion="Pass only one.",
373
+ )
374
+ objects = _resolve_paragraph_objects(doc, paragraphs)
375
+ body = doc.paragraphs
376
+ body_index = {paragraph.element: index for index, paragraph in enumerate(body)}
377
+ targets = [(body_index.get(paragraph.element), paragraph) for paragraph in objects]
378
+ else:
379
+ targets = list(_resolve_paragraph_targets(doc,
380
+ paragraph_index=paragraph_index,
381
+ paragraph_indexes=paragraph_indexes,
382
+ ))
383
+
220
384
  tab_pr_id: str | None = None
221
385
  if wants_tab_definition:
222
386
  converted_stops: list[dict[str, object]] = []
@@ -239,22 +403,13 @@ def set_paragraph_format(
239
403
  auto_tab_right=bool(auto_tab_right),
240
404
  )
241
405
 
242
- border: dict[str, str] | None = None
243
- if bottom_border:
244
- border_fill_id = header.ensure_border_fill(
245
- border_color=border_color,
246
- border_width=border_width,
247
- active_borders=("bottom",),
248
- )
249
- border = {
250
- "borderFillIDRef": border_fill_id,
251
- "offsetLeft": "0",
252
- "offsetRight": "0",
253
- "offsetTop": "0",
254
- "offsetBottom": "0",
255
- "connect": "0",
256
- "ignoreMargin": "0",
257
- }
406
+ border_attrs = _paragraph_border_attrs(
407
+ header,
408
+ border,
409
+ bottom_border=bottom_border,
410
+ border_color=border_color,
411
+ border_width=border_width,
412
+ )
258
413
 
259
414
  # column_break bypasses paraPr entirely (it's hp:p's own attribute, not
260
415
  # a shared style) -- only mint a new paraPr when one of the *other*
@@ -265,17 +420,12 @@ def set_paragraph_format(
265
420
  or line_spacing_percent is not None
266
421
  or bool(margins)
267
422
  or heading is not None
268
- or bottom_border
423
+ or border_attrs is not None
269
424
  or bool(break_setting)
270
425
  or wants_tab_definition
271
426
  )
272
427
 
273
- targets = _resolve_paragraph_targets(doc,
274
- paragraph_index=paragraph_index,
275
- paragraph_indexes=paragraph_indexes,
276
- )
277
- formatted: list[int] = []
278
- for index, paragraph in targets:
428
+ for paragraph in (paragraph for _, paragraph in targets):
279
429
  if wants_para_pr_change:
280
430
  para_pr_id = header.ensure_paragraph_format(
281
431
  base_para_pr_id=paragraph.para_pr_id_ref,
@@ -283,18 +433,17 @@ def set_paragraph_format(
283
433
  line_spacing_percent=line_spacing_percent,
284
434
  margins=margins,
285
435
  heading=heading,
286
- border=border,
436
+ border=border_attrs,
287
437
  break_setting=break_setting or None,
288
438
  tab_pr_id_ref=tab_pr_id,
289
439
  )
290
440
  paragraph.para_pr_id_ref = para_pr_id
291
441
  if column_break is not None:
292
442
  paragraph.column_break = column_break
293
- formatted.append(index)
294
443
 
295
444
  return ParagraphFormatResult(
296
- formatted=len(formatted),
297
- paragraphs=tuple(formatted),
445
+ formatted=len(targets),
446
+ paragraphs=tuple(index for index, _ in targets if index is not None),
298
447
  units=Units(indent="mm", paragraph_spacing="pt", line_spacing="%"),
299
448
  )
300
449
 
@@ -396,7 +545,12 @@ def set_page_setup(
396
545
  section: HwpxOxmlSection | None = None,
397
546
  section_index: int | None = None,
398
547
  ) -> PageSetup:
399
- """Set page size, margins, orientation, and optional columns in human units."""
548
+ """Set page size, margins, orientation, and optional columns in human units.
549
+
550
+ The page is written as Hancom writes it: ``WIDELY`` for portrait and
551
+ ``NARROWLY`` for landscape, both with the paper's portrait size. The
552
+ returned ``page_size`` reports the page as drawn (landscape is wider).
553
+ """
400
554
 
401
555
  normalized_orientation = _normalize_page_orientation(orientation)
402
556
  target_width_mm = width_mm
@@ -415,10 +569,11 @@ def set_page_setup(
415
569
  target_height_mm = paper_height if target_height_mm is None else target_height_mm
416
570
 
417
571
  if target_width_mm is not None and target_height_mm is not None:
418
- if normalized_orientation == "WIDELY" and target_width_mm < target_height_mm:
419
- target_width_mm, target_height_mm = target_height_mm, target_width_mm
420
- elif normalized_orientation == "PORTRAIT" and target_width_mm > target_height_mm:
421
- target_width_mm, target_height_mm = target_height_mm, target_width_mm
572
+ short_side, long_side = sorted((target_width_mm, target_height_mm))
573
+ if normalized_orientation == _PAGE_LANDSCAPE:
574
+ target_width_mm, target_height_mm = long_side, short_side
575
+ elif normalized_orientation == _PAGE_PORTRAIT:
576
+ target_width_mm, target_height_mm = short_side, long_side
422
577
 
423
578
  width = _mm_to_hwp_units(float(target_width_mm)) if target_width_mm is not None else None
424
579
  height = _mm_to_hwp_units(float(target_height_mm)) if target_height_mm is not None else None
@@ -510,10 +665,12 @@ def set_columns(
510
665
  section: HwpxOxmlSection | None = None,
511
666
  section_index: int | None = None,
512
667
  ) -> HwpxOxmlInlineObject:
513
- """Insert a column definition control.
668
+ """Set the columns of a section, or start new columns at a paragraph.
514
669
 
515
- This adds a ``<hp:ctrl><hp:colPr>`` element to the specified paragraph.
516
- Text that follows will be laid out in the specified number of columns.
670
+ Without ``paragraph`` this rewrites the section's own column layout (the
671
+ ``hp:colPr`` next to ``hp:secPr``) in place, so the whole section is laid
672
+ out in ``col_count`` columns. With ``paragraph`` it adds a column
673
+ definition control there, and the text from that paragraph on uses it.
517
674
 
518
675
  Args:
519
676
  col_count: Number of columns (1–255).
@@ -521,7 +678,28 @@ def set_columns(
521
678
  same_gap: Gap in HWPUNIT (7200 = 1 inch).
522
679
  separator_type: Optional column separator line type (e.g. ``SOLID``).
523
680
  """
681
+ if not 1 <= col_count <= 255:
682
+ raise HwpxValueError(
683
+ "col_count must be between 1 and 255",
684
+ code="page-columns-invalid",
685
+ context={"requested": col_count},
686
+ suggestion="Use columns=1 to remove columns.",
687
+ )
524
688
  if paragraph is None:
689
+ target_section = _resolve_section(doc, section=section, section_index=section_index)
690
+ ctrl = target_section.properties.set_columns(
691
+ col_count,
692
+ col_type=col_type,
693
+ layout=layout,
694
+ same_size=same_size,
695
+ same_gap=same_gap,
696
+ column_widths=column_widths,
697
+ separator_type=separator_type,
698
+ separator_width=separator_width,
699
+ separator_color=separator_color,
700
+ )
701
+ if ctrl is not None:
702
+ return HwpxOxmlInlineObject(ctrl, target_section.paragraphs[0])
525
703
  paragraph = doc.add_paragraph(
526
704
  "", section=section, section_index=section_index,
527
705
  include_run=False,
@@ -573,21 +751,12 @@ def add_hyperlink(
573
751
 
574
752
  The display text follows the Hancom convention (blue ``#0000FF`` text
575
753
  with a blue bottom underline — dominant styling across real-corpus
576
- hyperlinks) unless ``char_pr_id_ref`` overrides it.
754
+ hyperlinks) on the character look of the paragraph it goes into, unless
755
+ ``char_pr_id_ref`` overrides it. ``paragraph.add_hyperlink`` picks that
756
+ style, so a link looks the same whichever way it was added.
577
757
 
578
758
  Returns the ``<hp:ctrl>`` wrapper containing the ``<hp:fieldBegin>``.
579
759
  """
580
- if char_pr_id_ref is None:
581
- # `doc._root.ensure_run_style` rather than `doc.ensure_run_style` —
582
- # that facade name moved in 6.0 (design table row 52) and is a pure
583
- # passthrough to `_root`, so this is byte-identical minus the
584
- # DeprecationWarning it would otherwise fire on every hyperlink even
585
- # when reached via the new `doc.refs.add_hyperlink` namespace path.
586
- char_pr_id_ref = doc._root.ensure_run_style(
587
- underline=True,
588
- color="#0000FF",
589
- underline_color="#0000FF",
590
- )
591
760
  if paragraph is None:
592
761
  paragraph = doc.add_paragraph(
593
762
  "", section=section, section_index=section_index,
@@ -631,7 +800,7 @@ def set_page_size(
631
800
  target_section.properties.set_page_size(
632
801
  width=width,
633
802
  height=height,
634
- orientation=orientation,
803
+ orientation=_normalize_page_orientation(orientation),
635
804
  gutter_type=gutter_type,
636
805
  )
637
806
 
@@ -868,10 +1037,14 @@ def hide_page_elements(
868
1037
  fill: bool = False,
869
1038
  page_num: bool = False,
870
1039
  ) -> "HwpxOxmlInlineObject":
871
- """Hide the named page elements from *paragraph*'s page onward.
1040
+ """Hide the named page elements on *paragraph*'s page only.
872
1041
 
873
1042
  Inserts ``<hp:ctrl><hp:pageHiding .../></hp:ctrl>`` (``ParaList XML
874
1043
  schema.xml:148-163`` — six independent booleans, all default unhidden).
1044
+ Hancom applies it to that page alone (its "hide on the current page
1045
+ only"); the next page shows the elements again. *page_num* hides
1046
+ Hancom's page-number control, not the header/footer number that
1047
+ ``set_page_number`` writes -- hide that one with *footer* (or *header*).
875
1048
  """
876
1049
 
877
1050
  return paragraph.add_page_hiding(
@@ -920,3 +1093,57 @@ def remove_footer(
920
1093
  return
921
1094
  target_section = doc._root.sections[-1]
922
1095
  target_section.properties.remove_footer(page_type=page_type)
1096
+
1097
+
1098
+ def flow_table_taller_than_page(doc: "HwpxDocument", table: Any) -> None:
1099
+ """Let a new body *table* flow across pages when its rows alone outgrow a page.
1100
+
1101
+ Hancom never breaks a table laid out as a character (``treatAsChar``, the
1102
+ ``add_table`` default) across pages: one taller than the page body is cut
1103
+ off at the paper's edge. Such a table becomes a flowing one instead
1104
+ (``Table.set_treat_as_char(False)``), which Hancom breaks between rows.
1105
+ """
1106
+
1107
+ properties = table.paragraph.section.properties
1108
+ size, margins = properties.page_size, properties.page_margins
1109
+ body = size.drawn_height - margins.top - margins.bottom - margins.header - margins.footer
1110
+ if body > 0 and _table_min_height(doc, table.element) > body:
1111
+ table.set_treat_as_char(False)
1112
+
1113
+
1114
+ def _table_min_height(doc: "HwpxDocument", table: Any) -> int:
1115
+ """A lower bound of the drawn height: every row is at least its tallest
1116
+ single-row cell, and a cell at least one line of its text plus its top and
1117
+ bottom margins."""
1118
+
1119
+ total = 0
1120
+ for row in table.findall(f"{HP}tr"):
1121
+ tallest = 0
1122
+ for cell in row.findall(f"{HP}tc"):
1123
+ span = cell.find(f"{HP}cellSpan")
1124
+ if span is not None and span.get("rowSpan", "1") != "1":
1125
+ continue
1126
+ run = cell.find(f".//{HP}run")
1127
+ line = _char_height(doc, run.get("charPrIDRef") if run is not None else None)
1128
+ margin = cell.find(f"{HP}cellMargin")
1129
+ padding = _int_attr(margin, "top") + _int_attr(margin, "bottom")
1130
+ tallest = max(tallest, _int_attr(cell.find(f"{HP}cellSz"), "height"), line + padding)
1131
+ total += tallest
1132
+ return total
1133
+
1134
+
1135
+ def _char_height(doc: "HwpxDocument", char_pr_id_ref: str | None) -> int:
1136
+ style = doc._root.char_property(char_pr_id_ref if char_pr_id_ref is not None else "0")
1137
+ try:
1138
+ return int(style.attributes.get("height", "1000")) if style is not None else 1000
1139
+ except ValueError:
1140
+ return 1000
1141
+
1142
+
1143
+ def _int_attr(element: Any, name: str) -> int:
1144
+ if element is None:
1145
+ return 0
1146
+ try:
1147
+ return int(element.get(name, "0"))
1148
+ except ValueError:
1149
+ return 0