@amaster.ai/pi-lark 0.1.8 → 0.1.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (155) hide show
  1. package/README.md +1 -3
  2. package/package.json +2 -2
  3. package/skills/lark-apps/SKILL.md +3 -1
  4. package/skills/lark-apps/references/lark-apps-cache.md +38 -5
  5. package/skills/lark-apps/references/lark-apps-db.md +130 -2
  6. package/skills/lark-apps/references/lark-apps-user-id-convert.md +63 -0
  7. package/skills/lark-base/SKILL.md +172 -167
  8. package/skills/lark-base/references/{lark-base-role-guide.md → lark-base-advanced-permission-and-role.md} +5 -5
  9. package/skills/lark-base/references/lark-base-app-block-data-config.md +122 -0
  10. package/skills/lark-base/references/lark-base-app.md +243 -0
  11. package/skills/lark-base/references/lark-base-cell-value.md +26 -19
  12. package/skills/lark-base/references/{dashboard-block-data-config.md → lark-base-dashboard-block-config.md} +65 -6
  13. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +1 -1
  14. package/skills/lark-base/references/lark-base-dashboard.md +38 -20
  15. package/skills/lark-base/references/lark-base-data-analysis-pandas.md +93 -0
  16. package/skills/lark-base/references/lark-base-data-analysis-python-stdlib.md +120 -0
  17. package/skills/lark-base/references/lark-base-data-query.md +8 -11
  18. package/skills/lark-base/references/lark-base-field-create.md +7 -50
  19. package/skills/lark-base/references/{formula-field-guide.md → lark-base-field-formula.md} +1 -1
  20. package/skills/lark-base/references/{lookup-field-guide.md → lark-base-field-lookup.md} +1 -1
  21. package/skills/lark-base/references/{lark-base-field-json.md → lark-base-field-schema.md} +25 -101
  22. package/skills/lark-base/references/lark-base-field-update.md +13 -51
  23. package/skills/lark-base/references/lark-base-filter-condition.md +19 -31
  24. package/skills/lark-base/references/lark-base-form-questions-create.md +36 -5
  25. package/skills/lark-base/references/lark-base-record-batch-create.md +5 -1
  26. package/skills/lark-base/references/lark-base-record-batch-update.md +5 -2
  27. package/skills/lark-base/references/lark-base-record-history-list.md +19 -2
  28. package/skills/lark-base/references/lark-base-record-query-and-analysis-cloud-sop.md +145 -0
  29. package/skills/lark-base/references/lark-base-record-query-and-analysis-sop.md +233 -0
  30. package/skills/lark-base/references/{role-config.md → lark-base-role-config.md} +2 -2
  31. package/skills/lark-base/references/lark-base-template-center.md +195 -0
  32. package/skills/lark-base/references/lark-base-view-set-filter.md +1 -1
  33. package/skills/lark-base/references/lark-base-workflow-schema.md +2 -2
  34. package/skills/lark-base/references/{lark-base-workflow-guide.md → lark-base-workflow.md} +1 -1
  35. package/skills/lark-calendar/SKILL.md +11 -6
  36. package/skills/lark-calendar/references/lark-calendar-create.md +4 -3
  37. package/skills/lark-calendar/references/lark-calendar-transfer.md +89 -0
  38. package/skills/lark-doc/SKILL.md +3 -3
  39. package/skills/lark-doc/references/lark-doc-fetch.md +9 -4
  40. package/skills/lark-doc/references/lark-doc-update.md +12 -8
  41. package/skills/lark-drive/SKILL.md +5 -3
  42. package/skills/lark-drive/references/lark-drive-add-comment.md +2 -2
  43. package/skills/lark-drive/references/lark-drive-download.md +27 -1
  44. package/skills/lark-drive/references/lark-drive-export.md +1 -0
  45. package/skills/lark-drive/references/lark-drive-member-remove.md +59 -0
  46. package/skills/lark-drive/references/lark-drive-preview.md +21 -2
  47. package/skills/lark-drive/references/lark-drive-push.md +5 -1
  48. package/skills/lark-drive/references/lark-drive-search.md +2 -0
  49. package/skills/lark-im/SKILL.md +14 -3
  50. package/skills/lark-im/references/lark-im-message-read-status.md +96 -0
  51. package/skills/lark-mail/references/lark-mail-draft-create.md +12 -12
  52. package/skills/lark-mail/references/lark-mail-forward.md +17 -17
  53. package/skills/lark-mail/references/lark-mail-reply-all.md +8 -8
  54. package/skills/lark-mail/references/lark-mail-reply.md +6 -6
  55. package/skills/lark-mail/references/lark-mail-send.md +20 -20
  56. package/skills/lark-mail/references/lark-mail-template-create.md +7 -6
  57. package/skills/lark-mail/references/lark-mail-template-update.md +7 -6
  58. package/skills/lark-meeting/SKILL.md +146 -0
  59. package/skills/lark-meeting/references/lark-minutes-apply-permission.md +92 -0
  60. package/skills/lark-meeting/references/lark-minutes-detail.md +52 -0
  61. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-download.md +7 -7
  62. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-search.md +4 -34
  63. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-speaker-replace.md +3 -4
  64. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-summary.md +2 -5
  65. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-todo.md +5 -15
  66. package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-update.md +2 -3
  67. package/skills/lark-meeting/references/lark-minutes-upload.md +65 -0
  68. package/skills/lark-meeting/references/lark-note-detail.md +15 -0
  69. package/skills/{lark-note → lark-meeting}/references/lark-note-transcript.md +5 -9
  70. package/skills/{lark-vc-agent → lark-meeting}/references/lark-vc-agent-meeting-join.md +4 -55
  71. package/skills/{lark-vc-agent → lark-meeting}/references/lark-vc-agent-meeting-leave.md +2 -41
  72. package/skills/lark-meeting/references/lark-vc-detail.md +31 -0
  73. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-events.md → lark-meeting/references/lark-vc-meeting-events.md} +120 -109
  74. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-list-active.md → lark-meeting/references/lark-vc-meeting-list-active.md} +4 -29
  75. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-message-send.md → lark-meeting/references/lark-vc-meeting-message-send.md} +3 -5
  76. package/skills/{lark-vc → lark-meeting}/references/lark-vc-recording.md +5 -64
  77. package/skills/{lark-vc → lark-meeting}/references/lark-vc-search.md +9 -28
  78. package/skills/lark-meeting/scenes/create-and-edit-minutes.md +125 -0
  79. package/skills/lark-meeting/scenes/live-meeting-attend.md +107 -0
  80. package/skills/lark-meeting/scenes/live-meeting-interact.md +72 -0
  81. package/skills/lark-meeting/scenes/query-meeting-and-artifacts.md +90 -0
  82. package/skills/lark-meeting/scenes/query-minutes-and-artifacts.md +70 -0
  83. package/skills/lark-meeting/scenes/query-note-and-artifacts.md +127 -0
  84. package/skills/lark-minutes/SKILL.md +5 -197
  85. package/skills/lark-note/SKILL.md +5 -84
  86. package/skills/lark-shared/SKILL.md +25 -188
  87. package/skills/lark-shared/references/lark-shared-config-init.md +12 -0
  88. package/skills/lark-shared/references/lark-shared-high-risk-approval.md +38 -0
  89. package/skills/lark-shared/references/lark-shared-identity-and-permissions.md +105 -0
  90. package/skills/lark-shared/references/lark-shared-output-contract.md +17 -0
  91. package/skills/lark-shared/references/lark-shared-update-notice.md +23 -0
  92. package/skills/lark-slides/SKILL.md +56 -54
  93. package/skills/lark-slides/references/cli/lark-slides-add-slide.md +92 -0
  94. package/skills/lark-slides/references/cli/lark-slides-create.md +176 -0
  95. package/skills/lark-slides/references/cli/lark-slides-delete-slide.md +65 -0
  96. package/skills/lark-slides/references/cli/lark-slides-history.md +132 -0
  97. package/skills/lark-slides/references/cli/lark-slides-media-upload.md +103 -0
  98. package/skills/lark-slides/references/cli/lark-slides-replace-slide.md +259 -0
  99. package/skills/lark-slides/references/cli/lark-slides-screenshot.md +115 -0
  100. package/skills/lark-slides/references/{lark-slides-update-slide.md → cli/lark-slides-update-slide.md} +21 -4
  101. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-get.md +110 -0
  102. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-replace.md +188 -0
  103. package/skills/lark-slides/references/cli/lark-slides-xml-presentations-get.md +157 -0
  104. package/skills/lark-slides/references/iconpark-index.json +5 -41901
  105. package/skills/lark-slides/references/iconpark.md +3 -44
  106. package/skills/lark-slides/references/lark-slides-add-slide.md +3 -90
  107. package/skills/lark-slides/references/lark-slides-create.md +3 -174
  108. package/skills/lark-slides/references/lark-slides-delete-slide.md +3 -63
  109. package/skills/lark-slides/references/lark-slides-edit-workflows.md +3 -141
  110. package/skills/lark-slides/references/lark-slides-history.md +3 -130
  111. package/skills/lark-slides/references/lark-slides-media-upload.md +3 -102
  112. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +3 -83
  113. package/skills/lark-slides/references/lark-slides-replace-slide.md +3 -256
  114. package/skills/lark-slides/references/lark-slides-screenshot.md +3 -113
  115. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +3 -108
  116. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +3 -186
  117. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +3 -155
  118. package/skills/lark-slides/references/planning-layer.md +1 -1
  119. package/skills/lark-slides/references/slides_chart_demo.xml +5 -1415
  120. package/skills/lark-slides/references/slides_xml_schema_definition.xml +3 -3512
  121. package/skills/lark-slides/references/troubleshooting.md +3 -60
  122. package/skills/lark-slides/references/validation-checklist.md +3 -154
  123. package/skills/lark-slides/references/workflow/error-handling.md +62 -0
  124. package/skills/lark-slides/references/workflow/slides-editing.md +143 -0
  125. package/skills/lark-slides/references/workflow/template-editing.md +85 -0
  126. package/skills/lark-slides/references/workflow/validation-xml.md +156 -0
  127. package/skills/lark-slides/references/xml/iconpark-index.json +37458 -0
  128. package/skills/lark-slides/references/xml/iconpark.md +46 -0
  129. package/skills/lark-slides/references/xml/slides_chart_demo.xml +1415 -0
  130. package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +3601 -0
  131. package/skills/lark-slides/references/xml/xml-schema-quick-ref.md +497 -0
  132. package/skills/lark-slides/references/xml-schema-quick-ref.md +3 -495
  133. package/skills/lark-slides/scripts/iconpark_tool.py +1 -1
  134. package/skills/lark-slides/scripts/xml_lint.py +2989 -0
  135. package/skills/lark-slides/scripts/xml_lint_test.py +4720 -0
  136. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +3 -2975
  137. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +5 -4712
  138. package/skills/lark-task/SKILL.md +13 -1
  139. package/skills/lark-task/references/lark-task-create.md +3 -1
  140. package/skills/lark-vc/SKILL.md +5 -195
  141. package/skills/lark-vc-agent/SKILL.md +5 -191
  142. package/skills/lark-wiki/SKILL.md +3 -1
  143. package/skills/lark-wiki/references/lark-wiki-node-copy.md +5 -19
  144. package/skills/lark-wiki/references/lark-wiki-node-create.md +19 -2
  145. package/skills/lark-wiki/references/lark-wiki-node-get.md +15 -0
  146. package/skills/lark-wiki/references/lark-wiki-node-list.md +1 -1
  147. package/skills/lark-workflow-meeting-summary/SKILL.md +20 -13
  148. package/skills/lark-base/references/lark-base-data-analysis-sop.md +0 -210
  149. package/skills/lark-base/references/lark-base-data-query-guide.md +0 -69
  150. package/skills/lark-base/references/lark-base-record-upsert.md +0 -63
  151. package/skills/lark-minutes/references/lark-minutes-detail.md +0 -62
  152. package/skills/lark-minutes/references/lark-minutes-upload.md +0 -104
  153. package/skills/lark-note/references/lark-note-detail.md +0 -26
  154. package/skills/lark-vc/references/lark-vc-detail.md +0 -44
  155. package/skills/lark-vc/references/vc-domain-boundaries.md +0 -196
@@ -1,2984 +1,12 @@
1
1
  #!/usr/bin/env python3
2
2
  # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
3
  # SPDX-License-Identifier: MIT
4
- """Validate Slides XML structure and page layout through one release gate."""
4
+ """Compatibility entry point for the renamed xml_lint module."""
5
5
 
6
- from __future__ import annotations
7
-
8
- import copy
9
- import json
10
- import math
11
- import re
12
6
  import sys
13
- import unicodedata
14
- import xml.parsers.expat as expat
15
- import xml.etree.ElementTree as ET
16
- from difflib import SequenceMatcher, get_close_matches
17
- from pathlib import Path
18
- from typing import Any
19
-
20
- import sxsd_validator
21
-
22
-
23
- XS_NS = "{http://www.w3.org/2001/XMLSchema}"
24
- XML_NS = "{http://www.w3.org/XML/1998/namespace}"
25
- SVG_NS = "{http://www.w3.org/2000/svg}"
26
- SML_NAMESPACE = "https://www.larkoffice.com/sml/2.0"
27
- SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
28
- ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
29
- SXSD_TAG_ALIASES = {
30
- "textbox": "<shape type=\"text\">",
31
- "textBox": "<shape type=\"text\">",
32
- "image": "<img>",
33
- "picture": "<img>",
34
- }
35
- SXSD_ATTR_ALIASES = {
36
- "x": "topLeftX",
37
- "left": "topLeftX",
38
- "y": "topLeftY",
39
- "top": "topLeftY",
40
- "w": "width",
41
- "h": "height",
42
- "fontColor": "color",
43
- }
44
- SERVER_FILLED_SXSD_ATTRS = {"id"}
45
- ROUNDTRIP_SXSD_ATTRS = {
46
- ("chart", "updated"),
47
- ("chartData", "isStaticData"),
48
- }
49
- # Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
50
- # it is server-emitted and absent from the write schema, so it must not block page linting.
51
- ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
52
- DEFAULT_TABLE_COLUMN_WIDTH = 110
53
- DEFAULT_TABLE_ROW_HEIGHT = 37
54
- DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
55
- TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
56
- TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
57
- SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
58
- CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
59
- HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
60
- DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
61
- GHOST_TEXT_MIN_FONT_SIZE = 96
62
- GHOST_TEXT_MAX_ALPHA = 0.5
63
- GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
64
- GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
65
- # A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
66
- # glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
67
- # padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
68
- LINE_TEXT_GRAZE_MIN_PX = 2.0
69
- LINE_TEXT_GRAZE_FONT_RATIO = 0.12
70
- # A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
71
- LINE_MIN_VISIBLE_ALPHA = 0.08
72
- # Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
73
- # visible defect; keep this well under 1px so real overflow is still always caught.
74
- CANVAS_OVERFLOW_TOLERANCE = 0.5
75
- XML_PATH_HINT_PREFIX = "Locate via related_objects[].xml_path."
76
- _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
77
- _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
78
-
79
-
80
- class XmlLayoutLintError(Exception):
81
- pass
82
-
83
-
84
- def fail(message: str) -> None:
85
- raise XmlLayoutLintError(message)
86
-
87
-
88
- def read_file(file_path: str | Path) -> str:
89
- return Path(file_path).read_text(encoding="utf-8")
90
-
91
-
92
- def parse_args(argv: list[str]) -> dict[str, Any]:
93
- options: dict[str, Any] = {}
94
- index = 0
95
- while index < len(argv):
96
- token = argv[index]
97
- if not token.startswith("--"):
98
- fail(f"unexpected argument: {token}, need --input")
99
- key = token[2:]
100
- next_token = argv[index + 1] if index + 1 < len(argv) else None
101
- if next_token is None or next_token.startswith("--"):
102
- options[key] = True
103
- index += 1
104
- continue
105
- options[key] = next_token
106
- index += 2
107
- return options
108
-
109
-
110
- def extract_attribute(tag_source: str, name: str) -> str | None:
111
- match = re.search(
112
- fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
113
- )
114
- if not match:
115
- return None
116
- return match.group(1) if match.group(1) is not None else match.group(2)
117
-
118
-
119
- def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
120
- raw = extract_attribute(tag_source, name)
121
- if raw is None:
122
- return None
123
- try:
124
- value = float(raw)
125
- except ValueError:
126
- return None
127
- return int(value) if value.is_integer() else value
128
-
129
-
130
- def extract_bool_attribute(tag_source: str, name: str) -> bool:
131
- value = extract_attribute(tag_source, name)
132
- return value in {"true", "1", "yes"}
133
-
134
-
135
- def extract_color_alpha(color: str | None) -> int | float | None:
136
- if color is None:
137
- return None
138
- normalized = re.sub(r"\s+", "", color).lower()
139
- if normalized == "transparent":
140
- return 0
141
- rgba_match = re.fullmatch(
142
- r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
143
- normalized,
144
- )
145
- if rgba_match is None:
146
- return None
147
- try:
148
- alpha = float(rgba_match.group(1))
149
- except ValueError:
150
- return None
151
- return int(alpha) if alpha.is_integer() else alpha
152
-
153
-
154
- def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
155
- base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
156
- color_alpha = extract_color_alpha(text_color)
157
- if not isinstance(color_alpha, (int, float)):
158
- return base_alpha
159
- return base_alpha * color_alpha
160
-
161
-
162
- def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
163
- for tag_name in style_tags:
164
- if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
165
- return True
166
- return False
167
-
168
-
169
- def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
170
- for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
171
- if extract_bool_attribute(attrs, attr_name):
172
- return True
173
- return False
174
-
175
-
176
- def sum_sizes(sizes: list[int | float]) -> int | float:
177
- return sum(sizes)
178
-
179
-
180
- def is_filled_size(size: int | float | None) -> bool:
181
- return isinstance(size, (int, float)) and math.isfinite(size) and size > 0
182
-
183
-
184
- def fill_last_size_gap(sizes: list[int | float], target_size: int | float) -> list[int | float]:
185
- if not sizes:
186
- return sizes
187
- final_sizes = [
188
- size if index == len(sizes) - 1 else max(1, math.floor(size + 0.5))
189
- for index, size in enumerate(sizes)
190
- ]
191
- remaining_size = target_size - sum_sizes(final_sizes[:-1])
192
- if remaining_size >= 1:
193
- final_sizes[-1] = remaining_size
194
- return final_sizes
195
-
196
- size_to_redistribute = 1 - remaining_size
197
- for index in range(len(final_sizes) - 2, -1, -1):
198
- reduction = min(final_sizes[index] - 1, size_to_redistribute)
199
- final_sizes[index] -= reduction
200
- size_to_redistribute -= reduction
201
- if size_to_redistribute == 0:
202
- final_sizes[-1] = 1
203
- return final_sizes
204
-
205
- final_sizes[-1] = 1
206
- return final_sizes
207
-
208
-
209
- def solve_weighted_min_layout(
210
- input_sizes: list[int | float | None], default_size: int | float, target_min_size: int | float | None
211
- ) -> dict[str, Any]:
212
- filled_indexes: list[int] = []
213
- empty_indexes: list[int] = []
214
- base_sizes: list[int | float] = []
215
- for index, size in enumerate(input_sizes):
216
- if is_filled_size(size):
217
- filled_indexes.append(index)
218
- base_sizes.append(size)
219
- else:
220
- empty_indexes.append(index)
221
- base_sizes.append(0)
222
- filled_sum = sum_sizes(base_sizes)
223
-
224
- if target_min_size is None:
225
- final_sizes = [default_size if index in empty_indexes else size for index, size in enumerate(base_sizes)]
226
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
227
-
228
- if not filled_indexes:
229
- average_size = target_min_size / len(input_sizes)
230
- final_sizes = fill_last_size_gap([average_size] * len(input_sizes), target_min_size)
231
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
232
-
233
- if empty_indexes:
234
- remaining_size = target_min_size - filled_sum
235
- final_sizes = [*base_sizes]
236
- if remaining_size > 0:
237
- average_size = remaining_size / len(empty_indexes)
238
- empty_sizes = fill_last_size_gap([average_size] * len(empty_indexes), remaining_size)
239
- for index, empty_size in zip(empty_indexes, empty_sizes):
240
- final_sizes[index] = empty_size
241
- else:
242
- for index in empty_indexes:
243
- final_sizes[index] = default_size
244
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
245
-
246
- ratio = max(1, target_min_size / filled_sum)
247
- actual_size = max(target_min_size, filled_sum)
248
- if ratio == 1:
249
- return {"final_sizes": [*base_sizes], "actual_size": actual_size, "ratio": ratio}
250
- final_sizes = fill_last_size_gap([size * ratio for size in base_sizes], actual_size)
251
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
252
-
253
-
254
- def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
255
- stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
256
- if preserve_line_breaks:
257
- stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
258
- stripped = re.sub(r"<[^>]+>", " ", stripped)
259
- stripped = stripped.replace("&nbsp;", " ")
260
- stripped = stripped.replace("&amp;", "&")
261
- stripped = stripped.replace("&lt;", "<")
262
- stripped = stripped.replace("&gt;", ">")
263
- stripped = stripped.replace("&quot;", '"')
264
- stripped = stripped.replace("&#39;", "'")
265
- if preserve_line_breaks:
266
- return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
267
- return re.sub(r"\s+", " ", stripped).strip()
268
-
269
-
270
- def strip_xml_paragraphs(value: str) -> str:
271
- paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
272
- if paragraphs:
273
- return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
274
- return strip_xml(value, preserve_line_breaks=True)
275
-
276
-
277
- def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
278
- paragraphs = []
279
- for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
280
- paragraphs.append(
281
- {
282
- "text": strip_xml(body, preserve_line_breaks=True),
283
- "fontSize": extract_max_span_font_size(body, default_font_size),
284
- "textAlign": extract_attribute(attrs, "textAlign"),
285
- "lineSpacing": extract_attribute(attrs, "lineSpacing"),
286
- "beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
287
- "afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
288
- "letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
289
- }
290
- )
291
- return paragraphs
292
-
293
-
294
- def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
295
- font_sizes = [
296
- font_size
297
- for attrs in re.findall(r"<span\b([^>]*)>", value)
298
- if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
299
- ]
300
- return max([default_font_size, *font_sizes])
301
-
302
-
303
- def extract_tag_attributes(value: str, tag: str) -> str:
304
- match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
305
- return match.group(1) if match else ""
306
-
307
-
308
- def xml_local_name(tag: str) -> str:
309
- return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
310
-
311
-
312
- def xml_namespace(tag: str) -> str | None:
313
- return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
314
-
315
-
316
- def load_sxsd_tag_attributes() -> dict[str, set[str]]:
317
- global _SXSD_TAG_ATTRIBUTES_CACHE
318
- if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
319
- return _SXSD_TAG_ATTRIBUTES_CACHE
320
-
321
- _SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
322
- return _SXSD_TAG_ATTRIBUTES_CACHE
323
-
324
-
325
- def load_iconpark_icon_types() -> set[str]:
326
- global _ICONPARK_ICON_TYPES_CACHE
327
- if _ICONPARK_ICON_TYPES_CACHE is not None:
328
- return _ICONPARK_ICON_TYPES_CACHE
329
-
330
- try:
331
- index_data = json.loads(ICONPARK_INDEX_PATH.read_text(encoding="utf-8"))
332
- except json.JSONDecodeError as error:
333
- fail(f"invalid iconpark index JSON: {error}")
334
- icons = index_data.get("icons")
335
- if not isinstance(icons, list):
336
- fail("iconpark index must contain an icons array")
337
-
338
- icon_types = {
339
- icon["iconType"]
340
- for icon in icons
341
- if isinstance(icon, dict) and isinstance(icon.get("iconType"), str) and icon["iconType"]
342
- }
343
- _ICONPARK_ICON_TYPES_CACHE = icon_types
344
- return icon_types
345
-
346
-
347
- def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
348
- alias = SXSD_TAG_ALIASES.get(tag_name)
349
- if alias:
350
- return f"Use {alias} instead of <{tag_name}>."
351
- if tag_name == "svg":
352
- return 'Inside <embed> or <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
353
- close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
354
- if close_matches:
355
- return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
356
- return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
357
-
358
-
359
- def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
360
- alias = SXSD_ATTR_ALIASES.get(attr_name)
361
- if alias and alias in allowed_attrs:
362
- return [alias]
363
- return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
364
-
365
-
366
- def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
367
- suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
368
- if suggestions:
369
- if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
370
- return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
371
- return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
372
- allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
373
- if len(allowed_attrs) > 8:
374
- allowed_summary += ", ..."
375
- return f"Unsupported SXSD attribute for <{tag_name}>. Allowed attributes include: {allowed_summary}."
376
-
377
-
378
- def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
379
- return ("whiteboard" in ancestors or "embed" in ancestors) and xml_namespace(element.tag) == SVG_NS
380
-
381
-
382
- def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
383
- return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
384
-
385
-
386
- def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
387
- return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
388
-
389
-
390
- def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
391
- sanitized_root = copy.deepcopy(root)
392
-
393
- def sanitize(element: ET.Element) -> None:
394
- tag_name = xml_local_name(element.tag)
395
- for raw_attr_name in list(element.attrib):
396
- if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
397
- del element.attrib[raw_attr_name]
398
- for child in list(element):
399
- if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
400
- element.remove(child)
401
- continue
402
- sanitize(child)
403
-
404
- sanitize(sanitized_root)
405
- return sanitized_root
406
-
407
-
408
- def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
409
- tag_attributes = load_sxsd_tag_attributes()
410
- supported_tags = set(tag_attributes)
411
- issues: list[dict[str, Any]] = []
412
- suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
413
-
414
- def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
415
- if should_skip_sxsd_subtree(element, ancestors):
416
- return
417
-
418
- tag_name = xml_local_name(element.tag)
419
- current_path = f"{path}/{tag_name}" if path else tag_name
420
- parent_name = ancestors[-1] if ancestors else None
421
- if should_skip_sxsd_tag(parent_name, tag_name):
422
- return
423
- if tag_name not in supported_tags:
424
- issues.append(
425
- {
426
- "level": "error",
427
- "code": "sxsd_unsupported_tag",
428
- "tag": tag_name,
429
- "path": current_path,
430
- "message": f"unsupported SXSD tag <{tag_name}> at {current_path}",
431
- "hint": build_sxsd_tag_hint(tag_name, supported_tags),
432
- }
433
- )
434
- return
435
- else:
436
- allowed_attrs = tag_attributes[tag_name]
437
- for raw_attr_name in element.attrib:
438
- if raw_attr_name.startswith(XML_NS):
439
- continue
440
- attr_name = xml_local_name(raw_attr_name)
441
- if should_skip_sxsd_attribute(tag_name, attr_name):
442
- continue
443
- if attr_name in allowed_attrs:
444
- continue
445
- suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
446
- if suggestions:
447
- suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
448
- set(suggestions)
449
- )
450
- issues.append(
451
- {
452
- "level": "error",
453
- "code": "sxsd_unsupported_attr",
454
- "tag": tag_name,
455
- "attr": attr_name,
456
- "path": current_path,
457
- "message": f'unsupported SXSD attribute "{attr_name}" on <{tag_name}> at {current_path}',
458
- "hint": build_sxsd_attr_hint(tag_name, attr_name, allowed_attrs),
459
- }
460
- )
461
-
462
- for child in element:
463
- visit(child, [*ancestors, tag_name], current_path)
464
-
465
- visit(root, [], "")
466
- existing = {
467
- (issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
468
- for issue in issues
469
- }
470
- unsupported_tag_locations = {
471
- (issue.get("path"), issue.get("tag"))
472
- for issue in issues
473
- if issue.get("code") == "sxsd_unsupported_tag"
474
- }
475
- schema_issues = _validate_sxsd_schema_constraints(xml, root)
476
- missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
477
- for schema_issue in schema_issues:
478
- if schema_issue.get("code") != "sxsd_missing_required_attr":
479
- continue
480
- location = (schema_issue.get("path"), schema_issue.get("tag"))
481
- missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
482
-
483
- suggested_attrs: set[tuple[str, str, str]] = set()
484
- for location, candidate_groups in suggested_attr_candidates.items():
485
- missing_attrs = missing_attrs_by_location.get(location, set())
486
- for candidates in candidate_groups:
487
- matching_missing_attrs = candidates & missing_attrs
488
- if len(matching_missing_attrs) == 1:
489
- suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
490
-
491
- for schema_issue in schema_issues:
492
- if schema_issue.get("code") == "sxsd_unexpected_child" and (
493
- schema_issue.get("path"),
494
- schema_issue.get("tag"),
495
- ) in unsupported_tag_locations:
496
- continue
497
- if schema_issue.get("code") == "sxsd_missing_required_attr" and (
498
- schema_issue.get("path"),
499
- schema_issue.get("tag"),
500
- schema_issue.get("attr"),
501
- ) in suggested_attrs:
502
- continue
503
- key = (
504
- schema_issue.get("code"),
505
- schema_issue.get("path"),
506
- schema_issue.get("tag"),
507
- schema_issue.get("attr"),
508
- )
509
- if key not in existing:
510
- issues.append(schema_issue)
511
- return issues
512
-
513
-
514
- def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
515
- issues: list[dict[str, Any]] = []
516
- if re.match(r"^\s*<\?xml\b", xml):
517
- issues.append(
518
- {
519
- "level": "error",
520
- "code": "sxsd_unsupported_declaration",
521
- "path": xml_local_name(root.tag),
522
- "tag": xml_local_name(root.tag),
523
- "expected": "SXSD document without an XML declaration",
524
- "actual": "<?xml ...?>",
525
- "message": "XML declarations are not supported by the Slides SXSD write format",
526
- "hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
527
- }
528
- )
529
-
530
- issues.extend(
531
- sxsd_validator.validate_sxsd(
532
- without_server_filled_sxsd_fields(root),
533
- SXSD_SCHEMA_PATH,
534
- )
535
- )
536
- issues.extend(validate_embed_svg_roots(root))
537
- return issues
538
-
539
-
540
- def validate_embed_svg_roots(root: ET.Element) -> list[dict[str, Any]]:
541
- issues: list[dict[str, Any]] = []
542
- document_namespace = sxsd_validator.element_namespace(root.tag)
543
- is_bare_slide_fragment = (
544
- xml_local_name(root.tag) == "slide" and document_namespace is None
545
- )
546
- if (
547
- document_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES
548
- and not is_bare_slide_fragment
549
- ):
550
- return issues
551
-
552
- def visit(element: ET.Element, ancestors: list[str], parent_path: str) -> None:
553
- if should_skip_sxsd_subtree(element, ancestors):
554
- return
555
-
556
- tag_name = xml_local_name(element.tag)
557
- path = f"{parent_path}/{tag_name}" if parent_path else tag_name
558
- if (
559
- tag_name == "embed"
560
- and sxsd_validator.element_namespace(element.tag) == document_namespace
561
- ):
562
- for child in element:
563
- if xml_namespace(child.tag) != SVG_NS or xml_local_name(child.tag) == "svg":
564
- continue
565
- child_name = xml_local_name(child.tag)
566
- child_path = f"{path}/{child_name}"
567
- issues.append(
568
- {
569
- "level": "error",
570
- "code": "sxsd_unexpected_child",
571
- "path": child_path,
572
- "tag": child_name,
573
- "expected": '<svg xmlns="http://www.w3.org/2000/svg">',
574
- "actual": child_name,
575
- "message": f"embedded SVG content must use an <svg> root at {child_path}",
576
- "hint": 'Wrap the SVG content in <svg xmlns="http://www.w3.org/2000/svg">...</svg>.',
577
- }
578
- )
579
- for child in element:
580
- visit(child, [*ancestors, tag_name], path)
581
-
582
- visit(root, [], "")
583
- return issues
584
-
585
-
586
- def build_iconpark_icon_type_hint(icon_type: str, supported_icon_types: set[str]) -> str:
587
- close_matches = get_close_matches(icon_type, sorted(supported_icon_types), n=3, cutoff=0.58)
588
- if close_matches:
589
- return (
590
- "iconType must exist in iconpark-index.json. Did you mean "
591
- + ", ".join(f'"{match}"' for match in close_matches)
592
- + "?"
593
- )
594
- return "iconType must exist in iconpark-index.json. Use scripts/iconpark_tool.py to search supported icons."
595
-
596
-
597
- def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
598
- supported_icon_types: set[str] | None = None
599
- issues: list[dict[str, Any]] = []
600
-
601
- def direct_child(element: ET.Element, local_name: str) -> ET.Element | None:
602
- return next((child for child in element if xml_local_name(child.tag) == local_name), None)
603
-
604
- def is_transparent_color(color: str) -> bool:
605
- normalized = re.sub(r"\s+", "", color).lower()
606
- if normalized == "transparent":
607
- return True
608
- rgba_match = re.fullmatch(r"rgba\([^,]+,[^,]+,[^,]+,([0-9.]+)\)", normalized)
609
- if not rgba_match:
610
- return False
611
- try:
612
- return float(rgba_match.group(1)) <= 0
613
- except ValueError:
614
- return False
615
-
616
- def append_missing_fill_color_issue(current_path: str) -> None:
617
- issues.append(
618
- {
619
- "level": "error",
620
- "code": "icon_missing_fill_color",
621
- "tag": "icon",
622
- "path": current_path,
623
- "message": f"<icon> must set explicit non-transparent fillColor for visual visibility at {current_path}",
624
- "hint": 'Add <fill><fillColor color="rgba(R, G, B, 1)"/></fill> inside <icon>. This is a visual lint rule, not an SXSD required field.',
625
- }
626
- )
627
-
628
- def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
629
- nonlocal supported_icon_types
630
- if should_skip_sxsd_subtree(element, ancestors):
631
- return
632
-
633
- tag_name = xml_local_name(element.tag)
634
- current_path = f"{path}/{tag_name}" if path else tag_name
635
- if tag_name == "icon":
636
- icon_type = element.attrib.get("iconType")
637
- if icon_type is not None:
638
- if supported_icon_types is None:
639
- supported_icon_types = load_iconpark_icon_types()
640
- if icon_type not in supported_icon_types:
641
- issues.append(
642
- {
643
- "level": "error",
644
- "code": "iconpark_unsupported_icon_type",
645
- "tag": "icon",
646
- "attr": "iconType",
647
- "iconType": icon_type,
648
- "path": current_path,
649
- "message": f'unsupported iconpark iconType "{icon_type}" at {current_path}',
650
- "hint": build_iconpark_icon_type_hint(icon_type, supported_icon_types),
651
- }
652
- )
653
- fill = direct_child(element, "fill")
654
- fill_color = direct_child(fill, "fillColor") if fill is not None else None
655
- color = fill_color.attrib.get("color") if fill_color is not None else None
656
- if not color:
657
- append_missing_fill_color_issue(current_path)
658
- elif is_transparent_color(color):
659
- issues.append(
660
- {
661
- "level": "error",
662
- "code": "icon_transparent_fill_color",
663
- "tag": "icon",
664
- "attr": "fillColor",
665
- "path": current_path,
666
- "color": color,
667
- "message": f'<icon> fillColor must not be transparent for visual visibility at {current_path}: "{color}"',
668
- "hint": 'Use an opaque visible color, for example <fillColor color="rgba(37, 99, 235, 1)"/>.',
669
- }
670
- )
671
- for child in element:
672
- visit(child, [*ancestors, tag_name], current_path)
673
-
674
- visit(root, [], "")
675
- return issues
676
-
677
-
678
- def extract_error_context(xml: str, line: int | None, column: int | None, radius: int = 40) -> str | None:
679
- if line is None or column is None:
680
- return None
681
- lines = xml.splitlines()
682
- if line < 1 or line > len(lines):
683
- return None
684
- source_line = lines[line - 1]
685
- start = max(column - radius, 0)
686
- end = min(column + radius, len(source_line))
687
- return source_line[start:end].strip()
688
-
689
-
690
- def build_xml_error_issue(error: ET.ParseError, xml: str) -> dict[str, Any]:
691
- line, column = getattr(error, "position", (None, None))
692
- return {
693
- "level": "error",
694
- "code": "xml_not_well_formed",
695
- "message": f"XML is not well-formed: {error}",
696
- "line": line,
697
- "column": column,
698
- "context": extract_error_context(xml, line, column),
699
- "hint": (
700
- "Escape raw user text before placing it in XML. In text nodes and attribute values, bare & must be "
701
- "written as &amp;. In text nodes, write < as &lt; and > as &gt;. For attribute URLs, use a=1&amp;b=2."
702
- ),
703
- }
704
-
705
-
706
- def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
707
- namespace_map: dict[str, str] = {}
708
- pending_declarations: list[tuple[str, str | None]] = []
709
- declarations_by_element: list[list[tuple[str, str | None]]] = []
710
- element_stack: list[str] = []
711
- issues: list[dict[str, Any]] = []
712
-
713
- parser = expat.ParserCreate(namespace_separator="|")
714
- parser.namespace_prefixes = True
715
-
716
- def handle_namespace_decl(prefix: str | None, namespace: str) -> None:
717
- normalized_prefix = prefix or ""
718
- previous_namespace = namespace_map.get(normalized_prefix)
719
- namespace_map[normalized_prefix] = namespace
720
- pending_declarations.append((normalized_prefix, previous_namespace))
721
-
722
- def handle_start_element(name: str, _attrs: dict[str, str]) -> None:
723
- declarations_by_element.append(pending_declarations.copy())
724
- pending_declarations.clear()
725
- name_parts = name.rsplit("|", 2)
726
- if len(name_parts) == 3:
727
- _namespace, local_name, prefix = name_parts
728
- element_name = f"{prefix}:{local_name}"
729
- else:
730
- prefix = ""
731
- local_name = name_parts[-1]
732
- element_name = local_name
733
- element_stack.append(element_name)
734
- if not prefix:
735
- return
736
-
737
- actual_namespace = namespace_map.get(prefix)
738
- if actual_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES:
739
- return
740
- path = "/".join(element_stack)
741
- issues.append(
742
- {
743
- "level": "error",
744
- "code": "sml_prefixed_tag",
745
- "tag": element_name,
746
- "namespace": actual_namespace,
747
- "path": path,
748
- "line": parser.CurrentLineNumber,
749
- "column": parser.CurrentColumnNumber,
750
- "message": f"SML tag <{element_name}> must not use a namespace prefix at {path}",
751
- "hint": (
752
- f'Use <{local_name}> under the default namespace '
753
- f'<{local_name} xmlns="{SML_NAMESPACE}">, or use an unprefixed SML tag.'
754
- ),
755
- }
756
- )
757
-
758
- def handle_end_element(_name: str) -> None:
759
- for prefix, previous_namespace in reversed(declarations_by_element.pop()):
760
- if previous_namespace is None:
761
- namespace_map.pop(prefix, None)
762
- else:
763
- namespace_map[prefix] = previous_namespace
764
- element_stack.pop()
765
-
766
- parser.StartNamespaceDeclHandler = handle_namespace_decl
767
- parser.StartElementHandler = handle_start_element
768
- parser.EndElementHandler = handle_end_element
769
- parser.Parse(xml, True)
770
- return issues
771
-
772
-
773
- def parse_xml_root(xml: str) -> tuple[ET.Element | None, dict[str, Any] | None]:
774
- try:
775
- root = ET.fromstring(xml)
776
- except ET.ParseError as error:
777
- return None, build_xml_error_issue(error, xml)
778
-
779
- root_name = xml_local_name(root.tag)
780
- if root_name not in {"presentation", "slide"}:
781
- fail("input must contain a <presentation> or <slide> root")
782
- return root, None
783
-
784
-
785
- def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
786
- _, xml_error = parse_xml_root(xml)
787
- return xml_error
788
-
789
-
790
- def serialize_slide_for_layout(slide_root: ET.Element) -> str:
791
- slide_copy = copy.deepcopy(slide_root)
792
- for element in slide_copy.iter():
793
- if not isinstance(element.tag, str):
794
- continue
795
- element.tag = xml_local_name(element.tag)
796
- attributes = {
797
- xml_local_name(attribute_name): value
798
- for attribute_name, value in element.attrib.items()
799
- }
800
- element.attrib.clear()
801
- element.attrib.update(attributes)
802
- return ET.tostring(slide_copy, encoding="unicode")
803
-
804
-
805
- def parse_presentation(root: ET.Element) -> dict[str, Any]:
806
- root_name = xml_local_name(root.tag)
807
- if root_name == "slide":
808
- slide_roots = [root]
809
- width = 960
810
- height = 540
811
- elif root_name == "presentation":
812
- slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
813
- width = int(float(root.attrib.get("width", 960)))
814
- height = int(float(root.attrib.get("height", 540)))
815
- else:
816
- fail("input must contain a <presentation> or <slide> root")
817
- return {
818
- "width": width,
819
- "height": height,
820
- "slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
821
- "slide_roots": slide_roots,
822
- }
823
-
824
-
825
- def build_source_xml_paths(slide_xml: str, slide_number: int) -> dict[str, list[str]]:
826
- root = ET.fromstring(slide_xml)
827
- data = next((child for child in root if xml_local_name(child.tag) == "data"), None)
828
- paths: dict[str, list[str]] = {}
829
- if data is None:
830
- return paths
831
- counts: dict[str, int] = {}
832
- for child in data:
833
- kind = xml_local_name(child.tag)
834
- counts[kind] = counts.get(kind, 0) + 1
835
- paths.setdefault(kind, []).append(
836
- f"slide[{slide_number}]/data/{kind}[{counts[kind]}]"
837
- )
838
- return paths
839
-
840
-
841
- def attach_source_xml_paths(
842
- elements: list[dict[str, Any]], source_paths: dict[str, list[str]]
843
- ) -> None:
844
- offsets: dict[str, int] = {}
845
- for element in elements:
846
- kind = element["kind"]
847
- source_kind_index = element.get("_source_kind_index")
848
- offset = (
849
- source_kind_index - 1
850
- if isinstance(source_kind_index, int) and source_kind_index > 0
851
- else offsets.get(kind, 0)
852
- )
853
- kind_paths = source_paths.get(kind, [])
854
- if offset < len(kind_paths):
855
- element["xml_path"] = kind_paths[offset]
856
- element["_ref"] = kind_paths[offset]
857
- offsets[kind] = max(offsets.get(kind, 0), offset + 1)
858
-
859
-
860
- def extract_source_id_elements(slide_xml: str, slide_number: int) -> list[dict[str, Any]]:
861
- root = ET.fromstring(slide_xml)
862
- elements: list[dict[str, Any]] = []
863
- root_path = f"slide[{slide_number}]"
864
-
865
- def walk(parent: ET.Element, parent_path: str) -> None:
866
- child_counts: dict[str, int] = {}
867
- for child in parent:
868
- kind = xml_local_name(child.tag)
869
- child_counts[kind] = child_counts.get(kind, 0) + 1
870
- xml_path = (
871
- f"{parent_path}/data"
872
- if parent is root and kind == "data"
873
- else f"{parent_path}/{kind}[{child_counts[kind]}]"
874
- )
875
- source_id = extract_attribute(
876
- ET.tostring(child, encoding="unicode").split(">", 1)[0], "id"
877
- )
878
- if source_id:
879
- elements.append(
880
- {
881
- "id": source_id,
882
- "_source_id": source_id,
883
- "kind": kind,
884
- "type": child.attrib.get("type") or kind,
885
- "xml_path": xml_path,
886
- "_ref": xml_path,
887
- "_slide_number": slide_number,
888
- }
889
- )
890
- walk(child, xml_path)
891
-
892
- walk(root, root_path)
893
- return elements
894
-
895
-
896
- def element_ref(element: dict[str, Any]) -> str:
897
- ref = element.get("_ref") or element.get("xml_path")
898
- if isinstance(ref, str) and ref:
899
- return ref
900
- # Low-level detector tests and external callers may pass already-extracted objects.
901
- # The lint_xml pipeline always attaches the source path before issue detection.
902
- fallback = element.get("id")
903
- if isinstance(fallback, str) and fallback:
904
- return fallback
905
- raise AssertionError("lint element must have a source xml path or fallback id")
906
-
907
-
908
- def source_element_id(element: dict[str, Any]) -> str | None:
909
- value = element.get("_source_id")
910
- return value if isinstance(value, str) and value else None
911
-
912
-
913
- def element_label(element: dict[str, Any]) -> str:
914
- return source_element_id(element) or element_ref(element)
915
-
916
-
917
- def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
918
- elements: list[dict[str, Any]] = []
919
- source_kind_counts: dict[str, int] = {}
920
-
921
- for match in re.finditer(r"<(shape|img|table|chart|whiteboard|embed)\b([^>]*)>", slide_xml):
922
- kind, attrs = match.group(1), match.group(2)
923
- source_kind_counts[kind] = source_kind_counts.get(kind, 0) + 1
924
- source_kind_index = source_kind_counts[kind]
925
- is_self_closing = attrs.rstrip().endswith("/")
926
- content = ""
927
- if kind in {"shape", "table"} and not is_self_closing:
928
- close_index = slide_xml.find(f"</{kind}>", match.end())
929
- if close_index != -1:
930
- content = slide_xml[match.end() : close_index]
931
-
932
- source_id = extract_attribute(attrs, "id") or None
933
- element_id = source_id or f"{kind}-{len(elements) + 1}"
934
- x = extract_numeric_attribute(attrs, "topLeftX")
935
- y = extract_numeric_attribute(attrs, "topLeftY")
936
- width = extract_numeric_attribute(attrs, "width")
937
- height = extract_numeric_attribute(attrs, "height")
938
- rotation = extract_numeric_attribute(attrs, "rotation") or 0
939
- alpha = extract_numeric_attribute(attrs, "alpha")
940
- table_layouts: dict[str, dict[str, Any] | None] = {}
941
- if kind == "table":
942
- width, table_layouts["width"] = resolve_table_dimension(
943
- content, width, extract_table_column_sizes, DEFAULT_TABLE_COLUMN_WIDTH
944
- )
945
- height, table_layouts["height"] = resolve_table_dimension(
946
- content, height, extract_table_row_sizes, DEFAULT_TABLE_ROW_HEIGHT
947
- )
948
- if all(value is not None for value in [x, y, width, height]):
949
- element = {
950
- "id": element_id,
951
- "_source_id": source_id,
952
- "kind": kind,
953
- "type": extract_attribute(attrs, "type") or kind,
954
- "x": x,
955
- "y": y,
956
- "width": width,
957
- "height": height,
958
- "rotation": rotation,
959
- "alpha": alpha if alpha is not None else 1,
960
- "order": len(elements),
961
- "_source_kind_index": source_kind_index,
962
- }
963
- if kind == "table":
964
- element.update(
965
- {
966
- "declared_width": extract_numeric_attribute(attrs, "width"),
967
- "declared_height": extract_numeric_attribute(attrs, "height"),
968
- "table_layouts": table_layouts,
969
- }
970
- )
971
- if kind == "shape":
972
- content_attrs = extract_tag_attributes(content, "content")
973
- font_size = extract_numeric_attribute(content_attrs, "fontSize")
974
- if font_size is None:
975
- font_size = extract_numeric_attribute(attrs, "fontSize")
976
- font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
977
- text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
978
- bold = (
979
- extract_bool_attribute(content_attrs, "bold")
980
- or extract_bool_attribute(attrs, "bold")
981
- or detect_inline_style_presence(content, {"strong", "b"})
982
- or detect_any_span_bool_attribute(content, "bold")
983
- )
984
- italic = (
985
- extract_bool_attribute(content_attrs, "italic")
986
- or extract_bool_attribute(attrs, "italic")
987
- or detect_inline_style_presence(content, {"i", "em"})
988
- or detect_any_span_bool_attribute(content, "italic")
989
- )
990
- element.update(
991
- {
992
- "textType": extract_attribute(content_attrs, "textType"),
993
- "textAlign": extract_attribute(content_attrs, "textAlign"),
994
- "verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
995
- "vert": extract_attribute(attrs, "vert") or "horz",
996
- "autoFit": extract_attribute(content_attrs, "autoFit"),
997
- "wrap": extract_attribute(content_attrs, "wrap"),
998
- "lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
999
- "beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
1000
- "afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
1001
- "letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
1002
- "paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
1003
- "paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
1004
- "paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
1005
- "paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
1006
- "fontSize": font_size if font_size is not None else 16,
1007
- "fontFamily": font_family or "",
1008
- "color": text_color,
1009
- "textAlpha": effective_text_alpha(alpha, text_color),
1010
- "bold": bold,
1011
- "italic": italic,
1012
- "text": strip_xml_paragraphs(content),
1013
- "paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
1014
- }
1015
- )
1016
- elements.append(element)
1017
- return elements
1018
-
1019
-
1020
- def intersects(left: dict[str, Any], right: dict[str, Any]) -> bool:
1021
- return (
1022
- left["x"] < right["x"] + right["width"]
1023
- and left["x"] + left["width"] > right["x"]
1024
- and left["y"] < right["y"] + right["height"]
1025
- and left["y"] + left["height"] > right["y"]
1026
- )
1027
-
1028
-
1029
- def is_text_element(element: dict[str, Any]) -> bool:
1030
- return element["kind"] == "shape" and element["type"] == "text"
1031
-
1032
-
1033
- def is_whiteboard_element(element: dict[str, Any]) -> bool:
1034
- return element["kind"] == "whiteboard"
1035
-
1036
-
1037
- def has_text_content(element: dict[str, Any]) -> bool:
1038
- return bool(element.get("text"))
1039
-
1040
-
1041
- def is_vertical_text(element: dict[str, Any]) -> bool:
1042
- return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
1043
-
1044
-
1045
- def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1046
- issues: list[dict[str, Any]] = []
1047
- text_elements = [
1048
- element
1049
- for element in elements
1050
- if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
1051
- ]
1052
- image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
1053
- for text_element in text_elements:
1054
- for image_element in image_elements:
1055
- if image_element["order"] <= text_element["order"]:
1056
- continue
1057
- if is_vertical_text(text_element):
1058
- if intersects(image_element, text_element):
1059
- issues.append({
1060
- "level": "info",
1061
- "code": "image_may_cover_vertical_text",
1062
- "elements": [element_ref(image_element), element_ref(text_element)],
1063
- "message": (
1064
- f"image {element_label(image_element)} may cover vertical text shape "
1065
- f"{element_label(text_element)}"
1066
- ),
1067
- "hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
1068
- })
1069
- continue
1070
- text_visual_bbox = estimate_text_visual_bbox(text_element)
1071
- if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
1072
- issues.append({
1073
- "level": "error",
1074
- "code": "image_covers_text",
1075
- "elements": [element_ref(image_element), element_ref(text_element)],
1076
- "message": (
1077
- f"image {element_label(image_element)} covers text shape "
1078
- f"{element_label(text_element)}"
1079
- ),
1080
- "hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
1081
- })
1082
- return issues
1083
-
1084
-
1085
- def is_decorative_text(element: dict[str, Any]) -> bool:
1086
- text = element.get("text") or ""
1087
- return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
1088
-
1089
-
1090
- def normalize_text_for_overlap(text: str) -> str:
1091
- return re.sub(r"\s+", "", text)
1092
-
1093
-
1094
- SERIF_FONT_PATTERNS = {
1095
- "song", "songti", "simsun", "ming", "mincho",
1096
- "georgia", "times", "caslon", "garamond", "sourcehan-serif",
1097
- "source han serif", "思源宋体", "宋体", "明体",
1098
- }
1099
-
1100
- SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
1101
- "helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
1102
-
1103
-
1104
- def classify_font_family(font_family: str | None) -> str:
1105
- if not font_family:
1106
- return "sans"
1107
- family_lower = font_family.lower()
1108
- for marker in SANS_EXPLICIT_MARKERS:
1109
- if marker in family_lower:
1110
- return "sans"
1111
- serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
1112
- for pattern in serif_keywords:
1113
- if pattern in family_lower:
1114
- return "serif"
1115
- return "sans"
1116
-
1117
-
1118
- _FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
1119
- "sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
1120
- "serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
1121
- }
1122
-
1123
-
1124
- def estimate_character_width(
1125
- character: str,
1126
- font_size: int | float,
1127
- bold: bool = False,
1128
- font_family: str | None = None,
1129
- ) -> int | float:
1130
- bold_multiplier = 1.05 if bold else 1.0
1131
- if character.isspace():
1132
- return font_size * 0.33 * bold_multiplier
1133
- ea_width = unicodedata.east_asian_width(character)
1134
- if ea_width in {"F", "W"}:
1135
- return font_size * bold_multiplier
1136
- category = classify_font_family(font_family)
1137
- coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
1138
- if character.isupper():
1139
- return font_size * coeffs["upper"] * bold_multiplier
1140
- if character.islower():
1141
- return font_size * coeffs["lower"] * bold_multiplier
1142
- if character.isdigit():
1143
- return font_size * coeffs["digit"] * bold_multiplier
1144
- return font_size * coeffs["punct"] * bold_multiplier
1145
-
1146
-
1147
- def estimate_text_width(
1148
- text: str,
1149
- font_size: int | float,
1150
- letter_spacing: int | float = 0,
1151
- bold: bool = False,
1152
- font_family: str | None = None,
1153
- ) -> int | float:
1154
- base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
1155
- return base + max(len(text) - 1, 0) * letter_spacing
1156
-
1157
-
1158
- def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
1159
- if paragraph is not None:
1160
- value = paragraph.get("letterSpacing")
1161
- if isinstance(value, (int, float)):
1162
- return value
1163
- value = element.get("letterSpacing")
1164
- return value if isinstance(value, (int, float)) else 0
1165
-
1166
-
1167
- def text_wrap_width_tolerance() -> int | float:
1168
- return TEXT_WRAP_WIDTH_TOLERANCE_PX
1169
-
1170
-
1171
- def text_height_overflow_tolerance() -> int | float:
1172
- return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
1173
-
1174
-
1175
- def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
1176
- return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
1177
-
1178
-
1179
- def is_short_metric_text(text: str) -> bool:
1180
- compact = re.sub(r"\s+", "", text)
1181
- if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
1182
- return False
1183
- if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
1184
- return True
1185
- if re.search(r"[,.,+\-–—/%%]", compact) is None:
1186
- return False
1187
- return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
1188
-
1189
-
1190
- def is_single_line_visual_candidate(
1191
- element: dict[str, Any],
1192
- paragraph: dict[str, Any] | None,
1193
- text: str,
1194
- logical_width: int | float,
1195
- effective_width: int | float,
1196
- ) -> bool:
1197
- if "\n" in text or logical_width <= effective_width:
1198
- return False
1199
- if is_short_metric_text(text):
1200
- return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
1201
-
1202
- text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
1203
- compact_len = len(re.sub(r"\s+", "", text))
1204
- if text_align == "center" and compact_len <= 32:
1205
- return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
1206
-
1207
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1208
- if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
1209
- return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
1210
- return False
1211
-
1212
-
1213
- def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
1214
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1215
- bold = element.get("bold", False)
1216
- font_family = element.get("fontFamily", "")
1217
- letter_spacing = resolve_letter_spacing(element)
1218
- paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
1219
- return max(
1220
- [estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
1221
- or [1]
1222
- )
1223
-
1224
-
1225
- def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
1226
- left_text = normalize_text_for_overlap(left.get("text") or "")
1227
- right_text = normalize_text_for_overlap(right.get("text") or "")
1228
- if not left_text or not right_text:
1229
- return False
1230
- if left_text == right_text or left_text in right_text or right_text in left_text:
1231
- return True
1232
- return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
1233
-
1234
-
1235
- def estimate_text_line_count_for_text(
1236
- element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
1237
- ) -> int:
1238
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1239
- bold = element.get("bold", False)
1240
- font_family = element.get("fontFamily", "")
1241
- letter_spacing = resolve_letter_spacing(element, paragraph)
1242
- available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
1243
- hard_lines = text.split("\n")
1244
- if not text:
1245
- return 0
1246
- line_count = 0
1247
- for hard_line in hard_lines:
1248
- if element.get("wrap") in {"false", "0"}:
1249
- line_count += 1
1250
- continue
1251
- logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
1252
- effective_width = available_width + text_wrap_width_tolerance()
1253
- if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
1254
- line_count += 1
1255
- continue
1256
- line_count += max(1, math.ceil(logical_width / effective_width))
1257
- return line_count
1258
-
1259
-
1260
- def estimate_text_line_count(element: dict[str, Any]) -> int:
1261
- return max(estimate_text_line_count_for_text(element, element["text"]), 1)
1262
-
1263
-
1264
- def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
1265
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1266
- if line_spacing is None:
1267
- return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
1268
- match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
1269
- if match is None:
1270
- return None
1271
- spacing_type, value = match.groups()
1272
- return font_size * float(value) if spacing_type == "multiple" else float(value)
1273
-
1274
-
1275
- def adjust_dense_body_line_height(
1276
- element: dict[str, Any],
1277
- line_spacing: str | None,
1278
- line_height: int | float,
1279
- paragraph_count: int,
1280
- ) -> int | float:
1281
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1282
- if paragraph_count < 4 or font_size > 14 or not line_spacing:
1283
- return line_height
1284
- match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
1285
- if match is None:
1286
- return line_height
1287
- return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
1288
-
1289
-
1290
- def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1291
- issues: list[dict[str, Any]] = []
1292
- for element in elements:
1293
- if not is_text_element(element) or not has_text_content(element):
1294
- continue
1295
- if has_explicit_height_auto_fit(element):
1296
- continue
1297
-
1298
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1299
- paragraphs = element.get("paragraphs") or [
1300
- {
1301
- "text": element["text"],
1302
- "lineSpacing": None,
1303
- "beforeLineSpacing": None,
1304
- "afterLineSpacing": None,
1305
- }
1306
- ]
1307
- line_count = 0
1308
- estimated_height = 0.0
1309
- line_heights: list[int | float] = []
1310
- for paragraph in paragraphs:
1311
- paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
1312
- if paragraph_line_count == 0:
1313
- continue
1314
- resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
1315
- line_height = estimate_text_line_height(element, resolved_line_spacing)
1316
- before_spacing = estimate_text_line_height(
1317
- element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
1318
- )
1319
- after_spacing = estimate_text_line_height(
1320
- element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
1321
- )
1322
- if line_height is None or before_spacing is None or after_spacing is None:
1323
- line_count = 0
1324
- break
1325
- line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
1326
- first_line_height = font_size if line_count == 0 else line_height
1327
- line_count += paragraph_line_count
1328
- line_heights.append(line_height)
1329
- estimated_height += (
1330
- before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
1331
- )
1332
- if line_count == 0:
1333
- continue
1334
- available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
1335
- overflow = estimated_height - available_height
1336
- if overflow <= text_height_overflow_tolerance():
1337
- continue
1338
-
1339
- is_background = is_background_decorative_text(element, elements)
1340
- if is_background:
1341
- level = "info"
1342
- else:
1343
- level = "error" if overflow > 10 else "warning"
1344
- message = (
1345
- f"text shape {element_label(element)} may overflow its own content box "
1346
- f'(estimated {estimated_height:g}px, available {available_height:g}px); '
1347
- 'consider setting content wrap="true" autoFit="normal-auto-fit"'
1348
- )
1349
- if is_background:
1350
- message += " (likely background decoration: large font, low alpha, underneath other text)"
1351
- issues.append(
1352
- {
1353
- "level": level,
1354
- "code": "text_may_overflow_shape",
1355
- "elements": [element_ref(element)],
1356
- "line_count": line_count,
1357
- "line_height": max(line_heights),
1358
- "estimated_height": estimated_height,
1359
- "available_height": available_height,
1360
- "overflow": overflow,
1361
- "message": message,
1362
- "hint": (
1363
- "Increase shape.height, reduce the text, or set content wrap=\"true\" "
1364
- "autoFit=\"normal-auto-fit\". "
1365
- "This is an estimate based on font size, line spacing, and wrapped line count."
1366
- ),
1367
- }
1368
- )
1369
- return issues
1370
-
1371
-
1372
- def is_background_decorative_text(
1373
- element: dict[str, Any], elements: list[dict[str, Any]]
1374
- ) -> bool:
1375
- if not is_ghost_text(element):
1376
- return False
1377
- for other in elements:
1378
- if other is element:
1379
- continue
1380
- if not is_text_element(other) or not has_text_content(other):
1381
- continue
1382
- foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
1383
- if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
1384
- continue
1385
- if other["order"] <= element["order"]:
1386
- continue
1387
- if intersects(element, other):
1388
- return True
1389
- return False
1390
-
1391
-
1392
- def is_ghost_text(element: dict[str, Any]) -> bool:
1393
- if not is_text_element(element) or not has_text_content(element):
1394
- return False
1395
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1396
- text_alpha = element.get("textAlpha", element.get("alpha", 1))
1397
- if not isinstance(text_alpha, (int, float)):
1398
- return False
1399
- if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
1400
- return True
1401
- return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
1402
-
1403
-
1404
- def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
1405
- if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
1406
- return None
1407
-
1408
- padding_left = element.get("paddingLeft", 0)
1409
- padding_right = element.get("paddingRight", 0)
1410
- padding_top = element.get("paddingTop", 0)
1411
- padding_bottom = element.get("paddingBottom", 0)
1412
- content_width = max(element["width"] - padding_left - padding_right, 0)
1413
- content_height = max(element["height"] - padding_top - padding_bottom, 0)
1414
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1415
- line_count = estimate_text_line_count(element)
1416
- estimated_width = max(1, estimate_text_max_line_width(element))
1417
- visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
1418
- visual_height = min(content_height, max(1, line_count * font_size * 1.2))
1419
- x = element["x"] + padding_left
1420
- if element.get("textAlign") == "center":
1421
- x += (content_width - visual_width) / 2
1422
- elif element.get("textAlign") == "right":
1423
- x += content_width - visual_width
1424
- y = element["y"] + padding_top
1425
- if element.get("verticalAlign") == "middle":
1426
- y += (content_height - visual_height) / 2
1427
- elif element.get("verticalAlign") == "bottom":
1428
- y += content_height - visual_height
1429
- return {
1430
- "x": x,
1431
- "y": y,
1432
- "width": visual_width,
1433
- "height": visual_height,
1434
- }
1435
-
1436
-
1437
- def intersection_area(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1438
- width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
1439
- height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
1440
- if width <= 0 or height <= 0:
1441
- return 0
1442
- return width * height
1443
-
1444
-
1445
- def intersection_height(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1446
- height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
1447
- return max(height, 0)
1448
-
1449
-
1450
- def intersection_width(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1451
- width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
1452
- return max(width, 0)
1453
-
1454
-
1455
- def element_area(element: dict[str, Any]) -> int | float:
1456
- return max(element["width"], 0) * max(element["height"], 0)
1457
-
1458
-
1459
- def contains(outer: dict[str, Any], inner: dict[str, Any], tolerance: int | float = 2) -> bool:
1460
- return (
1461
- inner["x"] >= outer["x"] - tolerance
1462
- and inner["y"] >= outer["y"] - tolerance
1463
- and inner["x"] + inner["width"] <= outer["x"] + outer["width"] + tolerance
1464
- and inner["y"] + inner["height"] <= outer["y"] + outer["height"] + tolerance
1465
- )
1466
-
1467
-
1468
- def is_bottom_layer_full_slide_whiteboard(
1469
- whiteboard: dict[str, Any], other: dict[str, Any], slide_width: int | float, slide_height: int | float
1470
- ) -> bool:
1471
- return (
1472
- whiteboard["order"] < other["order"]
1473
- and whiteboard["x"] <= 2
1474
- and whiteboard["y"] <= 2
1475
- and whiteboard["width"] >= slide_width - 4
1476
- and whiteboard["height"] >= slide_height - 4
1477
- )
1478
-
1479
-
1480
- def is_background_container_for_whiteboard(container: dict[str, Any], whiteboard: dict[str, Any]) -> bool:
1481
- if container["order"] > whiteboard["order"]:
1482
- return False
1483
- if is_text_element(container):
1484
- return False
1485
- return contains(container, whiteboard)
1486
-
1487
-
1488
- def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
1489
- if not (is_text_element(left) and is_text_element(right)):
1490
- return False
1491
- if not (has_text_content(left) and has_text_content(right)):
1492
- return True
1493
- top, bottom = sorted([left, right], key=lambda element: element["y"])
1494
- top_type = top.get("textType")
1495
- bottom_type = bottom.get("textType")
1496
- allowed_pairs = {
1497
- ("title", "sub-headline"),
1498
- ("title", None),
1499
- ("headline", "headline"),
1500
- ("headline", None),
1501
- }
1502
- if (top_type, bottom_type) not in allowed_pairs:
1503
- return False
1504
- same_column = abs(top["x"] - bottom["x"]) <= 4
1505
- vertical_offset = bottom["y"] - top["y"]
1506
- top_font_size = float(top.get("fontSize", 16))
1507
- return same_column and vertical_offset >= top_font_size * 0.75
1508
-
1509
-
1510
- def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str, Any]) -> bool:
1511
- if not (is_text_element(left) and is_text_element(right)):
1512
- return False
1513
- if not (has_text_content(left) and has_text_content(right)):
1514
- return False
1515
- if is_ghost_text(left) or is_ghost_text(right):
1516
- return False
1517
- if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
1518
- return False
1519
-
1520
- source, target = sorted([left, right], key=lambda element: element["x"])
1521
- if source["x"] == target["x"]:
1522
- return False
1523
- wrap_enabled = source.get("wrap") not in {"false", "0"}
1524
- has_horizontal_gap = source["x"] + source["width"] <= target["x"]
1525
- if wrap_enabled and has_horizontal_gap:
1526
- return False
1527
- if source.get("autoFit") == "normal-auto-fit":
1528
- return False
1529
- if source.get("textAlign") in {"center", "right"}:
1530
- return False
1531
-
1532
- font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
1533
- padding_left = source.get("paddingLeft", 0)
1534
- padding_right = source.get("paddingRight", 0)
1535
- available_width = max(source["width"] - padding_left - padding_right, 1)
1536
- visual_width = estimate_text_max_line_width(source)
1537
- overflow_width = visual_width - available_width
1538
- min_overflow = max(font_size * 1.5, available_width * 0.08)
1539
- if overflow_width < min_overflow:
1540
- return False
1541
-
1542
- intrusion_width = source["x"] + padding_left + visual_width - target["x"]
1543
- min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
1544
- if intrusion_width < min_intrusion:
1545
- return False
1546
-
1547
- vertical_overlap = intersection_height(source, target)
1548
- min_vertical_overlap = min(source["height"], target["height"]) * 0.40
1549
- return vertical_overlap >= min_vertical_overlap
1550
-
1551
-
1552
- def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
1553
- source, target = sorted([left, right], key=lambda element: element["x"])
1554
- padding_left = source.get("paddingLeft", 0)
1555
- visual_width = estimate_text_max_line_width(source)
1556
- source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
1557
- width = intersection_width(source_visual_bbox, target)
1558
- height = intersection_height(source_visual_bbox, target)
1559
- return {
1560
- "intersection_width": round(width, 3),
1561
- "intersection_height": round(height, 3),
1562
- "intersection_area": round(width * height, 3),
1563
- }
1564
-
1565
-
1566
- def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
1567
- if is_text_element(left) and not has_text_content(left):
1568
- return False
1569
- if is_text_element(right) and not has_text_content(right):
1570
- return False
1571
- if is_ghost_text(left) or is_ghost_text(right):
1572
- return False
1573
- if is_template_text_stack(left, right):
1574
- return False
1575
- if is_text_element(left) and is_text_element(right):
1576
- if is_similar_text_overlay(left, right):
1577
- return False
1578
- left_visual = estimate_text_visual_bbox(left)
1579
- right_visual = estimate_text_visual_bbox(right)
1580
- if left_visual is None or right_visual is None:
1581
- return False
1582
- overlap_area = intersection_area(left_visual, right_visual)
1583
- if overlap_area <= 0:
1584
- return False
1585
- smaller_area = min(
1586
- left_visual["width"] * left_visual["height"],
1587
- right_visual["width"] * right_visual["height"],
1588
- )
1589
- return smaller_area > 0 and overlap_area / smaller_area >= 0.30
1590
- return False
1591
-
1592
-
1593
- def build_whiteboard_external_overlap_issue(
1594
- whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
1595
- ) -> dict[str, Any]:
1596
- element_refs = [detail["element"] for detail in overlap_details]
1597
- return {
1598
- "level": "warning",
1599
- "code": "whiteboard_external_overlap",
1600
- "elements": [element_ref(whiteboard), *element_refs],
1601
- "message": (
1602
- f"whiteboard {element_label(whiteboard)} overlaps {len(element_refs)} "
1603
- "sibling elements across its boundary"
1604
- ),
1605
- "hint": (
1606
- "Treat this as a static whiteboard container-bbox risk, not final visual proof. "
1607
- "After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
1608
- "the final authority because XML readback does not include whiteboard SVG/Mermaid internals."
1609
- ),
1610
- "overlaps": overlap_details,
1611
- }
1612
-
1613
-
1614
- def should_report_whiteboard_overlap(
1615
- whiteboard: dict[str, Any],
1616
- other: dict[str, Any],
1617
- slide_width: int | float,
1618
- slide_height: int | float,
1619
- ) -> dict[str, Any] | None:
1620
- if other is whiteboard or not intersects(whiteboard, other):
1621
- return None
1622
- if is_ghost_text(other):
1623
- return None
1624
- if contains(whiteboard, other):
1625
- return None
1626
- if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
1627
- return None
1628
- if is_background_container_for_whiteboard(other, whiteboard):
1629
- return None
1630
-
1631
- overlap_width = intersection_width(whiteboard, other)
1632
- overlap_height = intersection_height(whiteboard, other)
1633
- if overlap_width < 8 or overlap_height < 8:
1634
- return None
1635
-
1636
- other_area = element_area(other)
1637
- if other_area <= 0:
1638
- return None
1639
- overlap_area = overlap_width * overlap_height
1640
- overlap_ratio = overlap_area / other_area
1641
- if overlap_ratio < 0.15:
1642
- return None
1643
-
1644
- return {
1645
- "element": element_ref(other),
1646
- "kind": other["kind"],
1647
- "type": other.get("type"),
1648
- "overlap_width": overlap_width,
1649
- "overlap_height": overlap_height,
1650
- "target_overlap_ratio": round(overlap_ratio, 3),
1651
- }
1652
-
1653
-
1654
- def prune_contained_text_overlap_details(
1655
- overlap_details: list[dict[str, Any]], elements_by_ref: dict[str, dict[str, Any]]
1656
- ) -> list[dict[str, Any]]:
1657
- pruned: list[dict[str, Any]] = []
1658
- for detail in overlap_details:
1659
- element = elements_by_ref[detail["element"]]
1660
- if is_text_element(element):
1661
- has_reported_container = any(
1662
- detail["element"] != other_detail["element"]
1663
- and not is_text_element(elements_by_ref[other_detail["element"]])
1664
- and contains(elements_by_ref[other_detail["element"]], element)
1665
- for other_detail in overlap_details
1666
- )
1667
- if has_reported_container:
1668
- continue
1669
- pruned.append(detail)
1670
- return pruned
1671
-
1672
-
1673
- def detect_whiteboard_external_overlaps(
1674
- elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1675
- ) -> list[dict[str, Any]]:
1676
- issues: list[dict[str, Any]] = []
1677
- elements_by_ref = {element_ref(element): element for element in elements}
1678
- for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
1679
- overlap_details = [
1680
- detail
1681
- for element in elements
1682
- if (
1683
- detail := should_report_whiteboard_overlap(
1684
- whiteboard,
1685
- element,
1686
- slide_width,
1687
- slide_height,
1688
- )
1689
- )
1690
- is not None
1691
- ]
1692
- overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_ref)
1693
- if overlap_details:
1694
- issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
1695
- return issues
1696
-
1697
-
1698
- def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
1699
- bbox = {key: element[key] for key in ("x", "y", "width", "height")}
1700
- if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
1701
- return bbox
1702
- rotation = element["rotation"]
1703
- if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
1704
- rotation = 0
1705
- rotation %= 360
1706
- if math.isclose(rotation, 0, abs_tol=1e-9):
1707
- return bbox
1708
- radians = math.radians(rotation)
1709
- sine = abs(math.sin(radians))
1710
- cosine = abs(math.cos(radians))
1711
- sine = 0 if math.isclose(sine, 0, abs_tol=1e-12) else sine
1712
- cosine = 0 if math.isclose(cosine, 0, abs_tol=1e-12) else cosine
1713
- rotated_width = element["width"] * cosine + element["height"] * sine
1714
- rotated_height = element["width"] * sine + element["height"] * cosine
1715
- return {
1716
- "x": element["x"] - (rotated_width - element["width"]) / 2,
1717
- "y": element["y"] - (rotated_height - element["height"]) / 2,
1718
- "width": rotated_width,
1719
- "height": rotated_height,
1720
- }
1721
-
1722
-
1723
- def detect_elements_out_of_canvas(
1724
- elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1725
- ) -> list[dict[str, Any]]:
1726
- issues: list[dict[str, Any]] = []
1727
- for element in (
1728
- element
1729
- for element in elements
1730
- if element["kind"] in {"table", "chart"}
1731
- or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
1732
- ):
1733
- bbox = element_canvas_bbox(element)
1734
- overflow = {
1735
- "left": max(-bbox["x"], 0),
1736
- "top": max(-bbox["y"], 0),
1737
- "right": max(bbox["x"] + bbox["width"] - slide_width, 0),
1738
- "bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
1739
- }
1740
- overflow_details = [
1741
- f"{side} by {amount:g}px"
1742
- for side, amount in overflow.items()
1743
- if amount > CANVAS_OVERFLOW_TOLERANCE
1744
- ]
1745
- if not overflow_details:
1746
- continue
1747
- issues.append(
1748
- {
1749
- "level": "error",
1750
- "code": f'{element["kind"]}_out_of_canvas',
1751
- "elements": [element_ref(element)],
1752
- "canvas": {"width": slide_width, "height": slide_height},
1753
- "bbox": bbox,
1754
- "overflow": overflow,
1755
- "message": (
1756
- f'{element["kind"]} {element_label(element)} exceeds the {slide_width:g}x{slide_height:g} canvas '
1757
- f'({", ".join(overflow_details)})'
1758
- ),
1759
- "hint": (
1760
- "Move the table inside the canvas, reduce table.width/table.height, or split the table across "
1761
- "slides."
1762
- if element["kind"] == "table"
1763
- else f'Move the {element["kind"]} inside the canvas or reduce its width/height.'
1764
- ),
1765
- }
1766
- )
1767
- return issues
1768
-
1769
-
1770
- def extract_table_column_sizes(table_xml: str) -> list[int | float | None]:
1771
- sizes: list[int | float | None] = []
1772
- for match in re.finditer(r"<col\b([^>]*)/?>", table_xml):
1773
- attrs = match.group(1)
1774
- span = extract_numeric_attribute(attrs, "span") or 1
1775
- span_count = int(span) if math.isfinite(span) and span > 0 and float(span).is_integer() else 1
1776
- sizes.extend([extract_numeric_attribute(attrs, "width")] * span_count)
1777
- return sizes
1778
-
1779
-
1780
- def extract_table_row_sizes(table_xml: str) -> list[int | float | None]:
1781
- return [extract_numeric_attribute(match.group(1), "height") for match in re.finditer(r"<tr\b([^>]*)>", table_xml)]
1782
-
1783
-
1784
- def resolve_table_dimension(
1785
- table_xml: str,
1786
- declared_size: int | float | None,
1787
- extract_sizes: Any,
1788
- default_size: int | float,
1789
- ) -> tuple[int | float | None, dict[str, Any] | None]:
1790
- input_sizes = extract_sizes(table_xml)
1791
- if not input_sizes:
1792
- return declared_size, None
1793
- layout = solve_weighted_min_layout(
1794
- input_sizes, default_size, declared_size if is_filled_size(declared_size) else None
1795
- )
1796
- return layout["actual_size"], layout
1797
-
1798
-
1799
- def format_size(size: int | float) -> str:
1800
- return f"{size:g}"
1801
-
1802
-
1803
- def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1804
- issues: list[dict[str, Any]] = []
1805
- dimensions = {
1806
- "width": ("col", "column widths"),
1807
- "height": ("tr", "row heights"),
1808
- }
1809
- for table in (element for element in elements if element["kind"] == "table"):
1810
- for dimension, (child_tag, child_description) in dimensions.items():
1811
- target_size = table[f"declared_{dimension}"]
1812
- if not is_filled_size(target_size):
1813
- continue
1814
- layout = table["table_layouts"][dimension]
1815
- if layout is None:
1816
- continue
1817
- actual_size = layout["actual_size"]
1818
- if math.isclose(actual_size, target_size, rel_tol=1e-9, abs_tol=1e-9):
1819
- continue
1820
- issues.append(
1821
- {
1822
- "level": "info",
1823
- "code": "table_resolved_size_mismatch",
1824
- "elements": [element_ref(table)],
1825
- "dimension": dimension,
1826
- "declared_size": target_size,
1827
- "resolved_size": actual_size,
1828
- "resolved_sizes": layout["final_sizes"],
1829
- "message": (
1830
- f'table {element_label(table)} declares {dimension}={format_size(target_size)}px, but its '
1831
- f"{child_description} resolve to {format_size(actual_size)}px"
1832
- ),
1833
- "hint": (
1834
- f"Set table.{dimension} to {format_size(actual_size)}px, or adjust <{child_tag}> sizes "
1835
- f"so their resolved total matches {format_size(target_size)}px."
1836
- ),
1837
- }
1838
- )
1839
- return issues
1840
-
1841
-
1842
- def segment_intersects_rect(
1843
- x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
1844
- ) -> bool:
1845
- """True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
1846
- left = rect["x"]
1847
- top = rect["y"]
1848
- right = rect["x"] + rect["width"]
1849
- bottom = rect["y"] + rect["height"]
1850
- if right <= left or bottom <= top:
1851
- return False
1852
- dx = x2 - x1
1853
- dy = y2 - y1
1854
- if dx == 0 and dy == 0:
1855
- return left <= x1 <= right and top <= y1 <= bottom
1856
- t_enter, t_exit = 0.0, 1.0
1857
- for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
1858
- if delta == 0:
1859
- if distance < 0:
1860
- return False
1861
- continue
1862
- t = distance / delta
1863
- if delta < 0:
1864
- t_enter = max(t_enter, t)
1865
- else:
1866
- t_exit = min(t_exit, t)
1867
- if t_enter > t_exit:
1868
- return False
1869
- return True
1870
-
1871
-
1872
- def line_text_graze_margin(text_element: dict[str, Any]) -> float:
1873
- font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
1874
- return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
1875
-
1876
-
1877
- def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
1878
- width = rect["width"] - 2 * margin
1879
- height = rect["height"] - 2 * margin
1880
- if width <= 0 or height <= 0:
1881
- return None
1882
- return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
1883
-
1884
-
1885
- def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
1886
- if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
1887
- return False
1888
- if not is_text_element(text_element) or not has_text_content(text_element):
1889
- return False
1890
- if is_ghost_text(text_element) or is_decorative_text(text_element):
1891
- return False
1892
- glyph_bbox = estimate_text_visual_bbox(text_element)
1893
- if glyph_bbox is None:
1894
- return False
1895
- # Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
1896
- # frame is exempt; only a line that actually cuts through the letterforms is a crossing.
1897
- target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
1898
- if target is None:
1899
- return False
1900
- return segment_intersects_rect(
1901
- line["startX"], line["startY"], line["endX"], line["endY"], target
1902
- )
1903
-
1904
-
1905
- def detect_line_text_crossings(
1906
- slide_xml: str, elements: list[dict[str, Any]], slide_number: int
1907
- ) -> list[dict[str, Any]]:
1908
- lines = extract_line_elements(slide_xml)
1909
- if not lines:
1910
- return []
1911
- attach_source_xml_paths(lines, build_source_xml_paths(slide_xml, slide_number))
1912
- text_elements = [element for element in elements if is_text_element(element)]
1913
- issues: list[dict[str, Any]] = []
1914
- for line in lines:
1915
- for text_element in text_elements:
1916
- if not line_crosses_text(line, text_element):
1917
- continue
1918
- issues.append(
1919
- {
1920
- "level": "error",
1921
- "code": "bbox_overlap",
1922
- "elements": [element_ref(line), element_ref(text_element)],
1923
- "message": (
1924
- f"line {element_label(line)} crosses text {element_label(text_element)}"
1925
- ),
1926
- "hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
1927
- }
1928
- )
1929
- return issues
1930
-
1931
-
1932
- def lint_slide(
1933
- slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
1934
- ) -> dict[str, Any]:
1935
- elements = extract_elements(slide_xml)
1936
- attach_source_xml_paths(elements, build_source_xml_paths(slide_xml, slide_number))
1937
- issues: list[dict[str, Any]] = [
1938
- *detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
1939
- *detect_elements_out_of_canvas(elements, slide_width, slide_height),
1940
- *detect_table_layout_size_mismatches(elements),
1941
- *detect_text_may_overflow_shapes(elements),
1942
- *detect_image_text_occlusions(elements),
1943
- *detect_line_text_crossings(slide_xml, elements, slide_number),
1944
- ]
1945
-
1946
- for index, left in enumerate(elements):
1947
- for right in elements[index + 1 :]:
1948
- horizontal_overflow = should_flag_horizontal_text_overflow(left, right)
1949
- if not horizontal_overflow and (not intersects(left, right) or not should_flag_overlap(left, right)):
1950
- continue
1951
- issues.append(
1952
- {
1953
- "level": "error",
1954
- "code": "bbox_overlap",
1955
- "elements": [element_ref(left), element_ref(right)],
1956
- "message": f"{element_label(left)} overlaps {element_label(right)}",
1957
- "hint": "Move or resize the elements so their visual bounds no longer intersect.",
1958
- **(
1959
- {"measurement": horizontal_text_overflow_measurement(left, right)}
1960
- if horizontal_overflow
1961
- else {}
1962
- ),
1963
- }
1964
- )
1965
-
1966
- return {
1967
- "slide_number": slide_number,
1968
- "element_count": len(elements),
1969
- "elements": elements,
1970
- "issues": issues,
1971
- }
1972
-
1973
-
1974
-
1975
- MIN_CONTAINER_WIDTH = 140
1976
- MIN_CONTAINER_HEIGHT = 160
1977
- MIN_SHORT_CARD_HEIGHT = 80
1978
- MIN_CONTAINER_AREA = 20_000
1979
- MIN_CONTENT_COVERAGE_RATIO = 0.15
1980
- MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
1981
- MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
1982
- SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
1983
- MIN_SIMILAR_SHORT_CARD_COUNT = 2
1984
- LARGE_VISUAL_CHILD_RATIO = 0.35
1985
- LAYOUT_PANEL_SPAN_RATIO = 0.90
1986
- IMAGE_OVERLAY_MATCH_RATIO = 0.90
1987
- DENSITY_CONTAINMENT_TOLERANCE = 8
1988
-
1989
-
1990
- def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1991
- left = max(element["x"], container["x"])
1992
- top = max(element["y"], container["y"])
1993
- right = min(element["x"] + element["width"], container["x"] + container["width"])
1994
- bottom = min(element["y"] + element["height"], container["y"] + container["height"])
1995
- if right <= left or bottom <= top:
1996
- return None
1997
- return {"x": left, "y": top, "width": right - left, "height": bottom - top}
1998
-
1999
-
2000
- def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
2001
- x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
2002
- area = 0
2003
- for left, right in zip(x_coordinates, x_coordinates[1:]):
2004
- intervals = sorted(
2005
- (rect["y"], rect["y"] + rect["height"])
2006
- for rect in rectangles
2007
- if rect["x"] < right and rect["x"] + rect["width"] > left
2008
- )
2009
- covered_height = 0
2010
- interval_end: int | float | None = None
2011
- for top, bottom in intervals:
2012
- if interval_end is None:
2013
- covered_height += bottom - top
2014
- interval_end = bottom
2015
- elif bottom > interval_end:
2016
- covered_height += bottom - max(top, interval_end)
2017
- interval_end = bottom
2018
- area += (right - left) * covered_height
2019
- return area
2020
-
2021
-
2022
- def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
2023
- return sum(
2024
- other is not element
2025
- and is_visually_rendered(other)
2026
- and other["kind"] == "shape"
2027
- and other["type"] == "rect"
2028
- and other["width"] >= MIN_CONTAINER_WIDTH
2029
- and other["height"] >= MIN_SHORT_CARD_HEIGHT
2030
- and element_area(other) >= MIN_CONTAINER_AREA
2031
- and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
2032
- <= SHORT_CARD_SIZE_TOLERANCE_RATIO
2033
- and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
2034
- <= SHORT_CARD_SIZE_TOLERANCE_RATIO
2035
- for other in elements
2036
- ) >= MIN_SIMILAR_SHORT_CARD_COUNT
2037
-
2038
-
2039
- def is_layout_container(
2040
- element: dict[str, Any],
2041
- slide_width: int | float,
2042
- slide_height: int | float,
2043
- elements: list[dict[str, Any]] | None = None,
2044
- ) -> bool:
2045
- has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
2046
- elements is not None
2047
- and element["height"] >= MIN_SHORT_CARD_HEIGHT
2048
- and has_similar_short_card_peer(element, elements)
2049
- )
2050
- return (
2051
- element["kind"] == "shape"
2052
- and element["type"] == "rect"
2053
- and is_visually_rendered(element)
2054
- and element["width"] >= MIN_CONTAINER_WIDTH
2055
- and has_supported_height
2056
- and element_area(element) >= MIN_CONTAINER_AREA
2057
- and not (
2058
- element["x"] <= 2
2059
- and element["y"] <= 2
2060
- and element["width"] >= slide_width - 4
2061
- and element["height"] >= slide_height - 4
2062
- )
2063
- )
2064
-
2065
-
2066
- def is_edge_spanning_layout_panel(
2067
- element: dict[str, Any], slide_width: int | float, slide_height: int | float
2068
- ) -> bool:
2069
- touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
2070
- touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
2071
- return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
2072
- touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
2073
- )
2074
-
2075
-
2076
- def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
2077
- container_area = element_area(container)
2078
- return any(
2079
- element["kind"] == "img"
2080
- and is_visually_rendered(element)
2081
- and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
2082
- for element in elements
2083
- )
2084
-
2085
-
2086
- def is_nested_in_layout_panel(
2087
- container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
2088
- ) -> bool:
2089
- return any(
2090
- element is not container
2091
- and element["kind"] == "shape"
2092
- and element["type"] == "rect"
2093
- and is_visually_rendered(element)
2094
- and is_edge_spanning_layout_panel(element, slide_width, slide_height)
2095
- and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2096
- for element in elements
2097
- )
2098
-
2099
-
2100
- def extract_density_elements(slide_xml: str, slide_number: int = 1) -> list[dict[str, Any]]:
2101
- elements = extract_elements(slide_xml)
2102
- source_paths = build_source_xml_paths(slide_xml, slide_number)
2103
- attach_source_xml_paths(elements, source_paths)
2104
- shape_elements_by_index = {
2105
- element["_source_kind_index"]: element
2106
- for element in elements
2107
- if element["kind"] == "shape"
2108
- }
2109
- root = ET.fromstring(slide_xml)
2110
- shape_index = 0
2111
- for node in root.iter():
2112
- if xml_local_name(node.tag) != "shape":
2113
- continue
2114
- shape_index += 1
2115
- element = shape_elements_by_index.get(shape_index)
2116
- if element is None:
2117
- continue
2118
- content_node = next(
2119
- (child for child in node if xml_local_name(child.tag) == "content"),
2120
- None,
2121
- )
2122
- paragraphs = (
2123
- [
2124
- " ".join("".join(paragraph.itertext()).split())
2125
- for paragraph in content_node.iter()
2126
- if xml_local_name(paragraph.tag) == "p"
2127
- ]
2128
- if content_node is not None
2129
- else []
2130
- )
2131
- raw_font_size = (
2132
- content_node.attrib.get("fontSize") if content_node is not None else None
2133
- ) or node.attrib.get("fontSize")
2134
- try:
2135
- base_font_size = float(raw_font_size or 16)
2136
- except ValueError:
2137
- base_font_size = 16.0
2138
- element.update(
2139
- {
2140
- "textType": content_node.attrib.get("textType") if content_node is not None else None,
2141
- "textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
2142
- "autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
2143
- "fontSize": base_font_size,
2144
- "text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
2145
- }
2146
- )
2147
- if not has_text_content(element):
2148
- continue
2149
- declared_font_sizes = []
2150
- for descendant in node.iter():
2151
- raw_declared_font_size = descendant.attrib.get("fontSize")
2152
- if raw_declared_font_size is None:
2153
- continue
2154
- try:
2155
- declared_font_sizes.append(float(raw_declared_font_size))
2156
- except ValueError:
2157
- continue
2158
- if declared_font_sizes:
2159
- element["fontSize"] = max(declared_font_sizes)
2160
- for source_kind_index, match in enumerate(
2161
- re.finditer(r"<icon\b([^>]*)>", slide_xml), start=1
2162
- ):
2163
- attrs = match.group(1)
2164
- source_id = extract_attribute(attrs, "id") or None
2165
- x = extract_numeric_attribute(attrs, "topLeftX")
2166
- y = extract_numeric_attribute(attrs, "topLeftY")
2167
- width = extract_numeric_attribute(attrs, "width")
2168
- height = extract_numeric_attribute(attrs, "height")
2169
- if any(value is None for value in (x, y, width, height)):
2170
- continue
2171
- icon_alpha = extract_numeric_attribute(attrs, "alpha")
2172
- elements.append(
2173
- {
2174
- "id": source_id or f"icon-{len(elements) + 1}",
2175
- "_source_id": source_id,
2176
- "kind": "icon",
2177
- "type": "icon",
2178
- "x": x,
2179
- "y": y,
2180
- "width": width,
2181
- "height": height,
2182
- "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2183
- "alpha": icon_alpha if icon_alpha is not None else 1,
2184
- "order": len(elements),
2185
- "_source_kind_index": source_kind_index,
2186
- }
2187
- )
2188
- for source_kind_index, match in enumerate(
2189
- re.finditer(r"<polyline\b([^>]*)>", slide_xml), start=1
2190
- ):
2191
- attrs = match.group(1)
2192
- x = extract_numeric_attribute(attrs, "topLeftX")
2193
- y = extract_numeric_attribute(attrs, "topLeftY")
2194
- width = extract_numeric_attribute(attrs, "width")
2195
- height = extract_numeric_attribute(attrs, "height")
2196
- if any(value is None for value in (x, y, width, height)):
2197
- continue
2198
- polyline_alpha = extract_numeric_attribute(attrs, "alpha")
2199
- source_id = extract_attribute(attrs, "id") or None
2200
- elements.append(
2201
- {
2202
- "id": source_id or f"polyline-{len(elements) + 1}",
2203
- "_source_id": source_id,
2204
- "kind": "polyline",
2205
- "type": "polyline",
2206
- "x": x,
2207
- "y": y,
2208
- "width": width,
2209
- "height": height,
2210
- "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2211
- "alpha": polyline_alpha if polyline_alpha is not None else 1,
2212
- "order": len(elements),
2213
- "_source_kind_index": source_kind_index,
2214
- }
2215
- )
2216
- for line_element in extract_line_elements(slide_xml):
2217
- line_element["order"] = len(elements)
2218
- elements.append(line_element)
2219
- attach_source_xml_paths(elements, source_paths)
2220
- for element in elements:
2221
- element["_slide_number"] = slide_number
2222
- return elements
2223
-
2224
-
2225
- def is_visually_rendered(element: dict[str, Any]) -> bool:
2226
- return element.get("alpha", 1) > 0
2227
-
2228
-
2229
- def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
2230
- if not is_visually_rendered(element):
2231
- return None
2232
- if is_text_element(element):
2233
- estimated = estimate_text_visual_bbox(element)
2234
- return clipped_bbox(estimated, container) if estimated else None
2235
- return clipped_bbox(element, container)
2236
-
2237
-
2238
- def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
2239
- if container["kind"] != "shape" or not has_text_content(container):
2240
- return None
2241
- text_proxy = {**container, "type": "text"}
2242
- estimated = estimate_text_visual_bbox(text_proxy)
2243
- return clipped_bbox(estimated, container) if estimated else None
2244
-
2245
-
2246
- def slide_content_visual_bbox(
2247
- element: dict[str, Any], slide_bbox: dict[str, int | float]
2248
- ) -> dict[str, int | float] | None:
2249
- if not is_visually_rendered(element):
2250
- return None
2251
- if is_text_element(element):
2252
- estimated = estimate_text_visual_bbox(element)
2253
- return clipped_bbox(estimated, slide_bbox) if estimated else None
2254
- if element["kind"] == "shape" and has_text_content(element):
2255
- estimated = own_text_visual_bbox(element)
2256
- return clipped_bbox(estimated, slide_bbox) if estimated else None
2257
- if element["kind"] == "line":
2258
- # a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
2259
- # treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
2260
- return clipped_bbox(line_stroke_bbox(element), slide_bbox)
2261
- if element["kind"] in {"img", "chart", "table", "whiteboard", "embed", "icon", "polyline"}:
2262
- return clipped_bbox(element, slide_bbox)
2263
- return None
2264
-
2265
-
2266
- def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
2267
- return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
2268
-
2269
-
2270
- def is_slide_content_present(
2271
- element: dict[str, Any], slide_bbox: dict[str, int | float]
2272
- ) -> bool:
2273
- # Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
2274
- # *anything* rendered here", not the richer "counts toward meaningful content density" bar
2275
- # that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
2276
- # decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
2277
- # count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
2278
- # instead of maintaining an allow-list that silently treats unlisted kinds as blank.
2279
- if not is_visually_rendered(element):
2280
- return False
2281
- if (
2282
- element["kind"] == "shape"
2283
- and element["type"] == "rect"
2284
- and not has_text_content(element)
2285
- and element["x"] <= 2
2286
- and element["y"] <= 2
2287
- and element["width"] >= slide_bbox["width"] - 4
2288
- and element["height"] >= slide_bbox["height"] - 4
2289
- ):
2290
- # A full-canvas plain rect is a background panel, not content -- same reasoning as
2291
- # is_layout_container's existing background exclusion. A slide with nothing else on it
2292
- # is still effectively blank.
2293
- return False
2294
- bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
2295
- return clipped_bbox(bbox, slide_bbox) is not None
2296
-
2297
-
2298
- def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
2299
- if element["kind"] not in {"img", "chart", "table", "whiteboard", "embed"}:
2300
- return False
2301
- if not is_visually_rendered(element):
2302
- return False
2303
- return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
2304
-
2305
-
2306
- def detect_sparse_container_content(
2307
- elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2308
- ) -> list[dict[str, Any]]:
2309
- issues: list[dict[str, Any]] = []
2310
- for container in (
2311
- element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
2312
- ):
2313
- if (
2314
- is_edge_spanning_layout_panel(container, slide_width, slide_height)
2315
- or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
2316
- or has_matching_image_overlay(container, elements)
2317
- ):
2318
- continue
2319
- children = [
2320
- element
2321
- for element in elements
2322
- if element is not container
2323
- and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2324
- ]
2325
- if any(is_large_visual_child(child, container) for child in children):
2326
- continue
2327
- own_text_bbox = own_text_visual_bbox(container)
2328
- rectangles = ([own_text_bbox] if own_text_bbox else []) + [
2329
- bbox for child in children if (bbox := visual_bbox(child, container)) is not None
2330
- ]
2331
- content_area = rectangle_union_area(rectangles) if rectangles else 0
2332
- coverage_ratio = content_area / element_area(container)
2333
- if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
2334
- continue
2335
- issues.append(
2336
- {
2337
- "level": "warning",
2338
- "code": "sparse_container_content",
2339
- "target": {
2340
- "slide_number": slide_number,
2341
- **(
2342
- {"container_id": source_element_id(container)}
2343
- if source_element_id(container) is not None
2344
- else {}
2345
- ),
2346
- "container_xml_path": element_ref(container),
2347
- "container_type": container["type"],
2348
- "bbox": {key: container[key] for key in ("x", "y", "width", "height")},
2349
- },
2350
- "rule": {
2351
- "name": "large_container_visible_content_coverage",
2352
- "threshold": MIN_CONTENT_COVERAGE_RATIO,
2353
- "comparison": "content_coverage_ratio < threshold",
2354
- },
2355
- "measurement": {
2356
- "container_area": element_area(container),
2357
- "visible_content_area": round(content_area, 3),
2358
- "content_coverage_ratio": round(coverage_ratio, 3),
2359
- "content_element_count": len(children) + (1 if own_text_bbox else 0),
2360
- },
2361
- "elements": [
2362
- element_ref(container),
2363
- *[element_ref(child) for child in children],
2364
- ],
2365
- }
2366
- )
2367
- return issues
2368
-
2369
-
2370
- def detect_sparse_slide_content(
2371
- elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2372
- ) -> list[dict[str, Any]]:
2373
- slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2374
- content = [
2375
- (element, bbox)
2376
- for element in elements
2377
- if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
2378
- ]
2379
- if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
2380
- return []
2381
- content_area = rectangle_union_area([bbox for _, bbox in content])
2382
- slide_area = slide_width * slide_height
2383
- coverage_ratio = content_area / slide_area
2384
- if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
2385
- return []
2386
- return [
2387
- {
2388
- "level": "warning",
2389
- "code": "sparse_slide_content",
2390
- "target": {
2391
- "slide_number": slide_number,
2392
- "bbox": slide_bbox,
2393
- },
2394
- "rule": {
2395
- "name": "slide_visible_content_coverage",
2396
- "threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
2397
- "comparison": "content_coverage_ratio < threshold",
2398
- },
2399
- "measurement": {
2400
- "slide_area": slide_area,
2401
- "visible_content_area": round(content_area, 3),
2402
- "content_coverage_ratio": round(coverage_ratio, 3),
2403
- "content_element_count": len(content),
2404
- },
2405
- "elements": [element_ref(element) for element, _ in content],
2406
- }
2407
- ]
2408
-
2409
-
2410
- def detect_blank_slide(
2411
- elements: list[dict[str, Any]],
2412
- slide_number: int,
2413
- slide_width: int | float,
2414
- slide_height: int | float,
2415
- ) -> list[dict[str, Any]]:
2416
- slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2417
- visible_elements = [
2418
- element for element in elements if is_slide_content_present(element, slide_bbox)
2419
- ]
2420
- if visible_elements:
2421
- return []
2422
- return [
2423
- {
2424
- "level": "error",
2425
- "code": "blank_slide",
2426
- "schema_version": "2.0",
2427
- "target": {"slide_number": slide_number},
2428
- "rule": {
2429
- "name": "slide_has_visible_content",
2430
- "comparison": "visible_element_count == 0",
2431
- },
2432
- "measurement": {
2433
- "visible_element_count": 0,
2434
- "declared_element_count": len(elements),
2435
- },
2436
- "elements": [element_ref(element) for element in elements],
2437
- "message": "slide has no visible content beyond empty layout shapes",
2438
- "hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
2439
- }
2440
- ]
2441
-
2442
-
2443
- def detect_duplicate_element_ids(
2444
- elements: list[dict[str, Any]], *, cross_slide_only: bool = False
2445
- ) -> list[dict[str, Any]]:
2446
- elements_by_source_id: dict[str, list[dict[str, Any]]] = {}
2447
- for element in elements:
2448
- source_id = source_element_id(element)
2449
- if source_id is not None:
2450
- elements_by_source_id.setdefault(source_id, []).append(element)
2451
- return [
2452
- {
2453
- "level": "error",
2454
- "code": "duplicate_element_id",
2455
- "elements": [element_ref(element) for element in duplicates],
2456
- "measurement": {
2457
- "element_id": source_id,
2458
- "duplicate_count": len(duplicates),
2459
- },
2460
- "message": f'element id "{source_id}" is used by {len(duplicates)} elements',
2461
- "hint": (
2462
- "Do not invent replacement IDs. For newly authored elements, remove the id attribute. "
2463
- "When updating read-back XML, keep the server ID on the original element only and remove it "
2464
- "from copied or new elements."
2465
- ),
2466
- }
2467
- for source_id, duplicates in elements_by_source_id.items()
2468
- if len(duplicates) > 1
2469
- and (
2470
- not cross_slide_only
2471
- or len({element.get("_slide_number") for element in duplicates}) > 1
2472
- )
2473
- ]
2474
-
2475
-
2476
- RULE_METADATA: dict[str, dict[str, Any]] = {
2477
- "xml_not_well_formed": {
2478
- "name": "xml_is_well_formed",
2479
- "comparison": "xml_parse_error == false",
2480
- },
2481
- "sml_prefixed_tag": {
2482
- "name": "sml_uses_default_namespace",
2483
- "comparison": "prefixed_sml_tag_count == 0",
2484
- },
2485
- "sxsd_unsupported_tag": {
2486
- "name": "tag_is_supported_by_slides_xml_schema",
2487
- "comparison": "unsupported_tag_count == 0",
2488
- },
2489
- "sxsd_unsupported_attr": {
2490
- "name": "attribute_is_supported_by_slides_xml_schema",
2491
- "comparison": "unsupported_attribute_count == 0",
2492
- },
2493
- "icon_missing_fill_color": {
2494
- "name": "icon_has_visible_fill_color",
2495
- "comparison": "fill_color_present == true",
2496
- },
2497
- "icon_transparent_fill_color": {
2498
- "name": "icon_has_visible_fill_color",
2499
- "comparison": "fill_alpha > 0",
2500
- },
2501
- "iconpark_unsupported_icon_type": {
2502
- "name": "iconpark_type_is_supported",
2503
- "comparison": "icon_type in iconpark_index",
2504
- },
2505
- "bbox_overlap": {
2506
- "name": "text_visual_bounds_do_not_overlap",
2507
- "comparison": "intersection_area == 0",
2508
- },
2509
- "text_may_overflow_shape": {
2510
- "name": "estimated_text_fits_declared_shape",
2511
- "comparison": "estimated_height <= available_height",
2512
- },
2513
- "whiteboard_external_overlap": {
2514
- "name": "whiteboard_does_not_cross_sibling_content",
2515
- "comparison": "external_overlap_count == 0",
2516
- },
2517
- "image_covers_text": {
2518
- "name": "image_does_not_cover_text",
2519
- "comparison": "intersection_area == 0",
2520
- },
2521
- "image_may_cover_vertical_text": {
2522
- "name": "image_vertical_text_occlusion_requires_review",
2523
- "comparison": "intersection_area == 0",
2524
- },
2525
- "table_resolved_size_mismatch": {
2526
- "name": "table_declared_size_matches_resolved_grid",
2527
- "comparison": "declared_size == resolved_size",
2528
- },
2529
- "blank_slide": {
2530
- "name": "slide_has_visible_content",
2531
- "comparison": "visible_element_count > 0",
2532
- },
2533
- "duplicate_element_id": {
2534
- "name": "element_ids_are_unique",
2535
- "comparison": "duplicate_count == 0",
2536
- },
2537
- }
2538
-
2539
-
2540
- def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
2541
- if issue.get("rule"):
2542
- return {**issue["rule"], "id": issue["code"]}
2543
- if issue["code"].endswith("_out_of_canvas"):
2544
- return {
2545
- "id": issue["code"],
2546
- "name": "element_stays_within_slide_canvas",
2547
- "comparison": "max(left, top, right, bottom overflow) == 0",
2548
- }
2549
- return {
2550
- "id": issue["code"],
2551
- **RULE_METADATA.get(
2552
- issue["code"],
2553
- {"name": issue["code"], "comparison": "violation_count == 0"},
2554
- ),
2555
- }
2556
-
2557
-
2558
- def issue_measurement(
2559
- issue: dict[str, Any], elements_by_ref: dict[str, dict[str, Any]]
2560
- ) -> dict[str, Any]:
2561
- if issue.get("measurement") is not None:
2562
- return issue["measurement"]
2563
- if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
2564
- left = elements_by_ref.get(issue["elements"][0])
2565
- right = elements_by_ref.get(issue["elements"][1])
2566
- if left and right:
2567
- left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
2568
- right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
2569
- width = intersection_width(left_box, right_box)
2570
- height = intersection_height(left_box, right_box)
2571
- return {
2572
- "intersection_width": round(width, 3),
2573
- "intersection_height": round(height, 3),
2574
- "intersection_area": round(width * height, 3),
2575
- }
2576
- if issue["code"].endswith("_out_of_canvas"):
2577
- return {
2578
- "canvas": issue.get("canvas"),
2579
- "bbox": issue.get("bbox"),
2580
- "overflow": issue.get("overflow"),
2581
- }
2582
- measurement_keys = (
2583
- "line",
2584
- "column",
2585
- "tag",
2586
- "attr",
2587
- "iconType",
2588
- "line_count",
2589
- "line_height",
2590
- "estimated_height",
2591
- "available_height",
2592
- "overflow",
2593
- "dimension",
2594
- "declared_size",
2595
- "resolved_size",
2596
- "resolved_sizes",
2597
- "overlaps",
2598
- )
2599
- measured = {key: issue[key] for key in measurement_keys if key in issue}
2600
- return measured or {"violation_count": 1}
2601
-
2602
-
2603
- def related_object(element: dict[str, Any]) -> dict[str, Any]:
2604
- related = {
2605
- "kind": element["kind"],
2606
- "type": element["type"],
2607
- }
2608
- bbox_keys = ("x", "y", "width", "height")
2609
- if all(key in element for key in bbox_keys):
2610
- related["bbox"] = {key: element[key] for key in bbox_keys}
2611
- if source_element_id(element) is not None:
2612
- related["element_id"] = source_element_id(element)
2613
- if element.get("xml_path"):
2614
- related["xml_path"] = element["xml_path"]
2615
- return related
2616
-
2617
-
2618
- def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
2619
- elements: list[dict[str, Any]] = []
2620
- for source_kind_index, match in enumerate(
2621
- re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml), start=1
2622
- ):
2623
- attrs = match.group(1)
2624
- source_id = extract_attribute(attrs, "id") or None
2625
- start_x = extract_numeric_attribute(attrs, "startX")
2626
- start_y = extract_numeric_attribute(attrs, "startY")
2627
- end_x = extract_numeric_attribute(attrs, "endX")
2628
- end_y = extract_numeric_attribute(attrs, "endY")
2629
- if any(value is None for value in (start_x, start_y, end_x, end_y)):
2630
- continue
2631
- line_alpha = extract_numeric_attribute(attrs, "alpha")
2632
- base_alpha = line_alpha if line_alpha is not None else 1
2633
- border_alpha = 1
2634
- if match.group(2) != "/":
2635
- close_index = slide_xml.find("</line>", match.end())
2636
- body = slide_xml[match.end() : close_index] if close_index != -1 else ""
2637
- border_attrs = extract_tag_attributes(body, "border")
2638
- color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
2639
- if isinstance(color_alpha, (int, float)):
2640
- border_alpha = color_alpha
2641
- elements.append(
2642
- {
2643
- "id": source_id or f"line-{len(elements) + 1}",
2644
- "_source_id": source_id,
2645
- "kind": "line",
2646
- "type": "line",
2647
- "x": min(start_x, end_x),
2648
- "y": min(start_y, end_y),
2649
- "width": abs(end_x - start_x),
2650
- "height": abs(end_y - start_y),
2651
- "startX": start_x,
2652
- "startY": start_y,
2653
- "endX": end_x,
2654
- "endY": end_y,
2655
- "rotation": 0,
2656
- "alpha": base_alpha * border_alpha,
2657
- "order": len(elements),
2658
- "_source_kind_index": source_kind_index,
2659
- }
2660
- )
2661
- return elements
2662
-
2663
-
2664
- def normalize_issue(
2665
- issue: dict[str, Any],
2666
- slide_number: int | None,
2667
- elements_by_ref: dict[str, dict[str, Any]],
2668
- ) -> dict[str, Any]:
2669
- normalized = dict(issue)
2670
- element_refs = list(dict.fromkeys(normalized.get("elements", [])))
2671
- resolved_elements = [
2672
- elements_by_ref[element_ref]
2673
- for element_ref in element_refs
2674
- if element_ref in elements_by_ref
2675
- ]
2676
- element_locators = [
2677
- source_element_id(elements_by_ref[element_ref]) or element_ref
2678
- if element_ref in elements_by_ref
2679
- else element_ref
2680
- for element_ref in element_refs
2681
- ]
2682
- element_ids = [
2683
- source_id
2684
- for element in resolved_elements
2685
- if (source_id := source_element_id(element)) is not None
2686
- ]
2687
- normalized["schema_version"] = "2.0"
2688
- normalized["elements"] = element_locators
2689
- normalized["element_ids"] = element_ids
2690
- normalized["target"] = {
2691
- **({"slide_number": slide_number} if slide_number is not None else {}),
2692
- **normalized.get("target", {}),
2693
- }
2694
- normalized["rule"] = issue_rule(normalized)
2695
- normalized["measurement"] = issue_measurement(issue, elements_by_ref)
2696
- normalized["related_objects"] = [related_object(element) for element in resolved_elements]
2697
- if normalized["code"] == "sparse_container_content":
2698
- ratio = normalized["measurement"]["content_coverage_ratio"]
2699
- threshold = normalized["rule"]["threshold"]
2700
- container_locator = (
2701
- normalized["target"].get("container_id")
2702
- or normalized["target"].get("container_xml_path")
2703
- or "unknown"
2704
- )
2705
- normalized.setdefault(
2706
- "message",
2707
- f"large card {container_locator} content coverage {ratio:.1%} is below {threshold:.1%}",
2708
- )
2709
- normalized.setdefault(
2710
- "hint",
2711
- "Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
2712
- )
2713
- elif normalized["code"] == "sparse_slide_content":
2714
- ratio = normalized["measurement"]["content_coverage_ratio"]
2715
- threshold = normalized["rule"]["threshold"]
2716
- normalized.setdefault(
2717
- "message",
2718
- f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
2719
- )
2720
- normalized.setdefault(
2721
- "hint",
2722
- "Review the rendered screenshot to decide whether the page is intentionally sparse.",
2723
- )
2724
- else:
2725
- normalized.setdefault("message", normalized["code"].replace("_", " "))
2726
- normalized.setdefault(
2727
- "hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
2728
- )
2729
- if any(related.get("xml_path") for related in normalized["related_objects"]):
2730
- hint = normalized["hint"]
2731
- if not hint.startswith(XML_PATH_HINT_PREFIX):
2732
- normalized["hint"] = f"{XML_PATH_HINT_PREFIX} {hint}"
2733
- return normalized
2734
-
2735
-
2736
- def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
2737
- if errors:
2738
- return "blocked"
2739
- if warnings:
2740
- return "needs_screenshot_review"
2741
- return "passed"
2742
-
2743
-
2744
- def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
2745
- if issue.get("code") == "sxsd_unsupported_declaration":
2746
- return False
2747
- if root_name == "slide":
2748
- return True
2749
- path = issue.get("path")
2750
- if not isinstance(path, str):
2751
- return False
2752
- if path.startswith("presentation/slide/"):
2753
- return True
2754
- return path == "presentation/slide" and (
2755
- issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
2756
- )
2757
-
2758
-
2759
- def build_result(
2760
- source_path: str | None,
2761
- slide_size: dict[str, int | float],
2762
- top_level_issues: list[dict[str, Any]],
2763
- slides: list[dict[str, Any]],
2764
- ) -> dict[str, Any]:
2765
- document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
2766
- document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
2767
- document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
2768
- error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
2769
- warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
2770
- info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
2771
- all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
2772
- all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
2773
- status = slide_status(all_errors, all_warnings)
2774
- result: dict[str, Any] = {
2775
- "schema_version": "2.0",
2776
- "tool": "xml_text_overlap_lint",
2777
- "file": source_path,
2778
- "slide_size": slide_size,
2779
- "summary": {
2780
- "slide_count": len(slides),
2781
- "error_count": error_count,
2782
- "warning_count": warning_count,
2783
- "info_count": info_count,
2784
- "status": status,
2785
- "release_ready": error_count == 0,
2786
- "screenshot_review_required": warning_count > 0,
2787
- },
2788
- "document": {
2789
- "errors": document_errors,
2790
- "warnings": document_warnings,
2791
- "infos": document_infos,
2792
- },
2793
- "slides": slides,
2794
- }
2795
- if top_level_issues:
2796
- result["issues"] = top_level_issues
2797
- return result
2798
-
2799
-
2800
- def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
2801
- root, xml_error = parse_xml_root(xml)
2802
- if xml_error:
2803
- issue = normalize_issue(xml_error, None, {})
2804
- return build_result(
2805
- source_path,
2806
- {"width": 960, "height": 540},
2807
- [issue],
2808
- [],
2809
- )
2810
- if root is None:
2811
- raise AssertionError("parse_xml_root must return a root or error")
2812
-
2813
- namespace_issues = validate_sml_tag_prefixes(xml)
2814
- root_name = xml_local_name(root.tag)
2815
- sxsd_issues = validate_sxsd_document(xml, root)
2816
- iconpark_issues = validate_iconpark_icon_types(root)
2817
- top_level_issues = [
2818
- normalize_issue(issue, None, {})
2819
- for issue in [
2820
- *namespace_issues,
2821
- *[
2822
- issue
2823
- for issue in sxsd_issues
2824
- if not is_slide_scoped_sxsd_issue(issue, root_name)
2825
- ],
2826
- *iconpark_issues,
2827
- ]
2828
- ]
2829
- if any(issue["level"] == "error" for issue in top_level_issues):
2830
- return build_result(
2831
- source_path,
2832
- {"width": 960, "height": 540},
2833
- top_level_issues,
2834
- [],
2835
- )
2836
-
2837
- presentation = parse_presentation(root)
2838
- slide_roots = presentation["slide_roots"]
2839
- slides: list[dict[str, Any]] = []
2840
- presentation_id_elements: list[dict[str, Any]] = []
2841
- presentation_elements_by_ref: dict[str, dict[str, Any]] = {}
2842
- for index, slide_xml in enumerate(presentation["slides"]):
2843
- slide_number = index + 1
2844
- slide_root = slide_roots[index]
2845
- slide_sxsd_issues = [
2846
- normalize_issue(issue, slide_number, {})
2847
- for issue in validate_sxsd_document(slide_xml, slide_root)
2848
- ]
2849
- slide_sxsd_errors = [
2850
- issue for issue in slide_sxsd_issues if issue["level"] == "error"
2851
- ]
2852
- if slide_sxsd_errors:
2853
- slide_sxsd_warnings = [
2854
- issue for issue in slide_sxsd_issues if issue["level"] == "warning"
2855
- ]
2856
- slides.append(
2857
- {
2858
- "slide_number": slide_number,
2859
- "status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
2860
- "element_count": 0,
2861
- "errors": slide_sxsd_errors,
2862
- "warnings": slide_sxsd_warnings,
2863
- "infos": [],
2864
- "issues": slide_sxsd_issues,
2865
- }
2866
- )
2867
- continue
2868
-
2869
- geometry = lint_slide(
2870
- slide_xml,
2871
- slide_number,
2872
- presentation["width"],
2873
- presentation["height"],
2874
- )
2875
- density_elements = extract_density_elements(slide_xml, slide_number)
2876
- id_elements = extract_source_id_elements(slide_xml, slide_number)
2877
- presentation_id_elements.extend(id_elements)
2878
- extra_elements = [
2879
- element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
2880
- ]
2881
- elements_by_ref = {
2882
- element_ref(element): element for element in density_elements
2883
- }
2884
- visible_element_count = len(elements_by_ref)
2885
- for element in id_elements:
2886
- elements_by_ref.setdefault(element_ref(element), element)
2887
- # geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
2888
- # selected inside lint_slide; prefer them so measurement/related_objects stay consistent
2889
- # with whatever actually triggered the issue, instead of density_elements' separate re-parse.
2890
- elements_by_ref.update(
2891
- {element_ref(element): element for element in geometry["elements"]}
2892
- )
2893
- presentation_elements_by_ref.update(
2894
- {
2895
- element_ref(element): elements_by_ref[element_ref(element)]
2896
- for element in id_elements
2897
- }
2898
- )
2899
- extra_overflow_issues = detect_elements_out_of_canvas(
2900
- extra_elements,
2901
- presentation["width"],
2902
- presentation["height"],
2903
- )
2904
- raw_issues = [
2905
- *geometry["issues"],
2906
- *extra_overflow_issues,
2907
- *detect_duplicate_element_ids(id_elements),
2908
- *detect_blank_slide(
2909
- density_elements,
2910
- slide_number,
2911
- presentation["width"],
2912
- presentation["height"],
2913
- ),
2914
- *detect_sparse_container_content(
2915
- density_elements,
2916
- slide_number,
2917
- presentation["width"],
2918
- presentation["height"],
2919
- ),
2920
- *detect_sparse_slide_content(
2921
- density_elements,
2922
- slide_number,
2923
- presentation["width"],
2924
- presentation["height"],
2925
- ),
2926
- ]
2927
- issues = [
2928
- *slide_sxsd_issues,
2929
- *[
2930
- normalize_issue(issue, slide_number, elements_by_ref)
2931
- for issue in raw_issues
2932
- ],
2933
- ]
2934
- errors = [issue for issue in issues if issue["level"] == "error"]
2935
- warnings = [issue for issue in issues if issue["level"] == "warning"]
2936
- infos = [issue for issue in issues if issue["level"] == "info"]
2937
- slides.append(
2938
- {
2939
- "slide_number": slide_number,
2940
- "status": slide_status(errors, warnings),
2941
- "element_count": visible_element_count,
2942
- "errors": errors,
2943
- "warnings": warnings,
2944
- "infos": infos,
2945
- "issues": issues,
2946
- }
2947
- )
2948
-
2949
- top_level_issues.extend(
2950
- normalize_issue(issue, None, presentation_elements_by_ref)
2951
- for issue in detect_duplicate_element_ids(
2952
- presentation_id_elements, cross_slide_only=True
2953
- )
2954
- )
2955
-
2956
- return build_result(
2957
- source_path,
2958
- {"width": presentation["width"], "height": presentation["height"]},
2959
- top_level_issues,
2960
- slides,
2961
- )
2962
-
2963
-
2964
- def print_usage() -> None:
2965
- print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
2966
-
2967
7
 
2968
- def run_cli(argv: list[str] | None = None) -> None:
2969
- options = parse_args(argv or sys.argv[1:])
2970
- if options.get("help") or options.get("--help"):
2971
- print_usage()
2972
- raise SystemExit(0)
2973
- if not options.get("input"):
2974
- print_usage()
2975
- fail("--input is required")
2976
- requested_path = options["input"]
2977
- resolved_path = Path(requested_path).resolve()
2978
- result = lint_xml(read_file(resolved_path), requested_path)
2979
- print(json.dumps(result, ensure_ascii=False, indent=2))
2980
- if result["summary"]["error_count"] > 0:
2981
- raise SystemExit(1)
8
+ from xml_lint import * # noqa: F401,F403
9
+ from xml_lint import XmlLayoutLintError, run_cli
2982
10
 
2983
11
 
2984
12
  if __name__ == "__main__":