@amaster.ai/pi-lark 0.1.7 → 0.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (215) hide show
  1. package/package.json +2 -2
  2. package/skills/lark-apps/SKILL.md +41 -6
  3. package/skills/lark-apps/references/lark-apps-cloud-dev.md +5 -4
  4. package/skills/lark-apps/references/lark-apps-create.md +6 -3
  5. package/skills/lark-apps/references/lark-apps-db.md +130 -2
  6. package/skills/lark-apps/references/lark-apps-get.md +1 -1
  7. package/skills/lark-apps/references/lark-apps-list.md +1 -1
  8. package/skills/lark-apps/references/lark-apps-local-dev.md +27 -1
  9. package/skills/lark-apps/references/lark-apps-release-create.md +1 -1
  10. package/skills/lark-apps/references/lark-apps-user-id-convert.md +63 -0
  11. package/skills/lark-base/SKILL.md +155 -159
  12. package/skills/lark-base/references/{lark-base-role-guide.md → lark-base-advanced-permission-and-role.md} +5 -5
  13. package/skills/lark-base/references/lark-base-app-block-data-config.md +122 -0
  14. package/skills/lark-base/references/lark-base-app.md +225 -0
  15. package/skills/lark-base/references/lark-base-cell-value.md +26 -19
  16. package/skills/lark-base/references/{dashboard-block-data-config.md → lark-base-dashboard-block-config.md} +37 -5
  17. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +18 -2
  18. package/skills/lark-base/references/lark-base-dashboard.md +25 -12
  19. package/skills/lark-base/references/lark-base-data-analysis-pandas.md +93 -0
  20. package/skills/lark-base/references/lark-base-data-analysis-python-stdlib.md +120 -0
  21. package/skills/lark-base/references/lark-base-data-query.md +8 -11
  22. package/skills/lark-base/references/lark-base-field-create.md +13 -45
  23. package/skills/lark-base/references/{formula-field-guide.md → lark-base-field-formula.md} +1 -1
  24. package/skills/lark-base/references/{lookup-field-guide.md → lark-base-field-lookup.md} +1 -1
  25. package/skills/lark-base/references/{lark-base-field-json.md → lark-base-field-schema.md} +18 -100
  26. package/skills/lark-base/references/lark-base-field-update.md +13 -51
  27. package/skills/lark-base/references/lark-base-filter-condition.md +19 -31
  28. package/skills/lark-base/references/lark-base-record-batch-create.md +5 -1
  29. package/skills/lark-base/references/lark-base-record-batch-update.md +5 -2
  30. package/skills/lark-base/references/lark-base-record-query-and-analysis-cloud-sop.md +145 -0
  31. package/skills/lark-base/references/lark-base-record-query-and-analysis-sop.md +233 -0
  32. package/skills/lark-base/references/{role-config.md → lark-base-role-config.md} +2 -2
  33. package/skills/lark-base/references/lark-base-view-set-filter.md +1 -1
  34. package/skills/lark-base/references/lark-base-workflow-schema.md +2 -2
  35. package/skills/lark-base/references/{lark-base-workflow-guide.md → lark-base-workflow.md} +1 -1
  36. package/skills/lark-calendar/SKILL.md +3 -1
  37. package/skills/lark-calendar/references/lark-calendar-create.md +4 -3
  38. package/skills/lark-doc/SKILL.md +26 -61
  39. package/skills/lark-doc/references/genres/business-analysis.md +30 -0
  40. package/skills/lark-doc/references/genres/data-report.md +32 -0
  41. package/skills/lark-doc/references/genres/email.md +38 -0
  42. package/skills/lark-doc/references/genres/execution-plan.md +27 -0
  43. package/skills/lark-doc/references/genres/formal-doc.md +37 -0
  44. package/skills/lark-doc/references/genres/meeting-minutes.md +24 -0
  45. package/skills/lark-doc/references/genres/memo-brief.md +25 -0
  46. package/skills/lark-doc/references/genres/official-redhead.md +73 -0
  47. package/skills/lark-doc/references/genres/prd.md +26 -0
  48. package/skills/lark-doc/references/genres/proposal.md +24 -0
  49. package/skills/lark-doc/references/genres/research-report.md +32 -0
  50. package/skills/lark-doc/references/genres/retrospective.md +25 -0
  51. package/skills/lark-doc/references/genres/route-consumer.md +37 -0
  52. package/skills/lark-doc/references/genres/route-creative.md +36 -0
  53. package/skills/lark-doc/references/genres/route-knowledge.md +39 -0
  54. package/skills/lark-doc/references/genres/route-marketing.md +40 -0
  55. package/skills/lark-doc/references/genres/route-media.md +36 -0
  56. package/skills/lark-doc/references/genres/route-opinion.md +38 -0
  57. package/skills/lark-doc/references/genres/route-personal-brand.md +36 -0
  58. package/skills/lark-doc/references/genres/route-platform.md +9 -0
  59. package/skills/lark-doc/references/genres/route-report.md +10 -0
  60. package/skills/lark-doc/references/genres/route-workplace.md +17 -0
  61. package/skills/lark-doc/references/genres/sop-tutorial.md +41 -0
  62. package/skills/lark-doc/references/genres/technical-doc.md +39 -0
  63. package/skills/lark-doc/references/genres/wechat.md +39 -0
  64. package/skills/lark-doc/references/genres/weekly-report.md +24 -0
  65. package/skills/lark-doc/references/genres/white-paper.md +32 -0
  66. package/skills/lark-doc/references/genres/xiaohongshu.md +38 -0
  67. package/skills/lark-doc/references/lark-doc-create-workflow.md +121 -0
  68. package/skills/lark-doc/references/lark-doc-create.md +22 -48
  69. package/skills/lark-doc/references/lark-doc-fetch.md +80 -92
  70. package/skills/lark-doc/references/lark-doc-history.md +16 -15
  71. package/skills/lark-doc/references/lark-doc-md.md +5 -1
  72. package/skills/lark-doc/references/lark-doc-media-download.md +2 -1
  73. package/skills/lark-doc/references/lark-doc-script.md +76 -0
  74. package/skills/lark-doc/references/lark-doc-update.md +73 -221
  75. package/skills/lark-doc/references/lark-doc-whiteboard.md +5 -9
  76. package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +17 -12
  77. package/skills/lark-doc/references/lark-doc-xml.md +38 -167
  78. package/skills/lark-drive/SKILL.md +11 -7
  79. package/skills/lark-drive/references/lark-drive-apply-permission.md +1 -1
  80. package/skills/lark-drive/references/lark-drive-copy.md +87 -0
  81. package/skills/lark-drive/references/lark-drive-download.md +29 -2
  82. package/skills/lark-drive/references/lark-drive-export.md +4 -0
  83. package/skills/lark-drive/references/lark-drive-member-remove.md +59 -0
  84. package/skills/lark-drive/references/lark-drive-preview.md +21 -2
  85. package/skills/lark-drive/references/lark-drive-push.md +5 -1
  86. package/skills/lark-drive/references/lark-drive-search.md +2 -0
  87. package/skills/lark-drive/references/lark-drive-task-result.md +3 -0
  88. package/skills/lark-drive/references/lark-drive-update-title.md +78 -0
  89. package/skills/lark-event/SKILL.md +7 -4
  90. package/skills/lark-event/references/lark-event-vc.md +8 -2
  91. package/skills/lark-im/SKILL.md +14 -9
  92. package/skills/lark-im/references/lark-im-chat-list.md +9 -2
  93. package/skills/lark-im/references/lark-im-chat-members-list.md +7 -4
  94. package/skills/lark-im/references/lark-im-chat-messages-list.md +10 -3
  95. package/skills/lark-im/references/lark-im-chat-search.md +9 -2
  96. package/skills/lark-im/references/lark-im-feed-group-list-item.md +2 -2
  97. package/skills/lark-im/references/lark-im-feed-group-list.md +2 -2
  98. package/skills/lark-im/references/lark-im-feed-shortcut-list.md +1 -1
  99. package/skills/lark-im/references/lark-im-flag-list.md +2 -2
  100. package/skills/lark-im/references/lark-im-message-enrichment.md +1 -1
  101. package/skills/lark-im/references/lark-im-messages-resources-download.md +19 -25
  102. package/skills/lark-im/references/lark-im-messages-search.md +4 -5
  103. package/skills/lark-im/references/lark-im-threads-messages-list.md +8 -4
  104. package/skills/lark-mail/references/lark-mail-triage.md +19 -4
  105. package/skills/lark-minutes/SKILL.md +12 -6
  106. package/skills/lark-minutes/references/lark-minutes-apply-permission.md +95 -0
  107. package/skills/lark-minutes/references/lark-minutes-detail.md +7 -6
  108. package/skills/lark-minutes/references/lark-minutes-download.md +4 -2
  109. package/skills/lark-minutes/references/lark-minutes-search.md +6 -7
  110. package/skills/lark-note/SKILL.md +13 -9
  111. package/skills/lark-note/references/lark-note-detail.md +5 -2
  112. package/skills/lark-note/references/lark-note-transcript.md +2 -0
  113. package/skills/lark-shared/SKILL.md +39 -3
  114. package/skills/lark-sheets/SKILL.md +83 -82
  115. package/skills/lark-sheets/references/lark-sheets-batch-update.md +13 -58
  116. package/skills/lark-sheets/references/lark-sheets-chart.md +2 -1
  117. package/skills/lark-sheets/references/lark-sheets-conditional-format.md +1 -1
  118. package/skills/lark-sheets/references/lark-sheets-range-operations.md +5 -5
  119. package/skills/lark-sheets/references/lark-sheets-read-data.md +80 -6
  120. package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +21 -10
  121. package/skills/lark-sheets/references/lark-sheets-styles-put.md +93 -0
  122. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +2 -2
  123. package/skills/lark-sheets/references/lark-sheets-workbook.md +4 -3
  124. package/skills/lark-sheets/references/lark-sheets-write-cells.md +40 -12
  125. package/skills/lark-sheets/scripts/lark_detect_subtables.py +593 -0
  126. package/skills/lark-sheets/scripts/lark_inspect_workbook.py +188 -0
  127. package/skills/lark-sheets/scripts/lark_profile_table.py +614 -0
  128. package/skills/lark-sheets/scripts/lark_sheet_range.py +176 -0
  129. package/skills/lark-sheets/scripts/lark_sheet_read_cli.py +184 -0
  130. package/skills/lark-sheets/scripts/sheets_df.py +21 -3
  131. package/skills/lark-slides/SKILL.md +64 -81
  132. package/skills/lark-slides/references/cli/lark-slides-add-slide.md +92 -0
  133. package/skills/lark-slides/references/cli/lark-slides-create.md +176 -0
  134. package/skills/lark-slides/references/cli/lark-slides-delete-slide.md +65 -0
  135. package/skills/lark-slides/references/cli/lark-slides-history.md +132 -0
  136. package/skills/lark-slides/references/cli/lark-slides-media-upload.md +103 -0
  137. package/skills/lark-slides/references/cli/lark-slides-replace-slide.md +259 -0
  138. package/skills/lark-slides/references/cli/lark-slides-screenshot.md +115 -0
  139. package/skills/lark-slides/references/cli/lark-slides-update-slide.md +163 -0
  140. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-get.md +110 -0
  141. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-replace.md +188 -0
  142. package/skills/lark-slides/references/cli/lark-slides-xml-presentations-get.md +157 -0
  143. package/skills/lark-slides/references/iconpark-index.json +5 -41901
  144. package/skills/lark-slides/references/iconpark.md +3 -44
  145. package/skills/lark-slides/references/lark-slides-add-slide.md +5 -0
  146. package/skills/lark-slides/references/lark-slides-create.md +3 -162
  147. package/skills/lark-slides/references/lark-slides-delete-slide.md +5 -0
  148. package/skills/lark-slides/references/lark-slides-edit-workflows.md +3 -142
  149. package/skills/lark-slides/references/lark-slides-history.md +3 -130
  150. package/skills/lark-slides/references/lark-slides-media-upload.md +3 -124
  151. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +3 -83
  152. package/skills/lark-slides/references/lark-slides-replace-slide.md +3 -235
  153. package/skills/lark-slides/references/lark-slides-screenshot.md +3 -95
  154. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +3 -108
  155. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +3 -186
  156. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +3 -132
  157. package/skills/lark-slides/references/planning-layer.md +1 -1
  158. package/skills/lark-slides/references/slides_chart_demo.xml +5 -1416
  159. package/skills/lark-slides/references/slides_xml_schema_definition.xml +3 -3468
  160. package/skills/lark-slides/references/troubleshooting.md +3 -61
  161. package/skills/lark-slides/references/validation-checklist.md +3 -154
  162. package/skills/lark-slides/references/workflow/error-handling.md +62 -0
  163. package/skills/lark-slides/references/workflow/slides-editing.md +143 -0
  164. package/skills/lark-slides/references/workflow/template-editing.md +85 -0
  165. package/skills/lark-slides/references/workflow/validation-xml.md +156 -0
  166. package/skills/lark-slides/references/xml/iconpark-index.json +37458 -0
  167. package/skills/lark-slides/references/xml/iconpark.md +46 -0
  168. package/skills/lark-slides/references/xml/slides_chart_demo.xml +1415 -0
  169. package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +3514 -0
  170. package/skills/lark-slides/references/xml/xml-schema-quick-ref.md +497 -0
  171. package/skills/lark-slides/references/xml-schema-quick-ref.md +3 -483
  172. package/skills/lark-slides/scripts/iconpark_tool.py +1 -1
  173. package/skills/lark-slides/scripts/sxsd_validator.py +154 -10
  174. package/skills/lark-slides/scripts/xml_lint.py +2989 -0
  175. package/skills/lark-slides/scripts/xml_lint_test.py +4720 -0
  176. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +3 -2691
  177. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +5 -3788
  178. package/skills/lark-task/SKILL.md +12 -0
  179. package/skills/lark-task/references/lark-task-create.md +3 -1
  180. package/skills/lark-vc/SKILL.md +15 -5
  181. package/skills/lark-vc/references/lark-vc-detail.md +11 -6
  182. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-events.md → lark-vc/references/lark-vc-meeting-events.md} +121 -20
  183. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-list-active.md → lark-vc/references/lark-vc-meeting-list-active.md} +2 -2
  184. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-message-send.md → lark-vc/references/lark-vc-meeting-message-send.md} +3 -3
  185. package/skills/lark-vc/references/lark-vc-recording.md +8 -6
  186. package/skills/lark-vc/references/vc-domain-boundaries.md +8 -1
  187. package/skills/lark-vc-agent/SKILL.md +24 -9
  188. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-join.md +2 -2
  189. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-leave.md +2 -2
  190. package/skills/lark-whiteboard/SKILL.md +15 -8
  191. package/skills/lark-whiteboard/references/lark-whiteboard-export.md +4 -3
  192. package/skills/lark-whiteboard/references/lark-whiteboard-update.md +4 -4
  193. package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +19 -17
  194. package/skills/lark-whiteboard/routes/dsl.md +8 -2
  195. package/skills/lark-whiteboard/routes/mermaid.md +1 -1
  196. package/skills/lark-whiteboard/routes/svg-edit.md +5 -2
  197. package/skills/lark-whiteboard/routes/svg.md +3 -1
  198. package/skills/lark-whiteboard/scenes/mention.md +71 -0
  199. package/skills/lark-wiki/SKILL.md +8 -4
  200. package/skills/lark-wiki/references/lark-wiki-delete-space.md +6 -3
  201. package/skills/lark-wiki/references/lark-wiki-node-copy.md +5 -19
  202. package/skills/lark-wiki/references/lark-wiki-node-create.md +19 -2
  203. package/skills/lark-wiki/references/lark-wiki-node-get.md +15 -0
  204. package/skills/lark-wiki/references/lark-wiki-node-list.md +1 -1
  205. package/skills/lark-base/references/lark-base-data-analysis-sop.md +0 -210
  206. package/skills/lark-base/references/lark-base-data-query-guide.md +0 -61
  207. package/skills/lark-base/references/lark-base-record-upsert.md +0 -63
  208. package/skills/lark-doc/references/lark-doc-word-stat.md +0 -93
  209. package/skills/lark-doc/references/style/lark-doc-create-workflow.md +0 -47
  210. package/skills/lark-doc/references/style/lark-doc-style.md +0 -68
  211. package/skills/lark-doc/references/style/lark-doc-update-workflow.md +0 -48
  212. package/skills/lark-doc/scripts/doc_word_stat.py +0 -1243
  213. package/skills/lark-slides/references/lark-slides-replace-pages.md +0 -95
  214. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +0 -219
  215. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +0 -126
@@ -1,2700 +1,12 @@
1
1
  #!/usr/bin/env python3
2
2
  # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
3
  # SPDX-License-Identifier: MIT
4
- """Validate Slides XML structure and page layout through one release gate."""
4
+ """Compatibility entry point for the renamed xml_lint module."""
5
5
 
6
- from __future__ import annotations
7
-
8
- import copy
9
- import json
10
- import math
11
- import re
12
6
  import sys
13
- import unicodedata
14
- import xml.parsers.expat as expat
15
- import xml.etree.ElementTree as ET
16
- from difflib import SequenceMatcher, get_close_matches
17
- from pathlib import Path
18
- from typing import Any
19
-
20
- import sxsd_validator
21
-
22
-
23
- XS_NS = "{http://www.w3.org/2001/XMLSchema}"
24
- XML_NS = "{http://www.w3.org/XML/1998/namespace}"
25
- SVG_NS = "{http://www.w3.org/2000/svg}"
26
- SML_NAMESPACE = "http://www.larkoffice.com/sml/2.0"
27
- SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
28
- ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
29
- SXSD_TAG_ALIASES = {
30
- "textbox": "<shape type=\"text\">",
31
- "textBox": "<shape type=\"text\">",
32
- "image": "<img>",
33
- "picture": "<img>",
34
- }
35
- SXSD_ATTR_ALIASES = {
36
- "x": "topLeftX",
37
- "left": "topLeftX",
38
- "y": "topLeftY",
39
- "top": "topLeftY",
40
- "w": "width",
41
- "h": "height",
42
- "fontColor": "color",
43
- }
44
- SERVER_FILLED_SXSD_ATTRS = {"id"}
45
- ROUNDTRIP_SXSD_ATTRS = {
46
- ("chart", "updated"),
47
- ("chartData", "isStaticData"),
48
- }
49
- # Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
50
- # it is server-emitted and absent from the write schema, so it must not block page linting.
51
- ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
52
- DEFAULT_TABLE_COLUMN_WIDTH = 110
53
- DEFAULT_TABLE_ROW_HEIGHT = 37
54
- DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
55
- TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
56
- TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
57
- SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
58
- CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
59
- HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
60
- DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
61
- GHOST_TEXT_MIN_FONT_SIZE = 96
62
- GHOST_TEXT_MAX_ALPHA = 0.5
63
- GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
64
- GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
65
- # A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
66
- # glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
67
- # padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
68
- LINE_TEXT_GRAZE_MIN_PX = 2.0
69
- LINE_TEXT_GRAZE_FONT_RATIO = 0.12
70
- # A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
71
- LINE_MIN_VISIBLE_ALPHA = 0.08
72
- # Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
73
- # visible defect; keep this well under 1px so real overflow is still always caught.
74
- CANVAS_OVERFLOW_TOLERANCE = 0.5
75
- _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
76
- _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
77
-
78
-
79
- class XmlLayoutLintError(Exception):
80
- pass
81
-
82
-
83
- def fail(message: str) -> None:
84
- raise XmlLayoutLintError(message)
85
-
86
-
87
- def read_file(file_path: str | Path) -> str:
88
- return Path(file_path).read_text(encoding="utf-8")
89
-
90
-
91
- def parse_args(argv: list[str]) -> dict[str, Any]:
92
- options: dict[str, Any] = {}
93
- index = 0
94
- while index < len(argv):
95
- token = argv[index]
96
- if not token.startswith("--"):
97
- fail(f"unexpected argument: {token}, need --input")
98
- key = token[2:]
99
- next_token = argv[index + 1] if index + 1 < len(argv) else None
100
- if next_token is None or next_token.startswith("--"):
101
- options[key] = True
102
- index += 1
103
- continue
104
- options[key] = next_token
105
- index += 2
106
- return options
107
-
108
-
109
- def extract_attribute(tag_source: str, name: str) -> str | None:
110
- match = re.search(
111
- fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
112
- )
113
- if not match:
114
- return None
115
- return match.group(1) if match.group(1) is not None else match.group(2)
116
-
117
-
118
- def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
119
- raw = extract_attribute(tag_source, name)
120
- if raw is None:
121
- return None
122
- try:
123
- value = float(raw)
124
- except ValueError:
125
- return None
126
- return int(value) if value.is_integer() else value
127
-
128
-
129
- def extract_bool_attribute(tag_source: str, name: str) -> bool:
130
- value = extract_attribute(tag_source, name)
131
- return value in {"true", "1", "yes"}
132
-
133
-
134
- def extract_color_alpha(color: str | None) -> int | float | None:
135
- if color is None:
136
- return None
137
- normalized = re.sub(r"\s+", "", color).lower()
138
- if normalized == "transparent":
139
- return 0
140
- rgba_match = re.fullmatch(
141
- r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
142
- normalized,
143
- )
144
- if rgba_match is None:
145
- return None
146
- try:
147
- alpha = float(rgba_match.group(1))
148
- except ValueError:
149
- return None
150
- return int(alpha) if alpha.is_integer() else alpha
151
-
152
-
153
- def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
154
- base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
155
- color_alpha = extract_color_alpha(text_color)
156
- if not isinstance(color_alpha, (int, float)):
157
- return base_alpha
158
- return base_alpha * color_alpha
159
-
160
-
161
- def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
162
- for tag_name in style_tags:
163
- if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
164
- return True
165
- return False
166
-
167
-
168
- def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
169
- for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
170
- if extract_bool_attribute(attrs, attr_name):
171
- return True
172
- return False
173
-
174
-
175
- def sum_sizes(sizes: list[int | float]) -> int | float:
176
- return sum(sizes)
177
-
178
-
179
- def is_filled_size(size: int | float | None) -> bool:
180
- return isinstance(size, (int, float)) and math.isfinite(size) and size > 0
181
-
182
-
183
- def fill_last_size_gap(sizes: list[int | float], target_size: int | float) -> list[int | float]:
184
- if not sizes:
185
- return sizes
186
- final_sizes = [
187
- size if index == len(sizes) - 1 else max(1, math.floor(size + 0.5))
188
- for index, size in enumerate(sizes)
189
- ]
190
- remaining_size = target_size - sum_sizes(final_sizes[:-1])
191
- if remaining_size >= 1:
192
- final_sizes[-1] = remaining_size
193
- return final_sizes
194
-
195
- size_to_redistribute = 1 - remaining_size
196
- for index in range(len(final_sizes) - 2, -1, -1):
197
- reduction = min(final_sizes[index] - 1, size_to_redistribute)
198
- final_sizes[index] -= reduction
199
- size_to_redistribute -= reduction
200
- if size_to_redistribute == 0:
201
- final_sizes[-1] = 1
202
- return final_sizes
203
-
204
- final_sizes[-1] = 1
205
- return final_sizes
206
-
207
-
208
- def solve_weighted_min_layout(
209
- input_sizes: list[int | float | None], default_size: int | float, target_min_size: int | float | None
210
- ) -> dict[str, Any]:
211
- filled_indexes: list[int] = []
212
- empty_indexes: list[int] = []
213
- base_sizes: list[int | float] = []
214
- for index, size in enumerate(input_sizes):
215
- if is_filled_size(size):
216
- filled_indexes.append(index)
217
- base_sizes.append(size)
218
- else:
219
- empty_indexes.append(index)
220
- base_sizes.append(0)
221
- filled_sum = sum_sizes(base_sizes)
222
-
223
- if target_min_size is None:
224
- final_sizes = [default_size if index in empty_indexes else size for index, size in enumerate(base_sizes)]
225
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
226
-
227
- if not filled_indexes:
228
- average_size = target_min_size / len(input_sizes)
229
- final_sizes = fill_last_size_gap([average_size] * len(input_sizes), target_min_size)
230
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
231
-
232
- if empty_indexes:
233
- remaining_size = target_min_size - filled_sum
234
- final_sizes = [*base_sizes]
235
- if remaining_size > 0:
236
- average_size = remaining_size / len(empty_indexes)
237
- empty_sizes = fill_last_size_gap([average_size] * len(empty_indexes), remaining_size)
238
- for index, empty_size in zip(empty_indexes, empty_sizes):
239
- final_sizes[index] = empty_size
240
- else:
241
- for index in empty_indexes:
242
- final_sizes[index] = default_size
243
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
244
-
245
- ratio = max(1, target_min_size / filled_sum)
246
- actual_size = max(target_min_size, filled_sum)
247
- if ratio == 1:
248
- return {"final_sizes": [*base_sizes], "actual_size": actual_size, "ratio": ratio}
249
- final_sizes = fill_last_size_gap([size * ratio for size in base_sizes], actual_size)
250
- return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
251
-
252
-
253
- def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
254
- stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
255
- if preserve_line_breaks:
256
- stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
257
- stripped = re.sub(r"<[^>]+>", " ", stripped)
258
- stripped = stripped.replace("&nbsp;", " ")
259
- stripped = stripped.replace("&amp;", "&")
260
- stripped = stripped.replace("&lt;", "<")
261
- stripped = stripped.replace("&gt;", ">")
262
- stripped = stripped.replace("&quot;", '"')
263
- stripped = stripped.replace("&#39;", "'")
264
- if preserve_line_breaks:
265
- return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
266
- return re.sub(r"\s+", " ", stripped).strip()
267
-
268
-
269
- def strip_xml_paragraphs(value: str) -> str:
270
- paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
271
- if paragraphs:
272
- return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
273
- return strip_xml(value, preserve_line_breaks=True)
274
-
275
-
276
- def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
277
- paragraphs = []
278
- for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
279
- paragraphs.append(
280
- {
281
- "text": strip_xml(body, preserve_line_breaks=True),
282
- "fontSize": extract_max_span_font_size(body, default_font_size),
283
- "textAlign": extract_attribute(attrs, "textAlign"),
284
- "lineSpacing": extract_attribute(attrs, "lineSpacing"),
285
- "beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
286
- "afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
287
- "letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
288
- }
289
- )
290
- return paragraphs
291
-
292
-
293
- def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
294
- font_sizes = [
295
- font_size
296
- for attrs in re.findall(r"<span\b([^>]*)>", value)
297
- if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
298
- ]
299
- return max([default_font_size, *font_sizes])
300
-
301
-
302
- def extract_tag_attributes(value: str, tag: str) -> str:
303
- match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
304
- return match.group(1) if match else ""
305
-
306
-
307
- def xml_local_name(tag: str) -> str:
308
- return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
309
-
310
-
311
- def xml_namespace(tag: str) -> str | None:
312
- return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
313
-
314
-
315
- def load_sxsd_tag_attributes() -> dict[str, set[str]]:
316
- global _SXSD_TAG_ATTRIBUTES_CACHE
317
- if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
318
- return _SXSD_TAG_ATTRIBUTES_CACHE
319
-
320
- _SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
321
- return _SXSD_TAG_ATTRIBUTES_CACHE
322
-
323
-
324
- def load_iconpark_icon_types() -> set[str]:
325
- global _ICONPARK_ICON_TYPES_CACHE
326
- if _ICONPARK_ICON_TYPES_CACHE is not None:
327
- return _ICONPARK_ICON_TYPES_CACHE
328
-
329
- try:
330
- index_data = json.loads(ICONPARK_INDEX_PATH.read_text(encoding="utf-8"))
331
- except json.JSONDecodeError as error:
332
- fail(f"invalid iconpark index JSON: {error}")
333
- icons = index_data.get("icons")
334
- if not isinstance(icons, list):
335
- fail("iconpark index must contain an icons array")
336
-
337
- icon_types = {
338
- icon["iconType"]
339
- for icon in icons
340
- if isinstance(icon, dict) and isinstance(icon.get("iconType"), str) and icon["iconType"]
341
- }
342
- _ICONPARK_ICON_TYPES_CACHE = icon_types
343
- return icon_types
344
-
345
-
346
- def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
347
- alias = SXSD_TAG_ALIASES.get(tag_name)
348
- if alias:
349
- return f"Use {alias} instead of <{tag_name}>."
350
- if tag_name == "svg":
351
- return 'Inside <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
352
- close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
353
- if close_matches:
354
- return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
355
- return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
356
-
357
-
358
- def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
359
- alias = SXSD_ATTR_ALIASES.get(attr_name)
360
- if alias and alias in allowed_attrs:
361
- return [alias]
362
- return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
363
-
364
-
365
- def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
366
- suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
367
- if suggestions:
368
- if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
369
- return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
370
- return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
371
- allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
372
- if len(allowed_attrs) > 8:
373
- allowed_summary += ", ..."
374
- return f"Unsupported SXSD attribute for <{tag_name}>. Allowed attributes include: {allowed_summary}."
375
-
376
-
377
- def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
378
- return "whiteboard" in ancestors and xml_namespace(element.tag) == SVG_NS
379
-
380
-
381
- def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
382
- return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
383
-
384
-
385
- def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
386
- return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
387
-
388
-
389
- def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
390
- sanitized_root = copy.deepcopy(root)
391
-
392
- def sanitize(element: ET.Element) -> None:
393
- tag_name = xml_local_name(element.tag)
394
- for raw_attr_name in list(element.attrib):
395
- if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
396
- del element.attrib[raw_attr_name]
397
- for child in list(element):
398
- if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
399
- element.remove(child)
400
- continue
401
- sanitize(child)
402
-
403
- sanitize(sanitized_root)
404
- return sanitized_root
405
-
406
-
407
- def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
408
- tag_attributes = load_sxsd_tag_attributes()
409
- supported_tags = set(tag_attributes)
410
- issues: list[dict[str, Any]] = []
411
- suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
412
-
413
- def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
414
- if should_skip_sxsd_subtree(element, ancestors):
415
- return
416
-
417
- tag_name = xml_local_name(element.tag)
418
- current_path = f"{path}/{tag_name}" if path else tag_name
419
- parent_name = ancestors[-1] if ancestors else None
420
- if should_skip_sxsd_tag(parent_name, tag_name):
421
- return
422
- if tag_name not in supported_tags:
423
- issues.append(
424
- {
425
- "level": "error",
426
- "code": "sxsd_unsupported_tag",
427
- "tag": tag_name,
428
- "path": current_path,
429
- "message": f"unsupported SXSD tag <{tag_name}> at {current_path}",
430
- "hint": build_sxsd_tag_hint(tag_name, supported_tags),
431
- }
432
- )
433
- return
434
- else:
435
- allowed_attrs = tag_attributes[tag_name]
436
- for raw_attr_name in element.attrib:
437
- if raw_attr_name.startswith(XML_NS):
438
- continue
439
- attr_name = xml_local_name(raw_attr_name)
440
- if should_skip_sxsd_attribute(tag_name, attr_name):
441
- continue
442
- if attr_name in allowed_attrs:
443
- continue
444
- suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
445
- if suggestions:
446
- suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
447
- set(suggestions)
448
- )
449
- issues.append(
450
- {
451
- "level": "error",
452
- "code": "sxsd_unsupported_attr",
453
- "tag": tag_name,
454
- "attr": attr_name,
455
- "path": current_path,
456
- "message": f'unsupported SXSD attribute "{attr_name}" on <{tag_name}> at {current_path}',
457
- "hint": build_sxsd_attr_hint(tag_name, attr_name, allowed_attrs),
458
- }
459
- )
460
-
461
- for child in element:
462
- visit(child, [*ancestors, tag_name], current_path)
463
-
464
- visit(root, [], "")
465
- existing = {
466
- (issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
467
- for issue in issues
468
- }
469
- unsupported_tag_locations = {
470
- (issue.get("path"), issue.get("tag"))
471
- for issue in issues
472
- if issue.get("code") == "sxsd_unsupported_tag"
473
- }
474
- schema_issues = _validate_sxsd_schema_constraints(xml, root)
475
- missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
476
- for schema_issue in schema_issues:
477
- if schema_issue.get("code") != "sxsd_missing_required_attr":
478
- continue
479
- location = (schema_issue.get("path"), schema_issue.get("tag"))
480
- missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
481
-
482
- suggested_attrs: set[tuple[str, str, str]] = set()
483
- for location, candidate_groups in suggested_attr_candidates.items():
484
- missing_attrs = missing_attrs_by_location.get(location, set())
485
- for candidates in candidate_groups:
486
- matching_missing_attrs = candidates & missing_attrs
487
- if len(matching_missing_attrs) == 1:
488
- suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
489
-
490
- for schema_issue in schema_issues:
491
- if schema_issue.get("code") == "sxsd_unexpected_child" and (
492
- schema_issue.get("path"),
493
- schema_issue.get("tag"),
494
- ) in unsupported_tag_locations:
495
- continue
496
- if schema_issue.get("code") == "sxsd_missing_required_attr" and (
497
- schema_issue.get("path"),
498
- schema_issue.get("tag"),
499
- schema_issue.get("attr"),
500
- ) in suggested_attrs:
501
- continue
502
- key = (
503
- schema_issue.get("code"),
504
- schema_issue.get("path"),
505
- schema_issue.get("tag"),
506
- schema_issue.get("attr"),
507
- )
508
- if key not in existing:
509
- issues.append(schema_issue)
510
- return issues
511
-
512
-
513
- def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
514
- issues: list[dict[str, Any]] = []
515
- if re.match(r"^\s*<\?xml\b", xml):
516
- issues.append(
517
- {
518
- "level": "error",
519
- "code": "sxsd_unsupported_declaration",
520
- "path": xml_local_name(root.tag),
521
- "tag": xml_local_name(root.tag),
522
- "expected": "SXSD document without an XML declaration",
523
- "actual": "<?xml ...?>",
524
- "message": "XML declarations are not supported by the Slides SXSD write format",
525
- "hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
526
- }
527
- )
528
-
529
- issues.extend(
530
- sxsd_validator.validate_sxsd(
531
- without_server_filled_sxsd_fields(root),
532
- SXSD_SCHEMA_PATH,
533
- )
534
- )
535
- return issues
536
-
537
-
538
- def build_iconpark_icon_type_hint(icon_type: str, supported_icon_types: set[str]) -> str:
539
- close_matches = get_close_matches(icon_type, sorted(supported_icon_types), n=3, cutoff=0.58)
540
- if close_matches:
541
- return (
542
- "iconType must exist in iconpark-index.json. Did you mean "
543
- + ", ".join(f'"{match}"' for match in close_matches)
544
- + "?"
545
- )
546
- return "iconType must exist in iconpark-index.json. Use scripts/iconpark_tool.py to search supported icons."
547
-
548
-
549
- def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
550
- supported_icon_types: set[str] | None = None
551
- issues: list[dict[str, Any]] = []
552
-
553
- def direct_child(element: ET.Element, local_name: str) -> ET.Element | None:
554
- return next((child for child in element if xml_local_name(child.tag) == local_name), None)
555
-
556
- def is_transparent_color(color: str) -> bool:
557
- normalized = re.sub(r"\s+", "", color).lower()
558
- if normalized == "transparent":
559
- return True
560
- rgba_match = re.fullmatch(r"rgba\([^,]+,[^,]+,[^,]+,([0-9.]+)\)", normalized)
561
- if not rgba_match:
562
- return False
563
- try:
564
- return float(rgba_match.group(1)) <= 0
565
- except ValueError:
566
- return False
567
-
568
- def append_missing_fill_color_issue(current_path: str) -> None:
569
- issues.append(
570
- {
571
- "level": "error",
572
- "code": "icon_missing_fill_color",
573
- "tag": "icon",
574
- "path": current_path,
575
- "message": f"<icon> must set explicit non-transparent fillColor for visual visibility at {current_path}",
576
- "hint": 'Add <fill><fillColor color="rgba(R, G, B, 1)"/></fill> inside <icon>. This is a visual lint rule, not an SXSD required field.',
577
- }
578
- )
579
-
580
- def visit(element: ET.Element, path: str) -> None:
581
- nonlocal supported_icon_types
582
- tag_name = xml_local_name(element.tag)
583
- current_path = f"{path}/{tag_name}" if path else tag_name
584
- if tag_name == "icon":
585
- icon_type = element.attrib.get("iconType")
586
- if icon_type is not None:
587
- if supported_icon_types is None:
588
- supported_icon_types = load_iconpark_icon_types()
589
- if icon_type not in supported_icon_types:
590
- issues.append(
591
- {
592
- "level": "error",
593
- "code": "iconpark_unsupported_icon_type",
594
- "tag": "icon",
595
- "attr": "iconType",
596
- "iconType": icon_type,
597
- "path": current_path,
598
- "message": f'unsupported iconpark iconType "{icon_type}" at {current_path}',
599
- "hint": build_iconpark_icon_type_hint(icon_type, supported_icon_types),
600
- }
601
- )
602
- fill = direct_child(element, "fill")
603
- fill_color = direct_child(fill, "fillColor") if fill is not None else None
604
- color = fill_color.attrib.get("color") if fill_color is not None else None
605
- if not color:
606
- append_missing_fill_color_issue(current_path)
607
- elif is_transparent_color(color):
608
- issues.append(
609
- {
610
- "level": "error",
611
- "code": "icon_transparent_fill_color",
612
- "tag": "icon",
613
- "attr": "fillColor",
614
- "path": current_path,
615
- "color": color,
616
- "message": f'<icon> fillColor must not be transparent for visual visibility at {current_path}: "{color}"',
617
- "hint": 'Use an opaque visible color, for example <fillColor color="rgba(37, 99, 235, 1)"/>.',
618
- }
619
- )
620
- for child in element:
621
- visit(child, current_path)
622
-
623
- visit(root, "")
624
- return issues
625
-
626
-
627
- def extract_error_context(xml: str, line: int | None, column: int | None, radius: int = 40) -> str | None:
628
- if line is None or column is None:
629
- return None
630
- lines = xml.splitlines()
631
- if line < 1 or line > len(lines):
632
- return None
633
- source_line = lines[line - 1]
634
- start = max(column - radius, 0)
635
- end = min(column + radius, len(source_line))
636
- return source_line[start:end].strip()
637
-
638
-
639
- def build_xml_error_issue(error: ET.ParseError, xml: str) -> dict[str, Any]:
640
- line, column = getattr(error, "position", (None, None))
641
- return {
642
- "level": "error",
643
- "code": "xml_not_well_formed",
644
- "message": f"XML is not well-formed: {error}",
645
- "line": line,
646
- "column": column,
647
- "context": extract_error_context(xml, line, column),
648
- "hint": (
649
- "Escape raw user text before placing it in XML. In text nodes and attribute values, bare & must be "
650
- "written as &amp;. In text nodes, write < as &lt; and > as &gt;. For attribute URLs, use a=1&amp;b=2."
651
- ),
652
- }
653
-
654
-
655
- def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
656
- namespace_map: dict[str, str] = {}
657
- pending_declarations: list[tuple[str, str | None]] = []
658
- declarations_by_element: list[list[tuple[str, str | None]]] = []
659
- element_stack: list[str] = []
660
- issues: list[dict[str, Any]] = []
661
-
662
- parser = expat.ParserCreate(namespace_separator="|")
663
- parser.namespace_prefixes = True
664
-
665
- def handle_namespace_decl(prefix: str | None, namespace: str) -> None:
666
- normalized_prefix = prefix or ""
667
- previous_namespace = namespace_map.get(normalized_prefix)
668
- namespace_map[normalized_prefix] = namespace
669
- pending_declarations.append((normalized_prefix, previous_namespace))
670
-
671
- def handle_start_element(name: str, _attrs: dict[str, str]) -> None:
672
- declarations_by_element.append(pending_declarations.copy())
673
- pending_declarations.clear()
674
- name_parts = name.rsplit("|", 2)
675
- if len(name_parts) == 3:
676
- _namespace, local_name, prefix = name_parts
677
- element_name = f"{prefix}:{local_name}"
678
- else:
679
- prefix = ""
680
- local_name = name_parts[-1]
681
- element_name = local_name
682
- element_stack.append(element_name)
683
- if not prefix:
684
- return
685
-
686
- if namespace_map.get(prefix) != SML_NAMESPACE:
687
- return
688
- path = "/".join(element_stack)
689
- issues.append(
690
- {
691
- "level": "error",
692
- "code": "sml_prefixed_tag",
693
- "tag": element_name,
694
- "namespace": SML_NAMESPACE,
695
- "path": path,
696
- "line": parser.CurrentLineNumber,
697
- "column": parser.CurrentColumnNumber,
698
- "message": f"SML tag <{element_name}> must not use a namespace prefix at {path}",
699
- "hint": (
700
- f'Use <{local_name}> under the default namespace '
701
- f'<{local_name} xmlns="{SML_NAMESPACE}">, or use an unprefixed SML tag.'
702
- ),
703
- }
704
- )
705
-
706
- def handle_end_element(_name: str) -> None:
707
- for prefix, previous_namespace in reversed(declarations_by_element.pop()):
708
- if previous_namespace is None:
709
- namespace_map.pop(prefix, None)
710
- else:
711
- namespace_map[prefix] = previous_namespace
712
- element_stack.pop()
713
-
714
- parser.StartNamespaceDeclHandler = handle_namespace_decl
715
- parser.StartElementHandler = handle_start_element
716
- parser.EndElementHandler = handle_end_element
717
- parser.Parse(xml, True)
718
- return issues
719
-
720
-
721
- def parse_xml_root(xml: str) -> tuple[ET.Element | None, dict[str, Any] | None]:
722
- try:
723
- root = ET.fromstring(xml)
724
- except ET.ParseError as error:
725
- return None, build_xml_error_issue(error, xml)
726
-
727
- root_name = xml_local_name(root.tag)
728
- if root_name not in {"presentation", "slide"}:
729
- fail("input must contain a <presentation> or <slide> root")
730
- return root, None
731
-
732
-
733
- def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
734
- _, xml_error = parse_xml_root(xml)
735
- return xml_error
736
-
737
-
738
- def serialize_slide_for_layout(slide_root: ET.Element) -> str:
739
- slide_copy = copy.deepcopy(slide_root)
740
- for element in slide_copy.iter():
741
- if not isinstance(element.tag, str):
742
- continue
743
- element.tag = xml_local_name(element.tag)
744
- attributes = {
745
- xml_local_name(attribute_name): value
746
- for attribute_name, value in element.attrib.items()
747
- }
748
- element.attrib.clear()
749
- element.attrib.update(attributes)
750
- return ET.tostring(slide_copy, encoding="unicode")
751
-
752
-
753
- def parse_presentation(root: ET.Element) -> dict[str, Any]:
754
- root_name = xml_local_name(root.tag)
755
- if root_name == "slide":
756
- slide_roots = [root]
757
- width = 960
758
- height = 540
759
- elif root_name == "presentation":
760
- slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
761
- width = int(float(root.attrib.get("width", 960)))
762
- height = int(float(root.attrib.get("height", 540)))
763
- else:
764
- fail("input must contain a <presentation> or <slide> root")
765
- return {
766
- "width": width,
767
- "height": height,
768
- "slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
769
- "slide_roots": slide_roots,
770
- }
771
-
772
-
773
- def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
774
- elements: list[dict[str, Any]] = []
775
-
776
- for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
777
- kind, attrs = match.group(1), match.group(2)
778
- is_self_closing = attrs.rstrip().endswith("/")
779
- content = ""
780
- if kind in {"shape", "table"} and not is_self_closing:
781
- close_index = slide_xml.find(f"</{kind}>", match.end())
782
- if close_index != -1:
783
- content = slide_xml[match.end() : close_index]
784
-
785
- element_id = extract_attribute(attrs, "id") or f"{kind}-{len(elements) + 1}"
786
- x = extract_numeric_attribute(attrs, "topLeftX")
787
- y = extract_numeric_attribute(attrs, "topLeftY")
788
- width = extract_numeric_attribute(attrs, "width")
789
- height = extract_numeric_attribute(attrs, "height")
790
- rotation = extract_numeric_attribute(attrs, "rotation") or 0
791
- alpha = extract_numeric_attribute(attrs, "alpha")
792
- table_layouts: dict[str, dict[str, Any] | None] = {}
793
- if kind == "table":
794
- width, table_layouts["width"] = resolve_table_dimension(
795
- content, width, extract_table_column_sizes, DEFAULT_TABLE_COLUMN_WIDTH
796
- )
797
- height, table_layouts["height"] = resolve_table_dimension(
798
- content, height, extract_table_row_sizes, DEFAULT_TABLE_ROW_HEIGHT
799
- )
800
- if all(value is not None for value in [x, y, width, height]):
801
- element = {
802
- "id": element_id,
803
- "kind": kind,
804
- "type": extract_attribute(attrs, "type") or kind,
805
- "x": x,
806
- "y": y,
807
- "width": width,
808
- "height": height,
809
- "rotation": rotation,
810
- "alpha": alpha if alpha is not None else 1,
811
- "order": len(elements),
812
- }
813
- if kind == "table":
814
- element.update(
815
- {
816
- "declared_width": extract_numeric_attribute(attrs, "width"),
817
- "declared_height": extract_numeric_attribute(attrs, "height"),
818
- "table_layouts": table_layouts,
819
- }
820
- )
821
- if kind == "shape":
822
- content_attrs = extract_tag_attributes(content, "content")
823
- font_size = extract_numeric_attribute(content_attrs, "fontSize")
824
- if font_size is None:
825
- font_size = extract_numeric_attribute(attrs, "fontSize")
826
- font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
827
- text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
828
- bold = (
829
- extract_bool_attribute(content_attrs, "bold")
830
- or extract_bool_attribute(attrs, "bold")
831
- or detect_inline_style_presence(content, {"strong", "b"})
832
- or detect_any_span_bool_attribute(content, "bold")
833
- )
834
- italic = (
835
- extract_bool_attribute(content_attrs, "italic")
836
- or extract_bool_attribute(attrs, "italic")
837
- or detect_inline_style_presence(content, {"i", "em"})
838
- or detect_any_span_bool_attribute(content, "italic")
839
- )
840
- element.update(
841
- {
842
- "textType": extract_attribute(content_attrs, "textType"),
843
- "textAlign": extract_attribute(content_attrs, "textAlign"),
844
- "verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
845
- "vert": extract_attribute(attrs, "vert") or "horz",
846
- "autoFit": extract_attribute(content_attrs, "autoFit"),
847
- "wrap": extract_attribute(content_attrs, "wrap"),
848
- "lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
849
- "beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
850
- "afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
851
- "letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
852
- "paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
853
- "paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
854
- "paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
855
- "paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
856
- "fontSize": font_size if font_size is not None else 16,
857
- "fontFamily": font_family or "",
858
- "color": text_color,
859
- "textAlpha": effective_text_alpha(alpha, text_color),
860
- "bold": bold,
861
- "italic": italic,
862
- "text": strip_xml_paragraphs(content),
863
- "paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
864
- }
865
- )
866
- elements.append(element)
867
- return elements
868
-
869
-
870
- def intersects(left: dict[str, Any], right: dict[str, Any]) -> bool:
871
- return (
872
- left["x"] < right["x"] + right["width"]
873
- and left["x"] + left["width"] > right["x"]
874
- and left["y"] < right["y"] + right["height"]
875
- and left["y"] + left["height"] > right["y"]
876
- )
877
-
878
-
879
- def is_text_element(element: dict[str, Any]) -> bool:
880
- return element["kind"] == "shape" and element["type"] == "text"
881
-
882
-
883
- def is_whiteboard_element(element: dict[str, Any]) -> bool:
884
- return element["kind"] == "whiteboard"
885
-
886
-
887
- def has_text_content(element: dict[str, Any]) -> bool:
888
- return bool(element.get("text"))
889
-
890
-
891
- def is_vertical_text(element: dict[str, Any]) -> bool:
892
- return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
893
-
894
-
895
- def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
896
- issues: list[dict[str, Any]] = []
897
- text_elements = [
898
- element
899
- for element in elements
900
- if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
901
- ]
902
- image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
903
- for text_element in text_elements:
904
- for image_element in image_elements:
905
- if image_element["order"] <= text_element["order"]:
906
- continue
907
- if is_vertical_text(text_element):
908
- if intersects(image_element, text_element):
909
- issues.append({
910
- "level": "info",
911
- "code": "image_may_cover_vertical_text",
912
- "elements": [image_element["id"], text_element["id"]],
913
- "message": f'image {image_element["id"]} may cover vertical text shape {text_element["id"]}',
914
- "hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
915
- })
916
- continue
917
- text_visual_bbox = estimate_text_visual_bbox(text_element)
918
- if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
919
- issues.append({
920
- "level": "error",
921
- "code": "image_covers_text",
922
- "elements": [image_element["id"], text_element["id"]],
923
- "message": f'image {image_element["id"]} covers text shape {text_element["id"]}',
924
- "hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
925
- })
926
- return issues
927
-
928
-
929
- def is_decorative_text(element: dict[str, Any]) -> bool:
930
- text = element.get("text") or ""
931
- return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
932
-
933
-
934
- def normalize_text_for_overlap(text: str) -> str:
935
- return re.sub(r"\s+", "", text)
936
-
937
-
938
- SERIF_FONT_PATTERNS = {
939
- "song", "songti", "simsun", "ming", "mincho",
940
- "georgia", "times", "caslon", "garamond", "sourcehan-serif",
941
- "source han serif", "思源宋体", "宋体", "明体",
942
- }
943
-
944
- SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
945
- "helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
946
-
947
-
948
- def classify_font_family(font_family: str | None) -> str:
949
- if not font_family:
950
- return "sans"
951
- family_lower = font_family.lower()
952
- for marker in SANS_EXPLICIT_MARKERS:
953
- if marker in family_lower:
954
- return "sans"
955
- serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
956
- for pattern in serif_keywords:
957
- if pattern in family_lower:
958
- return "serif"
959
- return "sans"
960
-
961
-
962
- _FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
963
- "sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
964
- "serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
965
- }
966
-
967
-
968
- def estimate_character_width(
969
- character: str,
970
- font_size: int | float,
971
- bold: bool = False,
972
- font_family: str | None = None,
973
- ) -> int | float:
974
- bold_multiplier = 1.05 if bold else 1.0
975
- if character.isspace():
976
- return font_size * 0.33 * bold_multiplier
977
- ea_width = unicodedata.east_asian_width(character)
978
- if ea_width in {"F", "W"}:
979
- return font_size * bold_multiplier
980
- category = classify_font_family(font_family)
981
- coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
982
- if character.isupper():
983
- return font_size * coeffs["upper"] * bold_multiplier
984
- if character.islower():
985
- return font_size * coeffs["lower"] * bold_multiplier
986
- if character.isdigit():
987
- return font_size * coeffs["digit"] * bold_multiplier
988
- return font_size * coeffs["punct"] * bold_multiplier
989
-
990
-
991
- def estimate_text_width(
992
- text: str,
993
- font_size: int | float,
994
- letter_spacing: int | float = 0,
995
- bold: bool = False,
996
- font_family: str | None = None,
997
- ) -> int | float:
998
- base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
999
- return base + max(len(text) - 1, 0) * letter_spacing
1000
-
1001
-
1002
- def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
1003
- if paragraph is not None:
1004
- value = paragraph.get("letterSpacing")
1005
- if isinstance(value, (int, float)):
1006
- return value
1007
- value = element.get("letterSpacing")
1008
- return value if isinstance(value, (int, float)) else 0
1009
-
1010
-
1011
- def text_wrap_width_tolerance() -> int | float:
1012
- return TEXT_WRAP_WIDTH_TOLERANCE_PX
1013
-
1014
-
1015
- def text_height_overflow_tolerance() -> int | float:
1016
- return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
1017
-
1018
-
1019
- def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
1020
- return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
1021
-
1022
-
1023
- def is_short_metric_text(text: str) -> bool:
1024
- compact = re.sub(r"\s+", "", text)
1025
- if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
1026
- return False
1027
- if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
1028
- return True
1029
- if re.search(r"[,.,+\-–—/%%]", compact) is None:
1030
- return False
1031
- return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
1032
-
1033
-
1034
- def is_single_line_visual_candidate(
1035
- element: dict[str, Any],
1036
- paragraph: dict[str, Any] | None,
1037
- text: str,
1038
- logical_width: int | float,
1039
- effective_width: int | float,
1040
- ) -> bool:
1041
- if "\n" in text or logical_width <= effective_width:
1042
- return False
1043
- if is_short_metric_text(text):
1044
- return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
1045
-
1046
- text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
1047
- compact_len = len(re.sub(r"\s+", "", text))
1048
- if text_align == "center" and compact_len <= 32:
1049
- return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
1050
-
1051
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1052
- if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
1053
- return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
1054
- return False
1055
-
1056
-
1057
- def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
1058
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1059
- bold = element.get("bold", False)
1060
- font_family = element.get("fontFamily", "")
1061
- letter_spacing = resolve_letter_spacing(element)
1062
- paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
1063
- return max(
1064
- [estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
1065
- or [1]
1066
- )
1067
-
1068
-
1069
- def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
1070
- left_text = normalize_text_for_overlap(left.get("text") or "")
1071
- right_text = normalize_text_for_overlap(right.get("text") or "")
1072
- if not left_text or not right_text:
1073
- return False
1074
- if left_text == right_text or left_text in right_text or right_text in left_text:
1075
- return True
1076
- return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
1077
-
1078
-
1079
- def estimate_text_line_count_for_text(
1080
- element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
1081
- ) -> int:
1082
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1083
- bold = element.get("bold", False)
1084
- font_family = element.get("fontFamily", "")
1085
- letter_spacing = resolve_letter_spacing(element, paragraph)
1086
- available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
1087
- hard_lines = text.split("\n")
1088
- if not text:
1089
- return 0
1090
- line_count = 0
1091
- for hard_line in hard_lines:
1092
- if element.get("wrap") in {"false", "0"}:
1093
- line_count += 1
1094
- continue
1095
- logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
1096
- effective_width = available_width + text_wrap_width_tolerance()
1097
- if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
1098
- line_count += 1
1099
- continue
1100
- line_count += max(1, math.ceil(logical_width / effective_width))
1101
- return line_count
1102
-
1103
-
1104
- def estimate_text_line_count(element: dict[str, Any]) -> int:
1105
- return max(estimate_text_line_count_for_text(element, element["text"]), 1)
1106
-
1107
-
1108
- def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
1109
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1110
- if line_spacing is None:
1111
- return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
1112
- match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
1113
- if match is None:
1114
- return None
1115
- spacing_type, value = match.groups()
1116
- return font_size * float(value) if spacing_type == "multiple" else float(value)
1117
-
1118
-
1119
- def adjust_dense_body_line_height(
1120
- element: dict[str, Any],
1121
- line_spacing: str | None,
1122
- line_height: int | float,
1123
- paragraph_count: int,
1124
- ) -> int | float:
1125
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1126
- if paragraph_count < 4 or font_size > 14 or not line_spacing:
1127
- return line_height
1128
- match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
1129
- if match is None:
1130
- return line_height
1131
- return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
1132
-
1133
-
1134
- def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1135
- issues: list[dict[str, Any]] = []
1136
- for element in elements:
1137
- if not is_text_element(element) or not has_text_content(element):
1138
- continue
1139
- if has_explicit_height_auto_fit(element):
1140
- continue
1141
-
1142
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1143
- paragraphs = element.get("paragraphs") or [
1144
- {
1145
- "text": element["text"],
1146
- "lineSpacing": None,
1147
- "beforeLineSpacing": None,
1148
- "afterLineSpacing": None,
1149
- }
1150
- ]
1151
- line_count = 0
1152
- estimated_height = 0.0
1153
- line_heights: list[int | float] = []
1154
- for paragraph in paragraphs:
1155
- paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
1156
- if paragraph_line_count == 0:
1157
- continue
1158
- resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
1159
- line_height = estimate_text_line_height(element, resolved_line_spacing)
1160
- before_spacing = estimate_text_line_height(
1161
- element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
1162
- )
1163
- after_spacing = estimate_text_line_height(
1164
- element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
1165
- )
1166
- if line_height is None or before_spacing is None or after_spacing is None:
1167
- line_count = 0
1168
- break
1169
- line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
1170
- first_line_height = font_size if line_count == 0 else line_height
1171
- line_count += paragraph_line_count
1172
- line_heights.append(line_height)
1173
- estimated_height += (
1174
- before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
1175
- )
1176
- if line_count == 0:
1177
- continue
1178
- available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
1179
- overflow = estimated_height - available_height
1180
- if overflow <= text_height_overflow_tolerance():
1181
- continue
1182
-
1183
- is_background = is_background_decorative_text(element, elements)
1184
- if is_background:
1185
- level = "info"
1186
- else:
1187
- level = "error" if overflow > 10 else "warning"
1188
- message = (
1189
- f'text shape {element["id"]} may overflow its own content box '
1190
- f'(estimated {estimated_height:g}px, available {available_height:g}px); '
1191
- 'consider setting content wrap="true" autoFit="normal-auto-fit"'
1192
- )
1193
- if is_background:
1194
- message += " (likely background decoration: large font, low alpha, underneath other text)"
1195
- issues.append(
1196
- {
1197
- "level": level,
1198
- "code": "text_may_overflow_shape",
1199
- "elements": [element["id"]],
1200
- "line_count": line_count,
1201
- "line_height": max(line_heights),
1202
- "estimated_height": estimated_height,
1203
- "available_height": available_height,
1204
- "overflow": overflow,
1205
- "message": message,
1206
- "hint": (
1207
- "Increase shape.height, reduce the text, or set content wrap=\"true\" "
1208
- "autoFit=\"normal-auto-fit\". "
1209
- "This is an estimate based on font size, line spacing, and wrapped line count."
1210
- ),
1211
- }
1212
- )
1213
- return issues
1214
-
1215
-
1216
- def is_background_decorative_text(
1217
- element: dict[str, Any], elements: list[dict[str, Any]]
1218
- ) -> bool:
1219
- if not is_ghost_text(element):
1220
- return False
1221
- for other in elements:
1222
- if other is element:
1223
- continue
1224
- if not is_text_element(other) or not has_text_content(other):
1225
- continue
1226
- foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
1227
- if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
1228
- continue
1229
- if other["order"] <= element["order"]:
1230
- continue
1231
- if intersects(element, other):
1232
- return True
1233
- return False
1234
-
1235
-
1236
- def is_ghost_text(element: dict[str, Any]) -> bool:
1237
- if not is_text_element(element) or not has_text_content(element):
1238
- return False
1239
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1240
- text_alpha = element.get("textAlpha", element.get("alpha", 1))
1241
- if not isinstance(text_alpha, (int, float)):
1242
- return False
1243
- if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
1244
- return True
1245
- return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
1246
-
1247
-
1248
- def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
1249
- if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
1250
- return None
1251
-
1252
- padding_left = element.get("paddingLeft", 0)
1253
- padding_right = element.get("paddingRight", 0)
1254
- padding_top = element.get("paddingTop", 0)
1255
- padding_bottom = element.get("paddingBottom", 0)
1256
- content_width = max(element["width"] - padding_left - padding_right, 0)
1257
- content_height = max(element["height"] - padding_top - padding_bottom, 0)
1258
- font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1259
- line_count = estimate_text_line_count(element)
1260
- estimated_width = max(1, estimate_text_max_line_width(element))
1261
- visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
1262
- visual_height = min(content_height, max(1, line_count * font_size * 1.2))
1263
- x = element["x"] + padding_left
1264
- if element.get("textAlign") == "center":
1265
- x += (content_width - visual_width) / 2
1266
- elif element.get("textAlign") == "right":
1267
- x += content_width - visual_width
1268
- y = element["y"] + padding_top
1269
- if element.get("verticalAlign") == "middle":
1270
- y += (content_height - visual_height) / 2
1271
- elif element.get("verticalAlign") == "bottom":
1272
- y += content_height - visual_height
1273
- return {
1274
- "x": x,
1275
- "y": y,
1276
- "width": visual_width,
1277
- "height": visual_height,
1278
- }
1279
-
1280
-
1281
- def intersection_area(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1282
- width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
1283
- height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
1284
- if width <= 0 or height <= 0:
1285
- return 0
1286
- return width * height
1287
-
1288
-
1289
- def intersection_height(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1290
- height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
1291
- return max(height, 0)
1292
-
1293
-
1294
- def intersection_width(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1295
- width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
1296
- return max(width, 0)
1297
-
1298
-
1299
- def element_area(element: dict[str, Any]) -> int | float:
1300
- return max(element["width"], 0) * max(element["height"], 0)
1301
-
1302
-
1303
- def contains(outer: dict[str, Any], inner: dict[str, Any], tolerance: int | float = 2) -> bool:
1304
- return (
1305
- inner["x"] >= outer["x"] - tolerance
1306
- and inner["y"] >= outer["y"] - tolerance
1307
- and inner["x"] + inner["width"] <= outer["x"] + outer["width"] + tolerance
1308
- and inner["y"] + inner["height"] <= outer["y"] + outer["height"] + tolerance
1309
- )
1310
-
1311
-
1312
- def is_bottom_layer_full_slide_whiteboard(
1313
- whiteboard: dict[str, Any], other: dict[str, Any], slide_width: int | float, slide_height: int | float
1314
- ) -> bool:
1315
- return (
1316
- whiteboard["order"] < other["order"]
1317
- and whiteboard["x"] <= 2
1318
- and whiteboard["y"] <= 2
1319
- and whiteboard["width"] >= slide_width - 4
1320
- and whiteboard["height"] >= slide_height - 4
1321
- )
1322
-
1323
-
1324
- def is_background_container_for_whiteboard(container: dict[str, Any], whiteboard: dict[str, Any]) -> bool:
1325
- if container["order"] > whiteboard["order"]:
1326
- return False
1327
- if is_text_element(container):
1328
- return False
1329
- return contains(container, whiteboard)
1330
-
1331
-
1332
- def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
1333
- if not (is_text_element(left) and is_text_element(right)):
1334
- return False
1335
- if not (has_text_content(left) and has_text_content(right)):
1336
- return True
1337
- top, bottom = sorted([left, right], key=lambda element: element["y"])
1338
- top_type = top.get("textType")
1339
- bottom_type = bottom.get("textType")
1340
- allowed_pairs = {
1341
- ("title", "sub-headline"),
1342
- ("title", None),
1343
- ("headline", "headline"),
1344
- ("headline", None),
1345
- }
1346
- if (top_type, bottom_type) not in allowed_pairs:
1347
- return False
1348
- same_column = abs(top["x"] - bottom["x"]) <= 4
1349
- vertical_offset = bottom["y"] - top["y"]
1350
- top_font_size = float(top.get("fontSize", 16))
1351
- return same_column and vertical_offset >= top_font_size * 0.75
1352
-
1353
-
1354
- def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str, Any]) -> bool:
1355
- if not (is_text_element(left) and is_text_element(right)):
1356
- return False
1357
- if not (has_text_content(left) and has_text_content(right)):
1358
- return False
1359
- if is_ghost_text(left) or is_ghost_text(right):
1360
- return False
1361
- if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
1362
- return False
1363
-
1364
- source, target = sorted([left, right], key=lambda element: element["x"])
1365
- if source["x"] == target["x"]:
1366
- return False
1367
- wrap_enabled = source.get("wrap") not in {"false", "0"}
1368
- has_horizontal_gap = source["x"] + source["width"] <= target["x"]
1369
- if wrap_enabled and has_horizontal_gap:
1370
- return False
1371
- if source.get("autoFit") == "normal-auto-fit":
1372
- return False
1373
- if source.get("textAlign") in {"center", "right"}:
1374
- return False
1375
-
1376
- font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
1377
- padding_left = source.get("paddingLeft", 0)
1378
- padding_right = source.get("paddingRight", 0)
1379
- available_width = max(source["width"] - padding_left - padding_right, 1)
1380
- visual_width = estimate_text_max_line_width(source)
1381
- overflow_width = visual_width - available_width
1382
- min_overflow = max(font_size * 1.5, available_width * 0.08)
1383
- if overflow_width < min_overflow:
1384
- return False
1385
-
1386
- intrusion_width = source["x"] + padding_left + visual_width - target["x"]
1387
- min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
1388
- if intrusion_width < min_intrusion:
1389
- return False
1390
-
1391
- vertical_overlap = intersection_height(source, target)
1392
- min_vertical_overlap = min(source["height"], target["height"]) * 0.40
1393
- return vertical_overlap >= min_vertical_overlap
1394
-
1395
-
1396
- def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
1397
- source, target = sorted([left, right], key=lambda element: element["x"])
1398
- padding_left = source.get("paddingLeft", 0)
1399
- visual_width = estimate_text_max_line_width(source)
1400
- source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
1401
- width = intersection_width(source_visual_bbox, target)
1402
- height = intersection_height(source_visual_bbox, target)
1403
- return {
1404
- "intersection_width": round(width, 3),
1405
- "intersection_height": round(height, 3),
1406
- "intersection_area": round(width * height, 3),
1407
- }
1408
-
1409
-
1410
- def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
1411
- if is_text_element(left) and not has_text_content(left):
1412
- return False
1413
- if is_text_element(right) and not has_text_content(right):
1414
- return False
1415
- if is_ghost_text(left) or is_ghost_text(right):
1416
- return False
1417
- if is_template_text_stack(left, right):
1418
- return False
1419
- if is_text_element(left) and is_text_element(right):
1420
- if is_similar_text_overlay(left, right):
1421
- return False
1422
- left_visual = estimate_text_visual_bbox(left)
1423
- right_visual = estimate_text_visual_bbox(right)
1424
- if left_visual is None or right_visual is None:
1425
- return False
1426
- overlap_area = intersection_area(left_visual, right_visual)
1427
- if overlap_area <= 0:
1428
- return False
1429
- smaller_area = min(
1430
- left_visual["width"] * left_visual["height"],
1431
- right_visual["width"] * right_visual["height"],
1432
- )
1433
- return smaller_area > 0 and overlap_area / smaller_area >= 0.30
1434
- return False
1435
-
1436
-
1437
- def build_whiteboard_external_overlap_issue(
1438
- whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
1439
- ) -> dict[str, Any]:
1440
- element_ids = [detail["element"] for detail in overlap_details]
1441
- return {
1442
- "level": "warning",
1443
- "code": "whiteboard_external_overlap",
1444
- "elements": [whiteboard["id"], *element_ids],
1445
- "message": f'whiteboard {whiteboard["id"]} overlaps {len(element_ids)} sibling elements across its boundary',
1446
- "hint": (
1447
- "Treat this as a static whiteboard container-bbox risk, not final visual proof. "
1448
- "After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
1449
- "the final authority because XML readback does not include whiteboard SVG/Mermaid internals."
1450
- ),
1451
- "overlaps": overlap_details,
1452
- }
1453
-
1454
-
1455
- def should_report_whiteboard_overlap(
1456
- whiteboard: dict[str, Any],
1457
- other: dict[str, Any],
1458
- slide_width: int | float,
1459
- slide_height: int | float,
1460
- ) -> dict[str, Any] | None:
1461
- if other is whiteboard or not intersects(whiteboard, other):
1462
- return None
1463
- if is_ghost_text(other):
1464
- return None
1465
- if contains(whiteboard, other):
1466
- return None
1467
- if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
1468
- return None
1469
- if is_background_container_for_whiteboard(other, whiteboard):
1470
- return None
1471
-
1472
- overlap_width = intersection_width(whiteboard, other)
1473
- overlap_height = intersection_height(whiteboard, other)
1474
- if overlap_width < 8 or overlap_height < 8:
1475
- return None
1476
-
1477
- other_area = element_area(other)
1478
- if other_area <= 0:
1479
- return None
1480
- overlap_area = overlap_width * overlap_height
1481
- overlap_ratio = overlap_area / other_area
1482
- if overlap_ratio < 0.15:
1483
- return None
1484
-
1485
- return {
1486
- "element": other["id"],
1487
- "kind": other["kind"],
1488
- "type": other.get("type"),
1489
- "overlap_width": overlap_width,
1490
- "overlap_height": overlap_height,
1491
- "target_overlap_ratio": round(overlap_ratio, 3),
1492
- }
1493
-
1494
-
1495
- def prune_contained_text_overlap_details(
1496
- overlap_details: list[dict[str, Any]], elements_by_id: dict[str, dict[str, Any]]
1497
- ) -> list[dict[str, Any]]:
1498
- pruned: list[dict[str, Any]] = []
1499
- for detail in overlap_details:
1500
- element = elements_by_id[detail["element"]]
1501
- if is_text_element(element):
1502
- has_reported_container = any(
1503
- detail["element"] != other_detail["element"]
1504
- and not is_text_element(elements_by_id[other_detail["element"]])
1505
- and contains(elements_by_id[other_detail["element"]], element)
1506
- for other_detail in overlap_details
1507
- )
1508
- if has_reported_container:
1509
- continue
1510
- pruned.append(detail)
1511
- return pruned
1512
-
1513
-
1514
- def detect_whiteboard_external_overlaps(
1515
- elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1516
- ) -> list[dict[str, Any]]:
1517
- issues: list[dict[str, Any]] = []
1518
- elements_by_id = {element["id"]: element for element in elements}
1519
- for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
1520
- overlap_details = [
1521
- detail
1522
- for element in elements
1523
- if (
1524
- detail := should_report_whiteboard_overlap(
1525
- whiteboard,
1526
- element,
1527
- slide_width,
1528
- slide_height,
1529
- )
1530
- )
1531
- is not None
1532
- ]
1533
- overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_id)
1534
- if overlap_details:
1535
- issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
1536
- return issues
1537
-
1538
-
1539
- def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
1540
- bbox = {key: element[key] for key in ("x", "y", "width", "height")}
1541
- if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
1542
- return bbox
1543
- rotation = element["rotation"]
1544
- if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
1545
- rotation = 0
1546
- rotation %= 360
1547
- if math.isclose(rotation, 0, abs_tol=1e-9):
1548
- return bbox
1549
- radians = math.radians(rotation)
1550
- sine = abs(math.sin(radians))
1551
- cosine = abs(math.cos(radians))
1552
- sine = 0 if math.isclose(sine, 0, abs_tol=1e-12) else sine
1553
- cosine = 0 if math.isclose(cosine, 0, abs_tol=1e-12) else cosine
1554
- rotated_width = element["width"] * cosine + element["height"] * sine
1555
- rotated_height = element["width"] * sine + element["height"] * cosine
1556
- return {
1557
- "x": element["x"] - (rotated_width - element["width"]) / 2,
1558
- "y": element["y"] - (rotated_height - element["height"]) / 2,
1559
- "width": rotated_width,
1560
- "height": rotated_height,
1561
- }
1562
-
1563
-
1564
- def detect_elements_out_of_canvas(
1565
- elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1566
- ) -> list[dict[str, Any]]:
1567
- issues: list[dict[str, Any]] = []
1568
- for element in (
1569
- element
1570
- for element in elements
1571
- if element["kind"] in {"table", "chart"}
1572
- or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
1573
- ):
1574
- bbox = element_canvas_bbox(element)
1575
- overflow = {
1576
- "left": max(-bbox["x"], 0),
1577
- "top": max(-bbox["y"], 0),
1578
- "right": max(bbox["x"] + bbox["width"] - slide_width, 0),
1579
- "bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
1580
- }
1581
- overflow_details = [
1582
- f"{side} by {amount:g}px"
1583
- for side, amount in overflow.items()
1584
- if amount > CANVAS_OVERFLOW_TOLERANCE
1585
- ]
1586
- if not overflow_details:
1587
- continue
1588
- issues.append(
1589
- {
1590
- "level": "error",
1591
- "code": f'{element["kind"]}_out_of_canvas',
1592
- "elements": [element["id"]],
1593
- "canvas": {"width": slide_width, "height": slide_height},
1594
- "bbox": bbox,
1595
- "overflow": overflow,
1596
- "message": (
1597
- f'{element["kind"]} {element["id"]} exceeds the {slide_width:g}x{slide_height:g} canvas '
1598
- f'({", ".join(overflow_details)})'
1599
- ),
1600
- "hint": (
1601
- "Move the table inside the canvas, reduce table.width/table.height, or split the table across "
1602
- "slides."
1603
- if element["kind"] == "table"
1604
- else f'Move the {element["kind"]} inside the canvas or reduce its width/height.'
1605
- ),
1606
- }
1607
- )
1608
- return issues
1609
-
1610
-
1611
- def extract_table_column_sizes(table_xml: str) -> list[int | float | None]:
1612
- sizes: list[int | float | None] = []
1613
- for match in re.finditer(r"<col\b([^>]*)/?>", table_xml):
1614
- attrs = match.group(1)
1615
- span = extract_numeric_attribute(attrs, "span") or 1
1616
- span_count = int(span) if math.isfinite(span) and span > 0 and float(span).is_integer() else 1
1617
- sizes.extend([extract_numeric_attribute(attrs, "width")] * span_count)
1618
- return sizes
1619
-
1620
-
1621
- def extract_table_row_sizes(table_xml: str) -> list[int | float | None]:
1622
- return [extract_numeric_attribute(match.group(1), "height") for match in re.finditer(r"<tr\b([^>]*)>", table_xml)]
1623
-
1624
-
1625
- def resolve_table_dimension(
1626
- table_xml: str,
1627
- declared_size: int | float | None,
1628
- extract_sizes: Any,
1629
- default_size: int | float,
1630
- ) -> tuple[int | float | None, dict[str, Any] | None]:
1631
- input_sizes = extract_sizes(table_xml)
1632
- if not input_sizes:
1633
- return declared_size, None
1634
- layout = solve_weighted_min_layout(
1635
- input_sizes, default_size, declared_size if is_filled_size(declared_size) else None
1636
- )
1637
- return layout["actual_size"], layout
1638
-
1639
-
1640
- def format_size(size: int | float) -> str:
1641
- return f"{size:g}"
1642
-
1643
-
1644
- def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1645
- issues: list[dict[str, Any]] = []
1646
- dimensions = {
1647
- "width": ("col", "column widths"),
1648
- "height": ("tr", "row heights"),
1649
- }
1650
- for table in (element for element in elements if element["kind"] == "table"):
1651
- for dimension, (child_tag, child_description) in dimensions.items():
1652
- target_size = table[f"declared_{dimension}"]
1653
- if not is_filled_size(target_size):
1654
- continue
1655
- layout = table["table_layouts"][dimension]
1656
- if layout is None:
1657
- continue
1658
- actual_size = layout["actual_size"]
1659
- if math.isclose(actual_size, target_size, rel_tol=1e-9, abs_tol=1e-9):
1660
- continue
1661
- issues.append(
1662
- {
1663
- "level": "info",
1664
- "code": "table_resolved_size_mismatch",
1665
- "elements": [table["id"]],
1666
- "dimension": dimension,
1667
- "declared_size": target_size,
1668
- "resolved_size": actual_size,
1669
- "resolved_sizes": layout["final_sizes"],
1670
- "message": (
1671
- f'table {table["id"]} declares {dimension}={format_size(target_size)}px, but its '
1672
- f"{child_description} resolve to {format_size(actual_size)}px"
1673
- ),
1674
- "hint": (
1675
- f"Set table.{dimension} to {format_size(actual_size)}px, or adjust <{child_tag}> sizes "
1676
- f"so their resolved total matches {format_size(target_size)}px."
1677
- ),
1678
- }
1679
- )
1680
- return issues
1681
-
1682
-
1683
- def segment_intersects_rect(
1684
- x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
1685
- ) -> bool:
1686
- """True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
1687
- left = rect["x"]
1688
- top = rect["y"]
1689
- right = rect["x"] + rect["width"]
1690
- bottom = rect["y"] + rect["height"]
1691
- if right <= left or bottom <= top:
1692
- return False
1693
- dx = x2 - x1
1694
- dy = y2 - y1
1695
- if dx == 0 and dy == 0:
1696
- return left <= x1 <= right and top <= y1 <= bottom
1697
- t_enter, t_exit = 0.0, 1.0
1698
- for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
1699
- if delta == 0:
1700
- if distance < 0:
1701
- return False
1702
- continue
1703
- t = distance / delta
1704
- if delta < 0:
1705
- t_enter = max(t_enter, t)
1706
- else:
1707
- t_exit = min(t_exit, t)
1708
- if t_enter > t_exit:
1709
- return False
1710
- return True
1711
-
1712
-
1713
- def line_text_graze_margin(text_element: dict[str, Any]) -> float:
1714
- font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
1715
- return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
1716
-
1717
-
1718
- def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
1719
- width = rect["width"] - 2 * margin
1720
- height = rect["height"] - 2 * margin
1721
- if width <= 0 or height <= 0:
1722
- return None
1723
- return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
1724
-
1725
-
1726
- def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
1727
- if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
1728
- return False
1729
- if not is_text_element(text_element) or not has_text_content(text_element):
1730
- return False
1731
- if is_ghost_text(text_element) or is_decorative_text(text_element):
1732
- return False
1733
- glyph_bbox = estimate_text_visual_bbox(text_element)
1734
- if glyph_bbox is None:
1735
- return False
1736
- # Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
1737
- # frame is exempt; only a line that actually cuts through the letterforms is a crossing.
1738
- target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
1739
- if target is None:
1740
- return False
1741
- return segment_intersects_rect(
1742
- line["startX"], line["startY"], line["endX"], line["endY"], target
1743
- )
1744
-
1745
-
1746
- def detect_line_text_crossings(
1747
- slide_xml: str, elements: list[dict[str, Any]]
1748
- ) -> list[dict[str, Any]]:
1749
- lines = extract_line_elements(slide_xml)
1750
- if not lines:
1751
- return []
1752
- text_elements = [element for element in elements if is_text_element(element)]
1753
- issues: list[dict[str, Any]] = []
1754
- for line in lines:
1755
- for text_element in text_elements:
1756
- if not line_crosses_text(line, text_element):
1757
- continue
1758
- issues.append(
1759
- {
1760
- "level": "error",
1761
- "code": "bbox_overlap",
1762
- "elements": [line["id"], text_element["id"]],
1763
- "message": f'line {line["id"]} crosses text {text_element["id"]}',
1764
- "hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
1765
- }
1766
- )
1767
- return issues
1768
-
1769
-
1770
- def lint_slide(
1771
- slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
1772
- ) -> dict[str, Any]:
1773
- elements = extract_elements(slide_xml)
1774
- issues: list[dict[str, Any]] = [
1775
- *detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
1776
- *detect_elements_out_of_canvas(elements, slide_width, slide_height),
1777
- *detect_table_layout_size_mismatches(elements),
1778
- *detect_text_may_overflow_shapes(elements),
1779
- *detect_image_text_occlusions(elements),
1780
- *detect_line_text_crossings(slide_xml, elements),
1781
- ]
1782
-
1783
- for index, left in enumerate(elements):
1784
- for right in elements[index + 1 :]:
1785
- horizontal_overflow = should_flag_horizontal_text_overflow(left, right)
1786
- if not horizontal_overflow and (not intersects(left, right) or not should_flag_overlap(left, right)):
1787
- continue
1788
- issues.append(
1789
- {
1790
- "level": "error",
1791
- "code": "bbox_overlap",
1792
- "elements": [left["id"], right["id"]],
1793
- "message": f'{left["id"]} overlaps {right["id"]}',
1794
- "hint": "Move or resize the elements so their visual bounds no longer intersect.",
1795
- **(
1796
- {"measurement": horizontal_text_overflow_measurement(left, right)}
1797
- if horizontal_overflow
1798
- else {}
1799
- ),
1800
- }
1801
- )
1802
-
1803
- return {
1804
- "slide_number": slide_number,
1805
- "element_count": len(elements),
1806
- "elements": elements,
1807
- "issues": issues,
1808
- }
1809
-
1810
-
1811
-
1812
- MIN_CONTAINER_WIDTH = 140
1813
- MIN_CONTAINER_HEIGHT = 160
1814
- MIN_SHORT_CARD_HEIGHT = 80
1815
- MIN_CONTAINER_AREA = 20_000
1816
- MIN_CONTENT_COVERAGE_RATIO = 0.15
1817
- MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
1818
- MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
1819
- SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
1820
- MIN_SIMILAR_SHORT_CARD_COUNT = 2
1821
- LARGE_VISUAL_CHILD_RATIO = 0.35
1822
- LAYOUT_PANEL_SPAN_RATIO = 0.90
1823
- IMAGE_OVERLAY_MATCH_RATIO = 0.90
1824
- DENSITY_CONTAINMENT_TOLERANCE = 8
1825
-
1826
-
1827
- def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1828
- left = max(element["x"], container["x"])
1829
- top = max(element["y"], container["y"])
1830
- right = min(element["x"] + element["width"], container["x"] + container["width"])
1831
- bottom = min(element["y"] + element["height"], container["y"] + container["height"])
1832
- if right <= left or bottom <= top:
1833
- return None
1834
- return {"x": left, "y": top, "width": right - left, "height": bottom - top}
1835
-
1836
-
1837
- def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
1838
- x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
1839
- area = 0
1840
- for left, right in zip(x_coordinates, x_coordinates[1:]):
1841
- intervals = sorted(
1842
- (rect["y"], rect["y"] + rect["height"])
1843
- for rect in rectangles
1844
- if rect["x"] < right and rect["x"] + rect["width"] > left
1845
- )
1846
- covered_height = 0
1847
- interval_end: int | float | None = None
1848
- for top, bottom in intervals:
1849
- if interval_end is None:
1850
- covered_height += bottom - top
1851
- interval_end = bottom
1852
- elif bottom > interval_end:
1853
- covered_height += bottom - max(top, interval_end)
1854
- interval_end = bottom
1855
- area += (right - left) * covered_height
1856
- return area
1857
-
1858
-
1859
- def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1860
- return sum(
1861
- other is not element
1862
- and is_visually_rendered(other)
1863
- and other["kind"] == "shape"
1864
- and other["type"] == "rect"
1865
- and other["width"] >= MIN_CONTAINER_WIDTH
1866
- and other["height"] >= MIN_SHORT_CARD_HEIGHT
1867
- and element_area(other) >= MIN_CONTAINER_AREA
1868
- and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
1869
- <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1870
- and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
1871
- <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1872
- for other in elements
1873
- ) >= MIN_SIMILAR_SHORT_CARD_COUNT
1874
-
1875
-
1876
- def is_layout_container(
1877
- element: dict[str, Any],
1878
- slide_width: int | float,
1879
- slide_height: int | float,
1880
- elements: list[dict[str, Any]] | None = None,
1881
- ) -> bool:
1882
- has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
1883
- elements is not None
1884
- and element["height"] >= MIN_SHORT_CARD_HEIGHT
1885
- and has_similar_short_card_peer(element, elements)
1886
- )
1887
- return (
1888
- element["kind"] == "shape"
1889
- and element["type"] == "rect"
1890
- and is_visually_rendered(element)
1891
- and element["width"] >= MIN_CONTAINER_WIDTH
1892
- and has_supported_height
1893
- and element_area(element) >= MIN_CONTAINER_AREA
1894
- and not (
1895
- element["x"] <= 2
1896
- and element["y"] <= 2
1897
- and element["width"] >= slide_width - 4
1898
- and element["height"] >= slide_height - 4
1899
- )
1900
- )
1901
-
1902
-
1903
- def is_edge_spanning_layout_panel(
1904
- element: dict[str, Any], slide_width: int | float, slide_height: int | float
1905
- ) -> bool:
1906
- touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
1907
- touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
1908
- return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
1909
- touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
1910
- )
1911
-
1912
-
1913
- def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1914
- container_area = element_area(container)
1915
- return any(
1916
- element["kind"] == "img"
1917
- and is_visually_rendered(element)
1918
- and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
1919
- for element in elements
1920
- )
1921
-
1922
-
1923
- def is_nested_in_layout_panel(
1924
- container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1925
- ) -> bool:
1926
- return any(
1927
- element is not container
1928
- and element["kind"] == "shape"
1929
- and element["type"] == "rect"
1930
- and is_visually_rendered(element)
1931
- and is_edge_spanning_layout_panel(element, slide_width, slide_height)
1932
- and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
1933
- for element in elements
1934
- )
1935
-
1936
-
1937
- def extract_density_elements(slide_xml: str) -> list[dict[str, Any]]:
1938
- elements = extract_elements(slide_xml)
1939
- elements_by_id = {element["id"]: element for element in elements}
1940
- root = ET.fromstring(slide_xml)
1941
- for node in root.iter():
1942
- if xml_local_name(node.tag) != "shape":
1943
- continue
1944
- element = elements_by_id.get(node.attrib.get("id", ""))
1945
- if element is None:
1946
- continue
1947
- content_node = next(
1948
- (child for child in node if xml_local_name(child.tag) == "content"),
1949
- None,
1950
- )
1951
- paragraphs = (
1952
- [
1953
- " ".join("".join(paragraph.itertext()).split())
1954
- for paragraph in content_node.iter()
1955
- if xml_local_name(paragraph.tag) == "p"
1956
- ]
1957
- if content_node is not None
1958
- else []
1959
- )
1960
- raw_font_size = (
1961
- content_node.attrib.get("fontSize") if content_node is not None else None
1962
- ) or node.attrib.get("fontSize")
1963
- try:
1964
- base_font_size = float(raw_font_size or 16)
1965
- except ValueError:
1966
- base_font_size = 16.0
1967
- element.update(
1968
- {
1969
- "textType": content_node.attrib.get("textType") if content_node is not None else None,
1970
- "textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
1971
- "autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
1972
- "fontSize": base_font_size,
1973
- "text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
1974
- }
1975
- )
1976
- if not has_text_content(element):
1977
- continue
1978
- declared_font_sizes = []
1979
- for descendant in node.iter():
1980
- raw_declared_font_size = descendant.attrib.get("fontSize")
1981
- if raw_declared_font_size is None:
1982
- continue
1983
- try:
1984
- declared_font_sizes.append(float(raw_declared_font_size))
1985
- except ValueError:
1986
- continue
1987
- if declared_font_sizes:
1988
- element["fontSize"] = max(declared_font_sizes)
1989
- for match in re.finditer(r"<icon\b([^>]*)>", slide_xml):
1990
- attrs = match.group(1)
1991
- x = extract_numeric_attribute(attrs, "topLeftX")
1992
- y = extract_numeric_attribute(attrs, "topLeftY")
1993
- width = extract_numeric_attribute(attrs, "width")
1994
- height = extract_numeric_attribute(attrs, "height")
1995
- if any(value is None for value in (x, y, width, height)):
1996
- continue
1997
- icon_alpha = extract_numeric_attribute(attrs, "alpha")
1998
- elements.append(
1999
- {
2000
- "id": extract_attribute(attrs, "id") or f"icon-{len(elements) + 1}",
2001
- "kind": "icon",
2002
- "type": "icon",
2003
- "x": x,
2004
- "y": y,
2005
- "width": width,
2006
- "height": height,
2007
- "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2008
- "alpha": icon_alpha if icon_alpha is not None else 1,
2009
- "order": len(elements),
2010
- }
2011
- )
2012
- for match in re.finditer(r"<polyline\b([^>]*)>", slide_xml):
2013
- attrs = match.group(1)
2014
- x = extract_numeric_attribute(attrs, "topLeftX")
2015
- y = extract_numeric_attribute(attrs, "topLeftY")
2016
- width = extract_numeric_attribute(attrs, "width")
2017
- height = extract_numeric_attribute(attrs, "height")
2018
- if any(value is None for value in (x, y, width, height)):
2019
- continue
2020
- polyline_alpha = extract_numeric_attribute(attrs, "alpha")
2021
- elements.append(
2022
- {
2023
- "id": extract_attribute(attrs, "id") or f"polyline-{len(elements) + 1}",
2024
- "kind": "polyline",
2025
- "type": "polyline",
2026
- "x": x,
2027
- "y": y,
2028
- "width": width,
2029
- "height": height,
2030
- "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2031
- "alpha": polyline_alpha if polyline_alpha is not None else 1,
2032
- "order": len(elements),
2033
- }
2034
- )
2035
- for line_element in extract_line_elements(slide_xml):
2036
- line_element["order"] = len(elements)
2037
- elements.append(line_element)
2038
- return elements
2039
-
2040
-
2041
- def is_visually_rendered(element: dict[str, Any]) -> bool:
2042
- return element.get("alpha", 1) > 0
2043
-
2044
-
2045
- def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
2046
- if not is_visually_rendered(element):
2047
- return None
2048
- if is_text_element(element):
2049
- estimated = estimate_text_visual_bbox(element)
2050
- return clipped_bbox(estimated, container) if estimated else None
2051
- return clipped_bbox(element, container)
2052
-
2053
-
2054
- def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
2055
- if container["kind"] != "shape" or not has_text_content(container):
2056
- return None
2057
- text_proxy = {**container, "type": "text"}
2058
- estimated = estimate_text_visual_bbox(text_proxy)
2059
- return clipped_bbox(estimated, container) if estimated else None
2060
-
2061
-
2062
- def slide_content_visual_bbox(
2063
- element: dict[str, Any], slide_bbox: dict[str, int | float]
2064
- ) -> dict[str, int | float] | None:
2065
- if not is_visually_rendered(element):
2066
- return None
2067
- if is_text_element(element):
2068
- estimated = estimate_text_visual_bbox(element)
2069
- return clipped_bbox(estimated, slide_bbox) if estimated else None
2070
- if element["kind"] == "shape" and has_text_content(element):
2071
- estimated = own_text_visual_bbox(element)
2072
- return clipped_bbox(estimated, slide_bbox) if estimated else None
2073
- if element["kind"] == "line":
2074
- # a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
2075
- # treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
2076
- return clipped_bbox(line_stroke_bbox(element), slide_bbox)
2077
- if element["kind"] in {"img", "chart", "table", "whiteboard", "icon", "polyline"}:
2078
- return clipped_bbox(element, slide_bbox)
2079
- return None
2080
-
2081
-
2082
- def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
2083
- return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
2084
-
2085
-
2086
- def is_slide_content_present(
2087
- element: dict[str, Any], slide_bbox: dict[str, int | float]
2088
- ) -> bool:
2089
- # Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
2090
- # *anything* rendered here", not the richer "counts toward meaningful content density" bar
2091
- # that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
2092
- # decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
2093
- # count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
2094
- # instead of maintaining an allow-list that silently treats unlisted kinds as blank.
2095
- if not is_visually_rendered(element):
2096
- return False
2097
- if (
2098
- element["kind"] == "shape"
2099
- and element["type"] == "rect"
2100
- and not has_text_content(element)
2101
- and element["x"] <= 2
2102
- and element["y"] <= 2
2103
- and element["width"] >= slide_bbox["width"] - 4
2104
- and element["height"] >= slide_bbox["height"] - 4
2105
- ):
2106
- # A full-canvas plain rect is a background panel, not content -- same reasoning as
2107
- # is_layout_container's existing background exclusion. A slide with nothing else on it
2108
- # is still effectively blank.
2109
- return False
2110
- bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
2111
- return clipped_bbox(bbox, slide_bbox) is not None
2112
-
2113
-
2114
- def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
2115
- if element["kind"] not in {"img", "chart", "table", "whiteboard"}:
2116
- return False
2117
- if not is_visually_rendered(element):
2118
- return False
2119
- return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
2120
-
2121
-
2122
- def detect_sparse_container_content(
2123
- elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2124
- ) -> list[dict[str, Any]]:
2125
- issues: list[dict[str, Any]] = []
2126
- for container in (
2127
- element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
2128
- ):
2129
- if (
2130
- is_edge_spanning_layout_panel(container, slide_width, slide_height)
2131
- or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
2132
- or has_matching_image_overlay(container, elements)
2133
- ):
2134
- continue
2135
- children = [
2136
- element
2137
- for element in elements
2138
- if element is not container
2139
- and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2140
- ]
2141
- if any(is_large_visual_child(child, container) for child in children):
2142
- continue
2143
- own_text_bbox = own_text_visual_bbox(container)
2144
- rectangles = ([own_text_bbox] if own_text_bbox else []) + [
2145
- bbox for child in children if (bbox := visual_bbox(child, container)) is not None
2146
- ]
2147
- content_area = rectangle_union_area(rectangles) if rectangles else 0
2148
- coverage_ratio = content_area / element_area(container)
2149
- if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
2150
- continue
2151
- issues.append(
2152
- {
2153
- "level": "warning",
2154
- "code": "sparse_container_content",
2155
- "target": {
2156
- "slide_number": slide_number,
2157
- "container_id": container["id"],
2158
- "container_type": container["type"],
2159
- "bbox": {key: container[key] for key in ("x", "y", "width", "height")},
2160
- },
2161
- "rule": {
2162
- "name": "large_container_visible_content_coverage",
2163
- "threshold": MIN_CONTENT_COVERAGE_RATIO,
2164
- "comparison": "content_coverage_ratio < threshold",
2165
- },
2166
- "measurement": {
2167
- "container_area": element_area(container),
2168
- "visible_content_area": round(content_area, 3),
2169
- "content_coverage_ratio": round(coverage_ratio, 3),
2170
- "content_element_count": len(children) + (1 if own_text_bbox else 0),
2171
- },
2172
- "elements": [container["id"], *[child["id"] for child in children]],
2173
- }
2174
- )
2175
- return issues
2176
-
2177
-
2178
- def detect_sparse_slide_content(
2179
- elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2180
- ) -> list[dict[str, Any]]:
2181
- slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2182
- content = [
2183
- (element, bbox)
2184
- for element in elements
2185
- if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
2186
- ]
2187
- if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
2188
- return []
2189
- content_area = rectangle_union_area([bbox for _, bbox in content])
2190
- slide_area = slide_width * slide_height
2191
- coverage_ratio = content_area / slide_area
2192
- if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
2193
- return []
2194
- return [
2195
- {
2196
- "level": "warning",
2197
- "code": "sparse_slide_content",
2198
- "target": {
2199
- "slide_number": slide_number,
2200
- "bbox": slide_bbox,
2201
- },
2202
- "rule": {
2203
- "name": "slide_visible_content_coverage",
2204
- "threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
2205
- "comparison": "content_coverage_ratio < threshold",
2206
- },
2207
- "measurement": {
2208
- "slide_area": slide_area,
2209
- "visible_content_area": round(content_area, 3),
2210
- "content_coverage_ratio": round(coverage_ratio, 3),
2211
- "content_element_count": len(content),
2212
- },
2213
- "elements": [element["id"] for element, _ in content],
2214
- }
2215
- ]
2216
-
2217
-
2218
- def detect_blank_slide(
2219
- elements: list[dict[str, Any]],
2220
- slide_number: int,
2221
- slide_width: int | float,
2222
- slide_height: int | float,
2223
- ) -> list[dict[str, Any]]:
2224
- slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2225
- visible_elements = [
2226
- element for element in elements if is_slide_content_present(element, slide_bbox)
2227
- ]
2228
- if visible_elements:
2229
- return []
2230
- return [
2231
- {
2232
- "level": "error",
2233
- "code": "blank_slide",
2234
- "schema_version": "2.0",
2235
- "target": {"slide_number": slide_number},
2236
- "rule": {
2237
- "name": "slide_has_visible_content",
2238
- "comparison": "visible_element_count == 0",
2239
- },
2240
- "measurement": {
2241
- "visible_element_count": 0,
2242
- "declared_element_count": len(elements),
2243
- },
2244
- "elements": [element["id"] for element in elements],
2245
- "message": "slide has no visible content beyond empty layout shapes",
2246
- "hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
2247
- }
2248
- ]
2249
-
2250
-
2251
-
2252
- RULE_METADATA: dict[str, dict[str, Any]] = {
2253
- "xml_not_well_formed": {
2254
- "name": "xml_is_well_formed",
2255
- "comparison": "xml_parse_error == false",
2256
- },
2257
- "sml_prefixed_tag": {
2258
- "name": "sml_uses_default_namespace",
2259
- "comparison": "prefixed_sml_tag_count == 0",
2260
- },
2261
- "sxsd_unsupported_tag": {
2262
- "name": "tag_is_supported_by_slides_xml_schema",
2263
- "comparison": "unsupported_tag_count == 0",
2264
- },
2265
- "sxsd_unsupported_attr": {
2266
- "name": "attribute_is_supported_by_slides_xml_schema",
2267
- "comparison": "unsupported_attribute_count == 0",
2268
- },
2269
- "icon_missing_fill_color": {
2270
- "name": "icon_has_visible_fill_color",
2271
- "comparison": "fill_color_present == true",
2272
- },
2273
- "icon_transparent_fill_color": {
2274
- "name": "icon_has_visible_fill_color",
2275
- "comparison": "fill_alpha > 0",
2276
- },
2277
- "iconpark_unsupported_icon_type": {
2278
- "name": "iconpark_type_is_supported",
2279
- "comparison": "icon_type in iconpark_index",
2280
- },
2281
- "bbox_overlap": {
2282
- "name": "text_visual_bounds_do_not_overlap",
2283
- "comparison": "intersection_area == 0",
2284
- },
2285
- "text_may_overflow_shape": {
2286
- "name": "estimated_text_fits_declared_shape",
2287
- "comparison": "estimated_height <= available_height",
2288
- },
2289
- "whiteboard_external_overlap": {
2290
- "name": "whiteboard_does_not_cross_sibling_content",
2291
- "comparison": "external_overlap_count == 0",
2292
- },
2293
- "image_covers_text": {
2294
- "name": "image_does_not_cover_text",
2295
- "comparison": "intersection_area == 0",
2296
- },
2297
- "image_may_cover_vertical_text": {
2298
- "name": "image_vertical_text_occlusion_requires_review",
2299
- "comparison": "intersection_area == 0",
2300
- },
2301
- "table_resolved_size_mismatch": {
2302
- "name": "table_declared_size_matches_resolved_grid",
2303
- "comparison": "declared_size == resolved_size",
2304
- },
2305
- "blank_slide": {
2306
- "name": "slide_has_visible_content",
2307
- "comparison": "visible_element_count > 0",
2308
- },
2309
- }
2310
-
2311
-
2312
- def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
2313
- if issue.get("rule"):
2314
- return {**issue["rule"], "id": issue["code"]}
2315
- if issue["code"].endswith("_out_of_canvas"):
2316
- return {
2317
- "id": issue["code"],
2318
- "name": "element_stays_within_slide_canvas",
2319
- "comparison": "max(left, top, right, bottom overflow) == 0",
2320
- }
2321
- return {
2322
- "id": issue["code"],
2323
- **RULE_METADATA.get(
2324
- issue["code"],
2325
- {"name": issue["code"], "comparison": "violation_count == 0"},
2326
- ),
2327
- }
2328
-
2329
-
2330
- def issue_measurement(
2331
- issue: dict[str, Any], elements_by_id: dict[str, dict[str, Any]]
2332
- ) -> dict[str, Any]:
2333
- if issue.get("measurement") is not None:
2334
- return issue["measurement"]
2335
- if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
2336
- left = elements_by_id.get(issue["elements"][0])
2337
- right = elements_by_id.get(issue["elements"][1])
2338
- if left and right:
2339
- left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
2340
- right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
2341
- width = intersection_width(left_box, right_box)
2342
- height = intersection_height(left_box, right_box)
2343
- return {
2344
- "intersection_width": round(width, 3),
2345
- "intersection_height": round(height, 3),
2346
- "intersection_area": round(width * height, 3),
2347
- }
2348
- if issue["code"].endswith("_out_of_canvas"):
2349
- return {
2350
- "canvas": issue.get("canvas"),
2351
- "bbox": issue.get("bbox"),
2352
- "overflow": issue.get("overflow"),
2353
- }
2354
- measurement_keys = (
2355
- "line",
2356
- "column",
2357
- "tag",
2358
- "attr",
2359
- "iconType",
2360
- "line_count",
2361
- "line_height",
2362
- "estimated_height",
2363
- "available_height",
2364
- "overflow",
2365
- "dimension",
2366
- "declared_size",
2367
- "resolved_size",
2368
- "resolved_sizes",
2369
- "overlaps",
2370
- )
2371
- measured = {key: issue[key] for key in measurement_keys if key in issue}
2372
- return measured or {"violation_count": 1}
2373
-
2374
-
2375
- def related_object(element: dict[str, Any]) -> dict[str, Any]:
2376
- return {
2377
- "element_id": element["id"],
2378
- "kind": element["kind"],
2379
- "type": element["type"],
2380
- "bbox": {key: element[key] for key in ("x", "y", "width", "height")},
2381
- }
2382
-
2383
-
2384
- def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
2385
- elements: list[dict[str, Any]] = []
2386
- for match in re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml):
2387
- attrs = match.group(1)
2388
- start_x = extract_numeric_attribute(attrs, "startX")
2389
- start_y = extract_numeric_attribute(attrs, "startY")
2390
- end_x = extract_numeric_attribute(attrs, "endX")
2391
- end_y = extract_numeric_attribute(attrs, "endY")
2392
- if any(value is None for value in (start_x, start_y, end_x, end_y)):
2393
- continue
2394
- line_alpha = extract_numeric_attribute(attrs, "alpha")
2395
- base_alpha = line_alpha if line_alpha is not None else 1
2396
- border_alpha = 1
2397
- if match.group(2) != "/":
2398
- close_index = slide_xml.find("</line>", match.end())
2399
- body = slide_xml[match.end() : close_index] if close_index != -1 else ""
2400
- border_attrs = extract_tag_attributes(body, "border")
2401
- color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
2402
- if isinstance(color_alpha, (int, float)):
2403
- border_alpha = color_alpha
2404
- elements.append(
2405
- {
2406
- "id": extract_attribute(attrs, "id") or f"line-{len(elements) + 1}",
2407
- "kind": "line",
2408
- "type": "line",
2409
- "x": min(start_x, end_x),
2410
- "y": min(start_y, end_y),
2411
- "width": abs(end_x - start_x),
2412
- "height": abs(end_y - start_y),
2413
- "startX": start_x,
2414
- "startY": start_y,
2415
- "endX": end_x,
2416
- "endY": end_y,
2417
- "rotation": 0,
2418
- "alpha": base_alpha * border_alpha,
2419
- "order": len(elements),
2420
- }
2421
- )
2422
- return elements
2423
-
2424
-
2425
- def normalize_issue(
2426
- issue: dict[str, Any],
2427
- slide_number: int | None,
2428
- elements_by_id: dict[str, dict[str, Any]],
2429
- ) -> dict[str, Any]:
2430
- normalized = dict(issue)
2431
- element_ids = list(dict.fromkeys(normalized.get("elements", [])))
2432
- normalized["schema_version"] = "2.0"
2433
- normalized["element_ids"] = element_ids
2434
- normalized["target"] = {
2435
- **({"slide_number": slide_number} if slide_number is not None else {}),
2436
- **normalized.get("target", {}),
2437
- }
2438
- normalized["rule"] = issue_rule(normalized)
2439
- normalized["measurement"] = issue_measurement(normalized, elements_by_id)
2440
- normalized["related_objects"] = [
2441
- related_object(elements_by_id[element_id])
2442
- for element_id in element_ids
2443
- if element_id in elements_by_id
2444
- ]
2445
- if normalized["code"] == "sparse_container_content":
2446
- ratio = normalized["measurement"]["content_coverage_ratio"]
2447
- threshold = normalized["rule"]["threshold"]
2448
- container_id = normalized["target"].get("container_id", "unknown")
2449
- normalized.setdefault(
2450
- "message",
2451
- f"large card {container_id} content coverage {ratio:.1%} is below {threshold:.1%}",
2452
- )
2453
- normalized.setdefault(
2454
- "hint",
2455
- "Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
2456
- )
2457
- elif normalized["code"] == "sparse_slide_content":
2458
- ratio = normalized["measurement"]["content_coverage_ratio"]
2459
- threshold = normalized["rule"]["threshold"]
2460
- normalized.setdefault(
2461
- "message",
2462
- f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
2463
- )
2464
- normalized.setdefault(
2465
- "hint",
2466
- "Review the rendered screenshot to decide whether the page is intentionally sparse.",
2467
- )
2468
- else:
2469
- normalized.setdefault("message", normalized["code"].replace("_", " "))
2470
- normalized.setdefault(
2471
- "hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
2472
- )
2473
- return normalized
2474
-
2475
-
2476
- def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
2477
- if errors:
2478
- return "blocked"
2479
- if warnings:
2480
- return "needs_screenshot_review"
2481
- return "passed"
2482
-
2483
-
2484
- def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
2485
- if issue.get("code") == "sxsd_unsupported_declaration":
2486
- return False
2487
- if root_name == "slide":
2488
- return True
2489
- path = issue.get("path")
2490
- if not isinstance(path, str):
2491
- return False
2492
- if path.startswith("presentation/slide/"):
2493
- return True
2494
- return path == "presentation/slide" and (
2495
- issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
2496
- )
2497
-
2498
-
2499
- def build_result(
2500
- source_path: str | None,
2501
- slide_size: dict[str, int | float],
2502
- top_level_issues: list[dict[str, Any]],
2503
- slides: list[dict[str, Any]],
2504
- ) -> dict[str, Any]:
2505
- document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
2506
- document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
2507
- document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
2508
- error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
2509
- warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
2510
- info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
2511
- all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
2512
- all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
2513
- status = slide_status(all_errors, all_warnings)
2514
- result: dict[str, Any] = {
2515
- "schema_version": "2.0",
2516
- "tool": "xml_text_overlap_lint",
2517
- "file": source_path,
2518
- "slide_size": slide_size,
2519
- "summary": {
2520
- "slide_count": len(slides),
2521
- "error_count": error_count,
2522
- "warning_count": warning_count,
2523
- "info_count": info_count,
2524
- "status": status,
2525
- "release_ready": error_count == 0,
2526
- "screenshot_review_required": warning_count > 0,
2527
- },
2528
- "document": {
2529
- "errors": document_errors,
2530
- "warnings": document_warnings,
2531
- "infos": document_infos,
2532
- },
2533
- "slides": slides,
2534
- }
2535
- if top_level_issues:
2536
- result["issues"] = top_level_issues
2537
- return result
2538
-
2539
-
2540
- def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
2541
- root, xml_error = parse_xml_root(xml)
2542
- if xml_error:
2543
- issue = normalize_issue(xml_error, None, {})
2544
- return build_result(
2545
- source_path,
2546
- {"width": 960, "height": 540},
2547
- [issue],
2548
- [],
2549
- )
2550
- if root is None:
2551
- raise AssertionError("parse_xml_root must return a root or error")
2552
-
2553
- namespace_issues = validate_sml_tag_prefixes(xml)
2554
- root_name = xml_local_name(root.tag)
2555
- sxsd_issues = validate_sxsd_document(xml, root)
2556
- iconpark_issues = validate_iconpark_icon_types(root)
2557
- top_level_issues = [
2558
- normalize_issue(issue, None, {})
2559
- for issue in [
2560
- *namespace_issues,
2561
- *[
2562
- issue
2563
- for issue in sxsd_issues
2564
- if not is_slide_scoped_sxsd_issue(issue, root_name)
2565
- ],
2566
- *iconpark_issues,
2567
- ]
2568
- ]
2569
- if any(issue["level"] == "error" for issue in top_level_issues):
2570
- return build_result(
2571
- source_path,
2572
- {"width": 960, "height": 540},
2573
- top_level_issues,
2574
- [],
2575
- )
2576
-
2577
- presentation = parse_presentation(root)
2578
- slide_roots = presentation["slide_roots"]
2579
- slides: list[dict[str, Any]] = []
2580
- for index, slide_xml in enumerate(presentation["slides"]):
2581
- slide_number = index + 1
2582
- slide_root = slide_roots[index]
2583
- slide_sxsd_issues = [
2584
- normalize_issue(issue, slide_number, {})
2585
- for issue in validate_sxsd_document(slide_xml, slide_root)
2586
- ]
2587
- slide_sxsd_errors = [
2588
- issue for issue in slide_sxsd_issues if issue["level"] == "error"
2589
- ]
2590
- if slide_sxsd_errors:
2591
- slide_sxsd_warnings = [
2592
- issue for issue in slide_sxsd_issues if issue["level"] == "warning"
2593
- ]
2594
- slides.append(
2595
- {
2596
- "slide_number": slide_number,
2597
- "status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
2598
- "element_count": 0,
2599
- "errors": slide_sxsd_errors,
2600
- "warnings": slide_sxsd_warnings,
2601
- "infos": [],
2602
- "issues": slide_sxsd_issues,
2603
- }
2604
- )
2605
- continue
2606
-
2607
- geometry = lint_slide(
2608
- slide_xml,
2609
- slide_number,
2610
- presentation["width"],
2611
- presentation["height"],
2612
- )
2613
- density_elements = extract_density_elements(slide_xml)
2614
- extra_elements = [
2615
- element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
2616
- ]
2617
- elements_by_id = {
2618
- element["id"]: element for element in [*density_elements, *extra_elements]
2619
- }
2620
- # geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
2621
- # decided with inside lint_slide; prefer them so measurement/related_objects stay consistent
2622
- # with whatever actually triggered the issue, instead of density_elements' separate re-parse.
2623
- elements_by_id.update({element["id"]: element for element in geometry["elements"]})
2624
- extra_overflow_issues = detect_elements_out_of_canvas(
2625
- extra_elements,
2626
- presentation["width"],
2627
- presentation["height"],
2628
- )
2629
- raw_issues = [
2630
- *geometry["issues"],
2631
- *extra_overflow_issues,
2632
- *detect_blank_slide(
2633
- density_elements,
2634
- slide_number,
2635
- presentation["width"],
2636
- presentation["height"],
2637
- ),
2638
- *detect_sparse_container_content(
2639
- density_elements,
2640
- slide_number,
2641
- presentation["width"],
2642
- presentation["height"],
2643
- ),
2644
- *detect_sparse_slide_content(
2645
- density_elements,
2646
- slide_number,
2647
- presentation["width"],
2648
- presentation["height"],
2649
- ),
2650
- ]
2651
- issues = [
2652
- *slide_sxsd_issues,
2653
- *[
2654
- normalize_issue(issue, slide_number, elements_by_id)
2655
- for issue in raw_issues
2656
- ],
2657
- ]
2658
- errors = [issue for issue in issues if issue["level"] == "error"]
2659
- warnings = [issue for issue in issues if issue["level"] == "warning"]
2660
- infos = [issue for issue in issues if issue["level"] == "info"]
2661
- slides.append(
2662
- {
2663
- "slide_number": slide_number,
2664
- "status": slide_status(errors, warnings),
2665
- "element_count": len(elements_by_id),
2666
- "errors": errors,
2667
- "warnings": warnings,
2668
- "infos": infos,
2669
- "issues": issues,
2670
- }
2671
- )
2672
-
2673
- return build_result(
2674
- source_path,
2675
- {"width": presentation["width"], "height": presentation["height"]},
2676
- top_level_issues,
2677
- slides,
2678
- )
2679
-
2680
-
2681
- def print_usage() -> None:
2682
- print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
2683
-
2684
7
 
2685
- def run_cli(argv: list[str] | None = None) -> None:
2686
- options = parse_args(argv or sys.argv[1:])
2687
- if options.get("help") or options.get("--help"):
2688
- print_usage()
2689
- raise SystemExit(0)
2690
- if not options.get("input"):
2691
- print_usage()
2692
- fail("--input is required")
2693
- input_path = Path(options["input"]).resolve()
2694
- result = lint_xml(read_file(input_path), str(input_path))
2695
- print(json.dumps(result, ensure_ascii=False, indent=2))
2696
- if result["summary"]["error_count"] > 0:
2697
- raise SystemExit(1)
8
+ from xml_lint import * # noqa: F401,F403
9
+ from xml_lint import XmlLayoutLintError, run_cli
2698
10
 
2699
11
 
2700
12
  if __name__ == "__main__":