@amaster.ai/pi-lark 0.1.5 → 0.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (258) hide show
  1. package/README.md +5 -1
  2. package/dist/config.d.ts +1 -1
  3. package/dist/config.d.ts.map +1 -1
  4. package/dist/config.js +2 -2
  5. package/dist/config.js.map +1 -1
  6. package/dist/index.d.ts.map +1 -1
  7. package/dist/index.js +2 -1
  8. package/dist/index.js.map +1 -1
  9. package/package.json +4 -4
  10. package/skills/lark-approval/references/lark-approval-initiate.md +2 -5
  11. package/skills/lark-approval/references/lark-approval-instances-initiated.md +6 -0
  12. package/skills/lark-approval/references/lark-approval-tasks-query.md +9 -0
  13. package/skills/lark-approval/references/lark-approval-tasks-rollback.md +8 -2
  14. package/skills/lark-apps/SKILL.md +46 -16
  15. package/skills/lark-apps/creative-design/agents/assets/vision-probe.png +0 -0
  16. package/skills/lark-apps/creative-design/agents/fork-verifier-agent.md +71 -0
  17. package/skills/lark-apps/creative-design/agents/vision-probe-agent.md +41 -0
  18. package/skills/lark-apps/creative-design/assets/index.html +27 -0
  19. package/skills/lark-apps/creative-design/creative-design.md +239 -0
  20. package/skills/lark-apps/creative-design/references/aily.md +39 -0
  21. package/skills/lark-apps/creative-design/references/animated-video.md +34 -0
  22. package/skills/lark-apps/creative-design/references/charts.md +165 -0
  23. package/skills/lark-apps/creative-design/references/claude.md +36 -0
  24. package/skills/lark-apps/creative-design/references/codex.md +32 -0
  25. package/skills/lark-apps/creative-design/references/data-report.md +108 -0
  26. package/skills/lark-apps/creative-design/references/frontend-design.md +71 -0
  27. package/skills/lark-apps/creative-design/references/hi-fi-design.md +32 -0
  28. package/skills/lark-apps/creative-design/references/interactive-prototype.md +24 -0
  29. package/skills/lark-apps/creative-design/references/make-a-deck.md +133 -0
  30. package/skills/lark-apps/creative-design/references/visual-exposure.md +82 -0
  31. package/skills/lark-apps/creative-design/references/wireframe.md +14 -0
  32. package/skills/lark-apps/creative-design/starter-components/android-frame.jsx +188 -0
  33. package/skills/lark-apps/creative-design/starter-components/animations.jsx +773 -0
  34. package/skills/lark-apps/creative-design/starter-components/browser-window.jsx +122 -0
  35. package/skills/lark-apps/creative-design/starter-components/deck-stage.js +2483 -0
  36. package/skills/lark-apps/creative-design/starter-components/design-canvas.jsx +1432 -0
  37. package/skills/lark-apps/creative-design/starter-components/ios-frame.jsx +270 -0
  38. package/skills/lark-apps/creative-design/starter-components/macos-window.jsx +197 -0
  39. package/skills/lark-apps/creative-design/starter-components/tweaks-panel.jsx +752 -0
  40. package/skills/lark-apps/references/lark-apps-access-scope-set.md +1 -1
  41. package/skills/lark-apps/references/lark-apps-automation.md +242 -0
  42. package/skills/lark-apps/references/lark-apps-cache.md +61 -0
  43. package/skills/lark-apps/references/lark-apps-cloud-dev.md +0 -1
  44. package/skills/lark-apps/references/lark-apps-create.md +1 -2
  45. package/skills/lark-apps/references/lark-apps-db-execute.md +186 -2
  46. package/skills/lark-apps/references/lark-apps-db.md +4 -4
  47. package/skills/lark-apps/references/lark-apps-env-pull.md +1 -1
  48. package/skills/lark-apps/references/lark-apps-file.md +2 -2
  49. package/skills/lark-apps/references/lark-apps-get.md +43 -0
  50. package/skills/lark-apps/references/lark-apps-git-credential.md +1 -1
  51. package/skills/lark-apps/references/lark-apps-html-publish.md +5 -4
  52. package/skills/lark-apps/references/lark-apps-init.md +2 -3
  53. package/skills/lark-apps/references/lark-apps-list.md +1 -1
  54. package/skills/lark-apps/references/lark-apps-local-dev.md +54 -11
  55. package/skills/lark-apps/references/lark-apps-openapi-key.md +1 -1
  56. package/skills/lark-apps/references/lark-apps-release-create.md +5 -3
  57. package/skills/lark-apps/references/lark-apps-release-get.md +3 -3
  58. package/skills/lark-apps/references/lark-apps-role.md +133 -0
  59. package/skills/lark-base/SKILL.md +26 -15
  60. package/skills/lark-base/references/dashboard-block-data-config.md +28 -2
  61. package/skills/lark-base/references/lark-base-cell-value.md +12 -7
  62. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +7 -7
  63. package/skills/lark-base/references/lark-base-dashboard.md +11 -2
  64. package/skills/lark-base/references/lark-base-data-query.md +20 -11
  65. package/skills/lark-base/references/lark-base-field-create.md +8 -2
  66. package/skills/lark-base/references/lark-base-field-json.md +56 -19
  67. package/skills/lark-base/references/lark-base-field-update.md +21 -3
  68. package/skills/lark-base/references/lark-base-filter-condition.md +179 -0
  69. package/skills/lark-base/references/lark-base-form-questions-create.md +40 -7
  70. package/skills/lark-base/references/lark-base-form-questions-update.md +73 -20
  71. package/skills/lark-base/references/lark-base-form-submit.md +16 -7
  72. package/skills/lark-base/references/lark-base-record-batch-create.md +12 -10
  73. package/skills/lark-base/references/lark-base-record-batch-update.md +11 -9
  74. package/skills/lark-base/references/lark-base-record-upsert.md +1 -1
  75. package/skills/lark-base/references/lark-base-role-guide.md +11 -0
  76. package/skills/lark-base/references/lark-base-view-set-filter.md +14 -138
  77. package/skills/lark-base/references/role-config.md +31 -5
  78. package/skills/lark-calendar/SKILL.md +101 -37
  79. package/skills/lark-calendar/references/lark-calendar-create.md +13 -43
  80. package/skills/lark-calendar/references/lark-calendar-recurring.md +1 -0
  81. package/skills/lark-calendar/references/lark-calendar-room-find.md +7 -10
  82. package/skills/lark-calendar/references/lark-calendar-rsvp.md +1 -5
  83. package/skills/lark-calendar/references/lark-calendar-schedule-clear-time.md +60 -0
  84. package/skills/lark-calendar/references/lark-calendar-schedule-fuzzy-time.md +88 -0
  85. package/skills/lark-calendar/references/lark-calendar-schedule-meeting.md +67 -210
  86. package/skills/lark-calendar/references/lark-calendar-suggestion.md +2 -6
  87. package/skills/lark-calendar/references/lark-calendar-update.md +12 -11
  88. package/skills/lark-contact/SKILL.md +19 -3
  89. package/skills/lark-contact/references/lark-contact-search-bot.md +60 -0
  90. package/skills/lark-doc/SKILL.md +1 -1
  91. package/skills/lark-doc/references/lark-doc-fetch.md +14 -4
  92. package/skills/lark-doc/references/lark-doc-mindnote.md +17 -2
  93. package/skills/lark-doc/references/lark-doc-whiteboard.md +13 -8
  94. package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +76 -0
  95. package/skills/lark-doc/references/lark-doc-xml.md +6 -4
  96. package/skills/lark-drive/SKILL.md +35 -43
  97. package/skills/lark-drive/references/lark-drive-add-comment.md +2 -4
  98. package/skills/lark-drive/references/lark-drive-add-reply.md +47 -0
  99. package/skills/lark-drive/references/lark-drive-apply-permission.md +2 -2
  100. package/skills/lark-drive/references/lark-drive-batch-query-comments.md +46 -0
  101. package/skills/lark-drive/references/lark-drive-comment-content.md +50 -0
  102. package/skills/lark-drive/references/lark-drive-comment-location.md +18 -12
  103. package/skills/lark-drive/references/lark-drive-delete-reply.md +48 -0
  104. package/skills/lark-drive/references/lark-drive-delete.md +35 -11
  105. package/skills/lark-drive/references/lark-drive-download.md +5 -1
  106. package/skills/lark-drive/references/lark-drive-export.md +39 -10
  107. package/skills/lark-drive/references/lark-drive-files-list.md +27 -2
  108. package/skills/lark-drive/references/lark-drive-inspect.md +2 -0
  109. package/skills/lark-drive/references/lark-drive-list-comments.md +82 -0
  110. package/skills/lark-drive/references/lark-drive-list-replies.md +54 -0
  111. package/skills/lark-drive/references/lark-drive-member-add.md +3 -3
  112. package/skills/lark-drive/references/lark-drive-member-list.md +65 -0
  113. package/skills/lark-drive/references/lark-drive-move.md +5 -3
  114. package/skills/lark-drive/references/lark-drive-permission-get-setting.md +48 -0
  115. package/skills/lark-drive/references/lark-drive-permission-guide.md +12 -0
  116. package/skills/lark-drive/references/lark-drive-preview.md +11 -1
  117. package/skills/lark-drive/references/lark-drive-pull.md +3 -3
  118. package/skills/lark-drive/references/lark-drive-push.md +33 -6
  119. package/skills/lark-drive/references/lark-drive-react-reply.md +51 -0
  120. package/skills/lark-drive/references/lark-drive-reactions.md +27 -25
  121. package/skills/lark-drive/references/lark-drive-resolve-comment.md +45 -0
  122. package/skills/lark-drive/references/lark-drive-restore-comment.md +46 -0
  123. package/skills/lark-drive/references/lark-drive-search.md +7 -1
  124. package/skills/lark-drive/references/lark-drive-secure-label.md +1 -1
  125. package/skills/lark-drive/references/lark-drive-status.md +12 -14
  126. package/skills/lark-drive/references/lark-drive-task-result.md +58 -5
  127. package/skills/lark-drive/references/lark-drive-update-reply.md +46 -0
  128. package/skills/lark-drive/references/lark-drive-upload.md +1 -0
  129. package/skills/lark-drive/references/lark-drive-workflow-knowledge-organize.md +26 -20
  130. package/skills/lark-drive/references/lark-drive-workflow-permission-governance-commands.md +38 -8
  131. package/skills/lark-drive/references/lark-drive-workflow-permission-governance-outputs.md +10 -10
  132. package/skills/lark-drive/references/lark-drive-workflow-permission-governance.md +22 -20
  133. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-execute.md +273 -0
  134. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-recall.md +202 -0
  135. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-resolve-verify.md +231 -0
  136. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-review-plan.md +248 -0
  137. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-setup.md +174 -0
  138. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector.md +202 -0
  139. package/skills/lark-drive/references/lark-drive-workflow.md +5 -3
  140. package/skills/lark-event/SKILL.md +3 -1
  141. package/skills/lark-event/references/lark-event-application.md +38 -0
  142. package/skills/lark-event/references/lark-event-approval.md +170 -0
  143. package/skills/lark-im/SKILL.md +6 -5
  144. package/skills/lark-im/references/card/card-2.0-schema.md +1 -1
  145. package/skills/lark-im/references/card/lark-im-card-style.md +4 -4
  146. package/skills/lark-im/references/card/resource/icons.md +14 -0
  147. package/skills/lark-im/references/lark-im-flag-list.md +8 -7
  148. package/skills/lark-im/references/lark-im-messages-reply.md +1 -1
  149. package/skills/lark-im/references/lark-im-messages-send.md +1 -1
  150. package/skills/lark-mail/SKILL.md +12 -9
  151. package/skills/lark-mail/references/lark-mail-forward.md +1 -1
  152. package/skills/lark-mail/references/lark-mail-message-modify.md +48 -0
  153. package/skills/lark-mail/references/lark-mail-message-trash.md +41 -0
  154. package/skills/lark-mail/references/lark-mail-reply-all.md +1 -1
  155. package/skills/lark-mail/references/lark-mail-reply.md +1 -1
  156. package/skills/lark-mail/references/lark-mail-watch.md +1 -1
  157. package/skills/lark-markdown/SKILL.md +3 -2
  158. package/skills/lark-markdown/references/lark-markdown-create.md +22 -2
  159. package/skills/lark-minutes/SKILL.md +19 -4
  160. package/skills/lark-minutes/references/lark-minutes-download.md +0 -2
  161. package/skills/lark-minutes/references/lark-minutes-search.md +0 -2
  162. package/skills/lark-minutes/references/lark-minutes-speaker-replace.md +0 -2
  163. package/skills/lark-minutes/references/lark-minutes-summary.md +0 -2
  164. package/skills/lark-minutes/references/lark-minutes-todo.md +2 -4
  165. package/skills/lark-minutes/references/lark-minutes-update.md +0 -2
  166. package/skills/lark-minutes/references/lark-minutes-upload.md +10 -10
  167. package/skills/lark-okr/SKILL.md +71 -26
  168. package/skills/lark-okr/references/lark-okr-batch-create.md +19 -18
  169. package/skills/lark-okr/references/lark-okr-create.md +173 -0
  170. package/skills/lark-okr/references/lark-okr-cycle-list.md +17 -7
  171. package/skills/lark-okr/references/lark-okr-entities.md +1 -0
  172. package/skills/lark-okr/references/lark-okr-indicator-update.md +3 -1
  173. package/skills/lark-okr/references/lark-okr-indicators.md +61 -12
  174. package/skills/lark-okr/references/lark-okr-progress-list.md +21 -9
  175. package/skills/lark-shared/SKILL.md +26 -8
  176. package/skills/lark-sheets/SKILL.md +98 -29
  177. package/skills/lark-sheets/references/lark-sheets-batch-update.md +18 -9
  178. package/skills/lark-sheets/references/lark-sheets-changeset.md +105 -0
  179. package/skills/lark-sheets/references/lark-sheets-chart.md +4 -2
  180. package/skills/lark-sheets/references/lark-sheets-conditional-format.md +2 -0
  181. package/skills/lark-sheets/references/lark-sheets-filter-view.md +1 -1
  182. package/skills/lark-sheets/references/lark-sheets-float-image.md +6 -6
  183. package/skills/lark-sheets/references/lark-sheets-formula-translation.md +12 -3
  184. package/skills/lark-sheets/references/lark-sheets-formula-verify.md +77 -0
  185. package/skills/lark-sheets/references/lark-sheets-history.md +93 -0
  186. package/skills/lark-sheets/references/lark-sheets-pivot-table.md +7 -2
  187. package/skills/lark-sheets/references/lark-sheets-range-operations.md +44 -14
  188. package/skills/lark-sheets/references/lark-sheets-read-data.md +3 -3
  189. package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +4 -4
  190. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +4 -4
  191. package/skills/lark-sheets/references/lark-sheets-workbook.md +29 -4
  192. package/skills/lark-sheets/references/lark-sheets-write-cells.md +21 -11
  193. package/skills/lark-slides/SKILL.md +121 -63
  194. package/skills/lark-slides/references/asset-planning.md +18 -5
  195. package/skills/lark-slides/references/iconpark.md +3 -3
  196. package/skills/lark-slides/references/lark-slides-create.md +30 -3
  197. package/skills/lark-slides/references/lark-slides-history.md +132 -0
  198. package/skills/lark-slides/references/lark-slides-media-upload.md +1 -3
  199. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +85 -0
  200. package/skills/lark-slides/references/lark-slides-replace-pages.md +1 -1
  201. package/skills/lark-slides/references/lark-slides-replace-slide.md +1 -4
  202. package/skills/lark-slides/references/lark-slides-screenshot.md +11 -8
  203. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +5 -6
  204. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +5 -2
  205. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +5 -5
  206. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +14 -13
  207. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +67 -31
  208. package/skills/lark-slides/references/planning-layer.md +41 -10
  209. package/skills/lark-slides/references/slides_chart_demo.xml +1416 -0
  210. package/skills/lark-slides/references/slides_xml_schema_definition.xml +499 -78
  211. package/skills/lark-slides/references/troubleshooting.md +5 -5
  212. package/skills/lark-slides/references/validation-checklist.md +65 -19
  213. package/skills/lark-slides/references/visual-planning.md +26 -22
  214. package/skills/lark-slides/references/xml-schema-quick-ref.md +285 -45
  215. package/skills/lark-slides/scripts/sxsd_validator.py +908 -0
  216. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +2429 -91
  217. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +3567 -70
  218. package/skills/lark-task/SKILL.md +8 -0
  219. package/skills/lark-task/references/lark-task-complete.md +6 -2
  220. package/skills/lark-task/references/lark-task-create.md +23 -1
  221. package/skills/lark-task/references/lark-task-update.md +6 -2
  222. package/skills/lark-vc/SKILL.md +6 -3
  223. package/skills/lark-vc/references/lark-vc-recording.md +0 -2
  224. package/skills/lark-vc/references/vc-domain-boundaries.md +9 -1
  225. package/skills/lark-vc-agent/SKILL.md +25 -15
  226. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-events.md +65 -37
  227. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-leave.md +1 -1
  228. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-list-active.md +8 -8
  229. package/skills/lark-whiteboard/SKILL.md +13 -12
  230. package/skills/lark-whiteboard/elements/layout.md +1 -1
  231. package/skills/lark-whiteboard/elements/schema.md +2 -2
  232. package/skills/lark-whiteboard/references/{lark-whiteboard-query.md → lark-whiteboard-export.md} +15 -15
  233. package/skills/lark-whiteboard/references/lark-whiteboard-update.md +3 -3
  234. package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +12 -19
  235. package/skills/lark-whiteboard/routes/dsl.md +3 -3
  236. package/skills/lark-whiteboard/routes/mermaid.md +2 -2
  237. package/skills/lark-whiteboard/routes/svg-edit.md +4 -4
  238. package/skills/lark-whiteboard/routes/svg.md +11 -6
  239. package/skills/lark-whiteboard/scenes/bar-chart.md +1 -1
  240. package/skills/lark-whiteboard/scenes/fishbone.md +1 -1
  241. package/skills/lark-whiteboard/scenes/flywheel.md +1 -1
  242. package/skills/lark-whiteboard/scenes/line-chart.md +1 -1
  243. package/skills/lark-whiteboard/scenes/treemap.md +1 -1
  244. package/skills/lark-wiki/SKILL.md +8 -3
  245. package/skills/lark-wiki/references/lark-wiki-move-to-drive.md +122 -0
  246. package/skills/lark-wiki/references/lark-wiki-move.md +5 -3
  247. package/skills/lark-wiki/references/lark-wiki-node-get.md +1 -1
  248. package/skills/lark-wiki/references/lark-wiki-node-list.md +9 -2
  249. package/skills/lark-calendar/references/lark-calendar-agenda.md +0 -78
  250. package/skills/lark-calendar/references/lark-calendar-freebusy.md +0 -124
  251. package/skills/lark-calendar/references/lark-calendar-search-event.md +0 -29
  252. package/skills/lark-drive/references/lark-drive-comments-guide.md +0 -72
  253. package/skills/lark-sheets/references/lark-sheets-core-operations.md +0 -103
  254. package/skills/lark-slides/references/examples.md +0 -261
  255. package/skills/lark-slides/references/lark-slides-whiteboard.md +0 -330
  256. package/skills/lark-slides/references/slide-templates.md +0 -201
  257. package/skills/lark-slides/references/slides_demo.xml +0 -226
  258. package/skills/lark-slides/references/xml-format-guide.md +0 -369
@@ -1,24 +1,87 @@
1
1
  #!/usr/bin/env python3
2
2
  # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
3
  # SPDX-License-Identifier: MIT
4
+ """Validate Slides XML structure and page layout through one release gate."""
4
5
 
5
6
  from __future__ import annotations
6
7
 
8
+ import copy
7
9
  import json
10
+ import math
8
11
  import re
9
12
  import sys
13
+ import unicodedata
14
+ import xml.parsers.expat as expat
10
15
  import xml.etree.ElementTree as ET
11
- from difflib import SequenceMatcher
16
+ from difflib import SequenceMatcher, get_close_matches
12
17
  from pathlib import Path
13
18
  from typing import Any
14
19
 
15
-
16
- class XmlTextOverlapLintError(Exception):
20
+ import sxsd_validator
21
+
22
+
23
+ XS_NS = "{http://www.w3.org/2001/XMLSchema}"
24
+ XML_NS = "{http://www.w3.org/XML/1998/namespace}"
25
+ SVG_NS = "{http://www.w3.org/2000/svg}"
26
+ SML_NAMESPACE = "http://www.larkoffice.com/sml/2.0"
27
+ SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
28
+ ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
29
+ SXSD_TAG_ALIASES = {
30
+ "textbox": "<shape type=\"text\">",
31
+ "textBox": "<shape type=\"text\">",
32
+ "image": "<img>",
33
+ "picture": "<img>",
34
+ }
35
+ SXSD_ATTR_ALIASES = {
36
+ "x": "topLeftX",
37
+ "left": "topLeftX",
38
+ "y": "topLeftY",
39
+ "top": "topLeftY",
40
+ "w": "width",
41
+ "h": "height",
42
+ "fontColor": "color",
43
+ }
44
+ SERVER_FILLED_SXSD_ATTRS = {"id"}
45
+ ROUNDTRIP_SXSD_ATTRS = {
46
+ ("chart", "updated"),
47
+ ("chartData", "isStaticData"),
48
+ }
49
+ # Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
50
+ # it is server-emitted and absent from the write schema, so it must not block page linting.
51
+ ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
52
+ DEFAULT_TABLE_COLUMN_WIDTH = 110
53
+ DEFAULT_TABLE_ROW_HEIGHT = 37
54
+ DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
55
+ TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
56
+ TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
57
+ SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
58
+ CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
59
+ HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
60
+ DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
61
+ GHOST_TEXT_MIN_FONT_SIZE = 96
62
+ GHOST_TEXT_MAX_ALPHA = 0.5
63
+ GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
64
+ GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
65
+ # A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
66
+ # glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
67
+ # padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
68
+ LINE_TEXT_GRAZE_MIN_PX = 2.0
69
+ LINE_TEXT_GRAZE_FONT_RATIO = 0.12
70
+ # A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
71
+ LINE_MIN_VISIBLE_ALPHA = 0.08
72
+ # Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
73
+ # visible defect; keep this well under 1px so real overflow is still always caught.
74
+ CANVAS_OVERFLOW_TOLERANCE = 0.5
75
+ _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
76
+ _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
77
+
78
+
79
+ class XmlLayoutLintError(Exception):
17
80
  pass
18
81
 
19
82
 
20
83
  def fail(message: str) -> None:
21
- raise XmlTextOverlapLintError(message)
84
+ raise XmlLayoutLintError(message)
22
85
 
23
86
 
24
87
  def read_file(file_path: str | Path) -> str:
@@ -31,7 +94,7 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
31
94
  while index < len(argv):
32
95
  token = argv[index]
33
96
  if not token.startswith("--"):
34
- fail(f"unexpected argument: {token}")
97
+ fail(f"unexpected argument: {token}, need --input")
35
98
  key = token[2:]
36
99
  next_token = argv[index + 1] if index + 1 < len(argv) else None
37
100
  if next_token is None or next_token.startswith("--"):
@@ -44,8 +107,12 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
44
107
 
45
108
 
46
109
  def extract_attribute(tag_source: str, name: str) -> str | None:
47
- match = re.search(fr'{re.escape(name)}="([^"]+)"', tag_source)
48
- return match.group(1) if match else None
110
+ match = re.search(
111
+ fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
112
+ )
113
+ if not match:
114
+ return None
115
+ return match.group(1) if match.group(1) is not None else match.group(2)
49
116
 
50
117
 
51
118
  def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
@@ -59,8 +126,134 @@ def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
59
126
  return int(value) if value.is_integer() else value
60
127
 
61
128
 
62
- def strip_xml(value: str) -> str:
129
+ def extract_bool_attribute(tag_source: str, name: str) -> bool:
130
+ value = extract_attribute(tag_source, name)
131
+ return value in {"true", "1", "yes"}
132
+
133
+
134
+ def extract_color_alpha(color: str | None) -> int | float | None:
135
+ if color is None:
136
+ return None
137
+ normalized = re.sub(r"\s+", "", color).lower()
138
+ if normalized == "transparent":
139
+ return 0
140
+ rgba_match = re.fullmatch(
141
+ r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
142
+ normalized,
143
+ )
144
+ if rgba_match is None:
145
+ return None
146
+ try:
147
+ alpha = float(rgba_match.group(1))
148
+ except ValueError:
149
+ return None
150
+ return int(alpha) if alpha.is_integer() else alpha
151
+
152
+
153
+ def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
154
+ base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
155
+ color_alpha = extract_color_alpha(text_color)
156
+ if not isinstance(color_alpha, (int, float)):
157
+ return base_alpha
158
+ return base_alpha * color_alpha
159
+
160
+
161
+ def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
162
+ for tag_name in style_tags:
163
+ if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
164
+ return True
165
+ return False
166
+
167
+
168
+ def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
169
+ for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
170
+ if extract_bool_attribute(attrs, attr_name):
171
+ return True
172
+ return False
173
+
174
+
175
+ def sum_sizes(sizes: list[int | float]) -> int | float:
176
+ return sum(sizes)
177
+
178
+
179
+ def is_filled_size(size: int | float | None) -> bool:
180
+ return isinstance(size, (int, float)) and math.isfinite(size) and size > 0
181
+
182
+
183
+ def fill_last_size_gap(sizes: list[int | float], target_size: int | float) -> list[int | float]:
184
+ if not sizes:
185
+ return sizes
186
+ final_sizes = [
187
+ size if index == len(sizes) - 1 else max(1, math.floor(size + 0.5))
188
+ for index, size in enumerate(sizes)
189
+ ]
190
+ remaining_size = target_size - sum_sizes(final_sizes[:-1])
191
+ if remaining_size >= 1:
192
+ final_sizes[-1] = remaining_size
193
+ return final_sizes
194
+
195
+ size_to_redistribute = 1 - remaining_size
196
+ for index in range(len(final_sizes) - 2, -1, -1):
197
+ reduction = min(final_sizes[index] - 1, size_to_redistribute)
198
+ final_sizes[index] -= reduction
199
+ size_to_redistribute -= reduction
200
+ if size_to_redistribute == 0:
201
+ final_sizes[-1] = 1
202
+ return final_sizes
203
+
204
+ final_sizes[-1] = 1
205
+ return final_sizes
206
+
207
+
208
+ def solve_weighted_min_layout(
209
+ input_sizes: list[int | float | None], default_size: int | float, target_min_size: int | float | None
210
+ ) -> dict[str, Any]:
211
+ filled_indexes: list[int] = []
212
+ empty_indexes: list[int] = []
213
+ base_sizes: list[int | float] = []
214
+ for index, size in enumerate(input_sizes):
215
+ if is_filled_size(size):
216
+ filled_indexes.append(index)
217
+ base_sizes.append(size)
218
+ else:
219
+ empty_indexes.append(index)
220
+ base_sizes.append(0)
221
+ filled_sum = sum_sizes(base_sizes)
222
+
223
+ if target_min_size is None:
224
+ final_sizes = [default_size if index in empty_indexes else size for index, size in enumerate(base_sizes)]
225
+ return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
226
+
227
+ if not filled_indexes:
228
+ average_size = target_min_size / len(input_sizes)
229
+ final_sizes = fill_last_size_gap([average_size] * len(input_sizes), target_min_size)
230
+ return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
231
+
232
+ if empty_indexes:
233
+ remaining_size = target_min_size - filled_sum
234
+ final_sizes = [*base_sizes]
235
+ if remaining_size > 0:
236
+ average_size = remaining_size / len(empty_indexes)
237
+ empty_sizes = fill_last_size_gap([average_size] * len(empty_indexes), remaining_size)
238
+ for index, empty_size in zip(empty_indexes, empty_sizes):
239
+ final_sizes[index] = empty_size
240
+ else:
241
+ for index in empty_indexes:
242
+ final_sizes[index] = default_size
243
+ return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
244
+
245
+ ratio = max(1, target_min_size / filled_sum)
246
+ actual_size = max(target_min_size, filled_sum)
247
+ if ratio == 1:
248
+ return {"final_sizes": [*base_sizes], "actual_size": actual_size, "ratio": ratio}
249
+ final_sizes = fill_last_size_gap([size * ratio for size in base_sizes], actual_size)
250
+ return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
251
+
252
+
253
+ def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
63
254
  stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
255
+ if preserve_line_breaks:
256
+ stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
64
257
  stripped = re.sub(r"<[^>]+>", " ", stripped)
65
258
  stripped = stripped.replace("&nbsp;", " ")
66
259
  stripped = stripped.replace("&amp;", "&")
@@ -68,13 +261,369 @@ def strip_xml(value: str) -> str:
68
261
  stripped = stripped.replace("&gt;", ">")
69
262
  stripped = stripped.replace("&quot;", '"')
70
263
  stripped = stripped.replace("&#39;", "'")
264
+ if preserve_line_breaks:
265
+ return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
71
266
  return re.sub(r"\s+", " ", stripped).strip()
72
267
 
73
268
 
269
+ def strip_xml_paragraphs(value: str) -> str:
270
+ paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
271
+ if paragraphs:
272
+ return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
273
+ return strip_xml(value, preserve_line_breaks=True)
274
+
275
+
276
+ def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
277
+ paragraphs = []
278
+ for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
279
+ paragraphs.append(
280
+ {
281
+ "text": strip_xml(body, preserve_line_breaks=True),
282
+ "fontSize": extract_max_span_font_size(body, default_font_size),
283
+ "textAlign": extract_attribute(attrs, "textAlign"),
284
+ "lineSpacing": extract_attribute(attrs, "lineSpacing"),
285
+ "beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
286
+ "afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
287
+ "letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
288
+ }
289
+ )
290
+ return paragraphs
291
+
292
+
293
+ def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
294
+ font_sizes = [
295
+ font_size
296
+ for attrs in re.findall(r"<span\b([^>]*)>", value)
297
+ if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
298
+ ]
299
+ return max([default_font_size, *font_sizes])
300
+
301
+
302
+ def extract_tag_attributes(value: str, tag: str) -> str:
303
+ match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
304
+ return match.group(1) if match else ""
305
+
306
+
74
307
  def xml_local_name(tag: str) -> str:
75
308
  return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
76
309
 
77
310
 
311
+ def xml_namespace(tag: str) -> str | None:
312
+ return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
313
+
314
+
315
+ def load_sxsd_tag_attributes() -> dict[str, set[str]]:
316
+ global _SXSD_TAG_ATTRIBUTES_CACHE
317
+ if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
318
+ return _SXSD_TAG_ATTRIBUTES_CACHE
319
+
320
+ _SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
321
+ return _SXSD_TAG_ATTRIBUTES_CACHE
322
+
323
+
324
+ def load_iconpark_icon_types() -> set[str]:
325
+ global _ICONPARK_ICON_TYPES_CACHE
326
+ if _ICONPARK_ICON_TYPES_CACHE is not None:
327
+ return _ICONPARK_ICON_TYPES_CACHE
328
+
329
+ try:
330
+ index_data = json.loads(ICONPARK_INDEX_PATH.read_text(encoding="utf-8"))
331
+ except json.JSONDecodeError as error:
332
+ fail(f"invalid iconpark index JSON: {error}")
333
+ icons = index_data.get("icons")
334
+ if not isinstance(icons, list):
335
+ fail("iconpark index must contain an icons array")
336
+
337
+ icon_types = {
338
+ icon["iconType"]
339
+ for icon in icons
340
+ if isinstance(icon, dict) and isinstance(icon.get("iconType"), str) and icon["iconType"]
341
+ }
342
+ _ICONPARK_ICON_TYPES_CACHE = icon_types
343
+ return icon_types
344
+
345
+
346
+ def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
347
+ alias = SXSD_TAG_ALIASES.get(tag_name)
348
+ if alias:
349
+ return f"Use {alias} instead of <{tag_name}>."
350
+ if tag_name == "svg":
351
+ return 'Inside <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
352
+ close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
353
+ if close_matches:
354
+ return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
355
+ return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
356
+
357
+
358
+ def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
359
+ alias = SXSD_ATTR_ALIASES.get(attr_name)
360
+ if alias and alias in allowed_attrs:
361
+ return [alias]
362
+ return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
363
+
364
+
365
+ def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
366
+ suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
367
+ if suggestions:
368
+ if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
369
+ return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
370
+ return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
371
+ allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
372
+ if len(allowed_attrs) > 8:
373
+ allowed_summary += ", ..."
374
+ return f"Unsupported SXSD attribute for <{tag_name}>. Allowed attributes include: {allowed_summary}."
375
+
376
+
377
+ def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
378
+ return "whiteboard" in ancestors and xml_namespace(element.tag) == SVG_NS
379
+
380
+
381
+ def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
382
+ return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
383
+
384
+
385
+ def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
386
+ return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
387
+
388
+
389
+ def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
390
+ sanitized_root = copy.deepcopy(root)
391
+
392
+ def sanitize(element: ET.Element) -> None:
393
+ tag_name = xml_local_name(element.tag)
394
+ for raw_attr_name in list(element.attrib):
395
+ if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
396
+ del element.attrib[raw_attr_name]
397
+ for child in list(element):
398
+ if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
399
+ element.remove(child)
400
+ continue
401
+ sanitize(child)
402
+
403
+ sanitize(sanitized_root)
404
+ return sanitized_root
405
+
406
+
407
+ def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
408
+ tag_attributes = load_sxsd_tag_attributes()
409
+ supported_tags = set(tag_attributes)
410
+ issues: list[dict[str, Any]] = []
411
+ suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
412
+
413
+ def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
414
+ if should_skip_sxsd_subtree(element, ancestors):
415
+ return
416
+
417
+ tag_name = xml_local_name(element.tag)
418
+ current_path = f"{path}/{tag_name}" if path else tag_name
419
+ parent_name = ancestors[-1] if ancestors else None
420
+ if should_skip_sxsd_tag(parent_name, tag_name):
421
+ return
422
+ if tag_name not in supported_tags:
423
+ issues.append(
424
+ {
425
+ "level": "error",
426
+ "code": "sxsd_unsupported_tag",
427
+ "tag": tag_name,
428
+ "path": current_path,
429
+ "message": f"unsupported SXSD tag <{tag_name}> at {current_path}",
430
+ "hint": build_sxsd_tag_hint(tag_name, supported_tags),
431
+ }
432
+ )
433
+ return
434
+ else:
435
+ allowed_attrs = tag_attributes[tag_name]
436
+ for raw_attr_name in element.attrib:
437
+ if raw_attr_name.startswith(XML_NS):
438
+ continue
439
+ attr_name = xml_local_name(raw_attr_name)
440
+ if should_skip_sxsd_attribute(tag_name, attr_name):
441
+ continue
442
+ if attr_name in allowed_attrs:
443
+ continue
444
+ suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
445
+ if suggestions:
446
+ suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
447
+ set(suggestions)
448
+ )
449
+ issues.append(
450
+ {
451
+ "level": "error",
452
+ "code": "sxsd_unsupported_attr",
453
+ "tag": tag_name,
454
+ "attr": attr_name,
455
+ "path": current_path,
456
+ "message": f'unsupported SXSD attribute "{attr_name}" on <{tag_name}> at {current_path}',
457
+ "hint": build_sxsd_attr_hint(tag_name, attr_name, allowed_attrs),
458
+ }
459
+ )
460
+
461
+ for child in element:
462
+ visit(child, [*ancestors, tag_name], current_path)
463
+
464
+ visit(root, [], "")
465
+ existing = {
466
+ (issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
467
+ for issue in issues
468
+ }
469
+ unsupported_tag_locations = {
470
+ (issue.get("path"), issue.get("tag"))
471
+ for issue in issues
472
+ if issue.get("code") == "sxsd_unsupported_tag"
473
+ }
474
+ schema_issues = _validate_sxsd_schema_constraints(xml, root)
475
+ missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
476
+ for schema_issue in schema_issues:
477
+ if schema_issue.get("code") != "sxsd_missing_required_attr":
478
+ continue
479
+ location = (schema_issue.get("path"), schema_issue.get("tag"))
480
+ missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
481
+
482
+ suggested_attrs: set[tuple[str, str, str]] = set()
483
+ for location, candidate_groups in suggested_attr_candidates.items():
484
+ missing_attrs = missing_attrs_by_location.get(location, set())
485
+ for candidates in candidate_groups:
486
+ matching_missing_attrs = candidates & missing_attrs
487
+ if len(matching_missing_attrs) == 1:
488
+ suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
489
+
490
+ for schema_issue in schema_issues:
491
+ if schema_issue.get("code") == "sxsd_unexpected_child" and (
492
+ schema_issue.get("path"),
493
+ schema_issue.get("tag"),
494
+ ) in unsupported_tag_locations:
495
+ continue
496
+ if schema_issue.get("code") == "sxsd_missing_required_attr" and (
497
+ schema_issue.get("path"),
498
+ schema_issue.get("tag"),
499
+ schema_issue.get("attr"),
500
+ ) in suggested_attrs:
501
+ continue
502
+ key = (
503
+ schema_issue.get("code"),
504
+ schema_issue.get("path"),
505
+ schema_issue.get("tag"),
506
+ schema_issue.get("attr"),
507
+ )
508
+ if key not in existing:
509
+ issues.append(schema_issue)
510
+ return issues
511
+
512
+
513
+ def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
514
+ issues: list[dict[str, Any]] = []
515
+ if re.match(r"^\s*<\?xml\b", xml):
516
+ issues.append(
517
+ {
518
+ "level": "error",
519
+ "code": "sxsd_unsupported_declaration",
520
+ "path": xml_local_name(root.tag),
521
+ "tag": xml_local_name(root.tag),
522
+ "expected": "SXSD document without an XML declaration",
523
+ "actual": "<?xml ...?>",
524
+ "message": "XML declarations are not supported by the Slides SXSD write format",
525
+ "hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
526
+ }
527
+ )
528
+
529
+ issues.extend(
530
+ sxsd_validator.validate_sxsd(
531
+ without_server_filled_sxsd_fields(root),
532
+ SXSD_SCHEMA_PATH,
533
+ )
534
+ )
535
+ return issues
536
+
537
+
538
+ def build_iconpark_icon_type_hint(icon_type: str, supported_icon_types: set[str]) -> str:
539
+ close_matches = get_close_matches(icon_type, sorted(supported_icon_types), n=3, cutoff=0.58)
540
+ if close_matches:
541
+ return (
542
+ "iconType must exist in iconpark-index.json. Did you mean "
543
+ + ", ".join(f'"{match}"' for match in close_matches)
544
+ + "?"
545
+ )
546
+ return "iconType must exist in iconpark-index.json. Use scripts/iconpark_tool.py to search supported icons."
547
+
548
+
549
+ def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
550
+ supported_icon_types: set[str] | None = None
551
+ issues: list[dict[str, Any]] = []
552
+
553
+ def direct_child(element: ET.Element, local_name: str) -> ET.Element | None:
554
+ return next((child for child in element if xml_local_name(child.tag) == local_name), None)
555
+
556
+ def is_transparent_color(color: str) -> bool:
557
+ normalized = re.sub(r"\s+", "", color).lower()
558
+ if normalized == "transparent":
559
+ return True
560
+ rgba_match = re.fullmatch(r"rgba\([^,]+,[^,]+,[^,]+,([0-9.]+)\)", normalized)
561
+ if not rgba_match:
562
+ return False
563
+ try:
564
+ return float(rgba_match.group(1)) <= 0
565
+ except ValueError:
566
+ return False
567
+
568
+ def append_missing_fill_color_issue(current_path: str) -> None:
569
+ issues.append(
570
+ {
571
+ "level": "error",
572
+ "code": "icon_missing_fill_color",
573
+ "tag": "icon",
574
+ "path": current_path,
575
+ "message": f"<icon> must set explicit non-transparent fillColor for visual visibility at {current_path}",
576
+ "hint": 'Add <fill><fillColor color="rgba(R, G, B, 1)"/></fill> inside <icon>. This is a visual lint rule, not an SXSD required field.',
577
+ }
578
+ )
579
+
580
+ def visit(element: ET.Element, path: str) -> None:
581
+ nonlocal supported_icon_types
582
+ tag_name = xml_local_name(element.tag)
583
+ current_path = f"{path}/{tag_name}" if path else tag_name
584
+ if tag_name == "icon":
585
+ icon_type = element.attrib.get("iconType")
586
+ if icon_type is not None:
587
+ if supported_icon_types is None:
588
+ supported_icon_types = load_iconpark_icon_types()
589
+ if icon_type not in supported_icon_types:
590
+ issues.append(
591
+ {
592
+ "level": "error",
593
+ "code": "iconpark_unsupported_icon_type",
594
+ "tag": "icon",
595
+ "attr": "iconType",
596
+ "iconType": icon_type,
597
+ "path": current_path,
598
+ "message": f'unsupported iconpark iconType "{icon_type}" at {current_path}',
599
+ "hint": build_iconpark_icon_type_hint(icon_type, supported_icon_types),
600
+ }
601
+ )
602
+ fill = direct_child(element, "fill")
603
+ fill_color = direct_child(fill, "fillColor") if fill is not None else None
604
+ color = fill_color.attrib.get("color") if fill_color is not None else None
605
+ if not color:
606
+ append_missing_fill_color_issue(current_path)
607
+ elif is_transparent_color(color):
608
+ issues.append(
609
+ {
610
+ "level": "error",
611
+ "code": "icon_transparent_fill_color",
612
+ "tag": "icon",
613
+ "attr": "fillColor",
614
+ "path": current_path,
615
+ "color": color,
616
+ "message": f'<icon> fillColor must not be transparent for visual visibility at {current_path}: "{color}"',
617
+ "hint": 'Use an opaque visible color, for example <fillColor color="rgba(37, 99, 235, 1)"/>.',
618
+ }
619
+ )
620
+ for child in element:
621
+ visit(child, current_path)
622
+
623
+ visit(root, "")
624
+ return issues
625
+
626
+
78
627
  def extract_error_context(xml: str, line: int | None, column: int | None, radius: int = 40) -> str | None:
79
628
  if line is None or column is None:
80
629
  return None
@@ -103,75 +652,218 @@ def build_xml_error_issue(error: ET.ParseError, xml: str) -> dict[str, Any]:
103
652
  }
104
653
 
105
654
 
106
- def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
655
+ def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
656
+ namespace_map: dict[str, str] = {}
657
+ pending_declarations: list[tuple[str, str | None]] = []
658
+ declarations_by_element: list[list[tuple[str, str | None]]] = []
659
+ element_stack: list[str] = []
660
+ issues: list[dict[str, Any]] = []
661
+
662
+ parser = expat.ParserCreate(namespace_separator="|")
663
+ parser.namespace_prefixes = True
664
+
665
+ def handle_namespace_decl(prefix: str | None, namespace: str) -> None:
666
+ normalized_prefix = prefix or ""
667
+ previous_namespace = namespace_map.get(normalized_prefix)
668
+ namespace_map[normalized_prefix] = namespace
669
+ pending_declarations.append((normalized_prefix, previous_namespace))
670
+
671
+ def handle_start_element(name: str, _attrs: dict[str, str]) -> None:
672
+ declarations_by_element.append(pending_declarations.copy())
673
+ pending_declarations.clear()
674
+ name_parts = name.rsplit("|", 2)
675
+ if len(name_parts) == 3:
676
+ _namespace, local_name, prefix = name_parts
677
+ element_name = f"{prefix}:{local_name}"
678
+ else:
679
+ prefix = ""
680
+ local_name = name_parts[-1]
681
+ element_name = local_name
682
+ element_stack.append(element_name)
683
+ if not prefix:
684
+ return
685
+
686
+ if namespace_map.get(prefix) != SML_NAMESPACE:
687
+ return
688
+ path = "/".join(element_stack)
689
+ issues.append(
690
+ {
691
+ "level": "error",
692
+ "code": "sml_prefixed_tag",
693
+ "tag": element_name,
694
+ "namespace": SML_NAMESPACE,
695
+ "path": path,
696
+ "line": parser.CurrentLineNumber,
697
+ "column": parser.CurrentColumnNumber,
698
+ "message": f"SML tag <{element_name}> must not use a namespace prefix at {path}",
699
+ "hint": (
700
+ f'Use <{local_name}> under the default namespace '
701
+ f'<{local_name} xmlns="{SML_NAMESPACE}">, or use an unprefixed SML tag.'
702
+ ),
703
+ }
704
+ )
705
+
706
+ def handle_end_element(_name: str) -> None:
707
+ for prefix, previous_namespace in reversed(declarations_by_element.pop()):
708
+ if previous_namespace is None:
709
+ namespace_map.pop(prefix, None)
710
+ else:
711
+ namespace_map[prefix] = previous_namespace
712
+ element_stack.pop()
713
+
714
+ parser.StartNamespaceDeclHandler = handle_namespace_decl
715
+ parser.StartElementHandler = handle_start_element
716
+ parser.EndElementHandler = handle_end_element
717
+ parser.Parse(xml, True)
718
+ return issues
719
+
720
+
721
+ def parse_xml_root(xml: str) -> tuple[ET.Element | None, dict[str, Any] | None]:
107
722
  try:
108
723
  root = ET.fromstring(xml)
109
724
  except ET.ParseError as error:
110
- return build_xml_error_issue(error, xml)
725
+ return None, build_xml_error_issue(error, xml)
111
726
 
112
727
  root_name = xml_local_name(root.tag)
113
728
  if root_name not in {"presentation", "slide"}:
114
729
  fail("input must contain a <presentation> or <slide> root")
115
- return None
730
+ return root, None
116
731
 
117
732
 
118
- def parse_presentation(xml: str) -> dict[str, Any]:
119
- presentation_match = re.search(r"<presentation\b([^>]*)>", xml)
120
- if presentation_match:
121
- return {
122
- "width": int(float(extract_attribute(presentation_match.group(1), "width") or 960)),
123
- "height": int(float(extract_attribute(presentation_match.group(1), "height") or 540)),
124
- "slides": re.findall(r"<slide\b[\s\S]*?</slide>", xml),
733
+ def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
734
+ _, xml_error = parse_xml_root(xml)
735
+ return xml_error
736
+
737
+
738
+ def serialize_slide_for_layout(slide_root: ET.Element) -> str:
739
+ slide_copy = copy.deepcopy(slide_root)
740
+ for element in slide_copy.iter():
741
+ if not isinstance(element.tag, str):
742
+ continue
743
+ element.tag = xml_local_name(element.tag)
744
+ attributes = {
745
+ xml_local_name(attribute_name): value
746
+ for attribute_name, value in element.attrib.items()
125
747
  }
126
- slide_match = re.findall(r"<slide\b[\s\S]*?</slide>", xml)
127
- if slide_match:
128
- return {"width": 960, "height": 540, "slides": slide_match}
129
- fail("input must contain a <presentation> or <slide> root")
748
+ element.attrib.clear()
749
+ element.attrib.update(attributes)
750
+ return ET.tostring(slide_copy, encoding="unicode")
751
+
752
+
753
+ def parse_presentation(root: ET.Element) -> dict[str, Any]:
754
+ root_name = xml_local_name(root.tag)
755
+ if root_name == "slide":
756
+ slide_roots = [root]
757
+ width = 960
758
+ height = 540
759
+ elif root_name == "presentation":
760
+ slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
761
+ width = int(float(root.attrib.get("width", 960)))
762
+ height = int(float(root.attrib.get("height", 540)))
763
+ else:
764
+ fail("input must contain a <presentation> or <slide> root")
765
+ return {
766
+ "width": width,
767
+ "height": height,
768
+ "slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
769
+ "slide_roots": slide_roots,
770
+ }
130
771
 
131
772
 
132
773
  def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
133
774
  elements: list[dict[str, Any]] = []
134
- for match in re.finditer(r"<shape\b([^>]*)>([\s\S]*?)</shape>", slide_xml):
135
- attrs, content = match.group(1), match.group(2)
136
- x = extract_numeric_attribute(attrs, "topLeftX")
137
- y = extract_numeric_attribute(attrs, "topLeftY")
138
- width = extract_numeric_attribute(attrs, "width")
139
- height = extract_numeric_attribute(attrs, "height")
140
- if all(value is not None for value in [x, y, width, height]):
141
- font_size = float(extract_attribute(content, "fontSize") or extract_attribute(attrs, "fontSize") or 16)
142
- elements.append(
143
- {
144
- "id": f"shape-{len(elements) + 1}",
145
- "kind": "shape",
146
- "type": extract_attribute(attrs, "type") or "shape",
147
- "textType": extract_attribute(content, "textType"),
148
- "x": x,
149
- "y": y,
150
- "width": width,
151
- "height": height,
152
- "fontSize": font_size,
153
- "text": strip_xml(content),
154
- }
155
- )
156
775
 
157
- for match in re.finditer(r"<(img|table|chart)\b([^>]*)/?>", slide_xml):
158
- attrs = match.group(2)
776
+ for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
777
+ kind, attrs = match.group(1), match.group(2)
778
+ is_self_closing = attrs.rstrip().endswith("/")
779
+ content = ""
780
+ if kind in {"shape", "table"} and not is_self_closing:
781
+ close_index = slide_xml.find(f"</{kind}>", match.end())
782
+ if close_index != -1:
783
+ content = slide_xml[match.end() : close_index]
784
+
785
+ element_id = extract_attribute(attrs, "id") or f"{kind}-{len(elements) + 1}"
159
786
  x = extract_numeric_attribute(attrs, "topLeftX")
160
787
  y = extract_numeric_attribute(attrs, "topLeftY")
161
788
  width = extract_numeric_attribute(attrs, "width")
162
789
  height = extract_numeric_attribute(attrs, "height")
163
- if all(value is not None for value in [x, y, width, height]):
164
- elements.append(
165
- {
166
- "id": f"{match.group(1)}-{len(elements) + 1}",
167
- "kind": match.group(1),
168
- "type": match.group(1),
169
- "x": x,
170
- "y": y,
171
- "width": width,
172
- "height": height,
173
- }
790
+ rotation = extract_numeric_attribute(attrs, "rotation") or 0
791
+ alpha = extract_numeric_attribute(attrs, "alpha")
792
+ table_layouts: dict[str, dict[str, Any] | None] = {}
793
+ if kind == "table":
794
+ width, table_layouts["width"] = resolve_table_dimension(
795
+ content, width, extract_table_column_sizes, DEFAULT_TABLE_COLUMN_WIDTH
174
796
  )
797
+ height, table_layouts["height"] = resolve_table_dimension(
798
+ content, height, extract_table_row_sizes, DEFAULT_TABLE_ROW_HEIGHT
799
+ )
800
+ if all(value is not None for value in [x, y, width, height]):
801
+ element = {
802
+ "id": element_id,
803
+ "kind": kind,
804
+ "type": extract_attribute(attrs, "type") or kind,
805
+ "x": x,
806
+ "y": y,
807
+ "width": width,
808
+ "height": height,
809
+ "rotation": rotation,
810
+ "alpha": alpha if alpha is not None else 1,
811
+ "order": len(elements),
812
+ }
813
+ if kind == "table":
814
+ element.update(
815
+ {
816
+ "declared_width": extract_numeric_attribute(attrs, "width"),
817
+ "declared_height": extract_numeric_attribute(attrs, "height"),
818
+ "table_layouts": table_layouts,
819
+ }
820
+ )
821
+ if kind == "shape":
822
+ content_attrs = extract_tag_attributes(content, "content")
823
+ font_size = extract_numeric_attribute(content_attrs, "fontSize")
824
+ if font_size is None:
825
+ font_size = extract_numeric_attribute(attrs, "fontSize")
826
+ font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
827
+ text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
828
+ bold = (
829
+ extract_bool_attribute(content_attrs, "bold")
830
+ or extract_bool_attribute(attrs, "bold")
831
+ or detect_inline_style_presence(content, {"strong", "b"})
832
+ or detect_any_span_bool_attribute(content, "bold")
833
+ )
834
+ italic = (
835
+ extract_bool_attribute(content_attrs, "italic")
836
+ or extract_bool_attribute(attrs, "italic")
837
+ or detect_inline_style_presence(content, {"i", "em"})
838
+ or detect_any_span_bool_attribute(content, "italic")
839
+ )
840
+ element.update(
841
+ {
842
+ "textType": extract_attribute(content_attrs, "textType"),
843
+ "textAlign": extract_attribute(content_attrs, "textAlign"),
844
+ "verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
845
+ "vert": extract_attribute(attrs, "vert") or "horz",
846
+ "autoFit": extract_attribute(content_attrs, "autoFit"),
847
+ "wrap": extract_attribute(content_attrs, "wrap"),
848
+ "lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
849
+ "beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
850
+ "afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
851
+ "letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
852
+ "paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
853
+ "paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
854
+ "paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
855
+ "paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
856
+ "fontSize": font_size if font_size is not None else 16,
857
+ "fontFamily": font_family or "",
858
+ "color": text_color,
859
+ "textAlpha": effective_text_alpha(alpha, text_color),
860
+ "bold": bold,
861
+ "italic": italic,
862
+ "text": strip_xml_paragraphs(content),
863
+ "paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
864
+ }
865
+ )
866
+ elements.append(element)
175
867
  return elements
176
868
 
177
869
 
@@ -188,10 +880,52 @@ def is_text_element(element: dict[str, Any]) -> bool:
188
880
  return element["kind"] == "shape" and element["type"] == "text"
189
881
 
190
882
 
883
+ def is_whiteboard_element(element: dict[str, Any]) -> bool:
884
+ return element["kind"] == "whiteboard"
885
+
886
+
191
887
  def has_text_content(element: dict[str, Any]) -> bool:
192
888
  return bool(element.get("text"))
193
889
 
194
890
 
891
+ def is_vertical_text(element: dict[str, Any]) -> bool:
892
+ return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
893
+
894
+
895
+ def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
896
+ issues: list[dict[str, Any]] = []
897
+ text_elements = [
898
+ element
899
+ for element in elements
900
+ if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
901
+ ]
902
+ image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
903
+ for text_element in text_elements:
904
+ for image_element in image_elements:
905
+ if image_element["order"] <= text_element["order"]:
906
+ continue
907
+ if is_vertical_text(text_element):
908
+ if intersects(image_element, text_element):
909
+ issues.append({
910
+ "level": "info",
911
+ "code": "image_may_cover_vertical_text",
912
+ "elements": [image_element["id"], text_element["id"]],
913
+ "message": f'image {image_element["id"]} may cover vertical text shape {text_element["id"]}',
914
+ "hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
915
+ })
916
+ continue
917
+ text_visual_bbox = estimate_text_visual_bbox(text_element)
918
+ if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
919
+ issues.append({
920
+ "level": "error",
921
+ "code": "image_covers_text",
922
+ "elements": [image_element["id"], text_element["id"]],
923
+ "message": f'image {image_element["id"]} covers text shape {text_element["id"]}',
924
+ "hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
925
+ })
926
+ return issues
927
+
928
+
195
929
  def is_decorative_text(element: dict[str, Any]) -> bool:
196
930
  text = element.get("text") or ""
197
931
  return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
@@ -201,6 +935,137 @@ def normalize_text_for_overlap(text: str) -> str:
201
935
  return re.sub(r"\s+", "", text)
202
936
 
203
937
 
938
+ SERIF_FONT_PATTERNS = {
939
+ "song", "songti", "simsun", "ming", "mincho",
940
+ "georgia", "times", "caslon", "garamond", "sourcehan-serif",
941
+ "source han serif", "思源宋体", "宋体", "明体",
942
+ }
943
+
944
+ SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
945
+ "helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
946
+
947
+
948
+ def classify_font_family(font_family: str | None) -> str:
949
+ if not font_family:
950
+ return "sans"
951
+ family_lower = font_family.lower()
952
+ for marker in SANS_EXPLICIT_MARKERS:
953
+ if marker in family_lower:
954
+ return "sans"
955
+ serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
956
+ for pattern in serif_keywords:
957
+ if pattern in family_lower:
958
+ return "serif"
959
+ return "sans"
960
+
961
+
962
+ _FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
963
+ "sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
964
+ "serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
965
+ }
966
+
967
+
968
+ def estimate_character_width(
969
+ character: str,
970
+ font_size: int | float,
971
+ bold: bool = False,
972
+ font_family: str | None = None,
973
+ ) -> int | float:
974
+ bold_multiplier = 1.05 if bold else 1.0
975
+ if character.isspace():
976
+ return font_size * 0.33 * bold_multiplier
977
+ ea_width = unicodedata.east_asian_width(character)
978
+ if ea_width in {"F", "W"}:
979
+ return font_size * bold_multiplier
980
+ category = classify_font_family(font_family)
981
+ coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
982
+ if character.isupper():
983
+ return font_size * coeffs["upper"] * bold_multiplier
984
+ if character.islower():
985
+ return font_size * coeffs["lower"] * bold_multiplier
986
+ if character.isdigit():
987
+ return font_size * coeffs["digit"] * bold_multiplier
988
+ return font_size * coeffs["punct"] * bold_multiplier
989
+
990
+
991
+ def estimate_text_width(
992
+ text: str,
993
+ font_size: int | float,
994
+ letter_spacing: int | float = 0,
995
+ bold: bool = False,
996
+ font_family: str | None = None,
997
+ ) -> int | float:
998
+ base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
999
+ return base + max(len(text) - 1, 0) * letter_spacing
1000
+
1001
+
1002
+ def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
1003
+ if paragraph is not None:
1004
+ value = paragraph.get("letterSpacing")
1005
+ if isinstance(value, (int, float)):
1006
+ return value
1007
+ value = element.get("letterSpacing")
1008
+ return value if isinstance(value, (int, float)) else 0
1009
+
1010
+
1011
+ def text_wrap_width_tolerance() -> int | float:
1012
+ return TEXT_WRAP_WIDTH_TOLERANCE_PX
1013
+
1014
+
1015
+ def text_height_overflow_tolerance() -> int | float:
1016
+ return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
1017
+
1018
+
1019
+ def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
1020
+ return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
1021
+
1022
+
1023
+ def is_short_metric_text(text: str) -> bool:
1024
+ compact = re.sub(r"\s+", "", text)
1025
+ if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
1026
+ return False
1027
+ if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
1028
+ return True
1029
+ if re.search(r"[,.,+\-–—/%%]", compact) is None:
1030
+ return False
1031
+ return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
1032
+
1033
+
1034
+ def is_single_line_visual_candidate(
1035
+ element: dict[str, Any],
1036
+ paragraph: dict[str, Any] | None,
1037
+ text: str,
1038
+ logical_width: int | float,
1039
+ effective_width: int | float,
1040
+ ) -> bool:
1041
+ if "\n" in text or logical_width <= effective_width:
1042
+ return False
1043
+ if is_short_metric_text(text):
1044
+ return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
1045
+
1046
+ text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
1047
+ compact_len = len(re.sub(r"\s+", "", text))
1048
+ if text_align == "center" and compact_len <= 32:
1049
+ return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
1050
+
1051
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1052
+ if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
1053
+ return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
1054
+ return False
1055
+
1056
+
1057
+ def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
1058
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1059
+ bold = element.get("bold", False)
1060
+ font_family = element.get("fontFamily", "")
1061
+ letter_spacing = resolve_letter_spacing(element)
1062
+ paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
1063
+ return max(
1064
+ [estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
1065
+ or [1]
1066
+ )
1067
+
1068
+
204
1069
  def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
205
1070
  left_text = normalize_text_for_overlap(left.get("text") or "")
206
1071
  right_text = normalize_text_for_overlap(right.get("text") or "")
@@ -211,29 +1076,203 @@ def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool
211
1076
  return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
212
1077
 
213
1078
 
214
- def estimate_text_line_count(element: dict[str, Any]) -> int:
1079
+ def estimate_text_line_count_for_text(
1080
+ element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
1081
+ ) -> int:
215
1082
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
216
- chars_per_line = max(1, int(element["width"] // max(font_size * 0.55, 1)))
217
- paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
1083
+ bold = element.get("bold", False)
1084
+ font_family = element.get("fontFamily", "")
1085
+ letter_spacing = resolve_letter_spacing(element, paragraph)
1086
+ available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
1087
+ hard_lines = text.split("\n")
1088
+ if not text:
1089
+ return 0
218
1090
  line_count = 0
219
- for paragraph in paragraphs:
220
- logical_length = max(len(paragraph), 1)
221
- line_count += max(1, -(-logical_length // chars_per_line))
222
- return max(line_count, 1)
1091
+ for hard_line in hard_lines:
1092
+ if element.get("wrap") in {"false", "0"}:
1093
+ line_count += 1
1094
+ continue
1095
+ logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
1096
+ effective_width = available_width + text_wrap_width_tolerance()
1097
+ if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
1098
+ line_count += 1
1099
+ continue
1100
+ line_count += max(1, math.ceil(logical_width / effective_width))
1101
+ return line_count
1102
+
1103
+
1104
+ def estimate_text_line_count(element: dict[str, Any]) -> int:
1105
+ return max(estimate_text_line_count_for_text(element, element["text"]), 1)
1106
+
1107
+
1108
+ def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
1109
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1110
+ if line_spacing is None:
1111
+ return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
1112
+ match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
1113
+ if match is None:
1114
+ return None
1115
+ spacing_type, value = match.groups()
1116
+ return font_size * float(value) if spacing_type == "multiple" else float(value)
1117
+
1118
+
1119
+ def adjust_dense_body_line_height(
1120
+ element: dict[str, Any],
1121
+ line_spacing: str | None,
1122
+ line_height: int | float,
1123
+ paragraph_count: int,
1124
+ ) -> int | float:
1125
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1126
+ if paragraph_count < 4 or font_size > 14 or not line_spacing:
1127
+ return line_height
1128
+ match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
1129
+ if match is None:
1130
+ return line_height
1131
+ return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
1132
+
1133
+
1134
+ def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1135
+ issues: list[dict[str, Any]] = []
1136
+ for element in elements:
1137
+ if not is_text_element(element) or not has_text_content(element):
1138
+ continue
1139
+ if has_explicit_height_auto_fit(element):
1140
+ continue
1141
+
1142
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1143
+ paragraphs = element.get("paragraphs") or [
1144
+ {
1145
+ "text": element["text"],
1146
+ "lineSpacing": None,
1147
+ "beforeLineSpacing": None,
1148
+ "afterLineSpacing": None,
1149
+ }
1150
+ ]
1151
+ line_count = 0
1152
+ estimated_height = 0.0
1153
+ line_heights: list[int | float] = []
1154
+ for paragraph in paragraphs:
1155
+ paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
1156
+ if paragraph_line_count == 0:
1157
+ continue
1158
+ resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
1159
+ line_height = estimate_text_line_height(element, resolved_line_spacing)
1160
+ before_spacing = estimate_text_line_height(
1161
+ element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
1162
+ )
1163
+ after_spacing = estimate_text_line_height(
1164
+ element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
1165
+ )
1166
+ if line_height is None or before_spacing is None or after_spacing is None:
1167
+ line_count = 0
1168
+ break
1169
+ line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
1170
+ first_line_height = font_size if line_count == 0 else line_height
1171
+ line_count += paragraph_line_count
1172
+ line_heights.append(line_height)
1173
+ estimated_height += (
1174
+ before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
1175
+ )
1176
+ if line_count == 0:
1177
+ continue
1178
+ available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
1179
+ overflow = estimated_height - available_height
1180
+ if overflow <= text_height_overflow_tolerance():
1181
+ continue
1182
+
1183
+ is_background = is_background_decorative_text(element, elements)
1184
+ if is_background:
1185
+ level = "info"
1186
+ else:
1187
+ level = "error" if overflow > 10 else "warning"
1188
+ message = (
1189
+ f'text shape {element["id"]} may overflow its own content box '
1190
+ f'(estimated {estimated_height:g}px, available {available_height:g}px); '
1191
+ 'consider setting content wrap="true" autoFit="normal-auto-fit"'
1192
+ )
1193
+ if is_background:
1194
+ message += " (likely background decoration: large font, low alpha, underneath other text)"
1195
+ issues.append(
1196
+ {
1197
+ "level": level,
1198
+ "code": "text_may_overflow_shape",
1199
+ "elements": [element["id"]],
1200
+ "line_count": line_count,
1201
+ "line_height": max(line_heights),
1202
+ "estimated_height": estimated_height,
1203
+ "available_height": available_height,
1204
+ "overflow": overflow,
1205
+ "message": message,
1206
+ "hint": (
1207
+ "Increase shape.height, reduce the text, or set content wrap=\"true\" "
1208
+ "autoFit=\"normal-auto-fit\". "
1209
+ "This is an estimate based on font size, line spacing, and wrapped line count."
1210
+ ),
1211
+ }
1212
+ )
1213
+ return issues
1214
+
1215
+
1216
+ def is_background_decorative_text(
1217
+ element: dict[str, Any], elements: list[dict[str, Any]]
1218
+ ) -> bool:
1219
+ if not is_ghost_text(element):
1220
+ return False
1221
+ for other in elements:
1222
+ if other is element:
1223
+ continue
1224
+ if not is_text_element(other) or not has_text_content(other):
1225
+ continue
1226
+ foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
1227
+ if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
1228
+ continue
1229
+ if other["order"] <= element["order"]:
1230
+ continue
1231
+ if intersects(element, other):
1232
+ return True
1233
+ return False
1234
+
1235
+
1236
+ def is_ghost_text(element: dict[str, Any]) -> bool:
1237
+ if not is_text_element(element) or not has_text_content(element):
1238
+ return False
1239
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1240
+ text_alpha = element.get("textAlpha", element.get("alpha", 1))
1241
+ if not isinstance(text_alpha, (int, float)):
1242
+ return False
1243
+ if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
1244
+ return True
1245
+ return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
223
1246
 
224
1247
 
225
1248
  def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
226
1249
  if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
227
1250
  return None
228
1251
 
1252
+ padding_left = element.get("paddingLeft", 0)
1253
+ padding_right = element.get("paddingRight", 0)
1254
+ padding_top = element.get("paddingTop", 0)
1255
+ padding_bottom = element.get("paddingBottom", 0)
1256
+ content_width = max(element["width"] - padding_left - padding_right, 0)
1257
+ content_height = max(element["height"] - padding_top - padding_bottom, 0)
229
1258
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
230
- char_width = max(font_size * 0.55, 1)
231
1259
  line_count = estimate_text_line_count(element)
232
- visual_width = min(element["width"], max(1, len(element["text"]) * char_width))
233
- visual_height = min(element["height"], max(1, line_count * font_size * 1.2))
1260
+ estimated_width = max(1, estimate_text_max_line_width(element))
1261
+ visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
1262
+ visual_height = min(content_height, max(1, line_count * font_size * 1.2))
1263
+ x = element["x"] + padding_left
1264
+ if element.get("textAlign") == "center":
1265
+ x += (content_width - visual_width) / 2
1266
+ elif element.get("textAlign") == "right":
1267
+ x += content_width - visual_width
1268
+ y = element["y"] + padding_top
1269
+ if element.get("verticalAlign") == "middle":
1270
+ y += (content_height - visual_height) / 2
1271
+ elif element.get("verticalAlign") == "bottom":
1272
+ y += content_height - visual_height
234
1273
  return {
235
- "x": element["x"],
236
- "y": element["y"],
1274
+ "x": x,
1275
+ "y": y,
237
1276
  "width": visual_width,
238
1277
  "height": visual_height,
239
1278
  }
@@ -247,6 +1286,49 @@ def intersection_area(left: dict[str, Any], right: dict[str, Any]) -> int | floa
247
1286
  return width * height
248
1287
 
249
1288
 
1289
+ def intersection_height(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1290
+ height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
1291
+ return max(height, 0)
1292
+
1293
+
1294
+ def intersection_width(left: dict[str, Any], right: dict[str, Any]) -> int | float:
1295
+ width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
1296
+ return max(width, 0)
1297
+
1298
+
1299
+ def element_area(element: dict[str, Any]) -> int | float:
1300
+ return max(element["width"], 0) * max(element["height"], 0)
1301
+
1302
+
1303
+ def contains(outer: dict[str, Any], inner: dict[str, Any], tolerance: int | float = 2) -> bool:
1304
+ return (
1305
+ inner["x"] >= outer["x"] - tolerance
1306
+ and inner["y"] >= outer["y"] - tolerance
1307
+ and inner["x"] + inner["width"] <= outer["x"] + outer["width"] + tolerance
1308
+ and inner["y"] + inner["height"] <= outer["y"] + outer["height"] + tolerance
1309
+ )
1310
+
1311
+
1312
+ def is_bottom_layer_full_slide_whiteboard(
1313
+ whiteboard: dict[str, Any], other: dict[str, Any], slide_width: int | float, slide_height: int | float
1314
+ ) -> bool:
1315
+ return (
1316
+ whiteboard["order"] < other["order"]
1317
+ and whiteboard["x"] <= 2
1318
+ and whiteboard["y"] <= 2
1319
+ and whiteboard["width"] >= slide_width - 4
1320
+ and whiteboard["height"] >= slide_height - 4
1321
+ )
1322
+
1323
+
1324
+ def is_background_container_for_whiteboard(container: dict[str, Any], whiteboard: dict[str, Any]) -> bool:
1325
+ if container["order"] > whiteboard["order"]:
1326
+ return False
1327
+ if is_text_element(container):
1328
+ return False
1329
+ return contains(container, whiteboard)
1330
+
1331
+
250
1332
  def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
251
1333
  if not (is_text_element(left) and is_text_element(right)):
252
1334
  return False
@@ -269,11 +1351,69 @@ def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
269
1351
  return same_column and vertical_offset >= top_font_size * 0.75
270
1352
 
271
1353
 
1354
+ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str, Any]) -> bool:
1355
+ if not (is_text_element(left) and is_text_element(right)):
1356
+ return False
1357
+ if not (has_text_content(left) and has_text_content(right)):
1358
+ return False
1359
+ if is_ghost_text(left) or is_ghost_text(right):
1360
+ return False
1361
+ if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
1362
+ return False
1363
+
1364
+ source, target = sorted([left, right], key=lambda element: element["x"])
1365
+ if source["x"] == target["x"]:
1366
+ return False
1367
+ wrap_enabled = source.get("wrap") not in {"false", "0"}
1368
+ has_horizontal_gap = source["x"] + source["width"] <= target["x"]
1369
+ if wrap_enabled and has_horizontal_gap:
1370
+ return False
1371
+ if source.get("autoFit") == "normal-auto-fit":
1372
+ return False
1373
+ if source.get("textAlign") in {"center", "right"}:
1374
+ return False
1375
+
1376
+ font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
1377
+ padding_left = source.get("paddingLeft", 0)
1378
+ padding_right = source.get("paddingRight", 0)
1379
+ available_width = max(source["width"] - padding_left - padding_right, 1)
1380
+ visual_width = estimate_text_max_line_width(source)
1381
+ overflow_width = visual_width - available_width
1382
+ min_overflow = max(font_size * 1.5, available_width * 0.08)
1383
+ if overflow_width < min_overflow:
1384
+ return False
1385
+
1386
+ intrusion_width = source["x"] + padding_left + visual_width - target["x"]
1387
+ min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
1388
+ if intrusion_width < min_intrusion:
1389
+ return False
1390
+
1391
+ vertical_overlap = intersection_height(source, target)
1392
+ min_vertical_overlap = min(source["height"], target["height"]) * 0.40
1393
+ return vertical_overlap >= min_vertical_overlap
1394
+
1395
+
1396
+ def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
1397
+ source, target = sorted([left, right], key=lambda element: element["x"])
1398
+ padding_left = source.get("paddingLeft", 0)
1399
+ visual_width = estimate_text_max_line_width(source)
1400
+ source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
1401
+ width = intersection_width(source_visual_bbox, target)
1402
+ height = intersection_height(source_visual_bbox, target)
1403
+ return {
1404
+ "intersection_width": round(width, 3),
1405
+ "intersection_height": round(height, 3),
1406
+ "intersection_area": round(width * height, 3),
1407
+ }
1408
+
1409
+
272
1410
  def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
273
1411
  if is_text_element(left) and not has_text_content(left):
274
1412
  return False
275
1413
  if is_text_element(right) and not has_text_content(right):
276
1414
  return False
1415
+ if is_ghost_text(left) or is_ghost_text(right):
1416
+ return False
277
1417
  if is_template_text_stack(left, right):
278
1418
  return False
279
1419
  if is_text_element(left) and is_text_element(right):
@@ -294,13 +1434,356 @@ def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
294
1434
  return False
295
1435
 
296
1436
 
297
- def lint_slide(slide_xml: str, slide_number: int) -> dict[str, Any]:
298
- elements = extract_elements(slide_xml)
1437
+ def build_whiteboard_external_overlap_issue(
1438
+ whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
1439
+ ) -> dict[str, Any]:
1440
+ element_ids = [detail["element"] for detail in overlap_details]
1441
+ return {
1442
+ "level": "warning",
1443
+ "code": "whiteboard_external_overlap",
1444
+ "elements": [whiteboard["id"], *element_ids],
1445
+ "message": f'whiteboard {whiteboard["id"]} overlaps {len(element_ids)} sibling elements across its boundary',
1446
+ "hint": (
1447
+ "Treat this as a static whiteboard container-bbox risk, not final visual proof. "
1448
+ "After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
1449
+ "the final authority because XML readback does not include whiteboard SVG/Mermaid internals."
1450
+ ),
1451
+ "overlaps": overlap_details,
1452
+ }
1453
+
1454
+
1455
+ def should_report_whiteboard_overlap(
1456
+ whiteboard: dict[str, Any],
1457
+ other: dict[str, Any],
1458
+ slide_width: int | float,
1459
+ slide_height: int | float,
1460
+ ) -> dict[str, Any] | None:
1461
+ if other is whiteboard or not intersects(whiteboard, other):
1462
+ return None
1463
+ if is_ghost_text(other):
1464
+ return None
1465
+ if contains(whiteboard, other):
1466
+ return None
1467
+ if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
1468
+ return None
1469
+ if is_background_container_for_whiteboard(other, whiteboard):
1470
+ return None
1471
+
1472
+ overlap_width = intersection_width(whiteboard, other)
1473
+ overlap_height = intersection_height(whiteboard, other)
1474
+ if overlap_width < 8 or overlap_height < 8:
1475
+ return None
1476
+
1477
+ other_area = element_area(other)
1478
+ if other_area <= 0:
1479
+ return None
1480
+ overlap_area = overlap_width * overlap_height
1481
+ overlap_ratio = overlap_area / other_area
1482
+ if overlap_ratio < 0.15:
1483
+ return None
1484
+
1485
+ return {
1486
+ "element": other["id"],
1487
+ "kind": other["kind"],
1488
+ "type": other.get("type"),
1489
+ "overlap_width": overlap_width,
1490
+ "overlap_height": overlap_height,
1491
+ "target_overlap_ratio": round(overlap_ratio, 3),
1492
+ }
1493
+
1494
+
1495
+ def prune_contained_text_overlap_details(
1496
+ overlap_details: list[dict[str, Any]], elements_by_id: dict[str, dict[str, Any]]
1497
+ ) -> list[dict[str, Any]]:
1498
+ pruned: list[dict[str, Any]] = []
1499
+ for detail in overlap_details:
1500
+ element = elements_by_id[detail["element"]]
1501
+ if is_text_element(element):
1502
+ has_reported_container = any(
1503
+ detail["element"] != other_detail["element"]
1504
+ and not is_text_element(elements_by_id[other_detail["element"]])
1505
+ and contains(elements_by_id[other_detail["element"]], element)
1506
+ for other_detail in overlap_details
1507
+ )
1508
+ if has_reported_container:
1509
+ continue
1510
+ pruned.append(detail)
1511
+ return pruned
1512
+
1513
+
1514
+ def detect_whiteboard_external_overlaps(
1515
+ elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1516
+ ) -> list[dict[str, Any]]:
1517
+ issues: list[dict[str, Any]] = []
1518
+ elements_by_id = {element["id"]: element for element in elements}
1519
+ for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
1520
+ overlap_details = [
1521
+ detail
1522
+ for element in elements
1523
+ if (
1524
+ detail := should_report_whiteboard_overlap(
1525
+ whiteboard,
1526
+ element,
1527
+ slide_width,
1528
+ slide_height,
1529
+ )
1530
+ )
1531
+ is not None
1532
+ ]
1533
+ overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_id)
1534
+ if overlap_details:
1535
+ issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
1536
+ return issues
1537
+
1538
+
1539
+ def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
1540
+ bbox = {key: element[key] for key in ("x", "y", "width", "height")}
1541
+ if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
1542
+ return bbox
1543
+ rotation = element["rotation"]
1544
+ if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
1545
+ rotation = 0
1546
+ rotation %= 360
1547
+ if math.isclose(rotation, 0, abs_tol=1e-9):
1548
+ return bbox
1549
+ radians = math.radians(rotation)
1550
+ sine = abs(math.sin(radians))
1551
+ cosine = abs(math.cos(radians))
1552
+ sine = 0 if math.isclose(sine, 0, abs_tol=1e-12) else sine
1553
+ cosine = 0 if math.isclose(cosine, 0, abs_tol=1e-12) else cosine
1554
+ rotated_width = element["width"] * cosine + element["height"] * sine
1555
+ rotated_height = element["width"] * sine + element["height"] * cosine
1556
+ return {
1557
+ "x": element["x"] - (rotated_width - element["width"]) / 2,
1558
+ "y": element["y"] - (rotated_height - element["height"]) / 2,
1559
+ "width": rotated_width,
1560
+ "height": rotated_height,
1561
+ }
1562
+
1563
+
1564
+ def detect_elements_out_of_canvas(
1565
+ elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1566
+ ) -> list[dict[str, Any]]:
1567
+ issues: list[dict[str, Any]] = []
1568
+ for element in (
1569
+ element
1570
+ for element in elements
1571
+ if element["kind"] in {"table", "chart"}
1572
+ or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
1573
+ ):
1574
+ bbox = element_canvas_bbox(element)
1575
+ overflow = {
1576
+ "left": max(-bbox["x"], 0),
1577
+ "top": max(-bbox["y"], 0),
1578
+ "right": max(bbox["x"] + bbox["width"] - slide_width, 0),
1579
+ "bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
1580
+ }
1581
+ overflow_details = [
1582
+ f"{side} by {amount:g}px"
1583
+ for side, amount in overflow.items()
1584
+ if amount > CANVAS_OVERFLOW_TOLERANCE
1585
+ ]
1586
+ if not overflow_details:
1587
+ continue
1588
+ issues.append(
1589
+ {
1590
+ "level": "error",
1591
+ "code": f'{element["kind"]}_out_of_canvas',
1592
+ "elements": [element["id"]],
1593
+ "canvas": {"width": slide_width, "height": slide_height},
1594
+ "bbox": bbox,
1595
+ "overflow": overflow,
1596
+ "message": (
1597
+ f'{element["kind"]} {element["id"]} exceeds the {slide_width:g}x{slide_height:g} canvas '
1598
+ f'({", ".join(overflow_details)})'
1599
+ ),
1600
+ "hint": (
1601
+ "Move the table inside the canvas, reduce table.width/table.height, or split the table across "
1602
+ "slides."
1603
+ if element["kind"] == "table"
1604
+ else f'Move the {element["kind"]} inside the canvas or reduce its width/height.'
1605
+ ),
1606
+ }
1607
+ )
1608
+ return issues
1609
+
1610
+
1611
+ def extract_table_column_sizes(table_xml: str) -> list[int | float | None]:
1612
+ sizes: list[int | float | None] = []
1613
+ for match in re.finditer(r"<col\b([^>]*)/?>", table_xml):
1614
+ attrs = match.group(1)
1615
+ span = extract_numeric_attribute(attrs, "span") or 1
1616
+ span_count = int(span) if math.isfinite(span) and span > 0 and float(span).is_integer() else 1
1617
+ sizes.extend([extract_numeric_attribute(attrs, "width")] * span_count)
1618
+ return sizes
1619
+
1620
+
1621
+ def extract_table_row_sizes(table_xml: str) -> list[int | float | None]:
1622
+ return [extract_numeric_attribute(match.group(1), "height") for match in re.finditer(r"<tr\b([^>]*)>", table_xml)]
1623
+
1624
+
1625
+ def resolve_table_dimension(
1626
+ table_xml: str,
1627
+ declared_size: int | float | None,
1628
+ extract_sizes: Any,
1629
+ default_size: int | float,
1630
+ ) -> tuple[int | float | None, dict[str, Any] | None]:
1631
+ input_sizes = extract_sizes(table_xml)
1632
+ if not input_sizes:
1633
+ return declared_size, None
1634
+ layout = solve_weighted_min_layout(
1635
+ input_sizes, default_size, declared_size if is_filled_size(declared_size) else None
1636
+ )
1637
+ return layout["actual_size"], layout
1638
+
1639
+
1640
+ def format_size(size: int | float) -> str:
1641
+ return f"{size:g}"
1642
+
1643
+
1644
+ def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1645
+ issues: list[dict[str, Any]] = []
1646
+ dimensions = {
1647
+ "width": ("col", "column widths"),
1648
+ "height": ("tr", "row heights"),
1649
+ }
1650
+ for table in (element for element in elements if element["kind"] == "table"):
1651
+ for dimension, (child_tag, child_description) in dimensions.items():
1652
+ target_size = table[f"declared_{dimension}"]
1653
+ if not is_filled_size(target_size):
1654
+ continue
1655
+ layout = table["table_layouts"][dimension]
1656
+ if layout is None:
1657
+ continue
1658
+ actual_size = layout["actual_size"]
1659
+ if math.isclose(actual_size, target_size, rel_tol=1e-9, abs_tol=1e-9):
1660
+ continue
1661
+ issues.append(
1662
+ {
1663
+ "level": "info",
1664
+ "code": "table_resolved_size_mismatch",
1665
+ "elements": [table["id"]],
1666
+ "dimension": dimension,
1667
+ "declared_size": target_size,
1668
+ "resolved_size": actual_size,
1669
+ "resolved_sizes": layout["final_sizes"],
1670
+ "message": (
1671
+ f'table {table["id"]} declares {dimension}={format_size(target_size)}px, but its '
1672
+ f"{child_description} resolve to {format_size(actual_size)}px"
1673
+ ),
1674
+ "hint": (
1675
+ f"Set table.{dimension} to {format_size(actual_size)}px, or adjust <{child_tag}> sizes "
1676
+ f"so their resolved total matches {format_size(target_size)}px."
1677
+ ),
1678
+ }
1679
+ )
1680
+ return issues
1681
+
1682
+
1683
+ def segment_intersects_rect(
1684
+ x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
1685
+ ) -> bool:
1686
+ """True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
1687
+ left = rect["x"]
1688
+ top = rect["y"]
1689
+ right = rect["x"] + rect["width"]
1690
+ bottom = rect["y"] + rect["height"]
1691
+ if right <= left or bottom <= top:
1692
+ return False
1693
+ dx = x2 - x1
1694
+ dy = y2 - y1
1695
+ if dx == 0 and dy == 0:
1696
+ return left <= x1 <= right and top <= y1 <= bottom
1697
+ t_enter, t_exit = 0.0, 1.0
1698
+ for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
1699
+ if delta == 0:
1700
+ if distance < 0:
1701
+ return False
1702
+ continue
1703
+ t = distance / delta
1704
+ if delta < 0:
1705
+ t_enter = max(t_enter, t)
1706
+ else:
1707
+ t_exit = min(t_exit, t)
1708
+ if t_enter > t_exit:
1709
+ return False
1710
+ return True
1711
+
1712
+
1713
+ def line_text_graze_margin(text_element: dict[str, Any]) -> float:
1714
+ font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
1715
+ return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
1716
+
1717
+
1718
+ def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
1719
+ width = rect["width"] - 2 * margin
1720
+ height = rect["height"] - 2 * margin
1721
+ if width <= 0 or height <= 0:
1722
+ return None
1723
+ return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
1724
+
1725
+
1726
+ def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
1727
+ if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
1728
+ return False
1729
+ if not is_text_element(text_element) or not has_text_content(text_element):
1730
+ return False
1731
+ if is_ghost_text(text_element) or is_decorative_text(text_element):
1732
+ return False
1733
+ glyph_bbox = estimate_text_visual_bbox(text_element)
1734
+ if glyph_bbox is None:
1735
+ return False
1736
+ # Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
1737
+ # frame is exempt; only a line that actually cuts through the letterforms is a crossing.
1738
+ target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
1739
+ if target is None:
1740
+ return False
1741
+ return segment_intersects_rect(
1742
+ line["startX"], line["startY"], line["endX"], line["endY"], target
1743
+ )
1744
+
1745
+
1746
+ def detect_line_text_crossings(
1747
+ slide_xml: str, elements: list[dict[str, Any]]
1748
+ ) -> list[dict[str, Any]]:
1749
+ lines = extract_line_elements(slide_xml)
1750
+ if not lines:
1751
+ return []
1752
+ text_elements = [element for element in elements if is_text_element(element)]
299
1753
  issues: list[dict[str, Any]] = []
1754
+ for line in lines:
1755
+ for text_element in text_elements:
1756
+ if not line_crosses_text(line, text_element):
1757
+ continue
1758
+ issues.append(
1759
+ {
1760
+ "level": "error",
1761
+ "code": "bbox_overlap",
1762
+ "elements": [line["id"], text_element["id"]],
1763
+ "message": f'line {line["id"]} crosses text {text_element["id"]}',
1764
+ "hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
1765
+ }
1766
+ )
1767
+ return issues
1768
+
1769
+
1770
+ def lint_slide(
1771
+ slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
1772
+ ) -> dict[str, Any]:
1773
+ elements = extract_elements(slide_xml)
1774
+ issues: list[dict[str, Any]] = [
1775
+ *detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
1776
+ *detect_elements_out_of_canvas(elements, slide_width, slide_height),
1777
+ *detect_table_layout_size_mismatches(elements),
1778
+ *detect_text_may_overflow_shapes(elements),
1779
+ *detect_image_text_occlusions(elements),
1780
+ *detect_line_text_crossings(slide_xml, elements),
1781
+ ]
300
1782
 
301
1783
  for index, left in enumerate(elements):
302
1784
  for right in elements[index + 1 :]:
303
- if not intersects(left, right) or not should_flag_overlap(left, right):
1785
+ horizontal_overflow = should_flag_horizontal_text_overflow(left, right)
1786
+ if not horizontal_overflow and (not intersects(left, right) or not should_flag_overlap(left, right)):
304
1787
  continue
305
1788
  issues.append(
306
1789
  {
@@ -308,36 +1791,891 @@ def lint_slide(slide_xml: str, slide_number: int) -> dict[str, Any]:
308
1791
  "code": "bbox_overlap",
309
1792
  "elements": [left["id"], right["id"]],
310
1793
  "message": f'{left["id"]} overlaps {right["id"]}',
1794
+ "hint": "Move or resize the elements so their visual bounds no longer intersect.",
1795
+ **(
1796
+ {"measurement": horizontal_text_overflow_measurement(left, right)}
1797
+ if horizontal_overflow
1798
+ else {}
1799
+ ),
311
1800
  }
312
1801
  )
313
1802
 
314
- return {"slide_number": slide_number, "element_count": len(elements), "issues": issues}
1803
+ return {
1804
+ "slide_number": slide_number,
1805
+ "element_count": len(elements),
1806
+ "elements": elements,
1807
+ "issues": issues,
1808
+ }
315
1809
 
316
1810
 
317
- def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
318
- xml_error = validate_xml_well_formed(xml)
319
- if xml_error:
320
- return {
321
- "file": source_path,
322
- "slide_size": {"width": 960, "height": 540},
323
- "summary": {"slide_count": 0, "error_count": 1, "warning_count": 0},
324
- "issues": [xml_error],
325
- "slides": [],
1811
+
1812
+ MIN_CONTAINER_WIDTH = 140
1813
+ MIN_CONTAINER_HEIGHT = 160
1814
+ MIN_SHORT_CARD_HEIGHT = 80
1815
+ MIN_CONTAINER_AREA = 20_000
1816
+ MIN_CONTENT_COVERAGE_RATIO = 0.15
1817
+ MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
1818
+ MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
1819
+ SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
1820
+ MIN_SIMILAR_SHORT_CARD_COUNT = 2
1821
+ LARGE_VISUAL_CHILD_RATIO = 0.35
1822
+ LAYOUT_PANEL_SPAN_RATIO = 0.90
1823
+ IMAGE_OVERLAY_MATCH_RATIO = 0.90
1824
+ DENSITY_CONTAINMENT_TOLERANCE = 8
1825
+
1826
+
1827
+ def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1828
+ left = max(element["x"], container["x"])
1829
+ top = max(element["y"], container["y"])
1830
+ right = min(element["x"] + element["width"], container["x"] + container["width"])
1831
+ bottom = min(element["y"] + element["height"], container["y"] + container["height"])
1832
+ if right <= left or bottom <= top:
1833
+ return None
1834
+ return {"x": left, "y": top, "width": right - left, "height": bottom - top}
1835
+
1836
+
1837
+ def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
1838
+ x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
1839
+ area = 0
1840
+ for left, right in zip(x_coordinates, x_coordinates[1:]):
1841
+ intervals = sorted(
1842
+ (rect["y"], rect["y"] + rect["height"])
1843
+ for rect in rectangles
1844
+ if rect["x"] < right and rect["x"] + rect["width"] > left
1845
+ )
1846
+ covered_height = 0
1847
+ interval_end: int | float | None = None
1848
+ for top, bottom in intervals:
1849
+ if interval_end is None:
1850
+ covered_height += bottom - top
1851
+ interval_end = bottom
1852
+ elif bottom > interval_end:
1853
+ covered_height += bottom - max(top, interval_end)
1854
+ interval_end = bottom
1855
+ area += (right - left) * covered_height
1856
+ return area
1857
+
1858
+
1859
+ def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1860
+ return sum(
1861
+ other is not element
1862
+ and is_visually_rendered(other)
1863
+ and other["kind"] == "shape"
1864
+ and other["type"] == "rect"
1865
+ and other["width"] >= MIN_CONTAINER_WIDTH
1866
+ and other["height"] >= MIN_SHORT_CARD_HEIGHT
1867
+ and element_area(other) >= MIN_CONTAINER_AREA
1868
+ and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
1869
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1870
+ and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
1871
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1872
+ for other in elements
1873
+ ) >= MIN_SIMILAR_SHORT_CARD_COUNT
1874
+
1875
+
1876
+ def is_layout_container(
1877
+ element: dict[str, Any],
1878
+ slide_width: int | float,
1879
+ slide_height: int | float,
1880
+ elements: list[dict[str, Any]] | None = None,
1881
+ ) -> bool:
1882
+ has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
1883
+ elements is not None
1884
+ and element["height"] >= MIN_SHORT_CARD_HEIGHT
1885
+ and has_similar_short_card_peer(element, elements)
1886
+ )
1887
+ return (
1888
+ element["kind"] == "shape"
1889
+ and element["type"] == "rect"
1890
+ and is_visually_rendered(element)
1891
+ and element["width"] >= MIN_CONTAINER_WIDTH
1892
+ and has_supported_height
1893
+ and element_area(element) >= MIN_CONTAINER_AREA
1894
+ and not (
1895
+ element["x"] <= 2
1896
+ and element["y"] <= 2
1897
+ and element["width"] >= slide_width - 4
1898
+ and element["height"] >= slide_height - 4
1899
+ )
1900
+ )
1901
+
1902
+
1903
+ def is_edge_spanning_layout_panel(
1904
+ element: dict[str, Any], slide_width: int | float, slide_height: int | float
1905
+ ) -> bool:
1906
+ touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
1907
+ touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
1908
+ return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
1909
+ touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
1910
+ )
1911
+
1912
+
1913
+ def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1914
+ container_area = element_area(container)
1915
+ return any(
1916
+ element["kind"] == "img"
1917
+ and is_visually_rendered(element)
1918
+ and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
1919
+ for element in elements
1920
+ )
1921
+
1922
+
1923
+ def is_nested_in_layout_panel(
1924
+ container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1925
+ ) -> bool:
1926
+ return any(
1927
+ element is not container
1928
+ and element["kind"] == "shape"
1929
+ and element["type"] == "rect"
1930
+ and is_visually_rendered(element)
1931
+ and is_edge_spanning_layout_panel(element, slide_width, slide_height)
1932
+ and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
1933
+ for element in elements
1934
+ )
1935
+
1936
+
1937
+ def extract_density_elements(slide_xml: str) -> list[dict[str, Any]]:
1938
+ elements = extract_elements(slide_xml)
1939
+ elements_by_id = {element["id"]: element for element in elements}
1940
+ root = ET.fromstring(slide_xml)
1941
+ for node in root.iter():
1942
+ if xml_local_name(node.tag) != "shape":
1943
+ continue
1944
+ element = elements_by_id.get(node.attrib.get("id", ""))
1945
+ if element is None:
1946
+ continue
1947
+ content_node = next(
1948
+ (child for child in node if xml_local_name(child.tag) == "content"),
1949
+ None,
1950
+ )
1951
+ paragraphs = (
1952
+ [
1953
+ " ".join("".join(paragraph.itertext()).split())
1954
+ for paragraph in content_node.iter()
1955
+ if xml_local_name(paragraph.tag) == "p"
1956
+ ]
1957
+ if content_node is not None
1958
+ else []
1959
+ )
1960
+ raw_font_size = (
1961
+ content_node.attrib.get("fontSize") if content_node is not None else None
1962
+ ) or node.attrib.get("fontSize")
1963
+ try:
1964
+ base_font_size = float(raw_font_size or 16)
1965
+ except ValueError:
1966
+ base_font_size = 16.0
1967
+ element.update(
1968
+ {
1969
+ "textType": content_node.attrib.get("textType") if content_node is not None else None,
1970
+ "textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
1971
+ "autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
1972
+ "fontSize": base_font_size,
1973
+ "text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
1974
+ }
1975
+ )
1976
+ if not has_text_content(element):
1977
+ continue
1978
+ declared_font_sizes = []
1979
+ for descendant in node.iter():
1980
+ raw_declared_font_size = descendant.attrib.get("fontSize")
1981
+ if raw_declared_font_size is None:
1982
+ continue
1983
+ try:
1984
+ declared_font_sizes.append(float(raw_declared_font_size))
1985
+ except ValueError:
1986
+ continue
1987
+ if declared_font_sizes:
1988
+ element["fontSize"] = max(declared_font_sizes)
1989
+ for match in re.finditer(r"<icon\b([^>]*)>", slide_xml):
1990
+ attrs = match.group(1)
1991
+ x = extract_numeric_attribute(attrs, "topLeftX")
1992
+ y = extract_numeric_attribute(attrs, "topLeftY")
1993
+ width = extract_numeric_attribute(attrs, "width")
1994
+ height = extract_numeric_attribute(attrs, "height")
1995
+ if any(value is None for value in (x, y, width, height)):
1996
+ continue
1997
+ icon_alpha = extract_numeric_attribute(attrs, "alpha")
1998
+ elements.append(
1999
+ {
2000
+ "id": extract_attribute(attrs, "id") or f"icon-{len(elements) + 1}",
2001
+ "kind": "icon",
2002
+ "type": "icon",
2003
+ "x": x,
2004
+ "y": y,
2005
+ "width": width,
2006
+ "height": height,
2007
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2008
+ "alpha": icon_alpha if icon_alpha is not None else 1,
2009
+ "order": len(elements),
2010
+ }
2011
+ )
2012
+ for match in re.finditer(r"<polyline\b([^>]*)>", slide_xml):
2013
+ attrs = match.group(1)
2014
+ x = extract_numeric_attribute(attrs, "topLeftX")
2015
+ y = extract_numeric_attribute(attrs, "topLeftY")
2016
+ width = extract_numeric_attribute(attrs, "width")
2017
+ height = extract_numeric_attribute(attrs, "height")
2018
+ if any(value is None for value in (x, y, width, height)):
2019
+ continue
2020
+ polyline_alpha = extract_numeric_attribute(attrs, "alpha")
2021
+ elements.append(
2022
+ {
2023
+ "id": extract_attribute(attrs, "id") or f"polyline-{len(elements) + 1}",
2024
+ "kind": "polyline",
2025
+ "type": "polyline",
2026
+ "x": x,
2027
+ "y": y,
2028
+ "width": width,
2029
+ "height": height,
2030
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2031
+ "alpha": polyline_alpha if polyline_alpha is not None else 1,
2032
+ "order": len(elements),
2033
+ }
2034
+ )
2035
+ for line_element in extract_line_elements(slide_xml):
2036
+ line_element["order"] = len(elements)
2037
+ elements.append(line_element)
2038
+ return elements
2039
+
2040
+
2041
+ def is_visually_rendered(element: dict[str, Any]) -> bool:
2042
+ return element.get("alpha", 1) > 0
2043
+
2044
+
2045
+ def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
2046
+ if not is_visually_rendered(element):
2047
+ return None
2048
+ if is_text_element(element):
2049
+ estimated = estimate_text_visual_bbox(element)
2050
+ return clipped_bbox(estimated, container) if estimated else None
2051
+ return clipped_bbox(element, container)
2052
+
2053
+
2054
+ def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
2055
+ if container["kind"] != "shape" or not has_text_content(container):
2056
+ return None
2057
+ text_proxy = {**container, "type": "text"}
2058
+ estimated = estimate_text_visual_bbox(text_proxy)
2059
+ return clipped_bbox(estimated, container) if estimated else None
2060
+
2061
+
2062
+ def slide_content_visual_bbox(
2063
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
2064
+ ) -> dict[str, int | float] | None:
2065
+ if not is_visually_rendered(element):
2066
+ return None
2067
+ if is_text_element(element):
2068
+ estimated = estimate_text_visual_bbox(element)
2069
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
2070
+ if element["kind"] == "shape" and has_text_content(element):
2071
+ estimated = own_text_visual_bbox(element)
2072
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
2073
+ if element["kind"] == "line":
2074
+ # a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
2075
+ # treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
2076
+ return clipped_bbox(line_stroke_bbox(element), slide_bbox)
2077
+ if element["kind"] in {"img", "chart", "table", "whiteboard", "icon", "polyline"}:
2078
+ return clipped_bbox(element, slide_bbox)
2079
+ return None
2080
+
2081
+
2082
+ def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
2083
+ return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
2084
+
2085
+
2086
+ def is_slide_content_present(
2087
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
2088
+ ) -> bool:
2089
+ # Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
2090
+ # *anything* rendered here", not the richer "counts toward meaningful content density" bar
2091
+ # that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
2092
+ # decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
2093
+ # count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
2094
+ # instead of maintaining an allow-list that silently treats unlisted kinds as blank.
2095
+ if not is_visually_rendered(element):
2096
+ return False
2097
+ if (
2098
+ element["kind"] == "shape"
2099
+ and element["type"] == "rect"
2100
+ and not has_text_content(element)
2101
+ and element["x"] <= 2
2102
+ and element["y"] <= 2
2103
+ and element["width"] >= slide_bbox["width"] - 4
2104
+ and element["height"] >= slide_bbox["height"] - 4
2105
+ ):
2106
+ # A full-canvas plain rect is a background panel, not content -- same reasoning as
2107
+ # is_layout_container's existing background exclusion. A slide with nothing else on it
2108
+ # is still effectively blank.
2109
+ return False
2110
+ bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
2111
+ return clipped_bbox(bbox, slide_bbox) is not None
2112
+
2113
+
2114
+ def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
2115
+ if element["kind"] not in {"img", "chart", "table", "whiteboard"}:
2116
+ return False
2117
+ if not is_visually_rendered(element):
2118
+ return False
2119
+ return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
2120
+
2121
+
2122
+ def detect_sparse_container_content(
2123
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2124
+ ) -> list[dict[str, Any]]:
2125
+ issues: list[dict[str, Any]] = []
2126
+ for container in (
2127
+ element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
2128
+ ):
2129
+ if (
2130
+ is_edge_spanning_layout_panel(container, slide_width, slide_height)
2131
+ or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
2132
+ or has_matching_image_overlay(container, elements)
2133
+ ):
2134
+ continue
2135
+ children = [
2136
+ element
2137
+ for element in elements
2138
+ if element is not container
2139
+ and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2140
+ ]
2141
+ if any(is_large_visual_child(child, container) for child in children):
2142
+ continue
2143
+ own_text_bbox = own_text_visual_bbox(container)
2144
+ rectangles = ([own_text_bbox] if own_text_bbox else []) + [
2145
+ bbox for child in children if (bbox := visual_bbox(child, container)) is not None
2146
+ ]
2147
+ content_area = rectangle_union_area(rectangles) if rectangles else 0
2148
+ coverage_ratio = content_area / element_area(container)
2149
+ if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
2150
+ continue
2151
+ issues.append(
2152
+ {
2153
+ "level": "warning",
2154
+ "code": "sparse_container_content",
2155
+ "target": {
2156
+ "slide_number": slide_number,
2157
+ "container_id": container["id"],
2158
+ "container_type": container["type"],
2159
+ "bbox": {key: container[key] for key in ("x", "y", "width", "height")},
2160
+ },
2161
+ "rule": {
2162
+ "name": "large_container_visible_content_coverage",
2163
+ "threshold": MIN_CONTENT_COVERAGE_RATIO,
2164
+ "comparison": "content_coverage_ratio < threshold",
2165
+ },
2166
+ "measurement": {
2167
+ "container_area": element_area(container),
2168
+ "visible_content_area": round(content_area, 3),
2169
+ "content_coverage_ratio": round(coverage_ratio, 3),
2170
+ "content_element_count": len(children) + (1 if own_text_bbox else 0),
2171
+ },
2172
+ "elements": [container["id"], *[child["id"] for child in children]],
2173
+ }
2174
+ )
2175
+ return issues
2176
+
2177
+
2178
+ def detect_sparse_slide_content(
2179
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2180
+ ) -> list[dict[str, Any]]:
2181
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2182
+ content = [
2183
+ (element, bbox)
2184
+ for element in elements
2185
+ if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
2186
+ ]
2187
+ if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
2188
+ return []
2189
+ content_area = rectangle_union_area([bbox for _, bbox in content])
2190
+ slide_area = slide_width * slide_height
2191
+ coverage_ratio = content_area / slide_area
2192
+ if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
2193
+ return []
2194
+ return [
2195
+ {
2196
+ "level": "warning",
2197
+ "code": "sparse_slide_content",
2198
+ "target": {
2199
+ "slide_number": slide_number,
2200
+ "bbox": slide_bbox,
2201
+ },
2202
+ "rule": {
2203
+ "name": "slide_visible_content_coverage",
2204
+ "threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
2205
+ "comparison": "content_coverage_ratio < threshold",
2206
+ },
2207
+ "measurement": {
2208
+ "slide_area": slide_area,
2209
+ "visible_content_area": round(content_area, 3),
2210
+ "content_coverage_ratio": round(coverage_ratio, 3),
2211
+ "content_element_count": len(content),
2212
+ },
2213
+ "elements": [element["id"] for element, _ in content],
326
2214
  }
2215
+ ]
2216
+
327
2217
 
328
- presentation = parse_presentation(xml)
329
- slides = [
330
- lint_slide(slide_xml, index + 1)
331
- for index, slide_xml in enumerate(presentation["slides"])
2218
+ def detect_blank_slide(
2219
+ elements: list[dict[str, Any]],
2220
+ slide_number: int,
2221
+ slide_width: int | float,
2222
+ slide_height: int | float,
2223
+ ) -> list[dict[str, Any]]:
2224
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2225
+ visible_elements = [
2226
+ element for element in elements if is_slide_content_present(element, slide_bbox)
332
2227
  ]
333
- error_count = sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "error")
334
- warning_count = sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "warning")
2228
+ if visible_elements:
2229
+ return []
2230
+ return [
2231
+ {
2232
+ "level": "error",
2233
+ "code": "blank_slide",
2234
+ "schema_version": "2.0",
2235
+ "target": {"slide_number": slide_number},
2236
+ "rule": {
2237
+ "name": "slide_has_visible_content",
2238
+ "comparison": "visible_element_count == 0",
2239
+ },
2240
+ "measurement": {
2241
+ "visible_element_count": 0,
2242
+ "declared_element_count": len(elements),
2243
+ },
2244
+ "elements": [element["id"] for element in elements],
2245
+ "message": "slide has no visible content beyond empty layout shapes",
2246
+ "hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
2247
+ }
2248
+ ]
2249
+
2250
+
2251
+
2252
+ RULE_METADATA: dict[str, dict[str, Any]] = {
2253
+ "xml_not_well_formed": {
2254
+ "name": "xml_is_well_formed",
2255
+ "comparison": "xml_parse_error == false",
2256
+ },
2257
+ "sml_prefixed_tag": {
2258
+ "name": "sml_uses_default_namespace",
2259
+ "comparison": "prefixed_sml_tag_count == 0",
2260
+ },
2261
+ "sxsd_unsupported_tag": {
2262
+ "name": "tag_is_supported_by_slides_xml_schema",
2263
+ "comparison": "unsupported_tag_count == 0",
2264
+ },
2265
+ "sxsd_unsupported_attr": {
2266
+ "name": "attribute_is_supported_by_slides_xml_schema",
2267
+ "comparison": "unsupported_attribute_count == 0",
2268
+ },
2269
+ "icon_missing_fill_color": {
2270
+ "name": "icon_has_visible_fill_color",
2271
+ "comparison": "fill_color_present == true",
2272
+ },
2273
+ "icon_transparent_fill_color": {
2274
+ "name": "icon_has_visible_fill_color",
2275
+ "comparison": "fill_alpha > 0",
2276
+ },
2277
+ "iconpark_unsupported_icon_type": {
2278
+ "name": "iconpark_type_is_supported",
2279
+ "comparison": "icon_type in iconpark_index",
2280
+ },
2281
+ "bbox_overlap": {
2282
+ "name": "text_visual_bounds_do_not_overlap",
2283
+ "comparison": "intersection_area == 0",
2284
+ },
2285
+ "text_may_overflow_shape": {
2286
+ "name": "estimated_text_fits_declared_shape",
2287
+ "comparison": "estimated_height <= available_height",
2288
+ },
2289
+ "whiteboard_external_overlap": {
2290
+ "name": "whiteboard_does_not_cross_sibling_content",
2291
+ "comparison": "external_overlap_count == 0",
2292
+ },
2293
+ "image_covers_text": {
2294
+ "name": "image_does_not_cover_text",
2295
+ "comparison": "intersection_area == 0",
2296
+ },
2297
+ "image_may_cover_vertical_text": {
2298
+ "name": "image_vertical_text_occlusion_requires_review",
2299
+ "comparison": "intersection_area == 0",
2300
+ },
2301
+ "table_resolved_size_mismatch": {
2302
+ "name": "table_declared_size_matches_resolved_grid",
2303
+ "comparison": "declared_size == resolved_size",
2304
+ },
2305
+ "blank_slide": {
2306
+ "name": "slide_has_visible_content",
2307
+ "comparison": "visible_element_count > 0",
2308
+ },
2309
+ }
2310
+
2311
+
2312
+ def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
2313
+ if issue.get("rule"):
2314
+ return {**issue["rule"], "id": issue["code"]}
2315
+ if issue["code"].endswith("_out_of_canvas"):
2316
+ return {
2317
+ "id": issue["code"],
2318
+ "name": "element_stays_within_slide_canvas",
2319
+ "comparison": "max(left, top, right, bottom overflow) == 0",
2320
+ }
335
2321
  return {
2322
+ "id": issue["code"],
2323
+ **RULE_METADATA.get(
2324
+ issue["code"],
2325
+ {"name": issue["code"], "comparison": "violation_count == 0"},
2326
+ ),
2327
+ }
2328
+
2329
+
2330
+ def issue_measurement(
2331
+ issue: dict[str, Any], elements_by_id: dict[str, dict[str, Any]]
2332
+ ) -> dict[str, Any]:
2333
+ if issue.get("measurement") is not None:
2334
+ return issue["measurement"]
2335
+ if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
2336
+ left = elements_by_id.get(issue["elements"][0])
2337
+ right = elements_by_id.get(issue["elements"][1])
2338
+ if left and right:
2339
+ left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
2340
+ right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
2341
+ width = intersection_width(left_box, right_box)
2342
+ height = intersection_height(left_box, right_box)
2343
+ return {
2344
+ "intersection_width": round(width, 3),
2345
+ "intersection_height": round(height, 3),
2346
+ "intersection_area": round(width * height, 3),
2347
+ }
2348
+ if issue["code"].endswith("_out_of_canvas"):
2349
+ return {
2350
+ "canvas": issue.get("canvas"),
2351
+ "bbox": issue.get("bbox"),
2352
+ "overflow": issue.get("overflow"),
2353
+ }
2354
+ measurement_keys = (
2355
+ "line",
2356
+ "column",
2357
+ "tag",
2358
+ "attr",
2359
+ "iconType",
2360
+ "line_count",
2361
+ "line_height",
2362
+ "estimated_height",
2363
+ "available_height",
2364
+ "overflow",
2365
+ "dimension",
2366
+ "declared_size",
2367
+ "resolved_size",
2368
+ "resolved_sizes",
2369
+ "overlaps",
2370
+ )
2371
+ measured = {key: issue[key] for key in measurement_keys if key in issue}
2372
+ return measured or {"violation_count": 1}
2373
+
2374
+
2375
+ def related_object(element: dict[str, Any]) -> dict[str, Any]:
2376
+ return {
2377
+ "element_id": element["id"],
2378
+ "kind": element["kind"],
2379
+ "type": element["type"],
2380
+ "bbox": {key: element[key] for key in ("x", "y", "width", "height")},
2381
+ }
2382
+
2383
+
2384
+ def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
2385
+ elements: list[dict[str, Any]] = []
2386
+ for match in re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml):
2387
+ attrs = match.group(1)
2388
+ start_x = extract_numeric_attribute(attrs, "startX")
2389
+ start_y = extract_numeric_attribute(attrs, "startY")
2390
+ end_x = extract_numeric_attribute(attrs, "endX")
2391
+ end_y = extract_numeric_attribute(attrs, "endY")
2392
+ if any(value is None for value in (start_x, start_y, end_x, end_y)):
2393
+ continue
2394
+ line_alpha = extract_numeric_attribute(attrs, "alpha")
2395
+ base_alpha = line_alpha if line_alpha is not None else 1
2396
+ border_alpha = 1
2397
+ if match.group(2) != "/":
2398
+ close_index = slide_xml.find("</line>", match.end())
2399
+ body = slide_xml[match.end() : close_index] if close_index != -1 else ""
2400
+ border_attrs = extract_tag_attributes(body, "border")
2401
+ color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
2402
+ if isinstance(color_alpha, (int, float)):
2403
+ border_alpha = color_alpha
2404
+ elements.append(
2405
+ {
2406
+ "id": extract_attribute(attrs, "id") or f"line-{len(elements) + 1}",
2407
+ "kind": "line",
2408
+ "type": "line",
2409
+ "x": min(start_x, end_x),
2410
+ "y": min(start_y, end_y),
2411
+ "width": abs(end_x - start_x),
2412
+ "height": abs(end_y - start_y),
2413
+ "startX": start_x,
2414
+ "startY": start_y,
2415
+ "endX": end_x,
2416
+ "endY": end_y,
2417
+ "rotation": 0,
2418
+ "alpha": base_alpha * border_alpha,
2419
+ "order": len(elements),
2420
+ }
2421
+ )
2422
+ return elements
2423
+
2424
+
2425
+ def normalize_issue(
2426
+ issue: dict[str, Any],
2427
+ slide_number: int | None,
2428
+ elements_by_id: dict[str, dict[str, Any]],
2429
+ ) -> dict[str, Any]:
2430
+ normalized = dict(issue)
2431
+ element_ids = list(dict.fromkeys(normalized.get("elements", [])))
2432
+ normalized["schema_version"] = "2.0"
2433
+ normalized["element_ids"] = element_ids
2434
+ normalized["target"] = {
2435
+ **({"slide_number": slide_number} if slide_number is not None else {}),
2436
+ **normalized.get("target", {}),
2437
+ }
2438
+ normalized["rule"] = issue_rule(normalized)
2439
+ normalized["measurement"] = issue_measurement(normalized, elements_by_id)
2440
+ normalized["related_objects"] = [
2441
+ related_object(elements_by_id[element_id])
2442
+ for element_id in element_ids
2443
+ if element_id in elements_by_id
2444
+ ]
2445
+ if normalized["code"] == "sparse_container_content":
2446
+ ratio = normalized["measurement"]["content_coverage_ratio"]
2447
+ threshold = normalized["rule"]["threshold"]
2448
+ container_id = normalized["target"].get("container_id", "unknown")
2449
+ normalized.setdefault(
2450
+ "message",
2451
+ f"large card {container_id} content coverage {ratio:.1%} is below {threshold:.1%}",
2452
+ )
2453
+ normalized.setdefault(
2454
+ "hint",
2455
+ "Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
2456
+ )
2457
+ elif normalized["code"] == "sparse_slide_content":
2458
+ ratio = normalized["measurement"]["content_coverage_ratio"]
2459
+ threshold = normalized["rule"]["threshold"]
2460
+ normalized.setdefault(
2461
+ "message",
2462
+ f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
2463
+ )
2464
+ normalized.setdefault(
2465
+ "hint",
2466
+ "Review the rendered screenshot to decide whether the page is intentionally sparse.",
2467
+ )
2468
+ else:
2469
+ normalized.setdefault("message", normalized["code"].replace("_", " "))
2470
+ normalized.setdefault(
2471
+ "hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
2472
+ )
2473
+ return normalized
2474
+
2475
+
2476
+ def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
2477
+ if errors:
2478
+ return "blocked"
2479
+ if warnings:
2480
+ return "needs_screenshot_review"
2481
+ return "passed"
2482
+
2483
+
2484
+ def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
2485
+ if issue.get("code") == "sxsd_unsupported_declaration":
2486
+ return False
2487
+ if root_name == "slide":
2488
+ return True
2489
+ path = issue.get("path")
2490
+ if not isinstance(path, str):
2491
+ return False
2492
+ if path.startswith("presentation/slide/"):
2493
+ return True
2494
+ return path == "presentation/slide" and (
2495
+ issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
2496
+ )
2497
+
2498
+
2499
+ def build_result(
2500
+ source_path: str | None,
2501
+ slide_size: dict[str, int | float],
2502
+ top_level_issues: list[dict[str, Any]],
2503
+ slides: list[dict[str, Any]],
2504
+ ) -> dict[str, Any]:
2505
+ document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
2506
+ document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
2507
+ document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
2508
+ error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
2509
+ warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
2510
+ info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
2511
+ all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
2512
+ all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
2513
+ status = slide_status(all_errors, all_warnings)
2514
+ result: dict[str, Any] = {
2515
+ "schema_version": "2.0",
2516
+ "tool": "xml_text_overlap_lint",
336
2517
  "file": source_path,
337
- "slide_size": {"width": presentation["width"], "height": presentation["height"]},
338
- "summary": {"slide_count": len(slides), "error_count": error_count, "warning_count": warning_count},
2518
+ "slide_size": slide_size,
2519
+ "summary": {
2520
+ "slide_count": len(slides),
2521
+ "error_count": error_count,
2522
+ "warning_count": warning_count,
2523
+ "info_count": info_count,
2524
+ "status": status,
2525
+ "release_ready": error_count == 0,
2526
+ "screenshot_review_required": warning_count > 0,
2527
+ },
2528
+ "document": {
2529
+ "errors": document_errors,
2530
+ "warnings": document_warnings,
2531
+ "infos": document_infos,
2532
+ },
339
2533
  "slides": slides,
340
2534
  }
2535
+ if top_level_issues:
2536
+ result["issues"] = top_level_issues
2537
+ return result
2538
+
2539
+
2540
+ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
2541
+ root, xml_error = parse_xml_root(xml)
2542
+ if xml_error:
2543
+ issue = normalize_issue(xml_error, None, {})
2544
+ return build_result(
2545
+ source_path,
2546
+ {"width": 960, "height": 540},
2547
+ [issue],
2548
+ [],
2549
+ )
2550
+ if root is None:
2551
+ raise AssertionError("parse_xml_root must return a root or error")
2552
+
2553
+ namespace_issues = validate_sml_tag_prefixes(xml)
2554
+ root_name = xml_local_name(root.tag)
2555
+ sxsd_issues = validate_sxsd_document(xml, root)
2556
+ iconpark_issues = validate_iconpark_icon_types(root)
2557
+ top_level_issues = [
2558
+ normalize_issue(issue, None, {})
2559
+ for issue in [
2560
+ *namespace_issues,
2561
+ *[
2562
+ issue
2563
+ for issue in sxsd_issues
2564
+ if not is_slide_scoped_sxsd_issue(issue, root_name)
2565
+ ],
2566
+ *iconpark_issues,
2567
+ ]
2568
+ ]
2569
+ if any(issue["level"] == "error" for issue in top_level_issues):
2570
+ return build_result(
2571
+ source_path,
2572
+ {"width": 960, "height": 540},
2573
+ top_level_issues,
2574
+ [],
2575
+ )
2576
+
2577
+ presentation = parse_presentation(root)
2578
+ slide_roots = presentation["slide_roots"]
2579
+ slides: list[dict[str, Any]] = []
2580
+ for index, slide_xml in enumerate(presentation["slides"]):
2581
+ slide_number = index + 1
2582
+ slide_root = slide_roots[index]
2583
+ slide_sxsd_issues = [
2584
+ normalize_issue(issue, slide_number, {})
2585
+ for issue in validate_sxsd_document(slide_xml, slide_root)
2586
+ ]
2587
+ slide_sxsd_errors = [
2588
+ issue for issue in slide_sxsd_issues if issue["level"] == "error"
2589
+ ]
2590
+ if slide_sxsd_errors:
2591
+ slide_sxsd_warnings = [
2592
+ issue for issue in slide_sxsd_issues if issue["level"] == "warning"
2593
+ ]
2594
+ slides.append(
2595
+ {
2596
+ "slide_number": slide_number,
2597
+ "status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
2598
+ "element_count": 0,
2599
+ "errors": slide_sxsd_errors,
2600
+ "warnings": slide_sxsd_warnings,
2601
+ "infos": [],
2602
+ "issues": slide_sxsd_issues,
2603
+ }
2604
+ )
2605
+ continue
2606
+
2607
+ geometry = lint_slide(
2608
+ slide_xml,
2609
+ slide_number,
2610
+ presentation["width"],
2611
+ presentation["height"],
2612
+ )
2613
+ density_elements = extract_density_elements(slide_xml)
2614
+ extra_elements = [
2615
+ element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
2616
+ ]
2617
+ elements_by_id = {
2618
+ element["id"]: element for element in [*density_elements, *extra_elements]
2619
+ }
2620
+ # geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
2621
+ # decided with inside lint_slide; prefer them so measurement/related_objects stay consistent
2622
+ # with whatever actually triggered the issue, instead of density_elements' separate re-parse.
2623
+ elements_by_id.update({element["id"]: element for element in geometry["elements"]})
2624
+ extra_overflow_issues = detect_elements_out_of_canvas(
2625
+ extra_elements,
2626
+ presentation["width"],
2627
+ presentation["height"],
2628
+ )
2629
+ raw_issues = [
2630
+ *geometry["issues"],
2631
+ *extra_overflow_issues,
2632
+ *detect_blank_slide(
2633
+ density_elements,
2634
+ slide_number,
2635
+ presentation["width"],
2636
+ presentation["height"],
2637
+ ),
2638
+ *detect_sparse_container_content(
2639
+ density_elements,
2640
+ slide_number,
2641
+ presentation["width"],
2642
+ presentation["height"],
2643
+ ),
2644
+ *detect_sparse_slide_content(
2645
+ density_elements,
2646
+ slide_number,
2647
+ presentation["width"],
2648
+ presentation["height"],
2649
+ ),
2650
+ ]
2651
+ issues = [
2652
+ *slide_sxsd_issues,
2653
+ *[
2654
+ normalize_issue(issue, slide_number, elements_by_id)
2655
+ for issue in raw_issues
2656
+ ],
2657
+ ]
2658
+ errors = [issue for issue in issues if issue["level"] == "error"]
2659
+ warnings = [issue for issue in issues if issue["level"] == "warning"]
2660
+ infos = [issue for issue in issues if issue["level"] == "info"]
2661
+ slides.append(
2662
+ {
2663
+ "slide_number": slide_number,
2664
+ "status": slide_status(errors, warnings),
2665
+ "element_count": len(elements_by_id),
2666
+ "errors": errors,
2667
+ "warnings": warnings,
2668
+ "infos": infos,
2669
+ "issues": issues,
2670
+ }
2671
+ )
2672
+
2673
+ return build_result(
2674
+ source_path,
2675
+ {"width": presentation["width"], "height": presentation["height"]},
2676
+ top_level_issues,
2677
+ slides,
2678
+ )
341
2679
 
342
2680
 
343
2681
  def print_usage() -> None:
@@ -362,6 +2700,6 @@ def run_cli(argv: list[str] | None = None) -> None:
362
2700
  if __name__ == "__main__":
363
2701
  try:
364
2702
  run_cli()
365
- except XmlTextOverlapLintError as error:
2703
+ except XmlLayoutLintError as error:
366
2704
  print(f"xml-text-overlap-lint error: {error}", file=sys.stderr)
367
2705
  raise SystemExit(1) from error