@amaster.ai/pi-lark 0.1.6 → 0.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (266) hide show
  1. package/README.md +5 -1
  2. package/dist/config.d.ts +1 -1
  3. package/dist/config.d.ts.map +1 -1
  4. package/dist/config.js +2 -2
  5. package/dist/config.js.map +1 -1
  6. package/dist/index.d.ts.map +1 -1
  7. package/dist/index.js +2 -1
  8. package/dist/index.js.map +1 -1
  9. package/package.json +3 -3
  10. package/skills/lark-apps/SKILL.md +59 -14
  11. package/skills/lark-apps/creative-design/agents/assets/vision-probe.png +0 -0
  12. package/skills/lark-apps/creative-design/agents/fork-verifier-agent.md +71 -0
  13. package/skills/lark-apps/creative-design/agents/vision-probe-agent.md +41 -0
  14. package/skills/lark-apps/creative-design/assets/index.html +27 -0
  15. package/skills/lark-apps/creative-design/creative-design.md +239 -0
  16. package/skills/lark-apps/creative-design/references/aily.md +39 -0
  17. package/skills/lark-apps/creative-design/references/animated-video.md +34 -0
  18. package/skills/lark-apps/creative-design/references/charts.md +165 -0
  19. package/skills/lark-apps/creative-design/references/claude.md +36 -0
  20. package/skills/lark-apps/creative-design/references/codex.md +32 -0
  21. package/skills/lark-apps/creative-design/references/data-report.md +108 -0
  22. package/skills/lark-apps/creative-design/references/frontend-design.md +71 -0
  23. package/skills/lark-apps/creative-design/references/hi-fi-design.md +32 -0
  24. package/skills/lark-apps/creative-design/references/interactive-prototype.md +24 -0
  25. package/skills/lark-apps/creative-design/references/make-a-deck.md +133 -0
  26. package/skills/lark-apps/creative-design/references/visual-exposure.md +82 -0
  27. package/skills/lark-apps/creative-design/references/wireframe.md +14 -0
  28. package/skills/lark-apps/creative-design/starter-components/android-frame.jsx +188 -0
  29. package/skills/lark-apps/creative-design/starter-components/animations.jsx +773 -0
  30. package/skills/lark-apps/creative-design/starter-components/browser-window.jsx +122 -0
  31. package/skills/lark-apps/creative-design/starter-components/deck-stage.js +2483 -0
  32. package/skills/lark-apps/creative-design/starter-components/design-canvas.jsx +1432 -0
  33. package/skills/lark-apps/creative-design/starter-components/ios-frame.jsx +270 -0
  34. package/skills/lark-apps/creative-design/starter-components/macos-window.jsx +197 -0
  35. package/skills/lark-apps/creative-design/starter-components/tweaks-panel.jsx +752 -0
  36. package/skills/lark-apps/references/lark-apps-automation.md +80 -2
  37. package/skills/lark-apps/references/lark-apps-cache.md +61 -0
  38. package/skills/lark-apps/references/lark-apps-cloud-dev.md +5 -5
  39. package/skills/lark-apps/references/lark-apps-create.md +6 -4
  40. package/skills/lark-apps/references/lark-apps-db.md +1 -1
  41. package/skills/lark-apps/references/lark-apps-env-pull.md +1 -1
  42. package/skills/lark-apps/references/lark-apps-file.md +2 -2
  43. package/skills/lark-apps/references/lark-apps-get.md +1 -1
  44. package/skills/lark-apps/references/lark-apps-git-credential.md +1 -1
  45. package/skills/lark-apps/references/lark-apps-html-publish.md +4 -8
  46. package/skills/lark-apps/references/lark-apps-init.md +1 -1
  47. package/skills/lark-apps/references/lark-apps-list.md +2 -2
  48. package/skills/lark-apps/references/lark-apps-local-dev.md +80 -11
  49. package/skills/lark-apps/references/lark-apps-openapi-key.md +1 -1
  50. package/skills/lark-apps/references/lark-apps-release-create.md +2 -2
  51. package/skills/lark-apps/references/lark-apps-release-get.md +3 -3
  52. package/skills/lark-base/SKILL.md +34 -19
  53. package/skills/lark-base/references/lark-base-cell-value.md +3 -3
  54. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +17 -1
  55. package/skills/lark-base/references/lark-base-dashboard.md +17 -4
  56. package/skills/lark-base/references/lark-base-data-query-guide.md +8 -0
  57. package/skills/lark-base/references/lark-base-data-query.md +11 -4
  58. package/skills/lark-base/references/lark-base-field-create.md +21 -6
  59. package/skills/lark-base/references/lark-base-field-json.md +9 -6
  60. package/skills/lark-base/references/lark-base-field-update.md +17 -1
  61. package/skills/lark-base/references/lark-base-filter-condition.md +179 -0
  62. package/skills/lark-base/references/lark-base-form-questions-create.md +40 -7
  63. package/skills/lark-base/references/lark-base-form-questions-update.md +73 -20
  64. package/skills/lark-base/references/lark-base-form-submit.md +16 -7
  65. package/skills/lark-base/references/lark-base-record-batch-create.md +12 -10
  66. package/skills/lark-base/references/lark-base-record-batch-update.md +11 -9
  67. package/skills/lark-base/references/lark-base-record-upsert.md +1 -1
  68. package/skills/lark-base/references/lark-base-role-guide.md +11 -0
  69. package/skills/lark-base/references/lark-base-view-set-filter.md +11 -137
  70. package/skills/lark-base/references/role-config.md +31 -5
  71. package/skills/lark-calendar/SKILL.md +14 -8
  72. package/skills/lark-calendar/references/lark-calendar-create.md +6 -5
  73. package/skills/lark-calendar/references/lark-calendar-recurring.md +1 -0
  74. package/skills/lark-calendar/references/lark-calendar-room-find.md +2 -1
  75. package/skills/lark-calendar/references/lark-calendar-schedule-clear-time.md +1 -0
  76. package/skills/lark-calendar/references/lark-calendar-suggestion.md +1 -1
  77. package/skills/lark-calendar/references/lark-calendar-update.md +10 -4
  78. package/skills/lark-contact/SKILL.md +19 -3
  79. package/skills/lark-contact/references/lark-contact-search-bot.md +60 -0
  80. package/skills/lark-doc/SKILL.md +26 -61
  81. package/skills/lark-doc/references/genres/business-analysis.md +30 -0
  82. package/skills/lark-doc/references/genres/data-report.md +32 -0
  83. package/skills/lark-doc/references/genres/email.md +38 -0
  84. package/skills/lark-doc/references/genres/execution-plan.md +27 -0
  85. package/skills/lark-doc/references/genres/formal-doc.md +37 -0
  86. package/skills/lark-doc/references/genres/meeting-minutes.md +24 -0
  87. package/skills/lark-doc/references/genres/memo-brief.md +25 -0
  88. package/skills/lark-doc/references/genres/official-redhead.md +73 -0
  89. package/skills/lark-doc/references/genres/prd.md +26 -0
  90. package/skills/lark-doc/references/genres/proposal.md +24 -0
  91. package/skills/lark-doc/references/genres/research-report.md +32 -0
  92. package/skills/lark-doc/references/genres/retrospective.md +25 -0
  93. package/skills/lark-doc/references/genres/route-consumer.md +37 -0
  94. package/skills/lark-doc/references/genres/route-creative.md +36 -0
  95. package/skills/lark-doc/references/genres/route-knowledge.md +39 -0
  96. package/skills/lark-doc/references/genres/route-marketing.md +40 -0
  97. package/skills/lark-doc/references/genres/route-media.md +36 -0
  98. package/skills/lark-doc/references/genres/route-opinion.md +38 -0
  99. package/skills/lark-doc/references/genres/route-personal-brand.md +36 -0
  100. package/skills/lark-doc/references/genres/route-platform.md +9 -0
  101. package/skills/lark-doc/references/genres/route-report.md +10 -0
  102. package/skills/lark-doc/references/genres/route-workplace.md +17 -0
  103. package/skills/lark-doc/references/genres/sop-tutorial.md +41 -0
  104. package/skills/lark-doc/references/genres/technical-doc.md +39 -0
  105. package/skills/lark-doc/references/genres/wechat.md +39 -0
  106. package/skills/lark-doc/references/genres/weekly-report.md +24 -0
  107. package/skills/lark-doc/references/genres/white-paper.md +32 -0
  108. package/skills/lark-doc/references/genres/xiaohongshu.md +38 -0
  109. package/skills/lark-doc/references/lark-doc-create-workflow.md +121 -0
  110. package/skills/lark-doc/references/lark-doc-create.md +22 -48
  111. package/skills/lark-doc/references/lark-doc-fetch.md +84 -93
  112. package/skills/lark-doc/references/lark-doc-history.md +16 -15
  113. package/skills/lark-doc/references/lark-doc-md.md +5 -1
  114. package/skills/lark-doc/references/lark-doc-media-download.md +2 -1
  115. package/skills/lark-doc/references/lark-doc-script.md +76 -0
  116. package/skills/lark-doc/references/lark-doc-update.md +70 -222
  117. package/skills/lark-doc/references/lark-doc-whiteboard.md +14 -17
  118. package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +46 -0
  119. package/skills/lark-doc/references/lark-doc-xml.md +38 -166
  120. package/skills/lark-drive/SKILL.md +32 -50
  121. package/skills/lark-drive/references/lark-drive-add-comment.md +2 -4
  122. package/skills/lark-drive/references/lark-drive-add-reply.md +47 -0
  123. package/skills/lark-drive/references/lark-drive-apply-permission.md +3 -3
  124. package/skills/lark-drive/references/lark-drive-batch-query-comments.md +46 -0
  125. package/skills/lark-drive/references/lark-drive-comment-content.md +50 -0
  126. package/skills/lark-drive/references/lark-drive-comment-location.md +9 -15
  127. package/skills/lark-drive/references/lark-drive-copy.md +87 -0
  128. package/skills/lark-drive/references/lark-drive-delete-reply.md +48 -0
  129. package/skills/lark-drive/references/lark-drive-download.md +6 -1
  130. package/skills/lark-drive/references/lark-drive-export.md +3 -0
  131. package/skills/lark-drive/references/lark-drive-list-comments.md +25 -68
  132. package/skills/lark-drive/references/lark-drive-list-replies.md +54 -0
  133. package/skills/lark-drive/references/lark-drive-member-add.md +2 -2
  134. package/skills/lark-drive/references/lark-drive-member-list.md +65 -0
  135. package/skills/lark-drive/references/lark-drive-permission-get-setting.md +48 -0
  136. package/skills/lark-drive/references/lark-drive-preview.md +11 -1
  137. package/skills/lark-drive/references/lark-drive-react-reply.md +51 -0
  138. package/skills/lark-drive/references/lark-drive-reactions.md +27 -25
  139. package/skills/lark-drive/references/lark-drive-resolve-comment.md +45 -0
  140. package/skills/lark-drive/references/lark-drive-restore-comment.md +46 -0
  141. package/skills/lark-drive/references/lark-drive-search.md +7 -1
  142. package/skills/lark-drive/references/lark-drive-secure-label.md +1 -1
  143. package/skills/lark-drive/references/lark-drive-task-result.md +3 -0
  144. package/skills/lark-drive/references/lark-drive-update-reply.md +46 -0
  145. package/skills/lark-drive/references/lark-drive-update-title.md +78 -0
  146. package/skills/lark-drive/references/lark-drive-upload.md +1 -0
  147. package/skills/lark-drive/references/lark-drive-workflow-permission-governance-commands.md +38 -8
  148. package/skills/lark-drive/references/lark-drive-workflow-permission-governance-outputs.md +10 -10
  149. package/skills/lark-drive/references/lark-drive-workflow-permission-governance.md +22 -20
  150. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-execute.md +273 -0
  151. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-recall.md +202 -0
  152. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-resolve-verify.md +231 -0
  153. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-review-plan.md +248 -0
  154. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-setup.md +174 -0
  155. package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector.md +202 -0
  156. package/skills/lark-drive/references/lark-drive-workflow.md +5 -4
  157. package/skills/lark-event/SKILL.md +8 -4
  158. package/skills/lark-event/references/lark-event-application.md +38 -0
  159. package/skills/lark-event/references/lark-event-vc.md +8 -2
  160. package/skills/lark-im/SKILL.md +9 -9
  161. package/skills/lark-im/references/card/card-2.0-schema.md +1 -1
  162. package/skills/lark-im/references/card/lark-im-card-style.md +4 -4
  163. package/skills/lark-im/references/card/resource/icons.md +14 -0
  164. package/skills/lark-im/references/lark-im-chat-list.md +9 -2
  165. package/skills/lark-im/references/lark-im-chat-members-list.md +7 -4
  166. package/skills/lark-im/references/lark-im-chat-messages-list.md +10 -3
  167. package/skills/lark-im/references/lark-im-chat-search.md +9 -2
  168. package/skills/lark-im/references/lark-im-feed-group-list-item.md +2 -2
  169. package/skills/lark-im/references/lark-im-feed-group-list.md +2 -2
  170. package/skills/lark-im/references/lark-im-feed-shortcut-list.md +1 -1
  171. package/skills/lark-im/references/lark-im-flag-list.md +9 -8
  172. package/skills/lark-im/references/lark-im-message-enrichment.md +1 -1
  173. package/skills/lark-im/references/lark-im-messages-resources-download.md +19 -25
  174. package/skills/lark-im/references/lark-im-messages-search.md +4 -5
  175. package/skills/lark-im/references/lark-im-threads-messages-list.md +8 -4
  176. package/skills/lark-mail/references/lark-mail-triage.md +19 -4
  177. package/skills/lark-minutes/SKILL.md +1 -1
  178. package/skills/lark-minutes/references/lark-minutes-search.md +6 -7
  179. package/skills/lark-okr/SKILL.md +71 -26
  180. package/skills/lark-okr/references/lark-okr-batch-create.md +19 -18
  181. package/skills/lark-okr/references/lark-okr-create.md +173 -0
  182. package/skills/lark-okr/references/lark-okr-cycle-list.md +17 -7
  183. package/skills/lark-okr/references/lark-okr-entities.md +1 -0
  184. package/skills/lark-okr/references/lark-okr-indicator-update.md +3 -1
  185. package/skills/lark-okr/references/lark-okr-indicators.md +61 -12
  186. package/skills/lark-okr/references/lark-okr-progress-list.md +21 -9
  187. package/skills/lark-shared/SKILL.md +3 -3
  188. package/skills/lark-sheets/SKILL.md +83 -82
  189. package/skills/lark-sheets/references/lark-sheets-batch-update.md +13 -58
  190. package/skills/lark-sheets/references/lark-sheets-chart.md +2 -1
  191. package/skills/lark-sheets/references/lark-sheets-conditional-format.md +1 -1
  192. package/skills/lark-sheets/references/lark-sheets-range-operations.md +5 -5
  193. package/skills/lark-sheets/references/lark-sheets-read-data.md +80 -6
  194. package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +21 -10
  195. package/skills/lark-sheets/references/lark-sheets-styles-put.md +93 -0
  196. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +2 -2
  197. package/skills/lark-sheets/references/lark-sheets-workbook.md +4 -3
  198. package/skills/lark-sheets/references/lark-sheets-write-cells.md +40 -12
  199. package/skills/lark-sheets/scripts/lark_detect_subtables.py +593 -0
  200. package/skills/lark-sheets/scripts/lark_inspect_workbook.py +188 -0
  201. package/skills/lark-sheets/scripts/lark_profile_table.py +614 -0
  202. package/skills/lark-sheets/scripts/lark_sheet_range.py +176 -0
  203. package/skills/lark-sheets/scripts/lark_sheet_read_cli.py +184 -0
  204. package/skills/lark-sheets/scripts/sheets_df.py +21 -3
  205. package/skills/lark-slides/SKILL.md +134 -104
  206. package/skills/lark-slides/references/asset-planning.md +6 -4
  207. package/skills/lark-slides/references/iconpark.md +2 -2
  208. package/skills/lark-slides/references/lark-slides-add-slide.md +92 -0
  209. package/skills/lark-slides/references/lark-slides-create.md +86 -66
  210. package/skills/lark-slides/references/lark-slides-delete-slide.md +65 -0
  211. package/skills/lark-slides/references/lark-slides-edit-workflows.md +6 -7
  212. package/skills/lark-slides/references/lark-slides-history.md +132 -0
  213. package/skills/lark-slides/references/lark-slides-media-upload.md +4 -27
  214. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +7 -11
  215. package/skills/lark-slides/references/lark-slides-replace-slide.md +22 -4
  216. package/skills/lark-slides/references/lark-slides-screenshot.md +33 -15
  217. package/skills/lark-slides/references/lark-slides-update-slide.md +146 -0
  218. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +2 -2
  219. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +2 -3
  220. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +90 -32
  221. package/skills/lark-slides/references/planning-layer.md +11 -10
  222. package/skills/lark-slides/references/slides_chart_demo.xml +1415 -1
  223. package/skills/lark-slides/references/slides_xml_schema_definition.xml +539 -79
  224. package/skills/lark-slides/references/troubleshooting.md +26 -9
  225. package/skills/lark-slides/references/validation-checklist.md +55 -18
  226. package/skills/lark-slides/references/visual-planning.md +25 -22
  227. package/skills/lark-slides/references/xml-schema-quick-ref.md +299 -51
  228. package/skills/lark-slides/scripts/sxsd_validator.py +1052 -0
  229. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +1964 -195
  230. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +4051 -501
  231. package/skills/lark-task/SKILL.md +7 -0
  232. package/skills/lark-task/references/lark-task-complete.md +6 -2
  233. package/skills/lark-task/references/lark-task-create.md +9 -0
  234. package/skills/lark-task/references/lark-task-update.md +6 -2
  235. package/skills/lark-whiteboard/SKILL.md +21 -13
  236. package/skills/lark-whiteboard/elements/layout.md +1 -1
  237. package/skills/lark-whiteboard/elements/schema.md +2 -2
  238. package/skills/lark-whiteboard/references/{lark-whiteboard-query.md → lark-whiteboard-export.md} +17 -16
  239. package/skills/lark-whiteboard/references/lark-whiteboard-update.md +7 -7
  240. package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +23 -31
  241. package/skills/lark-whiteboard/routes/dsl.md +11 -5
  242. package/skills/lark-whiteboard/routes/mermaid.md +3 -3
  243. package/skills/lark-whiteboard/routes/svg-edit.md +9 -6
  244. package/skills/lark-whiteboard/routes/svg.md +14 -7
  245. package/skills/lark-whiteboard/scenes/bar-chart.md +1 -1
  246. package/skills/lark-whiteboard/scenes/fishbone.md +1 -1
  247. package/skills/lark-whiteboard/scenes/flywheel.md +1 -1
  248. package/skills/lark-whiteboard/scenes/line-chart.md +1 -1
  249. package/skills/lark-whiteboard/scenes/mention.md +71 -0
  250. package/skills/lark-whiteboard/scenes/treemap.md +1 -1
  251. package/skills/lark-wiki/SKILL.md +6 -3
  252. package/skills/lark-wiki/references/lark-wiki-delete-space.md +6 -3
  253. package/skills/lark-doc/references/lark-doc-word-stat.md +0 -93
  254. package/skills/lark-doc/references/style/lark-doc-create-workflow.md +0 -47
  255. package/skills/lark-doc/references/style/lark-doc-style.md +0 -68
  256. package/skills/lark-doc/references/style/lark-doc-update-workflow.md +0 -48
  257. package/skills/lark-doc/scripts/doc_word_stat.py +0 -1243
  258. package/skills/lark-drive/references/lark-drive-comments-guide.md +0 -80
  259. package/skills/lark-slides/references/examples.md +0 -91
  260. package/skills/lark-slides/references/lark-slides-replace-pages.md +0 -95
  261. package/skills/lark-slides/references/lark-slides-whiteboard.md +0 -331
  262. package/skills/lark-slides/references/lark-slides-xml-get.md +0 -100
  263. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +0 -125
  264. package/skills/lark-slides/references/slide-templates.md +0 -201
  265. package/skills/lark-slides/references/slides_demo.xml +0 -226
  266. package/skills/lark-slides/references/xml-format-guide.md +0 -433
@@ -1,9 +1,11 @@
1
1
  #!/usr/bin/env python3
2
2
  # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
3
  # SPDX-License-Identifier: MIT
4
+ """Validate Slides XML structure and page layout through one release gate."""
4
5
 
5
6
  from __future__ import annotations
6
7
 
8
+ import copy
7
9
  import json
8
10
  import math
9
11
  import re
@@ -15,11 +17,13 @@ from difflib import SequenceMatcher, get_close_matches
15
17
  from pathlib import Path
16
18
  from typing import Any
17
19
 
20
+ import sxsd_validator
21
+
18
22
 
19
23
  XS_NS = "{http://www.w3.org/2001/XMLSchema}"
20
24
  XML_NS = "{http://www.w3.org/XML/1998/namespace}"
21
25
  SVG_NS = "{http://www.w3.org/2000/svg}"
22
- SML_NAMESPACE = "http://www.larkoffice.com/sml/2.0"
26
+ SML_NAMESPACE = "https://www.larkoffice.com/sml/2.0"
23
27
  SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
24
28
  ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
25
29
  SXSD_TAG_ALIASES = {
@@ -38,18 +42,47 @@ SXSD_ATTR_ALIASES = {
38
42
  "fontColor": "color",
39
43
  }
40
44
  SERVER_FILLED_SXSD_ATTRS = {"id"}
45
+ ROUNDTRIP_SXSD_ATTRS = {
46
+ ("chart", "updated"),
47
+ ("chartData", "isStaticData"),
48
+ }
49
+ # Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
50
+ # it is server-emitted and absent from the write schema, so it must not block page linting.
51
+ ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
41
52
  DEFAULT_TABLE_COLUMN_WIDTH = 110
42
53
  DEFAULT_TABLE_ROW_HEIGHT = 37
54
+ DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
55
+ TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
56
+ TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
57
+ SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
58
+ CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
59
+ HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
60
+ DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
61
+ GHOST_TEXT_MIN_FONT_SIZE = 96
62
+ GHOST_TEXT_MAX_ALPHA = 0.5
63
+ GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
64
+ GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
65
+ # A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
66
+ # glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
67
+ # padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
68
+ LINE_TEXT_GRAZE_MIN_PX = 2.0
69
+ LINE_TEXT_GRAZE_FONT_RATIO = 0.12
70
+ # A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
71
+ LINE_MIN_VISIBLE_ALPHA = 0.08
72
+ # Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
73
+ # visible defect; keep this well under 1px so real overflow is still always caught.
74
+ CANVAS_OVERFLOW_TOLERANCE = 0.5
75
+ XML_PATH_HINT_PREFIX = "Locate via related_objects[].xml_path."
43
76
  _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
44
77
  _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
45
78
 
46
79
 
47
- class XmlTextOverlapLintError(Exception):
80
+ class XmlLayoutLintError(Exception):
48
81
  pass
49
82
 
50
83
 
51
84
  def fail(message: str) -> None:
52
- raise XmlTextOverlapLintError(message)
85
+ raise XmlLayoutLintError(message)
53
86
 
54
87
 
55
88
  def read_file(file_path: str | Path) -> str:
@@ -62,7 +95,7 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
62
95
  while index < len(argv):
63
96
  token = argv[index]
64
97
  if not token.startswith("--"):
65
- fail(f"unexpected argument: {token}")
98
+ fail(f"unexpected argument: {token}, need --input")
66
99
  key = token[2:]
67
100
  next_token = argv[index + 1] if index + 1 < len(argv) else None
68
101
  if next_token is None or next_token.startswith("--"):
@@ -75,8 +108,12 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
75
108
 
76
109
 
77
110
  def extract_attribute(tag_source: str, name: str) -> str | None:
78
- match = re.search(fr'{re.escape(name)}="([^"]+)"', tag_source)
79
- return match.group(1) if match else None
111
+ match = re.search(
112
+ fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
113
+ )
114
+ if not match:
115
+ return None
116
+ return match.group(1) if match.group(1) is not None else match.group(2)
80
117
 
81
118
 
82
119
  def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
@@ -90,6 +127,52 @@ def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
90
127
  return int(value) if value.is_integer() else value
91
128
 
92
129
 
130
+ def extract_bool_attribute(tag_source: str, name: str) -> bool:
131
+ value = extract_attribute(tag_source, name)
132
+ return value in {"true", "1", "yes"}
133
+
134
+
135
+ def extract_color_alpha(color: str | None) -> int | float | None:
136
+ if color is None:
137
+ return None
138
+ normalized = re.sub(r"\s+", "", color).lower()
139
+ if normalized == "transparent":
140
+ return 0
141
+ rgba_match = re.fullmatch(
142
+ r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
143
+ normalized,
144
+ )
145
+ if rgba_match is None:
146
+ return None
147
+ try:
148
+ alpha = float(rgba_match.group(1))
149
+ except ValueError:
150
+ return None
151
+ return int(alpha) if alpha.is_integer() else alpha
152
+
153
+
154
+ def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
155
+ base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
156
+ color_alpha = extract_color_alpha(text_color)
157
+ if not isinstance(color_alpha, (int, float)):
158
+ return base_alpha
159
+ return base_alpha * color_alpha
160
+
161
+
162
+ def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
163
+ for tag_name in style_tags:
164
+ if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
165
+ return True
166
+ return False
167
+
168
+
169
+ def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
170
+ for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
171
+ if extract_bool_attribute(attrs, attr_name):
172
+ return True
173
+ return False
174
+
175
+
93
176
  def sum_sizes(sizes: list[int | float]) -> int | float:
94
177
  return sum(sizes)
95
178
 
@@ -168,8 +251,10 @@ def solve_weighted_min_layout(
168
251
  return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
169
252
 
170
253
 
171
- def strip_xml(value: str) -> str:
254
+ def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
172
255
  stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
256
+ if preserve_line_breaks:
257
+ stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
173
258
  stripped = re.sub(r"<[^>]+>", " ", stripped)
174
259
  stripped = stripped.replace("&nbsp;", " ")
175
260
  stripped = stripped.replace("&amp;", "&")
@@ -177,32 +262,55 @@ def strip_xml(value: str) -> str:
177
262
  stripped = stripped.replace("&gt;", ">")
178
263
  stripped = stripped.replace("&quot;", '"')
179
264
  stripped = stripped.replace("&#39;", "'")
265
+ if preserve_line_breaks:
266
+ return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
180
267
  return re.sub(r"\s+", " ", stripped).strip()
181
268
 
182
269
 
183
270
  def strip_xml_paragraphs(value: str) -> str:
184
271
  paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
185
272
  if paragraphs:
186
- return "\n".join(strip_xml(paragraph) for paragraph in paragraphs)
187
- return strip_xml(value)
273
+ return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
274
+ return strip_xml(value, preserve_line_breaks=True)
188
275
 
189
276
 
190
- def xml_local_name(tag: str) -> str:
191
- return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
277
+ def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
278
+ paragraphs = []
279
+ for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
280
+ paragraphs.append(
281
+ {
282
+ "text": strip_xml(body, preserve_line_breaks=True),
283
+ "fontSize": extract_max_span_font_size(body, default_font_size),
284
+ "textAlign": extract_attribute(attrs, "textAlign"),
285
+ "lineSpacing": extract_attribute(attrs, "lineSpacing"),
286
+ "beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
287
+ "afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
288
+ "letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
289
+ }
290
+ )
291
+ return paragraphs
192
292
 
193
293
 
194
- def xml_namespace(tag: str) -> str | None:
195
- return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
294
+ def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
295
+ font_sizes = [
296
+ font_size
297
+ for attrs in re.findall(r"<span\b([^>]*)>", value)
298
+ if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
299
+ ]
300
+ return max([default_font_size, *font_sizes])
196
301
 
197
302
 
198
- def strip_xsd_prefix(value: str | None) -> str | None:
199
- if value is None:
200
- return None
201
- return value.rsplit(":", 1)[-1]
303
+ def extract_tag_attributes(value: str, tag: str) -> str:
304
+ match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
305
+ return match.group(1) if match else ""
306
+
307
+
308
+ def xml_local_name(tag: str) -> str:
309
+ return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
202
310
 
203
311
 
204
- def iter_direct_xsd_children(element: ET.Element, local_name: str) -> list[ET.Element]:
205
- return [child for child in element if child.tag == f"{XS_NS}{local_name}"]
312
+ def xml_namespace(tag: str) -> str | None:
313
+ return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
206
314
 
207
315
 
208
316
  def load_sxsd_tag_attributes() -> dict[str, set[str]]:
@@ -210,62 +318,8 @@ def load_sxsd_tag_attributes() -> dict[str, set[str]]:
210
318
  if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
211
319
  return _SXSD_TAG_ATTRIBUTES_CACHE
212
320
 
213
- schema_root = ET.parse(SXSD_SCHEMA_PATH).getroot()
214
- named_complex_types = {
215
- complex_type.attrib["name"]: complex_type
216
- for complex_type in schema_root.findall(f"{XS_NS}complexType")
217
- if complex_type.attrib.get("name")
218
- }
219
- resolving: set[str] = set()
220
-
221
- def attributes_for_complex_type(complex_type: ET.Element) -> set[str]:
222
- attrs: set[str] = {
223
- attribute.attrib["name"]
224
- for attribute in iter_direct_xsd_children(complex_type, "attribute")
225
- if attribute.attrib.get("name")
226
- }
227
- for content_name in ("simpleContent", "complexContent"):
228
- for complex_content in iter_direct_xsd_children(complex_type, content_name):
229
- for extension in iter_direct_xsd_children(complex_content, "extension"):
230
- base_type = strip_xsd_prefix(extension.attrib.get("base"))
231
- if base_type:
232
- attrs.update(attributes_for_type(base_type))
233
- attrs.update(
234
- attribute.attrib["name"]
235
- for attribute in iter_direct_xsd_children(extension, "attribute")
236
- if attribute.attrib.get("name")
237
- )
238
- return attrs
239
-
240
- def attributes_for_type(type_name: str) -> set[str]:
241
- if type_name in resolving:
242
- return set()
243
- complex_type = named_complex_types.get(type_name)
244
- if complex_type is None:
245
- return set()
246
- resolving.add(type_name)
247
- try:
248
- return attributes_for_complex_type(complex_type)
249
- finally:
250
- resolving.remove(type_name)
251
-
252
- tag_attributes: dict[str, set[str]] = {}
253
- for element in schema_root.iter(f"{XS_NS}element"):
254
- tag_name = element.attrib.get("name")
255
- if not tag_name:
256
- continue
257
-
258
- attrs: set[str] = set()
259
- type_name = strip_xsd_prefix(element.attrib.get("type"))
260
- if type_name:
261
- attrs.update(attributes_for_type(type_name))
262
- for complex_type in iter_direct_xsd_children(element, "complexType"):
263
- attrs.update(attributes_for_complex_type(complex_type))
264
-
265
- tag_attributes.setdefault(tag_name, set()).update(attrs)
266
-
267
- _SXSD_TAG_ATTRIBUTES_CACHE = tag_attributes
268
- return tag_attributes
321
+ _SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
322
+ return _SXSD_TAG_ATTRIBUTES_CACHE
269
323
 
270
324
 
271
325
  def load_iconpark_icon_types() -> set[str]:
@@ -295,20 +349,26 @@ def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
295
349
  if alias:
296
350
  return f"Use {alias} instead of <{tag_name}>."
297
351
  if tag_name == "svg":
298
- return 'Inside <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
352
+ return 'Inside <embed> or <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
299
353
  close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
300
354
  if close_matches:
301
355
  return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
302
356
  return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
303
357
 
304
358
 
305
- def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
359
+ def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
306
360
  alias = SXSD_ATTR_ALIASES.get(attr_name)
307
361
  if alias and alias in allowed_attrs:
308
- return f'Use "{alias}" on <{tag_name}> instead of "{attr_name}".'
309
- close_matches = get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
310
- if close_matches:
311
- return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in close_matches) + "?"
362
+ return [alias]
363
+ return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
364
+
365
+
366
+ def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
367
+ suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
368
+ if suggestions:
369
+ if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
370
+ return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
371
+ return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
312
372
  allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
313
373
  if len(allowed_attrs) > 8:
314
374
  allowed_summary += ", ..."
@@ -316,17 +376,40 @@ def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str])
316
376
 
317
377
 
318
378
  def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
319
- return "whiteboard" in ancestors and xml_namespace(element.tag) == SVG_NS
379
+ return ("whiteboard" in ancestors or "embed" in ancestors) and xml_namespace(element.tag) == SVG_NS
380
+
381
+
382
+ def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
383
+ return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
320
384
 
321
385
 
322
- def should_skip_sxsd_attribute(attr_name: str) -> bool:
323
- return attr_name in SERVER_FILLED_SXSD_ATTRS
386
+ def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
387
+ return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
324
388
 
325
389
 
326
- def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
390
+ def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
391
+ sanitized_root = copy.deepcopy(root)
392
+
393
+ def sanitize(element: ET.Element) -> None:
394
+ tag_name = xml_local_name(element.tag)
395
+ for raw_attr_name in list(element.attrib):
396
+ if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
397
+ del element.attrib[raw_attr_name]
398
+ for child in list(element):
399
+ if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
400
+ element.remove(child)
401
+ continue
402
+ sanitize(child)
403
+
404
+ sanitize(sanitized_root)
405
+ return sanitized_root
406
+
407
+
408
+ def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
327
409
  tag_attributes = load_sxsd_tag_attributes()
328
410
  supported_tags = set(tag_attributes)
329
411
  issues: list[dict[str, Any]] = []
412
+ suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
330
413
 
331
414
  def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
332
415
  if should_skip_sxsd_subtree(element, ancestors):
@@ -334,6 +417,9 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
334
417
 
335
418
  tag_name = xml_local_name(element.tag)
336
419
  current_path = f"{path}/{tag_name}" if path else tag_name
420
+ parent_name = ancestors[-1] if ancestors else None
421
+ if should_skip_sxsd_tag(parent_name, tag_name):
422
+ return
337
423
  if tag_name not in supported_tags:
338
424
  issues.append(
339
425
  {
@@ -352,10 +438,15 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
352
438
  if raw_attr_name.startswith(XML_NS):
353
439
  continue
354
440
  attr_name = xml_local_name(raw_attr_name)
355
- if should_skip_sxsd_attribute(attr_name):
441
+ if should_skip_sxsd_attribute(tag_name, attr_name):
356
442
  continue
357
443
  if attr_name in allowed_attrs:
358
444
  continue
445
+ suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
446
+ if suggestions:
447
+ suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
448
+ set(suggestions)
449
+ )
359
450
  issues.append(
360
451
  {
361
452
  "level": "error",
@@ -371,6 +462,123 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
371
462
  for child in element:
372
463
  visit(child, [*ancestors, tag_name], current_path)
373
464
 
465
+ visit(root, [], "")
466
+ existing = {
467
+ (issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
468
+ for issue in issues
469
+ }
470
+ unsupported_tag_locations = {
471
+ (issue.get("path"), issue.get("tag"))
472
+ for issue in issues
473
+ if issue.get("code") == "sxsd_unsupported_tag"
474
+ }
475
+ schema_issues = _validate_sxsd_schema_constraints(xml, root)
476
+ missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
477
+ for schema_issue in schema_issues:
478
+ if schema_issue.get("code") != "sxsd_missing_required_attr":
479
+ continue
480
+ location = (schema_issue.get("path"), schema_issue.get("tag"))
481
+ missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
482
+
483
+ suggested_attrs: set[tuple[str, str, str]] = set()
484
+ for location, candidate_groups in suggested_attr_candidates.items():
485
+ missing_attrs = missing_attrs_by_location.get(location, set())
486
+ for candidates in candidate_groups:
487
+ matching_missing_attrs = candidates & missing_attrs
488
+ if len(matching_missing_attrs) == 1:
489
+ suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
490
+
491
+ for schema_issue in schema_issues:
492
+ if schema_issue.get("code") == "sxsd_unexpected_child" and (
493
+ schema_issue.get("path"),
494
+ schema_issue.get("tag"),
495
+ ) in unsupported_tag_locations:
496
+ continue
497
+ if schema_issue.get("code") == "sxsd_missing_required_attr" and (
498
+ schema_issue.get("path"),
499
+ schema_issue.get("tag"),
500
+ schema_issue.get("attr"),
501
+ ) in suggested_attrs:
502
+ continue
503
+ key = (
504
+ schema_issue.get("code"),
505
+ schema_issue.get("path"),
506
+ schema_issue.get("tag"),
507
+ schema_issue.get("attr"),
508
+ )
509
+ if key not in existing:
510
+ issues.append(schema_issue)
511
+ return issues
512
+
513
+
514
+ def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
515
+ issues: list[dict[str, Any]] = []
516
+ if re.match(r"^\s*<\?xml\b", xml):
517
+ issues.append(
518
+ {
519
+ "level": "error",
520
+ "code": "sxsd_unsupported_declaration",
521
+ "path": xml_local_name(root.tag),
522
+ "tag": xml_local_name(root.tag),
523
+ "expected": "SXSD document without an XML declaration",
524
+ "actual": "<?xml ...?>",
525
+ "message": "XML declarations are not supported by the Slides SXSD write format",
526
+ "hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
527
+ }
528
+ )
529
+
530
+ issues.extend(
531
+ sxsd_validator.validate_sxsd(
532
+ without_server_filled_sxsd_fields(root),
533
+ SXSD_SCHEMA_PATH,
534
+ )
535
+ )
536
+ issues.extend(validate_embed_svg_roots(root))
537
+ return issues
538
+
539
+
540
+ def validate_embed_svg_roots(root: ET.Element) -> list[dict[str, Any]]:
541
+ issues: list[dict[str, Any]] = []
542
+ document_namespace = sxsd_validator.element_namespace(root.tag)
543
+ is_bare_slide_fragment = (
544
+ xml_local_name(root.tag) == "slide" and document_namespace is None
545
+ )
546
+ if (
547
+ document_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES
548
+ and not is_bare_slide_fragment
549
+ ):
550
+ return issues
551
+
552
+ def visit(element: ET.Element, ancestors: list[str], parent_path: str) -> None:
553
+ if should_skip_sxsd_subtree(element, ancestors):
554
+ return
555
+
556
+ tag_name = xml_local_name(element.tag)
557
+ path = f"{parent_path}/{tag_name}" if parent_path else tag_name
558
+ if (
559
+ tag_name == "embed"
560
+ and sxsd_validator.element_namespace(element.tag) == document_namespace
561
+ ):
562
+ for child in element:
563
+ if xml_namespace(child.tag) != SVG_NS or xml_local_name(child.tag) == "svg":
564
+ continue
565
+ child_name = xml_local_name(child.tag)
566
+ child_path = f"{path}/{child_name}"
567
+ issues.append(
568
+ {
569
+ "level": "error",
570
+ "code": "sxsd_unexpected_child",
571
+ "path": child_path,
572
+ "tag": child_name,
573
+ "expected": '<svg xmlns="http://www.w3.org/2000/svg">',
574
+ "actual": child_name,
575
+ "message": f"embedded SVG content must use an <svg> root at {child_path}",
576
+ "hint": 'Wrap the SVG content in <svg xmlns="http://www.w3.org/2000/svg">...</svg>.',
577
+ }
578
+ )
579
+ for child in element:
580
+ visit(child, [*ancestors, tag_name], path)
581
+
374
582
  visit(root, [], "")
375
583
  return issues
376
584
 
@@ -417,8 +625,11 @@ def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
417
625
  }
418
626
  )
419
627
 
420
- def visit(element: ET.Element, path: str) -> None:
628
+ def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
421
629
  nonlocal supported_icon_types
630
+ if should_skip_sxsd_subtree(element, ancestors):
631
+ return
632
+
422
633
  tag_name = xml_local_name(element.tag)
423
634
  current_path = f"{path}/{tag_name}" if path else tag_name
424
635
  if tag_name == "icon":
@@ -458,9 +669,9 @@ def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
458
669
  }
459
670
  )
460
671
  for child in element:
461
- visit(child, current_path)
672
+ visit(child, [*ancestors, tag_name], current_path)
462
673
 
463
- visit(root, "")
674
+ visit(root, [], "")
464
675
  return issues
465
676
 
466
677
 
@@ -523,7 +734,8 @@ def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
523
734
  if not prefix:
524
735
  return
525
736
 
526
- if namespace_map.get(prefix) != SML_NAMESPACE:
737
+ actual_namespace = namespace_map.get(prefix)
738
+ if actual_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES:
527
739
  return
528
740
  path = "/".join(element_stack)
529
741
  issues.append(
@@ -531,7 +743,7 @@ def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
531
743
  "level": "error",
532
744
  "code": "sml_prefixed_tag",
533
745
  "tag": element_name,
534
- "namespace": SML_NAMESPACE,
746
+ "namespace": actual_namespace,
535
747
  "path": path,
536
748
  "line": parser.CurrentLineNumber,
537
749
  "column": parser.CurrentColumnNumber,
@@ -575,37 +787,156 @@ def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
575
787
  return xml_error
576
788
 
577
789
 
578
- def parse_presentation(xml: str) -> dict[str, Any]:
579
- presentation_match = re.search(r"<presentation\b([^>]*)>", xml)
580
- if presentation_match:
581
- return {
582
- "width": int(float(extract_attribute(presentation_match.group(1), "width") or 960)),
583
- "height": int(float(extract_attribute(presentation_match.group(1), "height") or 540)),
584
- "slides": re.findall(r"<slide\b[\s\S]*?</slide>", xml),
790
+ def serialize_slide_for_layout(slide_root: ET.Element) -> str:
791
+ slide_copy = copy.deepcopy(slide_root)
792
+ for element in slide_copy.iter():
793
+ if not isinstance(element.tag, str):
794
+ continue
795
+ element.tag = xml_local_name(element.tag)
796
+ attributes = {
797
+ xml_local_name(attribute_name): value
798
+ for attribute_name, value in element.attrib.items()
585
799
  }
586
- slide_match = re.findall(r"<slide\b[\s\S]*?</slide>", xml)
587
- if slide_match:
588
- return {"width": 960, "height": 540, "slides": slide_match}
589
- fail("input must contain a <presentation> or <slide> root")
800
+ element.attrib.clear()
801
+ element.attrib.update(attributes)
802
+ return ET.tostring(slide_copy, encoding="unicode")
803
+
804
+
805
+ def parse_presentation(root: ET.Element) -> dict[str, Any]:
806
+ root_name = xml_local_name(root.tag)
807
+ if root_name == "slide":
808
+ slide_roots = [root]
809
+ width = 960
810
+ height = 540
811
+ elif root_name == "presentation":
812
+ slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
813
+ width = int(float(root.attrib.get("width", 960)))
814
+ height = int(float(root.attrib.get("height", 540)))
815
+ else:
816
+ fail("input must contain a <presentation> or <slide> root")
817
+ return {
818
+ "width": width,
819
+ "height": height,
820
+ "slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
821
+ "slide_roots": slide_roots,
822
+ }
823
+
824
+
825
+ def build_source_xml_paths(slide_xml: str, slide_number: int) -> dict[str, list[str]]:
826
+ root = ET.fromstring(slide_xml)
827
+ data = next((child for child in root if xml_local_name(child.tag) == "data"), None)
828
+ paths: dict[str, list[str]] = {}
829
+ if data is None:
830
+ return paths
831
+ counts: dict[str, int] = {}
832
+ for child in data:
833
+ kind = xml_local_name(child.tag)
834
+ counts[kind] = counts.get(kind, 0) + 1
835
+ paths.setdefault(kind, []).append(
836
+ f"slide[{slide_number}]/data/{kind}[{counts[kind]}]"
837
+ )
838
+ return paths
839
+
840
+
841
+ def attach_source_xml_paths(
842
+ elements: list[dict[str, Any]], source_paths: dict[str, list[str]]
843
+ ) -> None:
844
+ offsets: dict[str, int] = {}
845
+ for element in elements:
846
+ kind = element["kind"]
847
+ source_kind_index = element.get("_source_kind_index")
848
+ offset = (
849
+ source_kind_index - 1
850
+ if isinstance(source_kind_index, int) and source_kind_index > 0
851
+ else offsets.get(kind, 0)
852
+ )
853
+ kind_paths = source_paths.get(kind, [])
854
+ if offset < len(kind_paths):
855
+ element["xml_path"] = kind_paths[offset]
856
+ element["_ref"] = kind_paths[offset]
857
+ offsets[kind] = max(offsets.get(kind, 0), offset + 1)
858
+
859
+
860
+ def extract_source_id_elements(slide_xml: str, slide_number: int) -> list[dict[str, Any]]:
861
+ root = ET.fromstring(slide_xml)
862
+ elements: list[dict[str, Any]] = []
863
+ root_path = f"slide[{slide_number}]"
864
+
865
+ def walk(parent: ET.Element, parent_path: str) -> None:
866
+ child_counts: dict[str, int] = {}
867
+ for child in parent:
868
+ kind = xml_local_name(child.tag)
869
+ child_counts[kind] = child_counts.get(kind, 0) + 1
870
+ xml_path = (
871
+ f"{parent_path}/data"
872
+ if parent is root and kind == "data"
873
+ else f"{parent_path}/{kind}[{child_counts[kind]}]"
874
+ )
875
+ source_id = extract_attribute(
876
+ ET.tostring(child, encoding="unicode").split(">", 1)[0], "id"
877
+ )
878
+ if source_id:
879
+ elements.append(
880
+ {
881
+ "id": source_id,
882
+ "_source_id": source_id,
883
+ "kind": kind,
884
+ "type": child.attrib.get("type") or kind,
885
+ "xml_path": xml_path,
886
+ "_ref": xml_path,
887
+ "_slide_number": slide_number,
888
+ }
889
+ )
890
+ walk(child, xml_path)
891
+
892
+ walk(root, root_path)
893
+ return elements
894
+
895
+
896
+ def element_ref(element: dict[str, Any]) -> str:
897
+ ref = element.get("_ref") or element.get("xml_path")
898
+ if isinstance(ref, str) and ref:
899
+ return ref
900
+ # Low-level detector tests and external callers may pass already-extracted objects.
901
+ # The lint_xml pipeline always attaches the source path before issue detection.
902
+ fallback = element.get("id")
903
+ if isinstance(fallback, str) and fallback:
904
+ return fallback
905
+ raise AssertionError("lint element must have a source xml path or fallback id")
906
+
907
+
908
+ def source_element_id(element: dict[str, Any]) -> str | None:
909
+ value = element.get("_source_id")
910
+ return value if isinstance(value, str) and value else None
911
+
912
+
913
+ def element_label(element: dict[str, Any]) -> str:
914
+ return source_element_id(element) or element_ref(element)
590
915
 
591
916
 
592
917
  def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
593
918
  elements: list[dict[str, Any]] = []
919
+ source_kind_counts: dict[str, int] = {}
594
920
 
595
- for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
921
+ for match in re.finditer(r"<(shape|img|table|chart|whiteboard|embed)\b([^>]*)>", slide_xml):
596
922
  kind, attrs = match.group(1), match.group(2)
923
+ source_kind_counts[kind] = source_kind_counts.get(kind, 0) + 1
924
+ source_kind_index = source_kind_counts[kind]
925
+ is_self_closing = attrs.rstrip().endswith("/")
597
926
  content = ""
598
- if kind in {"shape", "table"}:
927
+ if kind in {"shape", "table"} and not is_self_closing:
599
928
  close_index = slide_xml.find(f"</{kind}>", match.end())
600
929
  if close_index != -1:
601
930
  content = slide_xml[match.end() : close_index]
602
931
 
603
- element_id = extract_attribute(attrs, "id") or f"{kind}-{len(elements) + 1}"
932
+ source_id = extract_attribute(attrs, "id") or None
933
+ element_id = source_id or f"{kind}-{len(elements) + 1}"
604
934
  x = extract_numeric_attribute(attrs, "topLeftX")
605
935
  y = extract_numeric_attribute(attrs, "topLeftY")
606
936
  width = extract_numeric_attribute(attrs, "width")
607
937
  height = extract_numeric_attribute(attrs, "height")
608
938
  rotation = extract_numeric_attribute(attrs, "rotation") or 0
939
+ alpha = extract_numeric_attribute(attrs, "alpha")
609
940
  table_layouts: dict[str, dict[str, Any] | None] = {}
610
941
  if kind == "table":
611
942
  width, table_layouts["width"] = resolve_table_dimension(
@@ -617,6 +948,7 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
617
948
  if all(value is not None for value in [x, y, width, height]):
618
949
  element = {
619
950
  "id": element_id,
951
+ "_source_id": source_id,
620
952
  "kind": kind,
621
953
  "type": extract_attribute(attrs, "type") or kind,
622
954
  "x": x,
@@ -624,7 +956,9 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
624
956
  "width": width,
625
957
  "height": height,
626
958
  "rotation": rotation,
959
+ "alpha": alpha if alpha is not None else 1,
627
960
  "order": len(elements),
961
+ "_source_kind_index": source_kind_index,
628
962
  }
629
963
  if kind == "table":
630
964
  element.update(
@@ -635,15 +969,48 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
635
969
  }
636
970
  )
637
971
  if kind == "shape":
972
+ content_attrs = extract_tag_attributes(content, "content")
973
+ font_size = extract_numeric_attribute(content_attrs, "fontSize")
974
+ if font_size is None:
975
+ font_size = extract_numeric_attribute(attrs, "fontSize")
976
+ font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
977
+ text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
978
+ bold = (
979
+ extract_bool_attribute(content_attrs, "bold")
980
+ or extract_bool_attribute(attrs, "bold")
981
+ or detect_inline_style_presence(content, {"strong", "b"})
982
+ or detect_any_span_bool_attribute(content, "bold")
983
+ )
984
+ italic = (
985
+ extract_bool_attribute(content_attrs, "italic")
986
+ or extract_bool_attribute(attrs, "italic")
987
+ or detect_inline_style_presence(content, {"i", "em"})
988
+ or detect_any_span_bool_attribute(content, "italic")
989
+ )
638
990
  element.update(
639
991
  {
640
- "textType": extract_attribute(content, "textType"),
641
- "textAlign": extract_attribute(content, "textAlign"),
642
- "autoFit": extract_attribute(content, "autoFit"),
643
- "fontSize": float(
644
- extract_attribute(content, "fontSize") or extract_attribute(attrs, "fontSize") or 16
645
- ),
992
+ "textType": extract_attribute(content_attrs, "textType"),
993
+ "textAlign": extract_attribute(content_attrs, "textAlign"),
994
+ "verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
995
+ "vert": extract_attribute(attrs, "vert") or "horz",
996
+ "autoFit": extract_attribute(content_attrs, "autoFit"),
997
+ "wrap": extract_attribute(content_attrs, "wrap"),
998
+ "lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
999
+ "beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
1000
+ "afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
1001
+ "letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
1002
+ "paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
1003
+ "paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
1004
+ "paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
1005
+ "paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
1006
+ "fontSize": font_size if font_size is not None else 16,
1007
+ "fontFamily": font_family or "",
1008
+ "color": text_color,
1009
+ "textAlpha": effective_text_alpha(alpha, text_color),
1010
+ "bold": bold,
1011
+ "italic": italic,
646
1012
  "text": strip_xml_paragraphs(content),
1013
+ "paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
647
1014
  }
648
1015
  )
649
1016
  elements.append(element)
@@ -671,6 +1038,50 @@ def has_text_content(element: dict[str, Any]) -> bool:
671
1038
  return bool(element.get("text"))
672
1039
 
673
1040
 
1041
+ def is_vertical_text(element: dict[str, Any]) -> bool:
1042
+ return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
1043
+
1044
+
1045
+ def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1046
+ issues: list[dict[str, Any]] = []
1047
+ text_elements = [
1048
+ element
1049
+ for element in elements
1050
+ if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
1051
+ ]
1052
+ image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
1053
+ for text_element in text_elements:
1054
+ for image_element in image_elements:
1055
+ if image_element["order"] <= text_element["order"]:
1056
+ continue
1057
+ if is_vertical_text(text_element):
1058
+ if intersects(image_element, text_element):
1059
+ issues.append({
1060
+ "level": "info",
1061
+ "code": "image_may_cover_vertical_text",
1062
+ "elements": [element_ref(image_element), element_ref(text_element)],
1063
+ "message": (
1064
+ f"image {element_label(image_element)} may cover vertical text shape "
1065
+ f"{element_label(text_element)}"
1066
+ ),
1067
+ "hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
1068
+ })
1069
+ continue
1070
+ text_visual_bbox = estimate_text_visual_bbox(text_element)
1071
+ if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
1072
+ issues.append({
1073
+ "level": "error",
1074
+ "code": "image_covers_text",
1075
+ "elements": [element_ref(image_element), element_ref(text_element)],
1076
+ "message": (
1077
+ f"image {element_label(image_element)} covers text shape "
1078
+ f"{element_label(text_element)}"
1079
+ ),
1080
+ "hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
1081
+ })
1082
+ return issues
1083
+
1084
+
674
1085
  def is_decorative_text(element: dict[str, Any]) -> bool:
675
1086
  text = element.get("text") or ""
676
1087
  return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
@@ -680,22 +1091,135 @@ def normalize_text_for_overlap(text: str) -> str:
680
1091
  return re.sub(r"\s+", "", text)
681
1092
 
682
1093
 
683
- def estimate_character_width(character: str, font_size: int | float) -> int | float:
1094
+ SERIF_FONT_PATTERNS = {
1095
+ "song", "songti", "simsun", "ming", "mincho",
1096
+ "georgia", "times", "caslon", "garamond", "sourcehan-serif",
1097
+ "source han serif", "思源宋体", "宋体", "明体",
1098
+ }
1099
+
1100
+ SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
1101
+ "helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
1102
+
1103
+
1104
+ def classify_font_family(font_family: str | None) -> str:
1105
+ if not font_family:
1106
+ return "sans"
1107
+ family_lower = font_family.lower()
1108
+ for marker in SANS_EXPLICIT_MARKERS:
1109
+ if marker in family_lower:
1110
+ return "sans"
1111
+ serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
1112
+ for pattern in serif_keywords:
1113
+ if pattern in family_lower:
1114
+ return "serif"
1115
+ return "sans"
1116
+
1117
+
1118
+ _FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
1119
+ "sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
1120
+ "serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
1121
+ }
1122
+
1123
+
1124
+ def estimate_character_width(
1125
+ character: str,
1126
+ font_size: int | float,
1127
+ bold: bool = False,
1128
+ font_family: str | None = None,
1129
+ ) -> int | float:
1130
+ bold_multiplier = 1.05 if bold else 1.0
684
1131
  if character.isspace():
685
- return font_size * 0.33
686
- if unicodedata.east_asian_width(character) in {"F", "W"}:
687
- return font_size
688
- return font_size * 0.55
1132
+ return font_size * 0.33 * bold_multiplier
1133
+ ea_width = unicodedata.east_asian_width(character)
1134
+ if ea_width in {"F", "W"}:
1135
+ return font_size * bold_multiplier
1136
+ category = classify_font_family(font_family)
1137
+ coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
1138
+ if character.isupper():
1139
+ return font_size * coeffs["upper"] * bold_multiplier
1140
+ if character.islower():
1141
+ return font_size * coeffs["lower"] * bold_multiplier
1142
+ if character.isdigit():
1143
+ return font_size * coeffs["digit"] * bold_multiplier
1144
+ return font_size * coeffs["punct"] * bold_multiplier
1145
+
1146
+
1147
+ def estimate_text_width(
1148
+ text: str,
1149
+ font_size: int | float,
1150
+ letter_spacing: int | float = 0,
1151
+ bold: bool = False,
1152
+ font_family: str | None = None,
1153
+ ) -> int | float:
1154
+ base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
1155
+ return base + max(len(text) - 1, 0) * letter_spacing
1156
+
1157
+
1158
+ def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
1159
+ if paragraph is not None:
1160
+ value = paragraph.get("letterSpacing")
1161
+ if isinstance(value, (int, float)):
1162
+ return value
1163
+ value = element.get("letterSpacing")
1164
+ return value if isinstance(value, (int, float)) else 0
1165
+
1166
+
1167
+ def text_wrap_width_tolerance() -> int | float:
1168
+ return TEXT_WRAP_WIDTH_TOLERANCE_PX
1169
+
1170
+
1171
+ def text_height_overflow_tolerance() -> int | float:
1172
+ return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
1173
+
1174
+
1175
+ def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
1176
+ return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
1177
+
1178
+
1179
+ def is_short_metric_text(text: str) -> bool:
1180
+ compact = re.sub(r"\s+", "", text)
1181
+ if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
1182
+ return False
1183
+ if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
1184
+ return True
1185
+ if re.search(r"[,.,+\-–—/%%]", compact) is None:
1186
+ return False
1187
+ return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
1188
+
1189
+
1190
+ def is_single_line_visual_candidate(
1191
+ element: dict[str, Any],
1192
+ paragraph: dict[str, Any] | None,
1193
+ text: str,
1194
+ logical_width: int | float,
1195
+ effective_width: int | float,
1196
+ ) -> bool:
1197
+ if "\n" in text or logical_width <= effective_width:
1198
+ return False
1199
+ if is_short_metric_text(text):
1200
+ return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
689
1201
 
1202
+ text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
1203
+ compact_len = len(re.sub(r"\s+", "", text))
1204
+ if text_align == "center" and compact_len <= 32:
1205
+ return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
690
1206
 
691
- def estimate_text_width(text: str, font_size: int | float) -> int | float:
692
- return sum(estimate_character_width(character, font_size) for character in text)
1207
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1208
+ if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
1209
+ return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
1210
+ return False
693
1211
 
694
1212
 
695
1213
  def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
696
1214
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1215
+ bold = element.get("bold", False)
1216
+ font_family = element.get("fontFamily", "")
1217
+ letter_spacing = resolve_letter_spacing(element)
697
1218
  paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
698
- return max([estimate_text_width(paragraph, font_size) for paragraph in paragraphs] or [1])
1219
+ return max(
1220
+ [estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
1221
+ or [1]
1222
+ )
699
1223
 
700
1224
 
701
1225
  def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
@@ -708,27 +1232,203 @@ def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool
708
1232
  return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
709
1233
 
710
1234
 
711
- def estimate_text_line_count(element: dict[str, Any]) -> int:
1235
+ def estimate_text_line_count_for_text(
1236
+ element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
1237
+ ) -> int:
712
1238
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
713
- paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
1239
+ bold = element.get("bold", False)
1240
+ font_family = element.get("fontFamily", "")
1241
+ letter_spacing = resolve_letter_spacing(element, paragraph)
1242
+ available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
1243
+ hard_lines = text.split("\n")
1244
+ if not text:
1245
+ return 0
714
1246
  line_count = 0
715
- for paragraph in paragraphs:
716
- logical_width = max(estimate_text_width(paragraph, font_size), 1)
717
- line_count += max(1, math.ceil(logical_width / max(element["width"], 1)))
718
- return max(line_count, 1)
1247
+ for hard_line in hard_lines:
1248
+ if element.get("wrap") in {"false", "0"}:
1249
+ line_count += 1
1250
+ continue
1251
+ logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
1252
+ effective_width = available_width + text_wrap_width_tolerance()
1253
+ if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
1254
+ line_count += 1
1255
+ continue
1256
+ line_count += max(1, math.ceil(logical_width / effective_width))
1257
+ return line_count
1258
+
1259
+
1260
+ def estimate_text_line_count(element: dict[str, Any]) -> int:
1261
+ return max(estimate_text_line_count_for_text(element, element["text"]), 1)
1262
+
1263
+
1264
+ def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
1265
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1266
+ if line_spacing is None:
1267
+ return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
1268
+ match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
1269
+ if match is None:
1270
+ return None
1271
+ spacing_type, value = match.groups()
1272
+ return font_size * float(value) if spacing_type == "multiple" else float(value)
1273
+
1274
+
1275
+ def adjust_dense_body_line_height(
1276
+ element: dict[str, Any],
1277
+ line_spacing: str | None,
1278
+ line_height: int | float,
1279
+ paragraph_count: int,
1280
+ ) -> int | float:
1281
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1282
+ if paragraph_count < 4 or font_size > 14 or not line_spacing:
1283
+ return line_height
1284
+ match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
1285
+ if match is None:
1286
+ return line_height
1287
+ return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
1288
+
1289
+
1290
+ def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
1291
+ issues: list[dict[str, Any]] = []
1292
+ for element in elements:
1293
+ if not is_text_element(element) or not has_text_content(element):
1294
+ continue
1295
+ if has_explicit_height_auto_fit(element):
1296
+ continue
1297
+
1298
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1299
+ paragraphs = element.get("paragraphs") or [
1300
+ {
1301
+ "text": element["text"],
1302
+ "lineSpacing": None,
1303
+ "beforeLineSpacing": None,
1304
+ "afterLineSpacing": None,
1305
+ }
1306
+ ]
1307
+ line_count = 0
1308
+ estimated_height = 0.0
1309
+ line_heights: list[int | float] = []
1310
+ for paragraph in paragraphs:
1311
+ paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
1312
+ if paragraph_line_count == 0:
1313
+ continue
1314
+ resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
1315
+ line_height = estimate_text_line_height(element, resolved_line_spacing)
1316
+ before_spacing = estimate_text_line_height(
1317
+ element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
1318
+ )
1319
+ after_spacing = estimate_text_line_height(
1320
+ element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
1321
+ )
1322
+ if line_height is None or before_spacing is None or after_spacing is None:
1323
+ line_count = 0
1324
+ break
1325
+ line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
1326
+ first_line_height = font_size if line_count == 0 else line_height
1327
+ line_count += paragraph_line_count
1328
+ line_heights.append(line_height)
1329
+ estimated_height += (
1330
+ before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
1331
+ )
1332
+ if line_count == 0:
1333
+ continue
1334
+ available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
1335
+ overflow = estimated_height - available_height
1336
+ if overflow <= text_height_overflow_tolerance():
1337
+ continue
1338
+
1339
+ is_background = is_background_decorative_text(element, elements)
1340
+ if is_background:
1341
+ level = "info"
1342
+ else:
1343
+ level = "error" if overflow > 10 else "warning"
1344
+ message = (
1345
+ f"text shape {element_label(element)} may overflow its own content box "
1346
+ f'(estimated {estimated_height:g}px, available {available_height:g}px); '
1347
+ 'consider setting content wrap="true" autoFit="normal-auto-fit"'
1348
+ )
1349
+ if is_background:
1350
+ message += " (likely background decoration: large font, low alpha, underneath other text)"
1351
+ issues.append(
1352
+ {
1353
+ "level": level,
1354
+ "code": "text_may_overflow_shape",
1355
+ "elements": [element_ref(element)],
1356
+ "line_count": line_count,
1357
+ "line_height": max(line_heights),
1358
+ "estimated_height": estimated_height,
1359
+ "available_height": available_height,
1360
+ "overflow": overflow,
1361
+ "message": message,
1362
+ "hint": (
1363
+ "Increase shape.height, reduce the text, or set content wrap=\"true\" "
1364
+ "autoFit=\"normal-auto-fit\". "
1365
+ "This is an estimate based on font size, line spacing, and wrapped line count."
1366
+ ),
1367
+ }
1368
+ )
1369
+ return issues
1370
+
1371
+
1372
+ def is_background_decorative_text(
1373
+ element: dict[str, Any], elements: list[dict[str, Any]]
1374
+ ) -> bool:
1375
+ if not is_ghost_text(element):
1376
+ return False
1377
+ for other in elements:
1378
+ if other is element:
1379
+ continue
1380
+ if not is_text_element(other) or not has_text_content(other):
1381
+ continue
1382
+ foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
1383
+ if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
1384
+ continue
1385
+ if other["order"] <= element["order"]:
1386
+ continue
1387
+ if intersects(element, other):
1388
+ return True
1389
+ return False
1390
+
1391
+
1392
+ def is_ghost_text(element: dict[str, Any]) -> bool:
1393
+ if not is_text_element(element) or not has_text_content(element):
1394
+ return False
1395
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
1396
+ text_alpha = element.get("textAlpha", element.get("alpha", 1))
1397
+ if not isinstance(text_alpha, (int, float)):
1398
+ return False
1399
+ if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
1400
+ return True
1401
+ return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
719
1402
 
720
1403
 
721
1404
  def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
722
1405
  if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
723
1406
  return None
724
1407
 
1408
+ padding_left = element.get("paddingLeft", 0)
1409
+ padding_right = element.get("paddingRight", 0)
1410
+ padding_top = element.get("paddingTop", 0)
1411
+ padding_bottom = element.get("paddingBottom", 0)
1412
+ content_width = max(element["width"] - padding_left - padding_right, 0)
1413
+ content_height = max(element["height"] - padding_top - padding_bottom, 0)
725
1414
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
726
1415
  line_count = estimate_text_line_count(element)
727
- visual_width = min(element["width"], max(1, estimate_text_max_line_width(element)))
728
- visual_height = min(element["height"], max(1, line_count * font_size * 1.2))
1416
+ estimated_width = max(1, estimate_text_max_line_width(element))
1417
+ visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
1418
+ visual_height = min(content_height, max(1, line_count * font_size * 1.2))
1419
+ x = element["x"] + padding_left
1420
+ if element.get("textAlign") == "center":
1421
+ x += (content_width - visual_width) / 2
1422
+ elif element.get("textAlign") == "right":
1423
+ x += content_width - visual_width
1424
+ y = element["y"] + padding_top
1425
+ if element.get("verticalAlign") == "middle":
1426
+ y += (content_height - visual_height) / 2
1427
+ elif element.get("verticalAlign") == "bottom":
1428
+ y += content_height - visual_height
729
1429
  return {
730
- "x": element["x"],
731
- "y": element["y"],
1430
+ "x": x,
1431
+ "y": y,
732
1432
  "width": visual_width,
733
1433
  "height": visual_height,
734
1434
  }
@@ -812,25 +1512,34 @@ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str,
812
1512
  return False
813
1513
  if not (has_text_content(left) and has_text_content(right)):
814
1514
  return False
1515
+ if is_ghost_text(left) or is_ghost_text(right):
1516
+ return False
815
1517
  if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
816
1518
  return False
817
1519
 
818
1520
  source, target = sorted([left, right], key=lambda element: element["x"])
819
1521
  if source["x"] == target["x"]:
820
1522
  return False
1523
+ wrap_enabled = source.get("wrap") not in {"false", "0"}
1524
+ has_horizontal_gap = source["x"] + source["width"] <= target["x"]
1525
+ if wrap_enabled and has_horizontal_gap:
1526
+ return False
821
1527
  if source.get("autoFit") == "normal-auto-fit":
822
1528
  return False
823
1529
  if source.get("textAlign") in {"center", "right"}:
824
1530
  return False
825
1531
 
826
1532
  font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
1533
+ padding_left = source.get("paddingLeft", 0)
1534
+ padding_right = source.get("paddingRight", 0)
1535
+ available_width = max(source["width"] - padding_left - padding_right, 1)
827
1536
  visual_width = estimate_text_max_line_width(source)
828
- overflow_width = visual_width - source["width"]
829
- min_overflow = max(font_size * 1.5, source["width"] * 0.08)
1537
+ overflow_width = visual_width - available_width
1538
+ min_overflow = max(font_size * 1.5, available_width * 0.08)
830
1539
  if overflow_width < min_overflow:
831
1540
  return False
832
1541
 
833
- intrusion_width = source["x"] + visual_width - target["x"]
1542
+ intrusion_width = source["x"] + padding_left + visual_width - target["x"]
834
1543
  min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
835
1544
  if intrusion_width < min_intrusion:
836
1545
  return False
@@ -840,11 +1549,27 @@ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str,
840
1549
  return vertical_overlap >= min_vertical_overlap
841
1550
 
842
1551
 
1552
+ def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
1553
+ source, target = sorted([left, right], key=lambda element: element["x"])
1554
+ padding_left = source.get("paddingLeft", 0)
1555
+ visual_width = estimate_text_max_line_width(source)
1556
+ source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
1557
+ width = intersection_width(source_visual_bbox, target)
1558
+ height = intersection_height(source_visual_bbox, target)
1559
+ return {
1560
+ "intersection_width": round(width, 3),
1561
+ "intersection_height": round(height, 3),
1562
+ "intersection_area": round(width * height, 3),
1563
+ }
1564
+
1565
+
843
1566
  def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
844
1567
  if is_text_element(left) and not has_text_content(left):
845
1568
  return False
846
1569
  if is_text_element(right) and not has_text_content(right):
847
1570
  return False
1571
+ if is_ghost_text(left) or is_ghost_text(right):
1572
+ return False
848
1573
  if is_template_text_stack(left, right):
849
1574
  return False
850
1575
  if is_text_element(left) and is_text_element(right):
@@ -868,12 +1593,15 @@ def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
868
1593
  def build_whiteboard_external_overlap_issue(
869
1594
  whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
870
1595
  ) -> dict[str, Any]:
871
- element_ids = [detail["element"] for detail in overlap_details]
1596
+ element_refs = [detail["element"] for detail in overlap_details]
872
1597
  return {
873
1598
  "level": "warning",
874
1599
  "code": "whiteboard_external_overlap",
875
- "elements": [whiteboard["id"], *element_ids],
876
- "message": f'whiteboard {whiteboard["id"]} overlaps {len(element_ids)} sibling elements across its boundary',
1600
+ "elements": [element_ref(whiteboard), *element_refs],
1601
+ "message": (
1602
+ f"whiteboard {element_label(whiteboard)} overlaps {len(element_refs)} "
1603
+ "sibling elements across its boundary"
1604
+ ),
877
1605
  "hint": (
878
1606
  "Treat this as a static whiteboard container-bbox risk, not final visual proof. "
879
1607
  "After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
@@ -891,6 +1619,8 @@ def should_report_whiteboard_overlap(
891
1619
  ) -> dict[str, Any] | None:
892
1620
  if other is whiteboard or not intersects(whiteboard, other):
893
1621
  return None
1622
+ if is_ghost_text(other):
1623
+ return None
894
1624
  if contains(whiteboard, other):
895
1625
  return None
896
1626
  if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
@@ -912,7 +1642,7 @@ def should_report_whiteboard_overlap(
912
1642
  return None
913
1643
 
914
1644
  return {
915
- "element": other["id"],
1645
+ "element": element_ref(other),
916
1646
  "kind": other["kind"],
917
1647
  "type": other.get("type"),
918
1648
  "overlap_width": overlap_width,
@@ -922,16 +1652,16 @@ def should_report_whiteboard_overlap(
922
1652
 
923
1653
 
924
1654
  def prune_contained_text_overlap_details(
925
- overlap_details: list[dict[str, Any]], elements_by_id: dict[str, dict[str, Any]]
1655
+ overlap_details: list[dict[str, Any]], elements_by_ref: dict[str, dict[str, Any]]
926
1656
  ) -> list[dict[str, Any]]:
927
1657
  pruned: list[dict[str, Any]] = []
928
1658
  for detail in overlap_details:
929
- element = elements_by_id[detail["element"]]
1659
+ element = elements_by_ref[detail["element"]]
930
1660
  if is_text_element(element):
931
1661
  has_reported_container = any(
932
1662
  detail["element"] != other_detail["element"]
933
- and not is_text_element(elements_by_id[other_detail["element"]])
934
- and contains(elements_by_id[other_detail["element"]], element)
1663
+ and not is_text_element(elements_by_ref[other_detail["element"]])
1664
+ and contains(elements_by_ref[other_detail["element"]], element)
935
1665
  for other_detail in overlap_details
936
1666
  )
937
1667
  if has_reported_container:
@@ -944,7 +1674,7 @@ def detect_whiteboard_external_overlaps(
944
1674
  elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
945
1675
  ) -> list[dict[str, Any]]:
946
1676
  issues: list[dict[str, Any]] = []
947
- elements_by_id = {element["id"]: element for element in elements}
1677
+ elements_by_ref = {element_ref(element): element for element in elements}
948
1678
  for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
949
1679
  overlap_details = [
950
1680
  detail
@@ -959,7 +1689,7 @@ def detect_whiteboard_external_overlaps(
959
1689
  )
960
1690
  is not None
961
1691
  ]
962
- overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_id)
1692
+ overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_ref)
963
1693
  if overlap_details:
964
1694
  issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
965
1695
  return issues
@@ -969,7 +1699,6 @@ def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
969
1699
  bbox = {key: element[key] for key in ("x", "y", "width", "height")}
970
1700
  if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
971
1701
  return bbox
972
-
973
1702
  rotation = element["rotation"]
974
1703
  if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
975
1704
  rotation = 0
@@ -999,7 +1728,7 @@ def detect_elements_out_of_canvas(
999
1728
  element
1000
1729
  for element in elements
1001
1730
  if element["kind"] in {"table", "chart"}
1002
- or (element["kind"] == "shape" and element["type"] == "text")
1731
+ or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
1003
1732
  ):
1004
1733
  bbox = element_canvas_bbox(element)
1005
1734
  overflow = {
@@ -1009,7 +1738,9 @@ def detect_elements_out_of_canvas(
1009
1738
  "bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
1010
1739
  }
1011
1740
  overflow_details = [
1012
- f"{side} by {amount:g}px" for side, amount in overflow.items() if amount > 0
1741
+ f"{side} by {amount:g}px"
1742
+ for side, amount in overflow.items()
1743
+ if amount > CANVAS_OVERFLOW_TOLERANCE
1013
1744
  ]
1014
1745
  if not overflow_details:
1015
1746
  continue
@@ -1017,12 +1748,12 @@ def detect_elements_out_of_canvas(
1017
1748
  {
1018
1749
  "level": "error",
1019
1750
  "code": f'{element["kind"]}_out_of_canvas',
1020
- "elements": [element["id"]],
1751
+ "elements": [element_ref(element)],
1021
1752
  "canvas": {"width": slide_width, "height": slide_height},
1022
1753
  "bbox": bbox,
1023
1754
  "overflow": overflow,
1024
1755
  "message": (
1025
- f'{element["kind"]} {element["id"]} exceeds the {slide_width:g}x{slide_height:g} canvas '
1756
+ f'{element["kind"]} {element_label(element)} exceeds the {slide_width:g}x{slide_height:g} canvas '
1026
1757
  f'({", ".join(overflow_details)})'
1027
1758
  ),
1028
1759
  "hint": (
@@ -1090,13 +1821,13 @@ def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[
1090
1821
  {
1091
1822
  "level": "info",
1092
1823
  "code": "table_resolved_size_mismatch",
1093
- "elements": [table["id"]],
1824
+ "elements": [element_ref(table)],
1094
1825
  "dimension": dimension,
1095
1826
  "declared_size": target_size,
1096
1827
  "resolved_size": actual_size,
1097
1828
  "resolved_sizes": layout["final_sizes"],
1098
1829
  "message": (
1099
- f'table {table["id"]} declares {dimension}={format_size(target_size)}px, but its '
1830
+ f'table {element_label(table)} declares {dimension}={format_size(target_size)}px, but its '
1100
1831
  f"{child_description} resolve to {format_size(actual_size)}px"
1101
1832
  ),
1102
1833
  "hint": (
@@ -1108,14 +1839,108 @@ def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[
1108
1839
  return issues
1109
1840
 
1110
1841
 
1111
- def lint_slide(
1112
- slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
1842
+ def segment_intersects_rect(
1843
+ x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
1844
+ ) -> bool:
1845
+ """True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
1846
+ left = rect["x"]
1847
+ top = rect["y"]
1848
+ right = rect["x"] + rect["width"]
1849
+ bottom = rect["y"] + rect["height"]
1850
+ if right <= left or bottom <= top:
1851
+ return False
1852
+ dx = x2 - x1
1853
+ dy = y2 - y1
1854
+ if dx == 0 and dy == 0:
1855
+ return left <= x1 <= right and top <= y1 <= bottom
1856
+ t_enter, t_exit = 0.0, 1.0
1857
+ for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
1858
+ if delta == 0:
1859
+ if distance < 0:
1860
+ return False
1861
+ continue
1862
+ t = distance / delta
1863
+ if delta < 0:
1864
+ t_enter = max(t_enter, t)
1865
+ else:
1866
+ t_exit = min(t_exit, t)
1867
+ if t_enter > t_exit:
1868
+ return False
1869
+ return True
1870
+
1871
+
1872
+ def line_text_graze_margin(text_element: dict[str, Any]) -> float:
1873
+ font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
1874
+ return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
1875
+
1876
+
1877
+ def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
1878
+ width = rect["width"] - 2 * margin
1879
+ height = rect["height"] - 2 * margin
1880
+ if width <= 0 or height <= 0:
1881
+ return None
1882
+ return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
1883
+
1884
+
1885
+ def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
1886
+ if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
1887
+ return False
1888
+ if not is_text_element(text_element) or not has_text_content(text_element):
1889
+ return False
1890
+ if is_ghost_text(text_element) or is_decorative_text(text_element):
1891
+ return False
1892
+ glyph_bbox = estimate_text_visual_bbox(text_element)
1893
+ if glyph_bbox is None:
1894
+ return False
1895
+ # Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
1896
+ # frame is exempt; only a line that actually cuts through the letterforms is a crossing.
1897
+ target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
1898
+ if target is None:
1899
+ return False
1900
+ return segment_intersects_rect(
1901
+ line["startX"], line["startY"], line["endX"], line["endY"], target
1902
+ )
1903
+
1904
+
1905
+ def detect_line_text_crossings(
1906
+ slide_xml: str, elements: list[dict[str, Any]], slide_number: int
1907
+ ) -> list[dict[str, Any]]:
1908
+ lines = extract_line_elements(slide_xml)
1909
+ if not lines:
1910
+ return []
1911
+ attach_source_xml_paths(lines, build_source_xml_paths(slide_xml, slide_number))
1912
+ text_elements = [element for element in elements if is_text_element(element)]
1913
+ issues: list[dict[str, Any]] = []
1914
+ for line in lines:
1915
+ for text_element in text_elements:
1916
+ if not line_crosses_text(line, text_element):
1917
+ continue
1918
+ issues.append(
1919
+ {
1920
+ "level": "error",
1921
+ "code": "bbox_overlap",
1922
+ "elements": [element_ref(line), element_ref(text_element)],
1923
+ "message": (
1924
+ f"line {element_label(line)} crosses text {element_label(text_element)}"
1925
+ ),
1926
+ "hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
1927
+ }
1928
+ )
1929
+ return issues
1930
+
1931
+
1932
+ def lint_slide(
1933
+ slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
1113
1934
  ) -> dict[str, Any]:
1114
1935
  elements = extract_elements(slide_xml)
1936
+ attach_source_xml_paths(elements, build_source_xml_paths(slide_xml, slide_number))
1115
1937
  issues: list[dict[str, Any]] = [
1116
1938
  *detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
1117
1939
  *detect_elements_out_of_canvas(elements, slide_width, slide_height),
1118
1940
  *detect_table_layout_size_mismatches(elements),
1941
+ *detect_text_may_overflow_shapes(elements),
1942
+ *detect_image_text_occlusions(elements),
1943
+ *detect_line_text_crossings(slide_xml, elements, slide_number),
1119
1944
  ]
1120
1945
 
1121
1946
  for index, left in enumerate(elements):
@@ -1127,64 +1952,843 @@ def lint_slide(
1127
1952
  {
1128
1953
  "level": "error",
1129
1954
  "code": "bbox_overlap",
1130
- "elements": [left["id"], right["id"]],
1131
- "message": f'{left["id"]} overlaps {right["id"]}',
1955
+ "elements": [element_ref(left), element_ref(right)],
1956
+ "message": f"{element_label(left)} overlaps {element_label(right)}",
1957
+ "hint": "Move or resize the elements so their visual bounds no longer intersect.",
1958
+ **(
1959
+ {"measurement": horizontal_text_overflow_measurement(left, right)}
1960
+ if horizontal_overflow
1961
+ else {}
1962
+ ),
1132
1963
  }
1133
1964
  )
1134
1965
 
1135
- return {"slide_number": slide_number, "element_count": len(elements), "issues": issues}
1966
+ return {
1967
+ "slide_number": slide_number,
1968
+ "element_count": len(elements),
1969
+ "elements": elements,
1970
+ "issues": issues,
1971
+ }
1136
1972
 
1137
1973
 
1138
- def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
1139
- root, xml_error = parse_xml_root(xml)
1140
- if xml_error:
1974
+
1975
+ MIN_CONTAINER_WIDTH = 140
1976
+ MIN_CONTAINER_HEIGHT = 160
1977
+ MIN_SHORT_CARD_HEIGHT = 80
1978
+ MIN_CONTAINER_AREA = 20_000
1979
+ MIN_CONTENT_COVERAGE_RATIO = 0.15
1980
+ MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
1981
+ MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
1982
+ SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
1983
+ MIN_SIMILAR_SHORT_CARD_COUNT = 2
1984
+ LARGE_VISUAL_CHILD_RATIO = 0.35
1985
+ LAYOUT_PANEL_SPAN_RATIO = 0.90
1986
+ IMAGE_OVERLAY_MATCH_RATIO = 0.90
1987
+ DENSITY_CONTAINMENT_TOLERANCE = 8
1988
+
1989
+
1990
+ def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1991
+ left = max(element["x"], container["x"])
1992
+ top = max(element["y"], container["y"])
1993
+ right = min(element["x"] + element["width"], container["x"] + container["width"])
1994
+ bottom = min(element["y"] + element["height"], container["y"] + container["height"])
1995
+ if right <= left or bottom <= top:
1996
+ return None
1997
+ return {"x": left, "y": top, "width": right - left, "height": bottom - top}
1998
+
1999
+
2000
+ def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
2001
+ x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
2002
+ area = 0
2003
+ for left, right in zip(x_coordinates, x_coordinates[1:]):
2004
+ intervals = sorted(
2005
+ (rect["y"], rect["y"] + rect["height"])
2006
+ for rect in rectangles
2007
+ if rect["x"] < right and rect["x"] + rect["width"] > left
2008
+ )
2009
+ covered_height = 0
2010
+ interval_end: int | float | None = None
2011
+ for top, bottom in intervals:
2012
+ if interval_end is None:
2013
+ covered_height += bottom - top
2014
+ interval_end = bottom
2015
+ elif bottom > interval_end:
2016
+ covered_height += bottom - max(top, interval_end)
2017
+ interval_end = bottom
2018
+ area += (right - left) * covered_height
2019
+ return area
2020
+
2021
+
2022
+ def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
2023
+ return sum(
2024
+ other is not element
2025
+ and is_visually_rendered(other)
2026
+ and other["kind"] == "shape"
2027
+ and other["type"] == "rect"
2028
+ and other["width"] >= MIN_CONTAINER_WIDTH
2029
+ and other["height"] >= MIN_SHORT_CARD_HEIGHT
2030
+ and element_area(other) >= MIN_CONTAINER_AREA
2031
+ and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
2032
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
2033
+ and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
2034
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
2035
+ for other in elements
2036
+ ) >= MIN_SIMILAR_SHORT_CARD_COUNT
2037
+
2038
+
2039
+ def is_layout_container(
2040
+ element: dict[str, Any],
2041
+ slide_width: int | float,
2042
+ slide_height: int | float,
2043
+ elements: list[dict[str, Any]] | None = None,
2044
+ ) -> bool:
2045
+ has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
2046
+ elements is not None
2047
+ and element["height"] >= MIN_SHORT_CARD_HEIGHT
2048
+ and has_similar_short_card_peer(element, elements)
2049
+ )
2050
+ return (
2051
+ element["kind"] == "shape"
2052
+ and element["type"] == "rect"
2053
+ and is_visually_rendered(element)
2054
+ and element["width"] >= MIN_CONTAINER_WIDTH
2055
+ and has_supported_height
2056
+ and element_area(element) >= MIN_CONTAINER_AREA
2057
+ and not (
2058
+ element["x"] <= 2
2059
+ and element["y"] <= 2
2060
+ and element["width"] >= slide_width - 4
2061
+ and element["height"] >= slide_height - 4
2062
+ )
2063
+ )
2064
+
2065
+
2066
+ def is_edge_spanning_layout_panel(
2067
+ element: dict[str, Any], slide_width: int | float, slide_height: int | float
2068
+ ) -> bool:
2069
+ touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
2070
+ touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
2071
+ return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
2072
+ touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
2073
+ )
2074
+
2075
+
2076
+ def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
2077
+ container_area = element_area(container)
2078
+ return any(
2079
+ element["kind"] == "img"
2080
+ and is_visually_rendered(element)
2081
+ and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
2082
+ for element in elements
2083
+ )
2084
+
2085
+
2086
+ def is_nested_in_layout_panel(
2087
+ container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
2088
+ ) -> bool:
2089
+ return any(
2090
+ element is not container
2091
+ and element["kind"] == "shape"
2092
+ and element["type"] == "rect"
2093
+ and is_visually_rendered(element)
2094
+ and is_edge_spanning_layout_panel(element, slide_width, slide_height)
2095
+ and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2096
+ for element in elements
2097
+ )
2098
+
2099
+
2100
+ def extract_density_elements(slide_xml: str, slide_number: int = 1) -> list[dict[str, Any]]:
2101
+ elements = extract_elements(slide_xml)
2102
+ source_paths = build_source_xml_paths(slide_xml, slide_number)
2103
+ attach_source_xml_paths(elements, source_paths)
2104
+ shape_elements_by_index = {
2105
+ element["_source_kind_index"]: element
2106
+ for element in elements
2107
+ if element["kind"] == "shape"
2108
+ }
2109
+ root = ET.fromstring(slide_xml)
2110
+ shape_index = 0
2111
+ for node in root.iter():
2112
+ if xml_local_name(node.tag) != "shape":
2113
+ continue
2114
+ shape_index += 1
2115
+ element = shape_elements_by_index.get(shape_index)
2116
+ if element is None:
2117
+ continue
2118
+ content_node = next(
2119
+ (child for child in node if xml_local_name(child.tag) == "content"),
2120
+ None,
2121
+ )
2122
+ paragraphs = (
2123
+ [
2124
+ " ".join("".join(paragraph.itertext()).split())
2125
+ for paragraph in content_node.iter()
2126
+ if xml_local_name(paragraph.tag) == "p"
2127
+ ]
2128
+ if content_node is not None
2129
+ else []
2130
+ )
2131
+ raw_font_size = (
2132
+ content_node.attrib.get("fontSize") if content_node is not None else None
2133
+ ) or node.attrib.get("fontSize")
2134
+ try:
2135
+ base_font_size = float(raw_font_size or 16)
2136
+ except ValueError:
2137
+ base_font_size = 16.0
2138
+ element.update(
2139
+ {
2140
+ "textType": content_node.attrib.get("textType") if content_node is not None else None,
2141
+ "textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
2142
+ "autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
2143
+ "fontSize": base_font_size,
2144
+ "text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
2145
+ }
2146
+ )
2147
+ if not has_text_content(element):
2148
+ continue
2149
+ declared_font_sizes = []
2150
+ for descendant in node.iter():
2151
+ raw_declared_font_size = descendant.attrib.get("fontSize")
2152
+ if raw_declared_font_size is None:
2153
+ continue
2154
+ try:
2155
+ declared_font_sizes.append(float(raw_declared_font_size))
2156
+ except ValueError:
2157
+ continue
2158
+ if declared_font_sizes:
2159
+ element["fontSize"] = max(declared_font_sizes)
2160
+ for source_kind_index, match in enumerate(
2161
+ re.finditer(r"<icon\b([^>]*)>", slide_xml), start=1
2162
+ ):
2163
+ attrs = match.group(1)
2164
+ source_id = extract_attribute(attrs, "id") or None
2165
+ x = extract_numeric_attribute(attrs, "topLeftX")
2166
+ y = extract_numeric_attribute(attrs, "topLeftY")
2167
+ width = extract_numeric_attribute(attrs, "width")
2168
+ height = extract_numeric_attribute(attrs, "height")
2169
+ if any(value is None for value in (x, y, width, height)):
2170
+ continue
2171
+ icon_alpha = extract_numeric_attribute(attrs, "alpha")
2172
+ elements.append(
2173
+ {
2174
+ "id": source_id or f"icon-{len(elements) + 1}",
2175
+ "_source_id": source_id,
2176
+ "kind": "icon",
2177
+ "type": "icon",
2178
+ "x": x,
2179
+ "y": y,
2180
+ "width": width,
2181
+ "height": height,
2182
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2183
+ "alpha": icon_alpha if icon_alpha is not None else 1,
2184
+ "order": len(elements),
2185
+ "_source_kind_index": source_kind_index,
2186
+ }
2187
+ )
2188
+ for source_kind_index, match in enumerate(
2189
+ re.finditer(r"<polyline\b([^>]*)>", slide_xml), start=1
2190
+ ):
2191
+ attrs = match.group(1)
2192
+ x = extract_numeric_attribute(attrs, "topLeftX")
2193
+ y = extract_numeric_attribute(attrs, "topLeftY")
2194
+ width = extract_numeric_attribute(attrs, "width")
2195
+ height = extract_numeric_attribute(attrs, "height")
2196
+ if any(value is None for value in (x, y, width, height)):
2197
+ continue
2198
+ polyline_alpha = extract_numeric_attribute(attrs, "alpha")
2199
+ source_id = extract_attribute(attrs, "id") or None
2200
+ elements.append(
2201
+ {
2202
+ "id": source_id or f"polyline-{len(elements) + 1}",
2203
+ "_source_id": source_id,
2204
+ "kind": "polyline",
2205
+ "type": "polyline",
2206
+ "x": x,
2207
+ "y": y,
2208
+ "width": width,
2209
+ "height": height,
2210
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
2211
+ "alpha": polyline_alpha if polyline_alpha is not None else 1,
2212
+ "order": len(elements),
2213
+ "_source_kind_index": source_kind_index,
2214
+ }
2215
+ )
2216
+ for line_element in extract_line_elements(slide_xml):
2217
+ line_element["order"] = len(elements)
2218
+ elements.append(line_element)
2219
+ attach_source_xml_paths(elements, source_paths)
2220
+ for element in elements:
2221
+ element["_slide_number"] = slide_number
2222
+ return elements
2223
+
2224
+
2225
+ def is_visually_rendered(element: dict[str, Any]) -> bool:
2226
+ return element.get("alpha", 1) > 0
2227
+
2228
+
2229
+ def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
2230
+ if not is_visually_rendered(element):
2231
+ return None
2232
+ if is_text_element(element):
2233
+ estimated = estimate_text_visual_bbox(element)
2234
+ return clipped_bbox(estimated, container) if estimated else None
2235
+ return clipped_bbox(element, container)
2236
+
2237
+
2238
+ def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
2239
+ if container["kind"] != "shape" or not has_text_content(container):
2240
+ return None
2241
+ text_proxy = {**container, "type": "text"}
2242
+ estimated = estimate_text_visual_bbox(text_proxy)
2243
+ return clipped_bbox(estimated, container) if estimated else None
2244
+
2245
+
2246
+ def slide_content_visual_bbox(
2247
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
2248
+ ) -> dict[str, int | float] | None:
2249
+ if not is_visually_rendered(element):
2250
+ return None
2251
+ if is_text_element(element):
2252
+ estimated = estimate_text_visual_bbox(element)
2253
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
2254
+ if element["kind"] == "shape" and has_text_content(element):
2255
+ estimated = own_text_visual_bbox(element)
2256
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
2257
+ if element["kind"] == "line":
2258
+ # a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
2259
+ # treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
2260
+ return clipped_bbox(line_stroke_bbox(element), slide_bbox)
2261
+ if element["kind"] in {"img", "chart", "table", "whiteboard", "embed", "icon", "polyline"}:
2262
+ return clipped_bbox(element, slide_bbox)
2263
+ return None
2264
+
2265
+
2266
+ def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
2267
+ return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
2268
+
2269
+
2270
+ def is_slide_content_present(
2271
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
2272
+ ) -> bool:
2273
+ # Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
2274
+ # *anything* rendered here", not the richer "counts toward meaningful content density" bar
2275
+ # that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
2276
+ # decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
2277
+ # count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
2278
+ # instead of maintaining an allow-list that silently treats unlisted kinds as blank.
2279
+ if not is_visually_rendered(element):
2280
+ return False
2281
+ if (
2282
+ element["kind"] == "shape"
2283
+ and element["type"] == "rect"
2284
+ and not has_text_content(element)
2285
+ and element["x"] <= 2
2286
+ and element["y"] <= 2
2287
+ and element["width"] >= slide_bbox["width"] - 4
2288
+ and element["height"] >= slide_bbox["height"] - 4
2289
+ ):
2290
+ # A full-canvas plain rect is a background panel, not content -- same reasoning as
2291
+ # is_layout_container's existing background exclusion. A slide with nothing else on it
2292
+ # is still effectively blank.
2293
+ return False
2294
+ bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
2295
+ return clipped_bbox(bbox, slide_bbox) is not None
2296
+
2297
+
2298
+ def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
2299
+ if element["kind"] not in {"img", "chart", "table", "whiteboard", "embed"}:
2300
+ return False
2301
+ if not is_visually_rendered(element):
2302
+ return False
2303
+ return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
2304
+
2305
+
2306
+ def detect_sparse_container_content(
2307
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2308
+ ) -> list[dict[str, Any]]:
2309
+ issues: list[dict[str, Any]] = []
2310
+ for container in (
2311
+ element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
2312
+ ):
2313
+ if (
2314
+ is_edge_spanning_layout_panel(container, slide_width, slide_height)
2315
+ or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
2316
+ or has_matching_image_overlay(container, elements)
2317
+ ):
2318
+ continue
2319
+ children = [
2320
+ element
2321
+ for element in elements
2322
+ if element is not container
2323
+ and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
2324
+ ]
2325
+ if any(is_large_visual_child(child, container) for child in children):
2326
+ continue
2327
+ own_text_bbox = own_text_visual_bbox(container)
2328
+ rectangles = ([own_text_bbox] if own_text_bbox else []) + [
2329
+ bbox for child in children if (bbox := visual_bbox(child, container)) is not None
2330
+ ]
2331
+ content_area = rectangle_union_area(rectangles) if rectangles else 0
2332
+ coverage_ratio = content_area / element_area(container)
2333
+ if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
2334
+ continue
2335
+ issues.append(
2336
+ {
2337
+ "level": "warning",
2338
+ "code": "sparse_container_content",
2339
+ "target": {
2340
+ "slide_number": slide_number,
2341
+ **(
2342
+ {"container_id": source_element_id(container)}
2343
+ if source_element_id(container) is not None
2344
+ else {}
2345
+ ),
2346
+ "container_xml_path": element_ref(container),
2347
+ "container_type": container["type"],
2348
+ "bbox": {key: container[key] for key in ("x", "y", "width", "height")},
2349
+ },
2350
+ "rule": {
2351
+ "name": "large_container_visible_content_coverage",
2352
+ "threshold": MIN_CONTENT_COVERAGE_RATIO,
2353
+ "comparison": "content_coverage_ratio < threshold",
2354
+ },
2355
+ "measurement": {
2356
+ "container_area": element_area(container),
2357
+ "visible_content_area": round(content_area, 3),
2358
+ "content_coverage_ratio": round(coverage_ratio, 3),
2359
+ "content_element_count": len(children) + (1 if own_text_bbox else 0),
2360
+ },
2361
+ "elements": [
2362
+ element_ref(container),
2363
+ *[element_ref(child) for child in children],
2364
+ ],
2365
+ }
2366
+ )
2367
+ return issues
2368
+
2369
+
2370
+ def detect_sparse_slide_content(
2371
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
2372
+ ) -> list[dict[str, Any]]:
2373
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2374
+ content = [
2375
+ (element, bbox)
2376
+ for element in elements
2377
+ if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
2378
+ ]
2379
+ if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
2380
+ return []
2381
+ content_area = rectangle_union_area([bbox for _, bbox in content])
2382
+ slide_area = slide_width * slide_height
2383
+ coverage_ratio = content_area / slide_area
2384
+ if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
2385
+ return []
2386
+ return [
2387
+ {
2388
+ "level": "warning",
2389
+ "code": "sparse_slide_content",
2390
+ "target": {
2391
+ "slide_number": slide_number,
2392
+ "bbox": slide_bbox,
2393
+ },
2394
+ "rule": {
2395
+ "name": "slide_visible_content_coverage",
2396
+ "threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
2397
+ "comparison": "content_coverage_ratio < threshold",
2398
+ },
2399
+ "measurement": {
2400
+ "slide_area": slide_area,
2401
+ "visible_content_area": round(content_area, 3),
2402
+ "content_coverage_ratio": round(coverage_ratio, 3),
2403
+ "content_element_count": len(content),
2404
+ },
2405
+ "elements": [element_ref(element) for element, _ in content],
2406
+ }
2407
+ ]
2408
+
2409
+
2410
+ def detect_blank_slide(
2411
+ elements: list[dict[str, Any]],
2412
+ slide_number: int,
2413
+ slide_width: int | float,
2414
+ slide_height: int | float,
2415
+ ) -> list[dict[str, Any]]:
2416
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
2417
+ visible_elements = [
2418
+ element for element in elements if is_slide_content_present(element, slide_bbox)
2419
+ ]
2420
+ if visible_elements:
2421
+ return []
2422
+ return [
2423
+ {
2424
+ "level": "error",
2425
+ "code": "blank_slide",
2426
+ "schema_version": "2.0",
2427
+ "target": {"slide_number": slide_number},
2428
+ "rule": {
2429
+ "name": "slide_has_visible_content",
2430
+ "comparison": "visible_element_count == 0",
2431
+ },
2432
+ "measurement": {
2433
+ "visible_element_count": 0,
2434
+ "declared_element_count": len(elements),
2435
+ },
2436
+ "elements": [element_ref(element) for element in elements],
2437
+ "message": "slide has no visible content beyond empty layout shapes",
2438
+ "hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
2439
+ }
2440
+ ]
2441
+
2442
+
2443
+ def detect_duplicate_element_ids(
2444
+ elements: list[dict[str, Any]], *, cross_slide_only: bool = False
2445
+ ) -> list[dict[str, Any]]:
2446
+ elements_by_source_id: dict[str, list[dict[str, Any]]] = {}
2447
+ for element in elements:
2448
+ source_id = source_element_id(element)
2449
+ if source_id is not None:
2450
+ elements_by_source_id.setdefault(source_id, []).append(element)
2451
+ return [
2452
+ {
2453
+ "level": "error",
2454
+ "code": "duplicate_element_id",
2455
+ "elements": [element_ref(element) for element in duplicates],
2456
+ "measurement": {
2457
+ "element_id": source_id,
2458
+ "duplicate_count": len(duplicates),
2459
+ },
2460
+ "message": f'element id "{source_id}" is used by {len(duplicates)} elements',
2461
+ "hint": (
2462
+ "Do not invent replacement IDs. For newly authored elements, remove the id attribute. "
2463
+ "When updating read-back XML, keep the server ID on the original element only and remove it "
2464
+ "from copied or new elements."
2465
+ ),
2466
+ }
2467
+ for source_id, duplicates in elements_by_source_id.items()
2468
+ if len(duplicates) > 1
2469
+ and (
2470
+ not cross_slide_only
2471
+ or len({element.get("_slide_number") for element in duplicates}) > 1
2472
+ )
2473
+ ]
2474
+
2475
+
2476
+ RULE_METADATA: dict[str, dict[str, Any]] = {
2477
+ "xml_not_well_formed": {
2478
+ "name": "xml_is_well_formed",
2479
+ "comparison": "xml_parse_error == false",
2480
+ },
2481
+ "sml_prefixed_tag": {
2482
+ "name": "sml_uses_default_namespace",
2483
+ "comparison": "prefixed_sml_tag_count == 0",
2484
+ },
2485
+ "sxsd_unsupported_tag": {
2486
+ "name": "tag_is_supported_by_slides_xml_schema",
2487
+ "comparison": "unsupported_tag_count == 0",
2488
+ },
2489
+ "sxsd_unsupported_attr": {
2490
+ "name": "attribute_is_supported_by_slides_xml_schema",
2491
+ "comparison": "unsupported_attribute_count == 0",
2492
+ },
2493
+ "icon_missing_fill_color": {
2494
+ "name": "icon_has_visible_fill_color",
2495
+ "comparison": "fill_color_present == true",
2496
+ },
2497
+ "icon_transparent_fill_color": {
2498
+ "name": "icon_has_visible_fill_color",
2499
+ "comparison": "fill_alpha > 0",
2500
+ },
2501
+ "iconpark_unsupported_icon_type": {
2502
+ "name": "iconpark_type_is_supported",
2503
+ "comparison": "icon_type in iconpark_index",
2504
+ },
2505
+ "bbox_overlap": {
2506
+ "name": "text_visual_bounds_do_not_overlap",
2507
+ "comparison": "intersection_area == 0",
2508
+ },
2509
+ "text_may_overflow_shape": {
2510
+ "name": "estimated_text_fits_declared_shape",
2511
+ "comparison": "estimated_height <= available_height",
2512
+ },
2513
+ "whiteboard_external_overlap": {
2514
+ "name": "whiteboard_does_not_cross_sibling_content",
2515
+ "comparison": "external_overlap_count == 0",
2516
+ },
2517
+ "image_covers_text": {
2518
+ "name": "image_does_not_cover_text",
2519
+ "comparison": "intersection_area == 0",
2520
+ },
2521
+ "image_may_cover_vertical_text": {
2522
+ "name": "image_vertical_text_occlusion_requires_review",
2523
+ "comparison": "intersection_area == 0",
2524
+ },
2525
+ "table_resolved_size_mismatch": {
2526
+ "name": "table_declared_size_matches_resolved_grid",
2527
+ "comparison": "declared_size == resolved_size",
2528
+ },
2529
+ "blank_slide": {
2530
+ "name": "slide_has_visible_content",
2531
+ "comparison": "visible_element_count > 0",
2532
+ },
2533
+ "duplicate_element_id": {
2534
+ "name": "element_ids_are_unique",
2535
+ "comparison": "duplicate_count == 0",
2536
+ },
2537
+ }
2538
+
2539
+
2540
+ def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
2541
+ if issue.get("rule"):
2542
+ return {**issue["rule"], "id": issue["code"]}
2543
+ if issue["code"].endswith("_out_of_canvas"):
1141
2544
  return {
1142
- "file": source_path,
1143
- "slide_size": {"width": 960, "height": 540},
1144
- "summary": {"slide_count": 0, "error_count": 1, "warning_count": 0, "info_count": 0},
1145
- "issues": [xml_error],
1146
- "slides": [],
2545
+ "id": issue["code"],
2546
+ "name": "element_stays_within_slide_canvas",
2547
+ "comparison": "max(left, top, right, bottom overflow) == 0",
1147
2548
  }
2549
+ return {
2550
+ "id": issue["code"],
2551
+ **RULE_METADATA.get(
2552
+ issue["code"],
2553
+ {"name": issue["code"], "comparison": "violation_count == 0"},
2554
+ ),
2555
+ }
1148
2556
 
1149
- namespace_issues = validate_sml_tag_prefixes(xml)
1150
- sxsd_issues = validate_sxsd_tag_attributes(root) if root is not None else []
1151
- iconpark_issues = validate_iconpark_icon_types(root) if root is not None else []
1152
- top_level_issues = [*namespace_issues, *sxsd_issues, *iconpark_issues]
1153
- if namespace_issues:
1154
- error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
1155
- warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
1156
- info_count = sum(1 for issue in top_level_issues if issue["level"] == "info")
2557
+
2558
+ def issue_measurement(
2559
+ issue: dict[str, Any], elements_by_ref: dict[str, dict[str, Any]]
2560
+ ) -> dict[str, Any]:
2561
+ if issue.get("measurement") is not None:
2562
+ return issue["measurement"]
2563
+ if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
2564
+ left = elements_by_ref.get(issue["elements"][0])
2565
+ right = elements_by_ref.get(issue["elements"][1])
2566
+ if left and right:
2567
+ left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
2568
+ right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
2569
+ width = intersection_width(left_box, right_box)
2570
+ height = intersection_height(left_box, right_box)
2571
+ return {
2572
+ "intersection_width": round(width, 3),
2573
+ "intersection_height": round(height, 3),
2574
+ "intersection_area": round(width * height, 3),
2575
+ }
2576
+ if issue["code"].endswith("_out_of_canvas"):
1157
2577
  return {
1158
- "file": source_path,
1159
- "slide_size": {"width": 960, "height": 540},
1160
- "summary": {
1161
- "slide_count": 0,
1162
- "error_count": error_count,
1163
- "warning_count": warning_count,
1164
- "info_count": info_count,
1165
- },
1166
- "issues": top_level_issues,
1167
- "slides": [],
2578
+ "canvas": issue.get("canvas"),
2579
+ "bbox": issue.get("bbox"),
2580
+ "overflow": issue.get("overflow"),
1168
2581
  }
1169
- presentation = parse_presentation(xml)
1170
- slides = [
1171
- lint_slide(slide_xml, index + 1, presentation["width"], presentation["height"])
1172
- for index, slide_xml in enumerate(presentation["slides"])
2582
+ measurement_keys = (
2583
+ "line",
2584
+ "column",
2585
+ "tag",
2586
+ "attr",
2587
+ "iconType",
2588
+ "line_count",
2589
+ "line_height",
2590
+ "estimated_height",
2591
+ "available_height",
2592
+ "overflow",
2593
+ "dimension",
2594
+ "declared_size",
2595
+ "resolved_size",
2596
+ "resolved_sizes",
2597
+ "overlaps",
2598
+ )
2599
+ measured = {key: issue[key] for key in measurement_keys if key in issue}
2600
+ return measured or {"violation_count": 1}
2601
+
2602
+
2603
+ def related_object(element: dict[str, Any]) -> dict[str, Any]:
2604
+ related = {
2605
+ "kind": element["kind"],
2606
+ "type": element["type"],
2607
+ }
2608
+ bbox_keys = ("x", "y", "width", "height")
2609
+ if all(key in element for key in bbox_keys):
2610
+ related["bbox"] = {key: element[key] for key in bbox_keys}
2611
+ if source_element_id(element) is not None:
2612
+ related["element_id"] = source_element_id(element)
2613
+ if element.get("xml_path"):
2614
+ related["xml_path"] = element["xml_path"]
2615
+ return related
2616
+
2617
+
2618
+ def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
2619
+ elements: list[dict[str, Any]] = []
2620
+ for source_kind_index, match in enumerate(
2621
+ re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml), start=1
2622
+ ):
2623
+ attrs = match.group(1)
2624
+ source_id = extract_attribute(attrs, "id") or None
2625
+ start_x = extract_numeric_attribute(attrs, "startX")
2626
+ start_y = extract_numeric_attribute(attrs, "startY")
2627
+ end_x = extract_numeric_attribute(attrs, "endX")
2628
+ end_y = extract_numeric_attribute(attrs, "endY")
2629
+ if any(value is None for value in (start_x, start_y, end_x, end_y)):
2630
+ continue
2631
+ line_alpha = extract_numeric_attribute(attrs, "alpha")
2632
+ base_alpha = line_alpha if line_alpha is not None else 1
2633
+ border_alpha = 1
2634
+ if match.group(2) != "/":
2635
+ close_index = slide_xml.find("</line>", match.end())
2636
+ body = slide_xml[match.end() : close_index] if close_index != -1 else ""
2637
+ border_attrs = extract_tag_attributes(body, "border")
2638
+ color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
2639
+ if isinstance(color_alpha, (int, float)):
2640
+ border_alpha = color_alpha
2641
+ elements.append(
2642
+ {
2643
+ "id": source_id or f"line-{len(elements) + 1}",
2644
+ "_source_id": source_id,
2645
+ "kind": "line",
2646
+ "type": "line",
2647
+ "x": min(start_x, end_x),
2648
+ "y": min(start_y, end_y),
2649
+ "width": abs(end_x - start_x),
2650
+ "height": abs(end_y - start_y),
2651
+ "startX": start_x,
2652
+ "startY": start_y,
2653
+ "endX": end_x,
2654
+ "endY": end_y,
2655
+ "rotation": 0,
2656
+ "alpha": base_alpha * border_alpha,
2657
+ "order": len(elements),
2658
+ "_source_kind_index": source_kind_index,
2659
+ }
2660
+ )
2661
+ return elements
2662
+
2663
+
2664
+ def normalize_issue(
2665
+ issue: dict[str, Any],
2666
+ slide_number: int | None,
2667
+ elements_by_ref: dict[str, dict[str, Any]],
2668
+ ) -> dict[str, Any]:
2669
+ normalized = dict(issue)
2670
+ element_refs = list(dict.fromkeys(normalized.get("elements", [])))
2671
+ resolved_elements = [
2672
+ elements_by_ref[element_ref]
2673
+ for element_ref in element_refs
2674
+ if element_ref in elements_by_ref
2675
+ ]
2676
+ element_locators = [
2677
+ source_element_id(elements_by_ref[element_ref]) or element_ref
2678
+ if element_ref in elements_by_ref
2679
+ else element_ref
2680
+ for element_ref in element_refs
2681
+ ]
2682
+ element_ids = [
2683
+ source_id
2684
+ for element in resolved_elements
2685
+ if (source_id := source_element_id(element)) is not None
1173
2686
  ]
1174
- error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
1175
- error_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "error")
1176
- warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
1177
- warning_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "warning")
1178
- info_count = sum(1 for issue in top_level_issues if issue["level"] == "info")
1179
- info_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "info")
1180
- result = {
2687
+ normalized["schema_version"] = "2.0"
2688
+ normalized["elements"] = element_locators
2689
+ normalized["element_ids"] = element_ids
2690
+ normalized["target"] = {
2691
+ **({"slide_number": slide_number} if slide_number is not None else {}),
2692
+ **normalized.get("target", {}),
2693
+ }
2694
+ normalized["rule"] = issue_rule(normalized)
2695
+ normalized["measurement"] = issue_measurement(issue, elements_by_ref)
2696
+ normalized["related_objects"] = [related_object(element) for element in resolved_elements]
2697
+ if normalized["code"] == "sparse_container_content":
2698
+ ratio = normalized["measurement"]["content_coverage_ratio"]
2699
+ threshold = normalized["rule"]["threshold"]
2700
+ container_locator = (
2701
+ normalized["target"].get("container_id")
2702
+ or normalized["target"].get("container_xml_path")
2703
+ or "unknown"
2704
+ )
2705
+ normalized.setdefault(
2706
+ "message",
2707
+ f"large card {container_locator} content coverage {ratio:.1%} is below {threshold:.1%}",
2708
+ )
2709
+ normalized.setdefault(
2710
+ "hint",
2711
+ "Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
2712
+ )
2713
+ elif normalized["code"] == "sparse_slide_content":
2714
+ ratio = normalized["measurement"]["content_coverage_ratio"]
2715
+ threshold = normalized["rule"]["threshold"]
2716
+ normalized.setdefault(
2717
+ "message",
2718
+ f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
2719
+ )
2720
+ normalized.setdefault(
2721
+ "hint",
2722
+ "Review the rendered screenshot to decide whether the page is intentionally sparse.",
2723
+ )
2724
+ else:
2725
+ normalized.setdefault("message", normalized["code"].replace("_", " "))
2726
+ normalized.setdefault(
2727
+ "hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
2728
+ )
2729
+ if any(related.get("xml_path") for related in normalized["related_objects"]):
2730
+ hint = normalized["hint"]
2731
+ if not hint.startswith(XML_PATH_HINT_PREFIX):
2732
+ normalized["hint"] = f"{XML_PATH_HINT_PREFIX} {hint}"
2733
+ return normalized
2734
+
2735
+
2736
+ def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
2737
+ if errors:
2738
+ return "blocked"
2739
+ if warnings:
2740
+ return "needs_screenshot_review"
2741
+ return "passed"
2742
+
2743
+
2744
+ def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
2745
+ if issue.get("code") == "sxsd_unsupported_declaration":
2746
+ return False
2747
+ if root_name == "slide":
2748
+ return True
2749
+ path = issue.get("path")
2750
+ if not isinstance(path, str):
2751
+ return False
2752
+ if path.startswith("presentation/slide/"):
2753
+ return True
2754
+ return path == "presentation/slide" and (
2755
+ issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
2756
+ )
2757
+
2758
+
2759
+ def build_result(
2760
+ source_path: str | None,
2761
+ slide_size: dict[str, int | float],
2762
+ top_level_issues: list[dict[str, Any]],
2763
+ slides: list[dict[str, Any]],
2764
+ ) -> dict[str, Any]:
2765
+ document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
2766
+ document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
2767
+ document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
2768
+ error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
2769
+ warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
2770
+ info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
2771
+ all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
2772
+ all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
2773
+ status = slide_status(all_errors, all_warnings)
2774
+ result: dict[str, Any] = {
2775
+ "schema_version": "2.0",
2776
+ "tool": "xml_text_overlap_lint",
1181
2777
  "file": source_path,
1182
- "slide_size": {"width": presentation["width"], "height": presentation["height"]},
2778
+ "slide_size": slide_size,
1183
2779
  "summary": {
1184
2780
  "slide_count": len(slides),
1185
2781
  "error_count": error_count,
1186
2782
  "warning_count": warning_count,
1187
2783
  "info_count": info_count,
2784
+ "status": status,
2785
+ "release_ready": error_count == 0,
2786
+ "screenshot_review_required": warning_count > 0,
2787
+ },
2788
+ "document": {
2789
+ "errors": document_errors,
2790
+ "warnings": document_warnings,
2791
+ "infos": document_infos,
1188
2792
  },
1189
2793
  "slides": slides,
1190
2794
  }
@@ -1193,6 +2797,170 @@ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
1193
2797
  return result
1194
2798
 
1195
2799
 
2800
+ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
2801
+ root, xml_error = parse_xml_root(xml)
2802
+ if xml_error:
2803
+ issue = normalize_issue(xml_error, None, {})
2804
+ return build_result(
2805
+ source_path,
2806
+ {"width": 960, "height": 540},
2807
+ [issue],
2808
+ [],
2809
+ )
2810
+ if root is None:
2811
+ raise AssertionError("parse_xml_root must return a root or error")
2812
+
2813
+ namespace_issues = validate_sml_tag_prefixes(xml)
2814
+ root_name = xml_local_name(root.tag)
2815
+ sxsd_issues = validate_sxsd_document(xml, root)
2816
+ iconpark_issues = validate_iconpark_icon_types(root)
2817
+ top_level_issues = [
2818
+ normalize_issue(issue, None, {})
2819
+ for issue in [
2820
+ *namespace_issues,
2821
+ *[
2822
+ issue
2823
+ for issue in sxsd_issues
2824
+ if not is_slide_scoped_sxsd_issue(issue, root_name)
2825
+ ],
2826
+ *iconpark_issues,
2827
+ ]
2828
+ ]
2829
+ if any(issue["level"] == "error" for issue in top_level_issues):
2830
+ return build_result(
2831
+ source_path,
2832
+ {"width": 960, "height": 540},
2833
+ top_level_issues,
2834
+ [],
2835
+ )
2836
+
2837
+ presentation = parse_presentation(root)
2838
+ slide_roots = presentation["slide_roots"]
2839
+ slides: list[dict[str, Any]] = []
2840
+ presentation_id_elements: list[dict[str, Any]] = []
2841
+ presentation_elements_by_ref: dict[str, dict[str, Any]] = {}
2842
+ for index, slide_xml in enumerate(presentation["slides"]):
2843
+ slide_number = index + 1
2844
+ slide_root = slide_roots[index]
2845
+ slide_sxsd_issues = [
2846
+ normalize_issue(issue, slide_number, {})
2847
+ for issue in validate_sxsd_document(slide_xml, slide_root)
2848
+ ]
2849
+ slide_sxsd_errors = [
2850
+ issue for issue in slide_sxsd_issues if issue["level"] == "error"
2851
+ ]
2852
+ if slide_sxsd_errors:
2853
+ slide_sxsd_warnings = [
2854
+ issue for issue in slide_sxsd_issues if issue["level"] == "warning"
2855
+ ]
2856
+ slides.append(
2857
+ {
2858
+ "slide_number": slide_number,
2859
+ "status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
2860
+ "element_count": 0,
2861
+ "errors": slide_sxsd_errors,
2862
+ "warnings": slide_sxsd_warnings,
2863
+ "infos": [],
2864
+ "issues": slide_sxsd_issues,
2865
+ }
2866
+ )
2867
+ continue
2868
+
2869
+ geometry = lint_slide(
2870
+ slide_xml,
2871
+ slide_number,
2872
+ presentation["width"],
2873
+ presentation["height"],
2874
+ )
2875
+ density_elements = extract_density_elements(slide_xml, slide_number)
2876
+ id_elements = extract_source_id_elements(slide_xml, slide_number)
2877
+ presentation_id_elements.extend(id_elements)
2878
+ extra_elements = [
2879
+ element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
2880
+ ]
2881
+ elements_by_ref = {
2882
+ element_ref(element): element for element in density_elements
2883
+ }
2884
+ visible_element_count = len(elements_by_ref)
2885
+ for element in id_elements:
2886
+ elements_by_ref.setdefault(element_ref(element), element)
2887
+ # geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
2888
+ # selected inside lint_slide; prefer them so measurement/related_objects stay consistent
2889
+ # with whatever actually triggered the issue, instead of density_elements' separate re-parse.
2890
+ elements_by_ref.update(
2891
+ {element_ref(element): element for element in geometry["elements"]}
2892
+ )
2893
+ presentation_elements_by_ref.update(
2894
+ {
2895
+ element_ref(element): elements_by_ref[element_ref(element)]
2896
+ for element in id_elements
2897
+ }
2898
+ )
2899
+ extra_overflow_issues = detect_elements_out_of_canvas(
2900
+ extra_elements,
2901
+ presentation["width"],
2902
+ presentation["height"],
2903
+ )
2904
+ raw_issues = [
2905
+ *geometry["issues"],
2906
+ *extra_overflow_issues,
2907
+ *detect_duplicate_element_ids(id_elements),
2908
+ *detect_blank_slide(
2909
+ density_elements,
2910
+ slide_number,
2911
+ presentation["width"],
2912
+ presentation["height"],
2913
+ ),
2914
+ *detect_sparse_container_content(
2915
+ density_elements,
2916
+ slide_number,
2917
+ presentation["width"],
2918
+ presentation["height"],
2919
+ ),
2920
+ *detect_sparse_slide_content(
2921
+ density_elements,
2922
+ slide_number,
2923
+ presentation["width"],
2924
+ presentation["height"],
2925
+ ),
2926
+ ]
2927
+ issues = [
2928
+ *slide_sxsd_issues,
2929
+ *[
2930
+ normalize_issue(issue, slide_number, elements_by_ref)
2931
+ for issue in raw_issues
2932
+ ],
2933
+ ]
2934
+ errors = [issue for issue in issues if issue["level"] == "error"]
2935
+ warnings = [issue for issue in issues if issue["level"] == "warning"]
2936
+ infos = [issue for issue in issues if issue["level"] == "info"]
2937
+ slides.append(
2938
+ {
2939
+ "slide_number": slide_number,
2940
+ "status": slide_status(errors, warnings),
2941
+ "element_count": visible_element_count,
2942
+ "errors": errors,
2943
+ "warnings": warnings,
2944
+ "infos": infos,
2945
+ "issues": issues,
2946
+ }
2947
+ )
2948
+
2949
+ top_level_issues.extend(
2950
+ normalize_issue(issue, None, presentation_elements_by_ref)
2951
+ for issue in detect_duplicate_element_ids(
2952
+ presentation_id_elements, cross_slide_only=True
2953
+ )
2954
+ )
2955
+
2956
+ return build_result(
2957
+ source_path,
2958
+ {"width": presentation["width"], "height": presentation["height"]},
2959
+ top_level_issues,
2960
+ slides,
2961
+ )
2962
+
2963
+
1196
2964
  def print_usage() -> None:
1197
2965
  print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
1198
2966
 
@@ -1205,8 +2973,9 @@ def run_cli(argv: list[str] | None = None) -> None:
1205
2973
  if not options.get("input"):
1206
2974
  print_usage()
1207
2975
  fail("--input is required")
1208
- input_path = Path(options["input"]).resolve()
1209
- result = lint_xml(read_file(input_path), str(input_path))
2976
+ requested_path = options["input"]
2977
+ resolved_path = Path(requested_path).resolve()
2978
+ result = lint_xml(read_file(resolved_path), requested_path)
1210
2979
  print(json.dumps(result, ensure_ascii=False, indent=2))
1211
2980
  if result["summary"]["error_count"] > 0:
1212
2981
  raise SystemExit(1)
@@ -1215,6 +2984,6 @@ def run_cli(argv: list[str] | None = None) -> None:
1215
2984
  if __name__ == "__main__":
1216
2985
  try:
1217
2986
  run_cli()
1218
- except XmlTextOverlapLintError as error:
2987
+ except XmlLayoutLintError as error:
1219
2988
  print(f"xml-text-overlap-lint error: {error}", file=sys.stderr)
1220
2989
  raise SystemExit(1) from error