@amaster.ai/pi-lark 0.1.6 → 0.1.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +2 -2
- package/dist/config.js.map +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/package.json +3 -3
- package/skills/lark-apps/SKILL.md +24 -12
- package/skills/lark-apps/creative-design/agents/assets/vision-probe.png +0 -0
- package/skills/lark-apps/creative-design/agents/fork-verifier-agent.md +71 -0
- package/skills/lark-apps/creative-design/agents/vision-probe-agent.md +41 -0
- package/skills/lark-apps/creative-design/assets/index.html +27 -0
- package/skills/lark-apps/creative-design/creative-design.md +239 -0
- package/skills/lark-apps/creative-design/references/aily.md +39 -0
- package/skills/lark-apps/creative-design/references/animated-video.md +34 -0
- package/skills/lark-apps/creative-design/references/charts.md +165 -0
- package/skills/lark-apps/creative-design/references/claude.md +36 -0
- package/skills/lark-apps/creative-design/references/codex.md +32 -0
- package/skills/lark-apps/creative-design/references/data-report.md +108 -0
- package/skills/lark-apps/creative-design/references/frontend-design.md +71 -0
- package/skills/lark-apps/creative-design/references/hi-fi-design.md +32 -0
- package/skills/lark-apps/creative-design/references/interactive-prototype.md +24 -0
- package/skills/lark-apps/creative-design/references/make-a-deck.md +133 -0
- package/skills/lark-apps/creative-design/references/visual-exposure.md +82 -0
- package/skills/lark-apps/creative-design/references/wireframe.md +14 -0
- package/skills/lark-apps/creative-design/starter-components/android-frame.jsx +188 -0
- package/skills/lark-apps/creative-design/starter-components/animations.jsx +773 -0
- package/skills/lark-apps/creative-design/starter-components/browser-window.jsx +122 -0
- package/skills/lark-apps/creative-design/starter-components/deck-stage.js +2483 -0
- package/skills/lark-apps/creative-design/starter-components/design-canvas.jsx +1432 -0
- package/skills/lark-apps/creative-design/starter-components/ios-frame.jsx +270 -0
- package/skills/lark-apps/creative-design/starter-components/macos-window.jsx +197 -0
- package/skills/lark-apps/creative-design/starter-components/tweaks-panel.jsx +752 -0
- package/skills/lark-apps/references/lark-apps-automation.md +80 -2
- package/skills/lark-apps/references/lark-apps-cache.md +61 -0
- package/skills/lark-apps/references/lark-apps-cloud-dev.md +0 -1
- package/skills/lark-apps/references/lark-apps-create.md +1 -2
- package/skills/lark-apps/references/lark-apps-db.md +1 -1
- package/skills/lark-apps/references/lark-apps-env-pull.md +1 -1
- package/skills/lark-apps/references/lark-apps-file.md +2 -2
- package/skills/lark-apps/references/lark-apps-git-credential.md +1 -1
- package/skills/lark-apps/references/lark-apps-html-publish.md +4 -8
- package/skills/lark-apps/references/lark-apps-init.md +1 -1
- package/skills/lark-apps/references/lark-apps-list.md +1 -1
- package/skills/lark-apps/references/lark-apps-local-dev.md +54 -11
- package/skills/lark-apps/references/lark-apps-openapi-key.md +1 -1
- package/skills/lark-apps/references/lark-apps-release-create.md +2 -2
- package/skills/lark-apps/references/lark-apps-release-get.md +3 -3
- package/skills/lark-base/SKILL.md +20 -13
- package/skills/lark-base/references/lark-base-cell-value.md +3 -3
- package/skills/lark-base/references/lark-base-data-query.md +11 -4
- package/skills/lark-base/references/lark-base-field-create.md +4 -0
- package/skills/lark-base/references/lark-base-field-json.md +4 -4
- package/skills/lark-base/references/lark-base-field-update.md +17 -1
- package/skills/lark-base/references/lark-base-filter-condition.md +179 -0
- package/skills/lark-base/references/lark-base-form-questions-create.md +40 -7
- package/skills/lark-base/references/lark-base-form-questions-update.md +73 -20
- package/skills/lark-base/references/lark-base-form-submit.md +16 -7
- package/skills/lark-base/references/lark-base-record-batch-create.md +12 -10
- package/skills/lark-base/references/lark-base-record-batch-update.md +11 -9
- package/skills/lark-base/references/lark-base-record-upsert.md +1 -1
- package/skills/lark-base/references/lark-base-role-guide.md +11 -0
- package/skills/lark-base/references/lark-base-view-set-filter.md +11 -137
- package/skills/lark-base/references/role-config.md +31 -5
- package/skills/lark-calendar/SKILL.md +14 -8
- package/skills/lark-calendar/references/lark-calendar-create.md +6 -5
- package/skills/lark-calendar/references/lark-calendar-recurring.md +1 -0
- package/skills/lark-calendar/references/lark-calendar-room-find.md +2 -1
- package/skills/lark-calendar/references/lark-calendar-schedule-clear-time.md +1 -0
- package/skills/lark-calendar/references/lark-calendar-suggestion.md +1 -1
- package/skills/lark-calendar/references/lark-calendar-update.md +10 -4
- package/skills/lark-contact/SKILL.md +19 -3
- package/skills/lark-contact/references/lark-contact-search-bot.md +60 -0
- package/skills/lark-doc/references/lark-doc-fetch.md +10 -2
- package/skills/lark-doc/references/lark-doc-whiteboard.md +9 -8
- package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +41 -0
- package/skills/lark-doc/references/lark-doc-xml.md +4 -3
- package/skills/lark-drive/SKILL.md +25 -45
- package/skills/lark-drive/references/lark-drive-add-comment.md +2 -4
- package/skills/lark-drive/references/lark-drive-add-reply.md +47 -0
- package/skills/lark-drive/references/lark-drive-apply-permission.md +2 -2
- package/skills/lark-drive/references/lark-drive-batch-query-comments.md +46 -0
- package/skills/lark-drive/references/lark-drive-comment-content.md +50 -0
- package/skills/lark-drive/references/lark-drive-comment-location.md +9 -15
- package/skills/lark-drive/references/lark-drive-delete-reply.md +48 -0
- package/skills/lark-drive/references/lark-drive-download.md +5 -1
- package/skills/lark-drive/references/lark-drive-list-comments.md +25 -68
- package/skills/lark-drive/references/lark-drive-list-replies.md +54 -0
- package/skills/lark-drive/references/lark-drive-member-add.md +2 -2
- package/skills/lark-drive/references/lark-drive-member-list.md +65 -0
- package/skills/lark-drive/references/lark-drive-permission-get-setting.md +48 -0
- package/skills/lark-drive/references/lark-drive-preview.md +11 -1
- package/skills/lark-drive/references/lark-drive-react-reply.md +51 -0
- package/skills/lark-drive/references/lark-drive-reactions.md +27 -25
- package/skills/lark-drive/references/lark-drive-resolve-comment.md +45 -0
- package/skills/lark-drive/references/lark-drive-restore-comment.md +46 -0
- package/skills/lark-drive/references/lark-drive-search.md +7 -1
- package/skills/lark-drive/references/lark-drive-secure-label.md +1 -1
- package/skills/lark-drive/references/lark-drive-update-reply.md +46 -0
- package/skills/lark-drive/references/lark-drive-upload.md +1 -0
- package/skills/lark-drive/references/lark-drive-workflow-permission-governance-commands.md +38 -8
- package/skills/lark-drive/references/lark-drive-workflow-permission-governance-outputs.md +10 -10
- package/skills/lark-drive/references/lark-drive-workflow-permission-governance.md +22 -20
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-execute.md +273 -0
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-recall.md +202 -0
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-resolve-verify.md +231 -0
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-review-plan.md +248 -0
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-setup.md +174 -0
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector.md +202 -0
- package/skills/lark-drive/references/lark-drive-workflow.md +5 -4
- package/skills/lark-event/SKILL.md +1 -0
- package/skills/lark-event/references/lark-event-application.md +38 -0
- package/skills/lark-im/SKILL.md +1 -1
- package/skills/lark-im/references/card/card-2.0-schema.md +1 -1
- package/skills/lark-im/references/card/lark-im-card-style.md +4 -4
- package/skills/lark-im/references/card/resource/icons.md +14 -0
- package/skills/lark-im/references/lark-im-flag-list.md +8 -7
- package/skills/lark-okr/SKILL.md +71 -26
- package/skills/lark-okr/references/lark-okr-batch-create.md +19 -18
- package/skills/lark-okr/references/lark-okr-create.md +173 -0
- package/skills/lark-okr/references/lark-okr-cycle-list.md +17 -7
- package/skills/lark-okr/references/lark-okr-entities.md +1 -0
- package/skills/lark-okr/references/lark-okr-indicator-update.md +3 -1
- package/skills/lark-okr/references/lark-okr-indicators.md +61 -12
- package/skills/lark-okr/references/lark-okr-progress-list.md +21 -9
- package/skills/lark-slides/SKILL.md +115 -68
- package/skills/lark-slides/references/asset-planning.md +6 -4
- package/skills/lark-slides/references/iconpark.md +2 -2
- package/skills/lark-slides/references/lark-slides-create.md +16 -8
- package/skills/lark-slides/references/lark-slides-history.md +132 -0
- package/skills/lark-slides/references/lark-slides-media-upload.md +2 -3
- package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +7 -11
- package/skills/lark-slides/references/lark-slides-replace-slide.md +0 -3
- package/skills/lark-slides/references/lark-slides-screenshot.md +4 -4
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +219 -0
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +6 -5
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +2 -2
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +2 -3
- package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +65 -30
- package/skills/lark-slides/references/planning-layer.md +11 -10
- package/skills/lark-slides/references/slides_chart_demo.xml +1416 -1
- package/skills/lark-slides/references/slides_xml_schema_definition.xml +492 -76
- package/skills/lark-slides/references/troubleshooting.md +25 -7
- package/skills/lark-slides/references/validation-checklist.md +53 -16
- package/skills/lark-slides/references/visual-planning.md +25 -22
- package/skills/lark-slides/references/xml-schema-quick-ref.md +281 -45
- package/skills/lark-slides/scripts/sxsd_validator.py +908 -0
- package/skills/lark-slides/scripts/xml_text_overlap_lint.py +1650 -165
- package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +3139 -513
- package/skills/lark-task/SKILL.md +7 -0
- package/skills/lark-task/references/lark-task-complete.md +6 -2
- package/skills/lark-task/references/lark-task-create.md +9 -0
- package/skills/lark-task/references/lark-task-update.md +6 -2
- package/skills/lark-whiteboard/SKILL.md +13 -12
- package/skills/lark-whiteboard/elements/layout.md +1 -1
- package/skills/lark-whiteboard/elements/schema.md +2 -2
- package/skills/lark-whiteboard/references/{lark-whiteboard-query.md → lark-whiteboard-export.md} +15 -15
- package/skills/lark-whiteboard/references/lark-whiteboard-update.md +3 -3
- package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +7 -17
- package/skills/lark-whiteboard/routes/dsl.md +3 -3
- package/skills/lark-whiteboard/routes/mermaid.md +2 -2
- package/skills/lark-whiteboard/routes/svg-edit.md +4 -4
- package/skills/lark-whiteboard/routes/svg.md +11 -6
- package/skills/lark-whiteboard/scenes/bar-chart.md +1 -1
- package/skills/lark-whiteboard/scenes/fishbone.md +1 -1
- package/skills/lark-whiteboard/scenes/flywheel.md +1 -1
- package/skills/lark-whiteboard/scenes/line-chart.md +1 -1
- package/skills/lark-whiteboard/scenes/treemap.md +1 -1
- package/skills/lark-wiki/SKILL.md +1 -0
- package/skills/lark-drive/references/lark-drive-comments-guide.md +0 -80
- package/skills/lark-slides/references/examples.md +0 -91
- package/skills/lark-slides/references/lark-slides-whiteboard.md +0 -331
- package/skills/lark-slides/references/lark-slides-xml-get.md +0 -100
- package/skills/lark-slides/references/slide-templates.md +0 -201
- package/skills/lark-slides/references/slides_demo.xml +0 -226
- package/skills/lark-slides/references/xml-format-guide.md +0 -433
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
# Copyright (c) 2026 Lark Technologies Pte. Ltd.
|
|
3
3
|
# SPDX-License-Identifier: MIT
|
|
4
|
+
"""Validate Slides XML structure and page layout through one release gate."""
|
|
4
5
|
|
|
5
6
|
from __future__ import annotations
|
|
6
7
|
|
|
8
|
+
import copy
|
|
7
9
|
import json
|
|
8
10
|
import math
|
|
9
11
|
import re
|
|
@@ -15,6 +17,8 @@ from difflib import SequenceMatcher, get_close_matches
|
|
|
15
17
|
from pathlib import Path
|
|
16
18
|
from typing import Any
|
|
17
19
|
|
|
20
|
+
import sxsd_validator
|
|
21
|
+
|
|
18
22
|
|
|
19
23
|
XS_NS = "{http://www.w3.org/2001/XMLSchema}"
|
|
20
24
|
XML_NS = "{http://www.w3.org/XML/1998/namespace}"
|
|
@@ -38,18 +42,46 @@ SXSD_ATTR_ALIASES = {
|
|
|
38
42
|
"fontColor": "color",
|
|
39
43
|
}
|
|
40
44
|
SERVER_FILLED_SXSD_ATTRS = {"id"}
|
|
45
|
+
ROUNDTRIP_SXSD_ATTRS = {
|
|
46
|
+
("chart", "updated"),
|
|
47
|
+
("chartData", "isStaticData"),
|
|
48
|
+
}
|
|
49
|
+
# Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
|
|
50
|
+
# it is server-emitted and absent from the write schema, so it must not block page linting.
|
|
51
|
+
ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
|
|
41
52
|
DEFAULT_TABLE_COLUMN_WIDTH = 110
|
|
42
53
|
DEFAULT_TABLE_ROW_HEIGHT = 37
|
|
54
|
+
DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
|
|
55
|
+
TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
|
|
56
|
+
TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
|
|
57
|
+
SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
|
|
58
|
+
CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
|
|
59
|
+
HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
|
|
60
|
+
DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
|
|
61
|
+
GHOST_TEXT_MIN_FONT_SIZE = 96
|
|
62
|
+
GHOST_TEXT_MAX_ALPHA = 0.5
|
|
63
|
+
GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
|
|
64
|
+
GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
|
|
65
|
+
# A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
|
|
66
|
+
# glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
|
|
67
|
+
# padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
|
|
68
|
+
LINE_TEXT_GRAZE_MIN_PX = 2.0
|
|
69
|
+
LINE_TEXT_GRAZE_FONT_RATIO = 0.12
|
|
70
|
+
# A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
|
|
71
|
+
LINE_MIN_VISIBLE_ALPHA = 0.08
|
|
72
|
+
# Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
|
|
73
|
+
# visible defect; keep this well under 1px so real overflow is still always caught.
|
|
74
|
+
CANVAS_OVERFLOW_TOLERANCE = 0.5
|
|
43
75
|
_SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
|
|
44
76
|
_ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
|
|
45
77
|
|
|
46
78
|
|
|
47
|
-
class
|
|
79
|
+
class XmlLayoutLintError(Exception):
|
|
48
80
|
pass
|
|
49
81
|
|
|
50
82
|
|
|
51
83
|
def fail(message: str) -> None:
|
|
52
|
-
raise
|
|
84
|
+
raise XmlLayoutLintError(message)
|
|
53
85
|
|
|
54
86
|
|
|
55
87
|
def read_file(file_path: str | Path) -> str:
|
|
@@ -62,7 +94,7 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
|
|
|
62
94
|
while index < len(argv):
|
|
63
95
|
token = argv[index]
|
|
64
96
|
if not token.startswith("--"):
|
|
65
|
-
fail(f"unexpected argument: {token}")
|
|
97
|
+
fail(f"unexpected argument: {token}, need --input")
|
|
66
98
|
key = token[2:]
|
|
67
99
|
next_token = argv[index + 1] if index + 1 < len(argv) else None
|
|
68
100
|
if next_token is None or next_token.startswith("--"):
|
|
@@ -75,8 +107,12 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
|
|
|
75
107
|
|
|
76
108
|
|
|
77
109
|
def extract_attribute(tag_source: str, name: str) -> str | None:
|
|
78
|
-
match = re.search(
|
|
79
|
-
|
|
110
|
+
match = re.search(
|
|
111
|
+
fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
|
|
112
|
+
)
|
|
113
|
+
if not match:
|
|
114
|
+
return None
|
|
115
|
+
return match.group(1) if match.group(1) is not None else match.group(2)
|
|
80
116
|
|
|
81
117
|
|
|
82
118
|
def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
|
|
@@ -90,6 +126,52 @@ def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
|
|
|
90
126
|
return int(value) if value.is_integer() else value
|
|
91
127
|
|
|
92
128
|
|
|
129
|
+
def extract_bool_attribute(tag_source: str, name: str) -> bool:
|
|
130
|
+
value = extract_attribute(tag_source, name)
|
|
131
|
+
return value in {"true", "1", "yes"}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def extract_color_alpha(color: str | None) -> int | float | None:
|
|
135
|
+
if color is None:
|
|
136
|
+
return None
|
|
137
|
+
normalized = re.sub(r"\s+", "", color).lower()
|
|
138
|
+
if normalized == "transparent":
|
|
139
|
+
return 0
|
|
140
|
+
rgba_match = re.fullmatch(
|
|
141
|
+
r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
|
|
142
|
+
normalized,
|
|
143
|
+
)
|
|
144
|
+
if rgba_match is None:
|
|
145
|
+
return None
|
|
146
|
+
try:
|
|
147
|
+
alpha = float(rgba_match.group(1))
|
|
148
|
+
except ValueError:
|
|
149
|
+
return None
|
|
150
|
+
return int(alpha) if alpha.is_integer() else alpha
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
|
|
154
|
+
base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
|
|
155
|
+
color_alpha = extract_color_alpha(text_color)
|
|
156
|
+
if not isinstance(color_alpha, (int, float)):
|
|
157
|
+
return base_alpha
|
|
158
|
+
return base_alpha * color_alpha
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
|
|
162
|
+
for tag_name in style_tags:
|
|
163
|
+
if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
|
|
164
|
+
return True
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
|
|
169
|
+
for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
|
|
170
|
+
if extract_bool_attribute(attrs, attr_name):
|
|
171
|
+
return True
|
|
172
|
+
return False
|
|
173
|
+
|
|
174
|
+
|
|
93
175
|
def sum_sizes(sizes: list[int | float]) -> int | float:
|
|
94
176
|
return sum(sizes)
|
|
95
177
|
|
|
@@ -168,8 +250,10 @@ def solve_weighted_min_layout(
|
|
|
168
250
|
return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
|
|
169
251
|
|
|
170
252
|
|
|
171
|
-
def strip_xml(value: str) -> str:
|
|
253
|
+
def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
|
|
172
254
|
stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
|
|
255
|
+
if preserve_line_breaks:
|
|
256
|
+
stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
|
|
173
257
|
stripped = re.sub(r"<[^>]+>", " ", stripped)
|
|
174
258
|
stripped = stripped.replace(" ", " ")
|
|
175
259
|
stripped = stripped.replace("&", "&")
|
|
@@ -177,32 +261,55 @@ def strip_xml(value: str) -> str:
|
|
|
177
261
|
stripped = stripped.replace(">", ">")
|
|
178
262
|
stripped = stripped.replace(""", '"')
|
|
179
263
|
stripped = stripped.replace("'", "'")
|
|
264
|
+
if preserve_line_breaks:
|
|
265
|
+
return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
|
|
180
266
|
return re.sub(r"\s+", " ", stripped).strip()
|
|
181
267
|
|
|
182
268
|
|
|
183
269
|
def strip_xml_paragraphs(value: str) -> str:
|
|
184
270
|
paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
|
|
185
271
|
if paragraphs:
|
|
186
|
-
return "\n".join(strip_xml(paragraph) for paragraph in paragraphs)
|
|
187
|
-
return strip_xml(value)
|
|
272
|
+
return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
|
|
273
|
+
return strip_xml(value, preserve_line_breaks=True)
|
|
188
274
|
|
|
189
275
|
|
|
190
|
-
def
|
|
191
|
-
|
|
276
|
+
def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
|
|
277
|
+
paragraphs = []
|
|
278
|
+
for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
|
|
279
|
+
paragraphs.append(
|
|
280
|
+
{
|
|
281
|
+
"text": strip_xml(body, preserve_line_breaks=True),
|
|
282
|
+
"fontSize": extract_max_span_font_size(body, default_font_size),
|
|
283
|
+
"textAlign": extract_attribute(attrs, "textAlign"),
|
|
284
|
+
"lineSpacing": extract_attribute(attrs, "lineSpacing"),
|
|
285
|
+
"beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
|
|
286
|
+
"afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
|
|
287
|
+
"letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
|
|
288
|
+
}
|
|
289
|
+
)
|
|
290
|
+
return paragraphs
|
|
192
291
|
|
|
193
292
|
|
|
194
|
-
def
|
|
195
|
-
|
|
293
|
+
def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
|
|
294
|
+
font_sizes = [
|
|
295
|
+
font_size
|
|
296
|
+
for attrs in re.findall(r"<span\b([^>]*)>", value)
|
|
297
|
+
if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
|
|
298
|
+
]
|
|
299
|
+
return max([default_font_size, *font_sizes])
|
|
196
300
|
|
|
197
301
|
|
|
198
|
-
def
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
302
|
+
def extract_tag_attributes(value: str, tag: str) -> str:
|
|
303
|
+
match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
|
|
304
|
+
return match.group(1) if match else ""
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def xml_local_name(tag: str) -> str:
|
|
308
|
+
return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
|
|
202
309
|
|
|
203
310
|
|
|
204
|
-
def
|
|
205
|
-
return [
|
|
311
|
+
def xml_namespace(tag: str) -> str | None:
|
|
312
|
+
return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
|
|
206
313
|
|
|
207
314
|
|
|
208
315
|
def load_sxsd_tag_attributes() -> dict[str, set[str]]:
|
|
@@ -210,62 +317,8 @@ def load_sxsd_tag_attributes() -> dict[str, set[str]]:
|
|
|
210
317
|
if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
|
|
211
318
|
return _SXSD_TAG_ATTRIBUTES_CACHE
|
|
212
319
|
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
complex_type.attrib["name"]: complex_type
|
|
216
|
-
for complex_type in schema_root.findall(f"{XS_NS}complexType")
|
|
217
|
-
if complex_type.attrib.get("name")
|
|
218
|
-
}
|
|
219
|
-
resolving: set[str] = set()
|
|
220
|
-
|
|
221
|
-
def attributes_for_complex_type(complex_type: ET.Element) -> set[str]:
|
|
222
|
-
attrs: set[str] = {
|
|
223
|
-
attribute.attrib["name"]
|
|
224
|
-
for attribute in iter_direct_xsd_children(complex_type, "attribute")
|
|
225
|
-
if attribute.attrib.get("name")
|
|
226
|
-
}
|
|
227
|
-
for content_name in ("simpleContent", "complexContent"):
|
|
228
|
-
for complex_content in iter_direct_xsd_children(complex_type, content_name):
|
|
229
|
-
for extension in iter_direct_xsd_children(complex_content, "extension"):
|
|
230
|
-
base_type = strip_xsd_prefix(extension.attrib.get("base"))
|
|
231
|
-
if base_type:
|
|
232
|
-
attrs.update(attributes_for_type(base_type))
|
|
233
|
-
attrs.update(
|
|
234
|
-
attribute.attrib["name"]
|
|
235
|
-
for attribute in iter_direct_xsd_children(extension, "attribute")
|
|
236
|
-
if attribute.attrib.get("name")
|
|
237
|
-
)
|
|
238
|
-
return attrs
|
|
239
|
-
|
|
240
|
-
def attributes_for_type(type_name: str) -> set[str]:
|
|
241
|
-
if type_name in resolving:
|
|
242
|
-
return set()
|
|
243
|
-
complex_type = named_complex_types.get(type_name)
|
|
244
|
-
if complex_type is None:
|
|
245
|
-
return set()
|
|
246
|
-
resolving.add(type_name)
|
|
247
|
-
try:
|
|
248
|
-
return attributes_for_complex_type(complex_type)
|
|
249
|
-
finally:
|
|
250
|
-
resolving.remove(type_name)
|
|
251
|
-
|
|
252
|
-
tag_attributes: dict[str, set[str]] = {}
|
|
253
|
-
for element in schema_root.iter(f"{XS_NS}element"):
|
|
254
|
-
tag_name = element.attrib.get("name")
|
|
255
|
-
if not tag_name:
|
|
256
|
-
continue
|
|
257
|
-
|
|
258
|
-
attrs: set[str] = set()
|
|
259
|
-
type_name = strip_xsd_prefix(element.attrib.get("type"))
|
|
260
|
-
if type_name:
|
|
261
|
-
attrs.update(attributes_for_type(type_name))
|
|
262
|
-
for complex_type in iter_direct_xsd_children(element, "complexType"):
|
|
263
|
-
attrs.update(attributes_for_complex_type(complex_type))
|
|
264
|
-
|
|
265
|
-
tag_attributes.setdefault(tag_name, set()).update(attrs)
|
|
266
|
-
|
|
267
|
-
_SXSD_TAG_ATTRIBUTES_CACHE = tag_attributes
|
|
268
|
-
return tag_attributes
|
|
320
|
+
_SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
|
|
321
|
+
return _SXSD_TAG_ATTRIBUTES_CACHE
|
|
269
322
|
|
|
270
323
|
|
|
271
324
|
def load_iconpark_icon_types() -> set[str]:
|
|
@@ -302,13 +355,19 @@ def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
|
|
|
302
355
|
return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
|
|
303
356
|
|
|
304
357
|
|
|
305
|
-
def
|
|
358
|
+
def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
|
|
306
359
|
alias = SXSD_ATTR_ALIASES.get(attr_name)
|
|
307
360
|
if alias and alias in allowed_attrs:
|
|
308
|
-
return
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
361
|
+
return [alias]
|
|
362
|
+
return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
|
|
366
|
+
suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
|
|
367
|
+
if suggestions:
|
|
368
|
+
if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
|
|
369
|
+
return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
|
|
370
|
+
return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
|
|
312
371
|
allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
|
|
313
372
|
if len(allowed_attrs) > 8:
|
|
314
373
|
allowed_summary += ", ..."
|
|
@@ -319,14 +378,37 @@ def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
|
|
|
319
378
|
return "whiteboard" in ancestors and xml_namespace(element.tag) == SVG_NS
|
|
320
379
|
|
|
321
380
|
|
|
322
|
-
def should_skip_sxsd_attribute(attr_name: str) -> bool:
|
|
323
|
-
return attr_name in SERVER_FILLED_SXSD_ATTRS
|
|
381
|
+
def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
|
|
382
|
+
return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
|
|
386
|
+
return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
|
|
387
|
+
|
|
324
388
|
|
|
389
|
+
def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
|
|
390
|
+
sanitized_root = copy.deepcopy(root)
|
|
325
391
|
|
|
326
|
-
def
|
|
392
|
+
def sanitize(element: ET.Element) -> None:
|
|
393
|
+
tag_name = xml_local_name(element.tag)
|
|
394
|
+
for raw_attr_name in list(element.attrib):
|
|
395
|
+
if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
|
|
396
|
+
del element.attrib[raw_attr_name]
|
|
397
|
+
for child in list(element):
|
|
398
|
+
if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
|
|
399
|
+
element.remove(child)
|
|
400
|
+
continue
|
|
401
|
+
sanitize(child)
|
|
402
|
+
|
|
403
|
+
sanitize(sanitized_root)
|
|
404
|
+
return sanitized_root
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
|
|
327
408
|
tag_attributes = load_sxsd_tag_attributes()
|
|
328
409
|
supported_tags = set(tag_attributes)
|
|
329
410
|
issues: list[dict[str, Any]] = []
|
|
411
|
+
suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
|
|
330
412
|
|
|
331
413
|
def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
|
|
332
414
|
if should_skip_sxsd_subtree(element, ancestors):
|
|
@@ -334,6 +416,9 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
|
|
|
334
416
|
|
|
335
417
|
tag_name = xml_local_name(element.tag)
|
|
336
418
|
current_path = f"{path}/{tag_name}" if path else tag_name
|
|
419
|
+
parent_name = ancestors[-1] if ancestors else None
|
|
420
|
+
if should_skip_sxsd_tag(parent_name, tag_name):
|
|
421
|
+
return
|
|
337
422
|
if tag_name not in supported_tags:
|
|
338
423
|
issues.append(
|
|
339
424
|
{
|
|
@@ -352,10 +437,15 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
|
|
|
352
437
|
if raw_attr_name.startswith(XML_NS):
|
|
353
438
|
continue
|
|
354
439
|
attr_name = xml_local_name(raw_attr_name)
|
|
355
|
-
if should_skip_sxsd_attribute(attr_name):
|
|
440
|
+
if should_skip_sxsd_attribute(tag_name, attr_name):
|
|
356
441
|
continue
|
|
357
442
|
if attr_name in allowed_attrs:
|
|
358
443
|
continue
|
|
444
|
+
suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
|
|
445
|
+
if suggestions:
|
|
446
|
+
suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
|
|
447
|
+
set(suggestions)
|
|
448
|
+
)
|
|
359
449
|
issues.append(
|
|
360
450
|
{
|
|
361
451
|
"level": "error",
|
|
@@ -372,6 +462,76 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
|
|
|
372
462
|
visit(child, [*ancestors, tag_name], current_path)
|
|
373
463
|
|
|
374
464
|
visit(root, [], "")
|
|
465
|
+
existing = {
|
|
466
|
+
(issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
|
|
467
|
+
for issue in issues
|
|
468
|
+
}
|
|
469
|
+
unsupported_tag_locations = {
|
|
470
|
+
(issue.get("path"), issue.get("tag"))
|
|
471
|
+
for issue in issues
|
|
472
|
+
if issue.get("code") == "sxsd_unsupported_tag"
|
|
473
|
+
}
|
|
474
|
+
schema_issues = _validate_sxsd_schema_constraints(xml, root)
|
|
475
|
+
missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
|
|
476
|
+
for schema_issue in schema_issues:
|
|
477
|
+
if schema_issue.get("code") != "sxsd_missing_required_attr":
|
|
478
|
+
continue
|
|
479
|
+
location = (schema_issue.get("path"), schema_issue.get("tag"))
|
|
480
|
+
missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
|
|
481
|
+
|
|
482
|
+
suggested_attrs: set[tuple[str, str, str]] = set()
|
|
483
|
+
for location, candidate_groups in suggested_attr_candidates.items():
|
|
484
|
+
missing_attrs = missing_attrs_by_location.get(location, set())
|
|
485
|
+
for candidates in candidate_groups:
|
|
486
|
+
matching_missing_attrs = candidates & missing_attrs
|
|
487
|
+
if len(matching_missing_attrs) == 1:
|
|
488
|
+
suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
|
|
489
|
+
|
|
490
|
+
for schema_issue in schema_issues:
|
|
491
|
+
if schema_issue.get("code") == "sxsd_unexpected_child" and (
|
|
492
|
+
schema_issue.get("path"),
|
|
493
|
+
schema_issue.get("tag"),
|
|
494
|
+
) in unsupported_tag_locations:
|
|
495
|
+
continue
|
|
496
|
+
if schema_issue.get("code") == "sxsd_missing_required_attr" and (
|
|
497
|
+
schema_issue.get("path"),
|
|
498
|
+
schema_issue.get("tag"),
|
|
499
|
+
schema_issue.get("attr"),
|
|
500
|
+
) in suggested_attrs:
|
|
501
|
+
continue
|
|
502
|
+
key = (
|
|
503
|
+
schema_issue.get("code"),
|
|
504
|
+
schema_issue.get("path"),
|
|
505
|
+
schema_issue.get("tag"),
|
|
506
|
+
schema_issue.get("attr"),
|
|
507
|
+
)
|
|
508
|
+
if key not in existing:
|
|
509
|
+
issues.append(schema_issue)
|
|
510
|
+
return issues
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
|
|
514
|
+
issues: list[dict[str, Any]] = []
|
|
515
|
+
if re.match(r"^\s*<\?xml\b", xml):
|
|
516
|
+
issues.append(
|
|
517
|
+
{
|
|
518
|
+
"level": "error",
|
|
519
|
+
"code": "sxsd_unsupported_declaration",
|
|
520
|
+
"path": xml_local_name(root.tag),
|
|
521
|
+
"tag": xml_local_name(root.tag),
|
|
522
|
+
"expected": "SXSD document without an XML declaration",
|
|
523
|
+
"actual": "<?xml ...?>",
|
|
524
|
+
"message": "XML declarations are not supported by the Slides SXSD write format",
|
|
525
|
+
"hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
|
|
526
|
+
}
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
issues.extend(
|
|
530
|
+
sxsd_validator.validate_sxsd(
|
|
531
|
+
without_server_filled_sxsd_fields(root),
|
|
532
|
+
SXSD_SCHEMA_PATH,
|
|
533
|
+
)
|
|
534
|
+
)
|
|
375
535
|
return issues
|
|
376
536
|
|
|
377
537
|
|
|
@@ -575,18 +735,39 @@ def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
|
|
|
575
735
|
return xml_error
|
|
576
736
|
|
|
577
737
|
|
|
578
|
-
def
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
738
|
+
def serialize_slide_for_layout(slide_root: ET.Element) -> str:
|
|
739
|
+
slide_copy = copy.deepcopy(slide_root)
|
|
740
|
+
for element in slide_copy.iter():
|
|
741
|
+
if not isinstance(element.tag, str):
|
|
742
|
+
continue
|
|
743
|
+
element.tag = xml_local_name(element.tag)
|
|
744
|
+
attributes = {
|
|
745
|
+
xml_local_name(attribute_name): value
|
|
746
|
+
for attribute_name, value in element.attrib.items()
|
|
585
747
|
}
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
748
|
+
element.attrib.clear()
|
|
749
|
+
element.attrib.update(attributes)
|
|
750
|
+
return ET.tostring(slide_copy, encoding="unicode")
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
def parse_presentation(root: ET.Element) -> dict[str, Any]:
|
|
754
|
+
root_name = xml_local_name(root.tag)
|
|
755
|
+
if root_name == "slide":
|
|
756
|
+
slide_roots = [root]
|
|
757
|
+
width = 960
|
|
758
|
+
height = 540
|
|
759
|
+
elif root_name == "presentation":
|
|
760
|
+
slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
|
|
761
|
+
width = int(float(root.attrib.get("width", 960)))
|
|
762
|
+
height = int(float(root.attrib.get("height", 540)))
|
|
763
|
+
else:
|
|
764
|
+
fail("input must contain a <presentation> or <slide> root")
|
|
765
|
+
return {
|
|
766
|
+
"width": width,
|
|
767
|
+
"height": height,
|
|
768
|
+
"slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
|
|
769
|
+
"slide_roots": slide_roots,
|
|
770
|
+
}
|
|
590
771
|
|
|
591
772
|
|
|
592
773
|
def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
@@ -594,8 +775,9 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
|
594
775
|
|
|
595
776
|
for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
|
|
596
777
|
kind, attrs = match.group(1), match.group(2)
|
|
778
|
+
is_self_closing = attrs.rstrip().endswith("/")
|
|
597
779
|
content = ""
|
|
598
|
-
if kind in {"shape", "table"}:
|
|
780
|
+
if kind in {"shape", "table"} and not is_self_closing:
|
|
599
781
|
close_index = slide_xml.find(f"</{kind}>", match.end())
|
|
600
782
|
if close_index != -1:
|
|
601
783
|
content = slide_xml[match.end() : close_index]
|
|
@@ -606,6 +788,7 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
|
606
788
|
width = extract_numeric_attribute(attrs, "width")
|
|
607
789
|
height = extract_numeric_attribute(attrs, "height")
|
|
608
790
|
rotation = extract_numeric_attribute(attrs, "rotation") or 0
|
|
791
|
+
alpha = extract_numeric_attribute(attrs, "alpha")
|
|
609
792
|
table_layouts: dict[str, dict[str, Any] | None] = {}
|
|
610
793
|
if kind == "table":
|
|
611
794
|
width, table_layouts["width"] = resolve_table_dimension(
|
|
@@ -624,6 +807,7 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
|
624
807
|
"width": width,
|
|
625
808
|
"height": height,
|
|
626
809
|
"rotation": rotation,
|
|
810
|
+
"alpha": alpha if alpha is not None else 1,
|
|
627
811
|
"order": len(elements),
|
|
628
812
|
}
|
|
629
813
|
if kind == "table":
|
|
@@ -635,15 +819,48 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
|
635
819
|
}
|
|
636
820
|
)
|
|
637
821
|
if kind == "shape":
|
|
822
|
+
content_attrs = extract_tag_attributes(content, "content")
|
|
823
|
+
font_size = extract_numeric_attribute(content_attrs, "fontSize")
|
|
824
|
+
if font_size is None:
|
|
825
|
+
font_size = extract_numeric_attribute(attrs, "fontSize")
|
|
826
|
+
font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
|
|
827
|
+
text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
|
|
828
|
+
bold = (
|
|
829
|
+
extract_bool_attribute(content_attrs, "bold")
|
|
830
|
+
or extract_bool_attribute(attrs, "bold")
|
|
831
|
+
or detect_inline_style_presence(content, {"strong", "b"})
|
|
832
|
+
or detect_any_span_bool_attribute(content, "bold")
|
|
833
|
+
)
|
|
834
|
+
italic = (
|
|
835
|
+
extract_bool_attribute(content_attrs, "italic")
|
|
836
|
+
or extract_bool_attribute(attrs, "italic")
|
|
837
|
+
or detect_inline_style_presence(content, {"i", "em"})
|
|
838
|
+
or detect_any_span_bool_attribute(content, "italic")
|
|
839
|
+
)
|
|
638
840
|
element.update(
|
|
639
841
|
{
|
|
640
|
-
"textType": extract_attribute(
|
|
641
|
-
"textAlign": extract_attribute(
|
|
642
|
-
"
|
|
643
|
-
"
|
|
644
|
-
|
|
645
|
-
),
|
|
842
|
+
"textType": extract_attribute(content_attrs, "textType"),
|
|
843
|
+
"textAlign": extract_attribute(content_attrs, "textAlign"),
|
|
844
|
+
"verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
|
|
845
|
+
"vert": extract_attribute(attrs, "vert") or "horz",
|
|
846
|
+
"autoFit": extract_attribute(content_attrs, "autoFit"),
|
|
847
|
+
"wrap": extract_attribute(content_attrs, "wrap"),
|
|
848
|
+
"lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
|
|
849
|
+
"beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
|
|
850
|
+
"afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
|
|
851
|
+
"letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
|
|
852
|
+
"paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
|
|
853
|
+
"paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
|
|
854
|
+
"paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
|
|
855
|
+
"paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
|
|
856
|
+
"fontSize": font_size if font_size is not None else 16,
|
|
857
|
+
"fontFamily": font_family or "",
|
|
858
|
+
"color": text_color,
|
|
859
|
+
"textAlpha": effective_text_alpha(alpha, text_color),
|
|
860
|
+
"bold": bold,
|
|
861
|
+
"italic": italic,
|
|
646
862
|
"text": strip_xml_paragraphs(content),
|
|
863
|
+
"paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
|
|
647
864
|
}
|
|
648
865
|
)
|
|
649
866
|
elements.append(element)
|
|
@@ -671,6 +888,44 @@ def has_text_content(element: dict[str, Any]) -> bool:
|
|
|
671
888
|
return bool(element.get("text"))
|
|
672
889
|
|
|
673
890
|
|
|
891
|
+
def is_vertical_text(element: dict[str, Any]) -> bool:
|
|
892
|
+
return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
|
|
893
|
+
|
|
894
|
+
|
|
895
|
+
def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
896
|
+
issues: list[dict[str, Any]] = []
|
|
897
|
+
text_elements = [
|
|
898
|
+
element
|
|
899
|
+
for element in elements
|
|
900
|
+
if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
|
|
901
|
+
]
|
|
902
|
+
image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
|
|
903
|
+
for text_element in text_elements:
|
|
904
|
+
for image_element in image_elements:
|
|
905
|
+
if image_element["order"] <= text_element["order"]:
|
|
906
|
+
continue
|
|
907
|
+
if is_vertical_text(text_element):
|
|
908
|
+
if intersects(image_element, text_element):
|
|
909
|
+
issues.append({
|
|
910
|
+
"level": "info",
|
|
911
|
+
"code": "image_may_cover_vertical_text",
|
|
912
|
+
"elements": [image_element["id"], text_element["id"]],
|
|
913
|
+
"message": f'image {image_element["id"]} may cover vertical text shape {text_element["id"]}',
|
|
914
|
+
"hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
|
|
915
|
+
})
|
|
916
|
+
continue
|
|
917
|
+
text_visual_bbox = estimate_text_visual_bbox(text_element)
|
|
918
|
+
if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
|
|
919
|
+
issues.append({
|
|
920
|
+
"level": "error",
|
|
921
|
+
"code": "image_covers_text",
|
|
922
|
+
"elements": [image_element["id"], text_element["id"]],
|
|
923
|
+
"message": f'image {image_element["id"]} covers text shape {text_element["id"]}',
|
|
924
|
+
"hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
|
|
925
|
+
})
|
|
926
|
+
return issues
|
|
927
|
+
|
|
928
|
+
|
|
674
929
|
def is_decorative_text(element: dict[str, Any]) -> bool:
|
|
675
930
|
text = element.get("text") or ""
|
|
676
931
|
return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
|
|
@@ -680,22 +935,135 @@ def normalize_text_for_overlap(text: str) -> str:
|
|
|
680
935
|
return re.sub(r"\s+", "", text)
|
|
681
936
|
|
|
682
937
|
|
|
683
|
-
|
|
938
|
+
SERIF_FONT_PATTERNS = {
|
|
939
|
+
"song", "songti", "simsun", "ming", "mincho",
|
|
940
|
+
"georgia", "times", "caslon", "garamond", "sourcehan-serif",
|
|
941
|
+
"source han serif", "思源宋体", "宋体", "明体",
|
|
942
|
+
}
|
|
943
|
+
|
|
944
|
+
SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
|
|
945
|
+
"helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
|
|
946
|
+
|
|
947
|
+
|
|
948
|
+
def classify_font_family(font_family: str | None) -> str:
|
|
949
|
+
if not font_family:
|
|
950
|
+
return "sans"
|
|
951
|
+
family_lower = font_family.lower()
|
|
952
|
+
for marker in SANS_EXPLICIT_MARKERS:
|
|
953
|
+
if marker in family_lower:
|
|
954
|
+
return "sans"
|
|
955
|
+
serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
|
|
956
|
+
for pattern in serif_keywords:
|
|
957
|
+
if pattern in family_lower:
|
|
958
|
+
return "serif"
|
|
959
|
+
return "sans"
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
_FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
|
|
963
|
+
"sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
|
|
964
|
+
"serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
|
|
965
|
+
}
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def estimate_character_width(
|
|
969
|
+
character: str,
|
|
970
|
+
font_size: int | float,
|
|
971
|
+
bold: bool = False,
|
|
972
|
+
font_family: str | None = None,
|
|
973
|
+
) -> int | float:
|
|
974
|
+
bold_multiplier = 1.05 if bold else 1.0
|
|
684
975
|
if character.isspace():
|
|
685
|
-
return font_size * 0.33
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
976
|
+
return font_size * 0.33 * bold_multiplier
|
|
977
|
+
ea_width = unicodedata.east_asian_width(character)
|
|
978
|
+
if ea_width in {"F", "W"}:
|
|
979
|
+
return font_size * bold_multiplier
|
|
980
|
+
category = classify_font_family(font_family)
|
|
981
|
+
coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
|
|
982
|
+
if character.isupper():
|
|
983
|
+
return font_size * coeffs["upper"] * bold_multiplier
|
|
984
|
+
if character.islower():
|
|
985
|
+
return font_size * coeffs["lower"] * bold_multiplier
|
|
986
|
+
if character.isdigit():
|
|
987
|
+
return font_size * coeffs["digit"] * bold_multiplier
|
|
988
|
+
return font_size * coeffs["punct"] * bold_multiplier
|
|
689
989
|
|
|
690
990
|
|
|
691
|
-
def estimate_text_width(
|
|
692
|
-
|
|
991
|
+
def estimate_text_width(
|
|
992
|
+
text: str,
|
|
993
|
+
font_size: int | float,
|
|
994
|
+
letter_spacing: int | float = 0,
|
|
995
|
+
bold: bool = False,
|
|
996
|
+
font_family: str | None = None,
|
|
997
|
+
) -> int | float:
|
|
998
|
+
base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
|
|
999
|
+
return base + max(len(text) - 1, 0) * letter_spacing
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
|
|
1003
|
+
if paragraph is not None:
|
|
1004
|
+
value = paragraph.get("letterSpacing")
|
|
1005
|
+
if isinstance(value, (int, float)):
|
|
1006
|
+
return value
|
|
1007
|
+
value = element.get("letterSpacing")
|
|
1008
|
+
return value if isinstance(value, (int, float)) else 0
|
|
1009
|
+
|
|
1010
|
+
|
|
1011
|
+
def text_wrap_width_tolerance() -> int | float:
|
|
1012
|
+
return TEXT_WRAP_WIDTH_TOLERANCE_PX
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def text_height_overflow_tolerance() -> int | float:
|
|
1016
|
+
return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
|
|
1017
|
+
|
|
1018
|
+
|
|
1019
|
+
def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
|
|
1020
|
+
return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
|
|
1021
|
+
|
|
1022
|
+
|
|
1023
|
+
def is_short_metric_text(text: str) -> bool:
|
|
1024
|
+
compact = re.sub(r"\s+", "", text)
|
|
1025
|
+
if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
|
|
1026
|
+
return False
|
|
1027
|
+
if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
|
|
1028
|
+
return True
|
|
1029
|
+
if re.search(r"[,.,+\-–—/%%]", compact) is None:
|
|
1030
|
+
return False
|
|
1031
|
+
return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
def is_single_line_visual_candidate(
|
|
1035
|
+
element: dict[str, Any],
|
|
1036
|
+
paragraph: dict[str, Any] | None,
|
|
1037
|
+
text: str,
|
|
1038
|
+
logical_width: int | float,
|
|
1039
|
+
effective_width: int | float,
|
|
1040
|
+
) -> bool:
|
|
1041
|
+
if "\n" in text or logical_width <= effective_width:
|
|
1042
|
+
return False
|
|
1043
|
+
if is_short_metric_text(text):
|
|
1044
|
+
return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
|
|
1045
|
+
|
|
1046
|
+
text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
|
|
1047
|
+
compact_len = len(re.sub(r"\s+", "", text))
|
|
1048
|
+
if text_align == "center" and compact_len <= 32:
|
|
1049
|
+
return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
|
|
1050
|
+
|
|
1051
|
+
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1052
|
+
if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
|
|
1053
|
+
return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
|
|
1054
|
+
return False
|
|
693
1055
|
|
|
694
1056
|
|
|
695
1057
|
def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
|
|
696
1058
|
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1059
|
+
bold = element.get("bold", False)
|
|
1060
|
+
font_family = element.get("fontFamily", "")
|
|
1061
|
+
letter_spacing = resolve_letter_spacing(element)
|
|
697
1062
|
paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
|
|
698
|
-
return max(
|
|
1063
|
+
return max(
|
|
1064
|
+
[estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
|
|
1065
|
+
or [1]
|
|
1066
|
+
)
|
|
699
1067
|
|
|
700
1068
|
|
|
701
1069
|
def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
@@ -708,27 +1076,203 @@ def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool
|
|
|
708
1076
|
return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
|
|
709
1077
|
|
|
710
1078
|
|
|
711
|
-
def
|
|
1079
|
+
def estimate_text_line_count_for_text(
|
|
1080
|
+
element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
|
|
1081
|
+
) -> int:
|
|
712
1082
|
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
713
|
-
|
|
1083
|
+
bold = element.get("bold", False)
|
|
1084
|
+
font_family = element.get("fontFamily", "")
|
|
1085
|
+
letter_spacing = resolve_letter_spacing(element, paragraph)
|
|
1086
|
+
available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
|
|
1087
|
+
hard_lines = text.split("\n")
|
|
1088
|
+
if not text:
|
|
1089
|
+
return 0
|
|
714
1090
|
line_count = 0
|
|
715
|
-
for
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
1091
|
+
for hard_line in hard_lines:
|
|
1092
|
+
if element.get("wrap") in {"false", "0"}:
|
|
1093
|
+
line_count += 1
|
|
1094
|
+
continue
|
|
1095
|
+
logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
|
|
1096
|
+
effective_width = available_width + text_wrap_width_tolerance()
|
|
1097
|
+
if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
|
|
1098
|
+
line_count += 1
|
|
1099
|
+
continue
|
|
1100
|
+
line_count += max(1, math.ceil(logical_width / effective_width))
|
|
1101
|
+
return line_count
|
|
1102
|
+
|
|
1103
|
+
|
|
1104
|
+
def estimate_text_line_count(element: dict[str, Any]) -> int:
|
|
1105
|
+
return max(estimate_text_line_count_for_text(element, element["text"]), 1)
|
|
1106
|
+
|
|
1107
|
+
|
|
1108
|
+
def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
|
|
1109
|
+
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1110
|
+
if line_spacing is None:
|
|
1111
|
+
return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
|
|
1112
|
+
match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
|
|
1113
|
+
if match is None:
|
|
1114
|
+
return None
|
|
1115
|
+
spacing_type, value = match.groups()
|
|
1116
|
+
return font_size * float(value) if spacing_type == "multiple" else float(value)
|
|
1117
|
+
|
|
1118
|
+
|
|
1119
|
+
def adjust_dense_body_line_height(
|
|
1120
|
+
element: dict[str, Any],
|
|
1121
|
+
line_spacing: str | None,
|
|
1122
|
+
line_height: int | float,
|
|
1123
|
+
paragraph_count: int,
|
|
1124
|
+
) -> int | float:
|
|
1125
|
+
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1126
|
+
if paragraph_count < 4 or font_size > 14 or not line_spacing:
|
|
1127
|
+
return line_height
|
|
1128
|
+
match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
|
|
1129
|
+
if match is None:
|
|
1130
|
+
return line_height
|
|
1131
|
+
return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1135
|
+
issues: list[dict[str, Any]] = []
|
|
1136
|
+
for element in elements:
|
|
1137
|
+
if not is_text_element(element) or not has_text_content(element):
|
|
1138
|
+
continue
|
|
1139
|
+
if has_explicit_height_auto_fit(element):
|
|
1140
|
+
continue
|
|
1141
|
+
|
|
1142
|
+
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1143
|
+
paragraphs = element.get("paragraphs") or [
|
|
1144
|
+
{
|
|
1145
|
+
"text": element["text"],
|
|
1146
|
+
"lineSpacing": None,
|
|
1147
|
+
"beforeLineSpacing": None,
|
|
1148
|
+
"afterLineSpacing": None,
|
|
1149
|
+
}
|
|
1150
|
+
]
|
|
1151
|
+
line_count = 0
|
|
1152
|
+
estimated_height = 0.0
|
|
1153
|
+
line_heights: list[int | float] = []
|
|
1154
|
+
for paragraph in paragraphs:
|
|
1155
|
+
paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
|
|
1156
|
+
if paragraph_line_count == 0:
|
|
1157
|
+
continue
|
|
1158
|
+
resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
|
|
1159
|
+
line_height = estimate_text_line_height(element, resolved_line_spacing)
|
|
1160
|
+
before_spacing = estimate_text_line_height(
|
|
1161
|
+
element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
|
|
1162
|
+
)
|
|
1163
|
+
after_spacing = estimate_text_line_height(
|
|
1164
|
+
element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
|
|
1165
|
+
)
|
|
1166
|
+
if line_height is None or before_spacing is None or after_spacing is None:
|
|
1167
|
+
line_count = 0
|
|
1168
|
+
break
|
|
1169
|
+
line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
|
|
1170
|
+
first_line_height = font_size if line_count == 0 else line_height
|
|
1171
|
+
line_count += paragraph_line_count
|
|
1172
|
+
line_heights.append(line_height)
|
|
1173
|
+
estimated_height += (
|
|
1174
|
+
before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
|
|
1175
|
+
)
|
|
1176
|
+
if line_count == 0:
|
|
1177
|
+
continue
|
|
1178
|
+
available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
|
|
1179
|
+
overflow = estimated_height - available_height
|
|
1180
|
+
if overflow <= text_height_overflow_tolerance():
|
|
1181
|
+
continue
|
|
1182
|
+
|
|
1183
|
+
is_background = is_background_decorative_text(element, elements)
|
|
1184
|
+
if is_background:
|
|
1185
|
+
level = "info"
|
|
1186
|
+
else:
|
|
1187
|
+
level = "error" if overflow > 10 else "warning"
|
|
1188
|
+
message = (
|
|
1189
|
+
f'text shape {element["id"]} may overflow its own content box '
|
|
1190
|
+
f'(estimated {estimated_height:g}px, available {available_height:g}px); '
|
|
1191
|
+
'consider setting content wrap="true" autoFit="normal-auto-fit"'
|
|
1192
|
+
)
|
|
1193
|
+
if is_background:
|
|
1194
|
+
message += " (likely background decoration: large font, low alpha, underneath other text)"
|
|
1195
|
+
issues.append(
|
|
1196
|
+
{
|
|
1197
|
+
"level": level,
|
|
1198
|
+
"code": "text_may_overflow_shape",
|
|
1199
|
+
"elements": [element["id"]],
|
|
1200
|
+
"line_count": line_count,
|
|
1201
|
+
"line_height": max(line_heights),
|
|
1202
|
+
"estimated_height": estimated_height,
|
|
1203
|
+
"available_height": available_height,
|
|
1204
|
+
"overflow": overflow,
|
|
1205
|
+
"message": message,
|
|
1206
|
+
"hint": (
|
|
1207
|
+
"Increase shape.height, reduce the text, or set content wrap=\"true\" "
|
|
1208
|
+
"autoFit=\"normal-auto-fit\". "
|
|
1209
|
+
"This is an estimate based on font size, line spacing, and wrapped line count."
|
|
1210
|
+
),
|
|
1211
|
+
}
|
|
1212
|
+
)
|
|
1213
|
+
return issues
|
|
1214
|
+
|
|
1215
|
+
|
|
1216
|
+
def is_background_decorative_text(
|
|
1217
|
+
element: dict[str, Any], elements: list[dict[str, Any]]
|
|
1218
|
+
) -> bool:
|
|
1219
|
+
if not is_ghost_text(element):
|
|
1220
|
+
return False
|
|
1221
|
+
for other in elements:
|
|
1222
|
+
if other is element:
|
|
1223
|
+
continue
|
|
1224
|
+
if not is_text_element(other) or not has_text_content(other):
|
|
1225
|
+
continue
|
|
1226
|
+
foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
|
|
1227
|
+
if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
|
|
1228
|
+
continue
|
|
1229
|
+
if other["order"] <= element["order"]:
|
|
1230
|
+
continue
|
|
1231
|
+
if intersects(element, other):
|
|
1232
|
+
return True
|
|
1233
|
+
return False
|
|
1234
|
+
|
|
1235
|
+
|
|
1236
|
+
def is_ghost_text(element: dict[str, Any]) -> bool:
|
|
1237
|
+
if not is_text_element(element) or not has_text_content(element):
|
|
1238
|
+
return False
|
|
1239
|
+
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1240
|
+
text_alpha = element.get("textAlpha", element.get("alpha", 1))
|
|
1241
|
+
if not isinstance(text_alpha, (int, float)):
|
|
1242
|
+
return False
|
|
1243
|
+
if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
|
|
1244
|
+
return True
|
|
1245
|
+
return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
|
|
719
1246
|
|
|
720
1247
|
|
|
721
1248
|
def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
|
|
722
1249
|
if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
|
|
723
1250
|
return None
|
|
724
1251
|
|
|
1252
|
+
padding_left = element.get("paddingLeft", 0)
|
|
1253
|
+
padding_right = element.get("paddingRight", 0)
|
|
1254
|
+
padding_top = element.get("paddingTop", 0)
|
|
1255
|
+
padding_bottom = element.get("paddingBottom", 0)
|
|
1256
|
+
content_width = max(element["width"] - padding_left - padding_right, 0)
|
|
1257
|
+
content_height = max(element["height"] - padding_top - padding_bottom, 0)
|
|
725
1258
|
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
726
1259
|
line_count = estimate_text_line_count(element)
|
|
727
|
-
|
|
728
|
-
|
|
1260
|
+
estimated_width = max(1, estimate_text_max_line_width(element))
|
|
1261
|
+
visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
|
|
1262
|
+
visual_height = min(content_height, max(1, line_count * font_size * 1.2))
|
|
1263
|
+
x = element["x"] + padding_left
|
|
1264
|
+
if element.get("textAlign") == "center":
|
|
1265
|
+
x += (content_width - visual_width) / 2
|
|
1266
|
+
elif element.get("textAlign") == "right":
|
|
1267
|
+
x += content_width - visual_width
|
|
1268
|
+
y = element["y"] + padding_top
|
|
1269
|
+
if element.get("verticalAlign") == "middle":
|
|
1270
|
+
y += (content_height - visual_height) / 2
|
|
1271
|
+
elif element.get("verticalAlign") == "bottom":
|
|
1272
|
+
y += content_height - visual_height
|
|
729
1273
|
return {
|
|
730
|
-
"x":
|
|
731
|
-
"y":
|
|
1274
|
+
"x": x,
|
|
1275
|
+
"y": y,
|
|
732
1276
|
"width": visual_width,
|
|
733
1277
|
"height": visual_height,
|
|
734
1278
|
}
|
|
@@ -812,25 +1356,34 @@ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str,
|
|
|
812
1356
|
return False
|
|
813
1357
|
if not (has_text_content(left) and has_text_content(right)):
|
|
814
1358
|
return False
|
|
1359
|
+
if is_ghost_text(left) or is_ghost_text(right):
|
|
1360
|
+
return False
|
|
815
1361
|
if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
|
|
816
1362
|
return False
|
|
817
1363
|
|
|
818
1364
|
source, target = sorted([left, right], key=lambda element: element["x"])
|
|
819
1365
|
if source["x"] == target["x"]:
|
|
820
1366
|
return False
|
|
1367
|
+
wrap_enabled = source.get("wrap") not in {"false", "0"}
|
|
1368
|
+
has_horizontal_gap = source["x"] + source["width"] <= target["x"]
|
|
1369
|
+
if wrap_enabled and has_horizontal_gap:
|
|
1370
|
+
return False
|
|
821
1371
|
if source.get("autoFit") == "normal-auto-fit":
|
|
822
1372
|
return False
|
|
823
1373
|
if source.get("textAlign") in {"center", "right"}:
|
|
824
1374
|
return False
|
|
825
1375
|
|
|
826
1376
|
font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
|
|
1377
|
+
padding_left = source.get("paddingLeft", 0)
|
|
1378
|
+
padding_right = source.get("paddingRight", 0)
|
|
1379
|
+
available_width = max(source["width"] - padding_left - padding_right, 1)
|
|
827
1380
|
visual_width = estimate_text_max_line_width(source)
|
|
828
|
-
overflow_width = visual_width -
|
|
829
|
-
min_overflow = max(font_size * 1.5,
|
|
1381
|
+
overflow_width = visual_width - available_width
|
|
1382
|
+
min_overflow = max(font_size * 1.5, available_width * 0.08)
|
|
830
1383
|
if overflow_width < min_overflow:
|
|
831
1384
|
return False
|
|
832
1385
|
|
|
833
|
-
intrusion_width = source["x"] + visual_width - target["x"]
|
|
1386
|
+
intrusion_width = source["x"] + padding_left + visual_width - target["x"]
|
|
834
1387
|
min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
|
|
835
1388
|
if intrusion_width < min_intrusion:
|
|
836
1389
|
return False
|
|
@@ -840,11 +1393,27 @@ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str,
|
|
|
840
1393
|
return vertical_overlap >= min_vertical_overlap
|
|
841
1394
|
|
|
842
1395
|
|
|
1396
|
+
def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
|
|
1397
|
+
source, target = sorted([left, right], key=lambda element: element["x"])
|
|
1398
|
+
padding_left = source.get("paddingLeft", 0)
|
|
1399
|
+
visual_width = estimate_text_max_line_width(source)
|
|
1400
|
+
source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
|
|
1401
|
+
width = intersection_width(source_visual_bbox, target)
|
|
1402
|
+
height = intersection_height(source_visual_bbox, target)
|
|
1403
|
+
return {
|
|
1404
|
+
"intersection_width": round(width, 3),
|
|
1405
|
+
"intersection_height": round(height, 3),
|
|
1406
|
+
"intersection_area": round(width * height, 3),
|
|
1407
|
+
}
|
|
1408
|
+
|
|
1409
|
+
|
|
843
1410
|
def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
844
1411
|
if is_text_element(left) and not has_text_content(left):
|
|
845
1412
|
return False
|
|
846
1413
|
if is_text_element(right) and not has_text_content(right):
|
|
847
1414
|
return False
|
|
1415
|
+
if is_ghost_text(left) or is_ghost_text(right):
|
|
1416
|
+
return False
|
|
848
1417
|
if is_template_text_stack(left, right):
|
|
849
1418
|
return False
|
|
850
1419
|
if is_text_element(left) and is_text_element(right):
|
|
@@ -891,6 +1460,8 @@ def should_report_whiteboard_overlap(
|
|
|
891
1460
|
) -> dict[str, Any] | None:
|
|
892
1461
|
if other is whiteboard or not intersects(whiteboard, other):
|
|
893
1462
|
return None
|
|
1463
|
+
if is_ghost_text(other):
|
|
1464
|
+
return None
|
|
894
1465
|
if contains(whiteboard, other):
|
|
895
1466
|
return None
|
|
896
1467
|
if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
|
|
@@ -969,7 +1540,6 @@ def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
|
|
|
969
1540
|
bbox = {key: element[key] for key in ("x", "y", "width", "height")}
|
|
970
1541
|
if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
|
|
971
1542
|
return bbox
|
|
972
|
-
|
|
973
1543
|
rotation = element["rotation"]
|
|
974
1544
|
if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
|
|
975
1545
|
rotation = 0
|
|
@@ -999,7 +1569,7 @@ def detect_elements_out_of_canvas(
|
|
|
999
1569
|
element
|
|
1000
1570
|
for element in elements
|
|
1001
1571
|
if element["kind"] in {"table", "chart"}
|
|
1002
|
-
or (element["kind"] == "shape" and element["type"]
|
|
1572
|
+
or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
|
|
1003
1573
|
):
|
|
1004
1574
|
bbox = element_canvas_bbox(element)
|
|
1005
1575
|
overflow = {
|
|
@@ -1009,7 +1579,9 @@ def detect_elements_out_of_canvas(
|
|
|
1009
1579
|
"bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
|
|
1010
1580
|
}
|
|
1011
1581
|
overflow_details = [
|
|
1012
|
-
f"{side} by {amount:g}px"
|
|
1582
|
+
f"{side} by {amount:g}px"
|
|
1583
|
+
for side, amount in overflow.items()
|
|
1584
|
+
if amount > CANVAS_OVERFLOW_TOLERANCE
|
|
1013
1585
|
]
|
|
1014
1586
|
if not overflow_details:
|
|
1015
1587
|
continue
|
|
@@ -1108,6 +1680,93 @@ def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[
|
|
|
1108
1680
|
return issues
|
|
1109
1681
|
|
|
1110
1682
|
|
|
1683
|
+
def segment_intersects_rect(
|
|
1684
|
+
x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
|
|
1685
|
+
) -> bool:
|
|
1686
|
+
"""True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
|
|
1687
|
+
left = rect["x"]
|
|
1688
|
+
top = rect["y"]
|
|
1689
|
+
right = rect["x"] + rect["width"]
|
|
1690
|
+
bottom = rect["y"] + rect["height"]
|
|
1691
|
+
if right <= left or bottom <= top:
|
|
1692
|
+
return False
|
|
1693
|
+
dx = x2 - x1
|
|
1694
|
+
dy = y2 - y1
|
|
1695
|
+
if dx == 0 and dy == 0:
|
|
1696
|
+
return left <= x1 <= right and top <= y1 <= bottom
|
|
1697
|
+
t_enter, t_exit = 0.0, 1.0
|
|
1698
|
+
for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
|
|
1699
|
+
if delta == 0:
|
|
1700
|
+
if distance < 0:
|
|
1701
|
+
return False
|
|
1702
|
+
continue
|
|
1703
|
+
t = distance / delta
|
|
1704
|
+
if delta < 0:
|
|
1705
|
+
t_enter = max(t_enter, t)
|
|
1706
|
+
else:
|
|
1707
|
+
t_exit = min(t_exit, t)
|
|
1708
|
+
if t_enter > t_exit:
|
|
1709
|
+
return False
|
|
1710
|
+
return True
|
|
1711
|
+
|
|
1712
|
+
|
|
1713
|
+
def line_text_graze_margin(text_element: dict[str, Any]) -> float:
|
|
1714
|
+
font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
|
|
1715
|
+
return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
|
|
1716
|
+
|
|
1717
|
+
|
|
1718
|
+
def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
|
|
1719
|
+
width = rect["width"] - 2 * margin
|
|
1720
|
+
height = rect["height"] - 2 * margin
|
|
1721
|
+
if width <= 0 or height <= 0:
|
|
1722
|
+
return None
|
|
1723
|
+
return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
|
|
1724
|
+
|
|
1725
|
+
|
|
1726
|
+
def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
|
|
1727
|
+
if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
|
|
1728
|
+
return False
|
|
1729
|
+
if not is_text_element(text_element) or not has_text_content(text_element):
|
|
1730
|
+
return False
|
|
1731
|
+
if is_ghost_text(text_element) or is_decorative_text(text_element):
|
|
1732
|
+
return False
|
|
1733
|
+
glyph_bbox = estimate_text_visual_bbox(text_element)
|
|
1734
|
+
if glyph_bbox is None:
|
|
1735
|
+
return False
|
|
1736
|
+
# Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
|
|
1737
|
+
# frame is exempt; only a line that actually cuts through the letterforms is a crossing.
|
|
1738
|
+
target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
|
|
1739
|
+
if target is None:
|
|
1740
|
+
return False
|
|
1741
|
+
return segment_intersects_rect(
|
|
1742
|
+
line["startX"], line["startY"], line["endX"], line["endY"], target
|
|
1743
|
+
)
|
|
1744
|
+
|
|
1745
|
+
|
|
1746
|
+
def detect_line_text_crossings(
|
|
1747
|
+
slide_xml: str, elements: list[dict[str, Any]]
|
|
1748
|
+
) -> list[dict[str, Any]]:
|
|
1749
|
+
lines = extract_line_elements(slide_xml)
|
|
1750
|
+
if not lines:
|
|
1751
|
+
return []
|
|
1752
|
+
text_elements = [element for element in elements if is_text_element(element)]
|
|
1753
|
+
issues: list[dict[str, Any]] = []
|
|
1754
|
+
for line in lines:
|
|
1755
|
+
for text_element in text_elements:
|
|
1756
|
+
if not line_crosses_text(line, text_element):
|
|
1757
|
+
continue
|
|
1758
|
+
issues.append(
|
|
1759
|
+
{
|
|
1760
|
+
"level": "error",
|
|
1761
|
+
"code": "bbox_overlap",
|
|
1762
|
+
"elements": [line["id"], text_element["id"]],
|
|
1763
|
+
"message": f'line {line["id"]} crosses text {text_element["id"]}',
|
|
1764
|
+
"hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
|
|
1765
|
+
}
|
|
1766
|
+
)
|
|
1767
|
+
return issues
|
|
1768
|
+
|
|
1769
|
+
|
|
1111
1770
|
def lint_slide(
|
|
1112
1771
|
slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
|
|
1113
1772
|
) -> dict[str, Any]:
|
|
@@ -1116,6 +1775,9 @@ def lint_slide(
|
|
|
1116
1775
|
*detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
|
|
1117
1776
|
*detect_elements_out_of_canvas(elements, slide_width, slide_height),
|
|
1118
1777
|
*detect_table_layout_size_mismatches(elements),
|
|
1778
|
+
*detect_text_may_overflow_shapes(elements),
|
|
1779
|
+
*detect_image_text_occlusions(elements),
|
|
1780
|
+
*detect_line_text_crossings(slide_xml, elements),
|
|
1119
1781
|
]
|
|
1120
1782
|
|
|
1121
1783
|
for index, left in enumerate(elements):
|
|
@@ -1129,62 +1791,744 @@ def lint_slide(
|
|
|
1129
1791
|
"code": "bbox_overlap",
|
|
1130
1792
|
"elements": [left["id"], right["id"]],
|
|
1131
1793
|
"message": f'{left["id"]} overlaps {right["id"]}',
|
|
1794
|
+
"hint": "Move or resize the elements so their visual bounds no longer intersect.",
|
|
1795
|
+
**(
|
|
1796
|
+
{"measurement": horizontal_text_overflow_measurement(left, right)}
|
|
1797
|
+
if horizontal_overflow
|
|
1798
|
+
else {}
|
|
1799
|
+
),
|
|
1132
1800
|
}
|
|
1133
1801
|
)
|
|
1134
1802
|
|
|
1135
|
-
return {
|
|
1803
|
+
return {
|
|
1804
|
+
"slide_number": slide_number,
|
|
1805
|
+
"element_count": len(elements),
|
|
1806
|
+
"elements": elements,
|
|
1807
|
+
"issues": issues,
|
|
1808
|
+
}
|
|
1136
1809
|
|
|
1137
1810
|
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1811
|
+
|
|
1812
|
+
MIN_CONTAINER_WIDTH = 140
|
|
1813
|
+
MIN_CONTAINER_HEIGHT = 160
|
|
1814
|
+
MIN_SHORT_CARD_HEIGHT = 80
|
|
1815
|
+
MIN_CONTAINER_AREA = 20_000
|
|
1816
|
+
MIN_CONTENT_COVERAGE_RATIO = 0.15
|
|
1817
|
+
MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
|
|
1818
|
+
MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
|
|
1819
|
+
SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
|
|
1820
|
+
MIN_SIMILAR_SHORT_CARD_COUNT = 2
|
|
1821
|
+
LARGE_VISUAL_CHILD_RATIO = 0.35
|
|
1822
|
+
LAYOUT_PANEL_SPAN_RATIO = 0.90
|
|
1823
|
+
IMAGE_OVERLAY_MATCH_RATIO = 0.90
|
|
1824
|
+
DENSITY_CONTAINMENT_TOLERANCE = 8
|
|
1825
|
+
|
|
1826
|
+
|
|
1827
|
+
def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
1828
|
+
left = max(element["x"], container["x"])
|
|
1829
|
+
top = max(element["y"], container["y"])
|
|
1830
|
+
right = min(element["x"] + element["width"], container["x"] + container["width"])
|
|
1831
|
+
bottom = min(element["y"] + element["height"], container["y"] + container["height"])
|
|
1832
|
+
if right <= left or bottom <= top:
|
|
1833
|
+
return None
|
|
1834
|
+
return {"x": left, "y": top, "width": right - left, "height": bottom - top}
|
|
1835
|
+
|
|
1836
|
+
|
|
1837
|
+
def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
|
|
1838
|
+
x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
|
|
1839
|
+
area = 0
|
|
1840
|
+
for left, right in zip(x_coordinates, x_coordinates[1:]):
|
|
1841
|
+
intervals = sorted(
|
|
1842
|
+
(rect["y"], rect["y"] + rect["height"])
|
|
1843
|
+
for rect in rectangles
|
|
1844
|
+
if rect["x"] < right and rect["x"] + rect["width"] > left
|
|
1845
|
+
)
|
|
1846
|
+
covered_height = 0
|
|
1847
|
+
interval_end: int | float | None = None
|
|
1848
|
+
for top, bottom in intervals:
|
|
1849
|
+
if interval_end is None:
|
|
1850
|
+
covered_height += bottom - top
|
|
1851
|
+
interval_end = bottom
|
|
1852
|
+
elif bottom > interval_end:
|
|
1853
|
+
covered_height += bottom - max(top, interval_end)
|
|
1854
|
+
interval_end = bottom
|
|
1855
|
+
area += (right - left) * covered_height
|
|
1856
|
+
return area
|
|
1857
|
+
|
|
1858
|
+
|
|
1859
|
+
def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
|
|
1860
|
+
return sum(
|
|
1861
|
+
other is not element
|
|
1862
|
+
and is_visually_rendered(other)
|
|
1863
|
+
and other["kind"] == "shape"
|
|
1864
|
+
and other["type"] == "rect"
|
|
1865
|
+
and other["width"] >= MIN_CONTAINER_WIDTH
|
|
1866
|
+
and other["height"] >= MIN_SHORT_CARD_HEIGHT
|
|
1867
|
+
and element_area(other) >= MIN_CONTAINER_AREA
|
|
1868
|
+
and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
|
|
1869
|
+
<= SHORT_CARD_SIZE_TOLERANCE_RATIO
|
|
1870
|
+
and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
|
|
1871
|
+
<= SHORT_CARD_SIZE_TOLERANCE_RATIO
|
|
1872
|
+
for other in elements
|
|
1873
|
+
) >= MIN_SIMILAR_SHORT_CARD_COUNT
|
|
1874
|
+
|
|
1875
|
+
|
|
1876
|
+
def is_layout_container(
|
|
1877
|
+
element: dict[str, Any],
|
|
1878
|
+
slide_width: int | float,
|
|
1879
|
+
slide_height: int | float,
|
|
1880
|
+
elements: list[dict[str, Any]] | None = None,
|
|
1881
|
+
) -> bool:
|
|
1882
|
+
has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
|
|
1883
|
+
elements is not None
|
|
1884
|
+
and element["height"] >= MIN_SHORT_CARD_HEIGHT
|
|
1885
|
+
and has_similar_short_card_peer(element, elements)
|
|
1886
|
+
)
|
|
1887
|
+
return (
|
|
1888
|
+
element["kind"] == "shape"
|
|
1889
|
+
and element["type"] == "rect"
|
|
1890
|
+
and is_visually_rendered(element)
|
|
1891
|
+
and element["width"] >= MIN_CONTAINER_WIDTH
|
|
1892
|
+
and has_supported_height
|
|
1893
|
+
and element_area(element) >= MIN_CONTAINER_AREA
|
|
1894
|
+
and not (
|
|
1895
|
+
element["x"] <= 2
|
|
1896
|
+
and element["y"] <= 2
|
|
1897
|
+
and element["width"] >= slide_width - 4
|
|
1898
|
+
and element["height"] >= slide_height - 4
|
|
1899
|
+
)
|
|
1900
|
+
)
|
|
1901
|
+
|
|
1902
|
+
|
|
1903
|
+
def is_edge_spanning_layout_panel(
|
|
1904
|
+
element: dict[str, Any], slide_width: int | float, slide_height: int | float
|
|
1905
|
+
) -> bool:
|
|
1906
|
+
touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
|
|
1907
|
+
touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
|
|
1908
|
+
return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
|
|
1909
|
+
touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
|
|
1910
|
+
)
|
|
1911
|
+
|
|
1912
|
+
|
|
1913
|
+
def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
|
|
1914
|
+
container_area = element_area(container)
|
|
1915
|
+
return any(
|
|
1916
|
+
element["kind"] == "img"
|
|
1917
|
+
and is_visually_rendered(element)
|
|
1918
|
+
and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
|
|
1919
|
+
for element in elements
|
|
1920
|
+
)
|
|
1921
|
+
|
|
1922
|
+
|
|
1923
|
+
def is_nested_in_layout_panel(
|
|
1924
|
+
container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
|
|
1925
|
+
) -> bool:
|
|
1926
|
+
return any(
|
|
1927
|
+
element is not container
|
|
1928
|
+
and element["kind"] == "shape"
|
|
1929
|
+
and element["type"] == "rect"
|
|
1930
|
+
and is_visually_rendered(element)
|
|
1931
|
+
and is_edge_spanning_layout_panel(element, slide_width, slide_height)
|
|
1932
|
+
and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
|
|
1933
|
+
for element in elements
|
|
1934
|
+
)
|
|
1935
|
+
|
|
1936
|
+
|
|
1937
|
+
def extract_density_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
1938
|
+
elements = extract_elements(slide_xml)
|
|
1939
|
+
elements_by_id = {element["id"]: element for element in elements}
|
|
1940
|
+
root = ET.fromstring(slide_xml)
|
|
1941
|
+
for node in root.iter():
|
|
1942
|
+
if xml_local_name(node.tag) != "shape":
|
|
1943
|
+
continue
|
|
1944
|
+
element = elements_by_id.get(node.attrib.get("id", ""))
|
|
1945
|
+
if element is None:
|
|
1946
|
+
continue
|
|
1947
|
+
content_node = next(
|
|
1948
|
+
(child for child in node if xml_local_name(child.tag) == "content"),
|
|
1949
|
+
None,
|
|
1950
|
+
)
|
|
1951
|
+
paragraphs = (
|
|
1952
|
+
[
|
|
1953
|
+
" ".join("".join(paragraph.itertext()).split())
|
|
1954
|
+
for paragraph in content_node.iter()
|
|
1955
|
+
if xml_local_name(paragraph.tag) == "p"
|
|
1956
|
+
]
|
|
1957
|
+
if content_node is not None
|
|
1958
|
+
else []
|
|
1959
|
+
)
|
|
1960
|
+
raw_font_size = (
|
|
1961
|
+
content_node.attrib.get("fontSize") if content_node is not None else None
|
|
1962
|
+
) or node.attrib.get("fontSize")
|
|
1963
|
+
try:
|
|
1964
|
+
base_font_size = float(raw_font_size or 16)
|
|
1965
|
+
except ValueError:
|
|
1966
|
+
base_font_size = 16.0
|
|
1967
|
+
element.update(
|
|
1968
|
+
{
|
|
1969
|
+
"textType": content_node.attrib.get("textType") if content_node is not None else None,
|
|
1970
|
+
"textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
|
|
1971
|
+
"autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
|
|
1972
|
+
"fontSize": base_font_size,
|
|
1973
|
+
"text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
|
|
1974
|
+
}
|
|
1975
|
+
)
|
|
1976
|
+
if not has_text_content(element):
|
|
1977
|
+
continue
|
|
1978
|
+
declared_font_sizes = []
|
|
1979
|
+
for descendant in node.iter():
|
|
1980
|
+
raw_declared_font_size = descendant.attrib.get("fontSize")
|
|
1981
|
+
if raw_declared_font_size is None:
|
|
1982
|
+
continue
|
|
1983
|
+
try:
|
|
1984
|
+
declared_font_sizes.append(float(raw_declared_font_size))
|
|
1985
|
+
except ValueError:
|
|
1986
|
+
continue
|
|
1987
|
+
if declared_font_sizes:
|
|
1988
|
+
element["fontSize"] = max(declared_font_sizes)
|
|
1989
|
+
for match in re.finditer(r"<icon\b([^>]*)>", slide_xml):
|
|
1990
|
+
attrs = match.group(1)
|
|
1991
|
+
x = extract_numeric_attribute(attrs, "topLeftX")
|
|
1992
|
+
y = extract_numeric_attribute(attrs, "topLeftY")
|
|
1993
|
+
width = extract_numeric_attribute(attrs, "width")
|
|
1994
|
+
height = extract_numeric_attribute(attrs, "height")
|
|
1995
|
+
if any(value is None for value in (x, y, width, height)):
|
|
1996
|
+
continue
|
|
1997
|
+
icon_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
1998
|
+
elements.append(
|
|
1999
|
+
{
|
|
2000
|
+
"id": extract_attribute(attrs, "id") or f"icon-{len(elements) + 1}",
|
|
2001
|
+
"kind": "icon",
|
|
2002
|
+
"type": "icon",
|
|
2003
|
+
"x": x,
|
|
2004
|
+
"y": y,
|
|
2005
|
+
"width": width,
|
|
2006
|
+
"height": height,
|
|
2007
|
+
"rotation": extract_numeric_attribute(attrs, "rotation") or 0,
|
|
2008
|
+
"alpha": icon_alpha if icon_alpha is not None else 1,
|
|
2009
|
+
"order": len(elements),
|
|
2010
|
+
}
|
|
2011
|
+
)
|
|
2012
|
+
for match in re.finditer(r"<polyline\b([^>]*)>", slide_xml):
|
|
2013
|
+
attrs = match.group(1)
|
|
2014
|
+
x = extract_numeric_attribute(attrs, "topLeftX")
|
|
2015
|
+
y = extract_numeric_attribute(attrs, "topLeftY")
|
|
2016
|
+
width = extract_numeric_attribute(attrs, "width")
|
|
2017
|
+
height = extract_numeric_attribute(attrs, "height")
|
|
2018
|
+
if any(value is None for value in (x, y, width, height)):
|
|
2019
|
+
continue
|
|
2020
|
+
polyline_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
2021
|
+
elements.append(
|
|
2022
|
+
{
|
|
2023
|
+
"id": extract_attribute(attrs, "id") or f"polyline-{len(elements) + 1}",
|
|
2024
|
+
"kind": "polyline",
|
|
2025
|
+
"type": "polyline",
|
|
2026
|
+
"x": x,
|
|
2027
|
+
"y": y,
|
|
2028
|
+
"width": width,
|
|
2029
|
+
"height": height,
|
|
2030
|
+
"rotation": extract_numeric_attribute(attrs, "rotation") or 0,
|
|
2031
|
+
"alpha": polyline_alpha if polyline_alpha is not None else 1,
|
|
2032
|
+
"order": len(elements),
|
|
2033
|
+
}
|
|
2034
|
+
)
|
|
2035
|
+
for line_element in extract_line_elements(slide_xml):
|
|
2036
|
+
line_element["order"] = len(elements)
|
|
2037
|
+
elements.append(line_element)
|
|
2038
|
+
return elements
|
|
2039
|
+
|
|
2040
|
+
|
|
2041
|
+
def is_visually_rendered(element: dict[str, Any]) -> bool:
|
|
2042
|
+
return element.get("alpha", 1) > 0
|
|
2043
|
+
|
|
2044
|
+
|
|
2045
|
+
def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
2046
|
+
if not is_visually_rendered(element):
|
|
2047
|
+
return None
|
|
2048
|
+
if is_text_element(element):
|
|
2049
|
+
estimated = estimate_text_visual_bbox(element)
|
|
2050
|
+
return clipped_bbox(estimated, container) if estimated else None
|
|
2051
|
+
return clipped_bbox(element, container)
|
|
2052
|
+
|
|
2053
|
+
|
|
2054
|
+
def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
2055
|
+
if container["kind"] != "shape" or not has_text_content(container):
|
|
2056
|
+
return None
|
|
2057
|
+
text_proxy = {**container, "type": "text"}
|
|
2058
|
+
estimated = estimate_text_visual_bbox(text_proxy)
|
|
2059
|
+
return clipped_bbox(estimated, container) if estimated else None
|
|
2060
|
+
|
|
2061
|
+
|
|
2062
|
+
def slide_content_visual_bbox(
|
|
2063
|
+
element: dict[str, Any], slide_bbox: dict[str, int | float]
|
|
2064
|
+
) -> dict[str, int | float] | None:
|
|
2065
|
+
if not is_visually_rendered(element):
|
|
2066
|
+
return None
|
|
2067
|
+
if is_text_element(element):
|
|
2068
|
+
estimated = estimate_text_visual_bbox(element)
|
|
2069
|
+
return clipped_bbox(estimated, slide_bbox) if estimated else None
|
|
2070
|
+
if element["kind"] == "shape" and has_text_content(element):
|
|
2071
|
+
estimated = own_text_visual_bbox(element)
|
|
2072
|
+
return clipped_bbox(estimated, slide_bbox) if estimated else None
|
|
2073
|
+
if element["kind"] == "line":
|
|
2074
|
+
# a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
|
|
2075
|
+
# treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
|
|
2076
|
+
return clipped_bbox(line_stroke_bbox(element), slide_bbox)
|
|
2077
|
+
if element["kind"] in {"img", "chart", "table", "whiteboard", "icon", "polyline"}:
|
|
2078
|
+
return clipped_bbox(element, slide_bbox)
|
|
2079
|
+
return None
|
|
2080
|
+
|
|
2081
|
+
|
|
2082
|
+
def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
|
|
2083
|
+
return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
|
|
2084
|
+
|
|
2085
|
+
|
|
2086
|
+
def is_slide_content_present(
|
|
2087
|
+
element: dict[str, Any], slide_bbox: dict[str, int | float]
|
|
2088
|
+
) -> bool:
|
|
2089
|
+
# Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
|
|
2090
|
+
# *anything* rendered here", not the richer "counts toward meaningful content density" bar
|
|
2091
|
+
# that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
|
|
2092
|
+
# decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
|
|
2093
|
+
# count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
|
|
2094
|
+
# instead of maintaining an allow-list that silently treats unlisted kinds as blank.
|
|
2095
|
+
if not is_visually_rendered(element):
|
|
2096
|
+
return False
|
|
2097
|
+
if (
|
|
2098
|
+
element["kind"] == "shape"
|
|
2099
|
+
and element["type"] == "rect"
|
|
2100
|
+
and not has_text_content(element)
|
|
2101
|
+
and element["x"] <= 2
|
|
2102
|
+
and element["y"] <= 2
|
|
2103
|
+
and element["width"] >= slide_bbox["width"] - 4
|
|
2104
|
+
and element["height"] >= slide_bbox["height"] - 4
|
|
2105
|
+
):
|
|
2106
|
+
# A full-canvas plain rect is a background panel, not content -- same reasoning as
|
|
2107
|
+
# is_layout_container's existing background exclusion. A slide with nothing else on it
|
|
2108
|
+
# is still effectively blank.
|
|
2109
|
+
return False
|
|
2110
|
+
bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
|
|
2111
|
+
return clipped_bbox(bbox, slide_bbox) is not None
|
|
2112
|
+
|
|
2113
|
+
|
|
2114
|
+
def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
|
|
2115
|
+
if element["kind"] not in {"img", "chart", "table", "whiteboard"}:
|
|
2116
|
+
return False
|
|
2117
|
+
if not is_visually_rendered(element):
|
|
2118
|
+
return False
|
|
2119
|
+
return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
|
|
2120
|
+
|
|
2121
|
+
|
|
2122
|
+
def detect_sparse_container_content(
|
|
2123
|
+
elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
|
|
2124
|
+
) -> list[dict[str, Any]]:
|
|
2125
|
+
issues: list[dict[str, Any]] = []
|
|
2126
|
+
for container in (
|
|
2127
|
+
element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
|
|
2128
|
+
):
|
|
2129
|
+
if (
|
|
2130
|
+
is_edge_spanning_layout_panel(container, slide_width, slide_height)
|
|
2131
|
+
or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
|
|
2132
|
+
or has_matching_image_overlay(container, elements)
|
|
2133
|
+
):
|
|
2134
|
+
continue
|
|
2135
|
+
children = [
|
|
2136
|
+
element
|
|
2137
|
+
for element in elements
|
|
2138
|
+
if element is not container
|
|
2139
|
+
and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
|
|
2140
|
+
]
|
|
2141
|
+
if any(is_large_visual_child(child, container) for child in children):
|
|
2142
|
+
continue
|
|
2143
|
+
own_text_bbox = own_text_visual_bbox(container)
|
|
2144
|
+
rectangles = ([own_text_bbox] if own_text_bbox else []) + [
|
|
2145
|
+
bbox for child in children if (bbox := visual_bbox(child, container)) is not None
|
|
2146
|
+
]
|
|
2147
|
+
content_area = rectangle_union_area(rectangles) if rectangles else 0
|
|
2148
|
+
coverage_ratio = content_area / element_area(container)
|
|
2149
|
+
if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
|
|
2150
|
+
continue
|
|
2151
|
+
issues.append(
|
|
2152
|
+
{
|
|
2153
|
+
"level": "warning",
|
|
2154
|
+
"code": "sparse_container_content",
|
|
2155
|
+
"target": {
|
|
2156
|
+
"slide_number": slide_number,
|
|
2157
|
+
"container_id": container["id"],
|
|
2158
|
+
"container_type": container["type"],
|
|
2159
|
+
"bbox": {key: container[key] for key in ("x", "y", "width", "height")},
|
|
2160
|
+
},
|
|
2161
|
+
"rule": {
|
|
2162
|
+
"name": "large_container_visible_content_coverage",
|
|
2163
|
+
"threshold": MIN_CONTENT_COVERAGE_RATIO,
|
|
2164
|
+
"comparison": "content_coverage_ratio < threshold",
|
|
2165
|
+
},
|
|
2166
|
+
"measurement": {
|
|
2167
|
+
"container_area": element_area(container),
|
|
2168
|
+
"visible_content_area": round(content_area, 3),
|
|
2169
|
+
"content_coverage_ratio": round(coverage_ratio, 3),
|
|
2170
|
+
"content_element_count": len(children) + (1 if own_text_bbox else 0),
|
|
2171
|
+
},
|
|
2172
|
+
"elements": [container["id"], *[child["id"] for child in children]],
|
|
2173
|
+
}
|
|
2174
|
+
)
|
|
2175
|
+
return issues
|
|
2176
|
+
|
|
2177
|
+
|
|
2178
|
+
def detect_sparse_slide_content(
|
|
2179
|
+
elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
|
|
2180
|
+
) -> list[dict[str, Any]]:
|
|
2181
|
+
slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
|
|
2182
|
+
content = [
|
|
2183
|
+
(element, bbox)
|
|
2184
|
+
for element in elements
|
|
2185
|
+
if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
|
|
2186
|
+
]
|
|
2187
|
+
if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
|
|
2188
|
+
return []
|
|
2189
|
+
content_area = rectangle_union_area([bbox for _, bbox in content])
|
|
2190
|
+
slide_area = slide_width * slide_height
|
|
2191
|
+
coverage_ratio = content_area / slide_area
|
|
2192
|
+
if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
|
|
2193
|
+
return []
|
|
2194
|
+
return [
|
|
2195
|
+
{
|
|
2196
|
+
"level": "warning",
|
|
2197
|
+
"code": "sparse_slide_content",
|
|
2198
|
+
"target": {
|
|
2199
|
+
"slide_number": slide_number,
|
|
2200
|
+
"bbox": slide_bbox,
|
|
2201
|
+
},
|
|
2202
|
+
"rule": {
|
|
2203
|
+
"name": "slide_visible_content_coverage",
|
|
2204
|
+
"threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
|
|
2205
|
+
"comparison": "content_coverage_ratio < threshold",
|
|
2206
|
+
},
|
|
2207
|
+
"measurement": {
|
|
2208
|
+
"slide_area": slide_area,
|
|
2209
|
+
"visible_content_area": round(content_area, 3),
|
|
2210
|
+
"content_coverage_ratio": round(coverage_ratio, 3),
|
|
2211
|
+
"content_element_count": len(content),
|
|
2212
|
+
},
|
|
2213
|
+
"elements": [element["id"] for element, _ in content],
|
|
2214
|
+
}
|
|
2215
|
+
]
|
|
2216
|
+
|
|
2217
|
+
|
|
2218
|
+
def detect_blank_slide(
|
|
2219
|
+
elements: list[dict[str, Any]],
|
|
2220
|
+
slide_number: int,
|
|
2221
|
+
slide_width: int | float,
|
|
2222
|
+
slide_height: int | float,
|
|
2223
|
+
) -> list[dict[str, Any]]:
|
|
2224
|
+
slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
|
|
2225
|
+
visible_elements = [
|
|
2226
|
+
element for element in elements if is_slide_content_present(element, slide_bbox)
|
|
2227
|
+
]
|
|
2228
|
+
if visible_elements:
|
|
2229
|
+
return []
|
|
2230
|
+
return [
|
|
2231
|
+
{
|
|
2232
|
+
"level": "error",
|
|
2233
|
+
"code": "blank_slide",
|
|
2234
|
+
"schema_version": "2.0",
|
|
2235
|
+
"target": {"slide_number": slide_number},
|
|
2236
|
+
"rule": {
|
|
2237
|
+
"name": "slide_has_visible_content",
|
|
2238
|
+
"comparison": "visible_element_count == 0",
|
|
2239
|
+
},
|
|
2240
|
+
"measurement": {
|
|
2241
|
+
"visible_element_count": 0,
|
|
2242
|
+
"declared_element_count": len(elements),
|
|
2243
|
+
},
|
|
2244
|
+
"elements": [element["id"] for element in elements],
|
|
2245
|
+
"message": "slide has no visible content beyond empty layout shapes",
|
|
2246
|
+
"hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
|
|
2247
|
+
}
|
|
2248
|
+
]
|
|
2249
|
+
|
|
2250
|
+
|
|
2251
|
+
|
|
2252
|
+
RULE_METADATA: dict[str, dict[str, Any]] = {
|
|
2253
|
+
"xml_not_well_formed": {
|
|
2254
|
+
"name": "xml_is_well_formed",
|
|
2255
|
+
"comparison": "xml_parse_error == false",
|
|
2256
|
+
},
|
|
2257
|
+
"sml_prefixed_tag": {
|
|
2258
|
+
"name": "sml_uses_default_namespace",
|
|
2259
|
+
"comparison": "prefixed_sml_tag_count == 0",
|
|
2260
|
+
},
|
|
2261
|
+
"sxsd_unsupported_tag": {
|
|
2262
|
+
"name": "tag_is_supported_by_slides_xml_schema",
|
|
2263
|
+
"comparison": "unsupported_tag_count == 0",
|
|
2264
|
+
},
|
|
2265
|
+
"sxsd_unsupported_attr": {
|
|
2266
|
+
"name": "attribute_is_supported_by_slides_xml_schema",
|
|
2267
|
+
"comparison": "unsupported_attribute_count == 0",
|
|
2268
|
+
},
|
|
2269
|
+
"icon_missing_fill_color": {
|
|
2270
|
+
"name": "icon_has_visible_fill_color",
|
|
2271
|
+
"comparison": "fill_color_present == true",
|
|
2272
|
+
},
|
|
2273
|
+
"icon_transparent_fill_color": {
|
|
2274
|
+
"name": "icon_has_visible_fill_color",
|
|
2275
|
+
"comparison": "fill_alpha > 0",
|
|
2276
|
+
},
|
|
2277
|
+
"iconpark_unsupported_icon_type": {
|
|
2278
|
+
"name": "iconpark_type_is_supported",
|
|
2279
|
+
"comparison": "icon_type in iconpark_index",
|
|
2280
|
+
},
|
|
2281
|
+
"bbox_overlap": {
|
|
2282
|
+
"name": "text_visual_bounds_do_not_overlap",
|
|
2283
|
+
"comparison": "intersection_area == 0",
|
|
2284
|
+
},
|
|
2285
|
+
"text_may_overflow_shape": {
|
|
2286
|
+
"name": "estimated_text_fits_declared_shape",
|
|
2287
|
+
"comparison": "estimated_height <= available_height",
|
|
2288
|
+
},
|
|
2289
|
+
"whiteboard_external_overlap": {
|
|
2290
|
+
"name": "whiteboard_does_not_cross_sibling_content",
|
|
2291
|
+
"comparison": "external_overlap_count == 0",
|
|
2292
|
+
},
|
|
2293
|
+
"image_covers_text": {
|
|
2294
|
+
"name": "image_does_not_cover_text",
|
|
2295
|
+
"comparison": "intersection_area == 0",
|
|
2296
|
+
},
|
|
2297
|
+
"image_may_cover_vertical_text": {
|
|
2298
|
+
"name": "image_vertical_text_occlusion_requires_review",
|
|
2299
|
+
"comparison": "intersection_area == 0",
|
|
2300
|
+
},
|
|
2301
|
+
"table_resolved_size_mismatch": {
|
|
2302
|
+
"name": "table_declared_size_matches_resolved_grid",
|
|
2303
|
+
"comparison": "declared_size == resolved_size",
|
|
2304
|
+
},
|
|
2305
|
+
"blank_slide": {
|
|
2306
|
+
"name": "slide_has_visible_content",
|
|
2307
|
+
"comparison": "visible_element_count > 0",
|
|
2308
|
+
},
|
|
2309
|
+
}
|
|
2310
|
+
|
|
2311
|
+
|
|
2312
|
+
def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
|
|
2313
|
+
if issue.get("rule"):
|
|
2314
|
+
return {**issue["rule"], "id": issue["code"]}
|
|
2315
|
+
if issue["code"].endswith("_out_of_canvas"):
|
|
1141
2316
|
return {
|
|
1142
|
-
"
|
|
1143
|
-
"
|
|
1144
|
-
"
|
|
1145
|
-
"issues": [xml_error],
|
|
1146
|
-
"slides": [],
|
|
2317
|
+
"id": issue["code"],
|
|
2318
|
+
"name": "element_stays_within_slide_canvas",
|
|
2319
|
+
"comparison": "max(left, top, right, bottom overflow) == 0",
|
|
1147
2320
|
}
|
|
2321
|
+
return {
|
|
2322
|
+
"id": issue["code"],
|
|
2323
|
+
**RULE_METADATA.get(
|
|
2324
|
+
issue["code"],
|
|
2325
|
+
{"name": issue["code"], "comparison": "violation_count == 0"},
|
|
2326
|
+
),
|
|
2327
|
+
}
|
|
1148
2328
|
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
if
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
2329
|
+
|
|
2330
|
+
def issue_measurement(
|
|
2331
|
+
issue: dict[str, Any], elements_by_id: dict[str, dict[str, Any]]
|
|
2332
|
+
) -> dict[str, Any]:
|
|
2333
|
+
if issue.get("measurement") is not None:
|
|
2334
|
+
return issue["measurement"]
|
|
2335
|
+
if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
|
|
2336
|
+
left = elements_by_id.get(issue["elements"][0])
|
|
2337
|
+
right = elements_by_id.get(issue["elements"][1])
|
|
2338
|
+
if left and right:
|
|
2339
|
+
left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
|
|
2340
|
+
right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
|
|
2341
|
+
width = intersection_width(left_box, right_box)
|
|
2342
|
+
height = intersection_height(left_box, right_box)
|
|
2343
|
+
return {
|
|
2344
|
+
"intersection_width": round(width, 3),
|
|
2345
|
+
"intersection_height": round(height, 3),
|
|
2346
|
+
"intersection_area": round(width * height, 3),
|
|
2347
|
+
}
|
|
2348
|
+
if issue["code"].endswith("_out_of_canvas"):
|
|
1157
2349
|
return {
|
|
1158
|
-
"
|
|
1159
|
-
"
|
|
1160
|
-
"
|
|
1161
|
-
"slide_count": 0,
|
|
1162
|
-
"error_count": error_count,
|
|
1163
|
-
"warning_count": warning_count,
|
|
1164
|
-
"info_count": info_count,
|
|
1165
|
-
},
|
|
1166
|
-
"issues": top_level_issues,
|
|
1167
|
-
"slides": [],
|
|
2350
|
+
"canvas": issue.get("canvas"),
|
|
2351
|
+
"bbox": issue.get("bbox"),
|
|
2352
|
+
"overflow": issue.get("overflow"),
|
|
1168
2353
|
}
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
2354
|
+
measurement_keys = (
|
|
2355
|
+
"line",
|
|
2356
|
+
"column",
|
|
2357
|
+
"tag",
|
|
2358
|
+
"attr",
|
|
2359
|
+
"iconType",
|
|
2360
|
+
"line_count",
|
|
2361
|
+
"line_height",
|
|
2362
|
+
"estimated_height",
|
|
2363
|
+
"available_height",
|
|
2364
|
+
"overflow",
|
|
2365
|
+
"dimension",
|
|
2366
|
+
"declared_size",
|
|
2367
|
+
"resolved_size",
|
|
2368
|
+
"resolved_sizes",
|
|
2369
|
+
"overlaps",
|
|
2370
|
+
)
|
|
2371
|
+
measured = {key: issue[key] for key in measurement_keys if key in issue}
|
|
2372
|
+
return measured or {"violation_count": 1}
|
|
2373
|
+
|
|
2374
|
+
|
|
2375
|
+
def related_object(element: dict[str, Any]) -> dict[str, Any]:
|
|
2376
|
+
return {
|
|
2377
|
+
"element_id": element["id"],
|
|
2378
|
+
"kind": element["kind"],
|
|
2379
|
+
"type": element["type"],
|
|
2380
|
+
"bbox": {key: element[key] for key in ("x", "y", "width", "height")},
|
|
2381
|
+
}
|
|
2382
|
+
|
|
2383
|
+
|
|
2384
|
+
def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
2385
|
+
elements: list[dict[str, Any]] = []
|
|
2386
|
+
for match in re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml):
|
|
2387
|
+
attrs = match.group(1)
|
|
2388
|
+
start_x = extract_numeric_attribute(attrs, "startX")
|
|
2389
|
+
start_y = extract_numeric_attribute(attrs, "startY")
|
|
2390
|
+
end_x = extract_numeric_attribute(attrs, "endX")
|
|
2391
|
+
end_y = extract_numeric_attribute(attrs, "endY")
|
|
2392
|
+
if any(value is None for value in (start_x, start_y, end_x, end_y)):
|
|
2393
|
+
continue
|
|
2394
|
+
line_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
2395
|
+
base_alpha = line_alpha if line_alpha is not None else 1
|
|
2396
|
+
border_alpha = 1
|
|
2397
|
+
if match.group(2) != "/":
|
|
2398
|
+
close_index = slide_xml.find("</line>", match.end())
|
|
2399
|
+
body = slide_xml[match.end() : close_index] if close_index != -1 else ""
|
|
2400
|
+
border_attrs = extract_tag_attributes(body, "border")
|
|
2401
|
+
color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
|
|
2402
|
+
if isinstance(color_alpha, (int, float)):
|
|
2403
|
+
border_alpha = color_alpha
|
|
2404
|
+
elements.append(
|
|
2405
|
+
{
|
|
2406
|
+
"id": extract_attribute(attrs, "id") or f"line-{len(elements) + 1}",
|
|
2407
|
+
"kind": "line",
|
|
2408
|
+
"type": "line",
|
|
2409
|
+
"x": min(start_x, end_x),
|
|
2410
|
+
"y": min(start_y, end_y),
|
|
2411
|
+
"width": abs(end_x - start_x),
|
|
2412
|
+
"height": abs(end_y - start_y),
|
|
2413
|
+
"startX": start_x,
|
|
2414
|
+
"startY": start_y,
|
|
2415
|
+
"endX": end_x,
|
|
2416
|
+
"endY": end_y,
|
|
2417
|
+
"rotation": 0,
|
|
2418
|
+
"alpha": base_alpha * border_alpha,
|
|
2419
|
+
"order": len(elements),
|
|
2420
|
+
}
|
|
2421
|
+
)
|
|
2422
|
+
return elements
|
|
2423
|
+
|
|
2424
|
+
|
|
2425
|
+
def normalize_issue(
|
|
2426
|
+
issue: dict[str, Any],
|
|
2427
|
+
slide_number: int | None,
|
|
2428
|
+
elements_by_id: dict[str, dict[str, Any]],
|
|
2429
|
+
) -> dict[str, Any]:
|
|
2430
|
+
normalized = dict(issue)
|
|
2431
|
+
element_ids = list(dict.fromkeys(normalized.get("elements", [])))
|
|
2432
|
+
normalized["schema_version"] = "2.0"
|
|
2433
|
+
normalized["element_ids"] = element_ids
|
|
2434
|
+
normalized["target"] = {
|
|
2435
|
+
**({"slide_number": slide_number} if slide_number is not None else {}),
|
|
2436
|
+
**normalized.get("target", {}),
|
|
2437
|
+
}
|
|
2438
|
+
normalized["rule"] = issue_rule(normalized)
|
|
2439
|
+
normalized["measurement"] = issue_measurement(normalized, elements_by_id)
|
|
2440
|
+
normalized["related_objects"] = [
|
|
2441
|
+
related_object(elements_by_id[element_id])
|
|
2442
|
+
for element_id in element_ids
|
|
2443
|
+
if element_id in elements_by_id
|
|
1173
2444
|
]
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
2445
|
+
if normalized["code"] == "sparse_container_content":
|
|
2446
|
+
ratio = normalized["measurement"]["content_coverage_ratio"]
|
|
2447
|
+
threshold = normalized["rule"]["threshold"]
|
|
2448
|
+
container_id = normalized["target"].get("container_id", "unknown")
|
|
2449
|
+
normalized.setdefault(
|
|
2450
|
+
"message",
|
|
2451
|
+
f"large card {container_id} content coverage {ratio:.1%} is below {threshold:.1%}",
|
|
2452
|
+
)
|
|
2453
|
+
normalized.setdefault(
|
|
2454
|
+
"hint",
|
|
2455
|
+
"Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
|
|
2456
|
+
)
|
|
2457
|
+
elif normalized["code"] == "sparse_slide_content":
|
|
2458
|
+
ratio = normalized["measurement"]["content_coverage_ratio"]
|
|
2459
|
+
threshold = normalized["rule"]["threshold"]
|
|
2460
|
+
normalized.setdefault(
|
|
2461
|
+
"message",
|
|
2462
|
+
f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
|
|
2463
|
+
)
|
|
2464
|
+
normalized.setdefault(
|
|
2465
|
+
"hint",
|
|
2466
|
+
"Review the rendered screenshot to decide whether the page is intentionally sparse.",
|
|
2467
|
+
)
|
|
2468
|
+
else:
|
|
2469
|
+
normalized.setdefault("message", normalized["code"].replace("_", " "))
|
|
2470
|
+
normalized.setdefault(
|
|
2471
|
+
"hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
|
|
2472
|
+
)
|
|
2473
|
+
return normalized
|
|
2474
|
+
|
|
2475
|
+
|
|
2476
|
+
def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
|
|
2477
|
+
if errors:
|
|
2478
|
+
return "blocked"
|
|
2479
|
+
if warnings:
|
|
2480
|
+
return "needs_screenshot_review"
|
|
2481
|
+
return "passed"
|
|
2482
|
+
|
|
2483
|
+
|
|
2484
|
+
def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
|
|
2485
|
+
if issue.get("code") == "sxsd_unsupported_declaration":
|
|
2486
|
+
return False
|
|
2487
|
+
if root_name == "slide":
|
|
2488
|
+
return True
|
|
2489
|
+
path = issue.get("path")
|
|
2490
|
+
if not isinstance(path, str):
|
|
2491
|
+
return False
|
|
2492
|
+
if path.startswith("presentation/slide/"):
|
|
2493
|
+
return True
|
|
2494
|
+
return path == "presentation/slide" and (
|
|
2495
|
+
issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
|
|
2496
|
+
)
|
|
2497
|
+
|
|
2498
|
+
|
|
2499
|
+
def build_result(
|
|
2500
|
+
source_path: str | None,
|
|
2501
|
+
slide_size: dict[str, int | float],
|
|
2502
|
+
top_level_issues: list[dict[str, Any]],
|
|
2503
|
+
slides: list[dict[str, Any]],
|
|
2504
|
+
) -> dict[str, Any]:
|
|
2505
|
+
document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
|
|
2506
|
+
document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
|
|
2507
|
+
document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
|
|
2508
|
+
error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
|
|
2509
|
+
warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
|
|
2510
|
+
info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
|
|
2511
|
+
all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
|
|
2512
|
+
all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
|
|
2513
|
+
status = slide_status(all_errors, all_warnings)
|
|
2514
|
+
result: dict[str, Any] = {
|
|
2515
|
+
"schema_version": "2.0",
|
|
2516
|
+
"tool": "xml_text_overlap_lint",
|
|
1181
2517
|
"file": source_path,
|
|
1182
|
-
"slide_size":
|
|
2518
|
+
"slide_size": slide_size,
|
|
1183
2519
|
"summary": {
|
|
1184
2520
|
"slide_count": len(slides),
|
|
1185
2521
|
"error_count": error_count,
|
|
1186
2522
|
"warning_count": warning_count,
|
|
1187
2523
|
"info_count": info_count,
|
|
2524
|
+
"status": status,
|
|
2525
|
+
"release_ready": error_count == 0,
|
|
2526
|
+
"screenshot_review_required": warning_count > 0,
|
|
2527
|
+
},
|
|
2528
|
+
"document": {
|
|
2529
|
+
"errors": document_errors,
|
|
2530
|
+
"warnings": document_warnings,
|
|
2531
|
+
"infos": document_infos,
|
|
1188
2532
|
},
|
|
1189
2533
|
"slides": slides,
|
|
1190
2534
|
}
|
|
@@ -1193,6 +2537,147 @@ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
|
|
|
1193
2537
|
return result
|
|
1194
2538
|
|
|
1195
2539
|
|
|
2540
|
+
def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
|
|
2541
|
+
root, xml_error = parse_xml_root(xml)
|
|
2542
|
+
if xml_error:
|
|
2543
|
+
issue = normalize_issue(xml_error, None, {})
|
|
2544
|
+
return build_result(
|
|
2545
|
+
source_path,
|
|
2546
|
+
{"width": 960, "height": 540},
|
|
2547
|
+
[issue],
|
|
2548
|
+
[],
|
|
2549
|
+
)
|
|
2550
|
+
if root is None:
|
|
2551
|
+
raise AssertionError("parse_xml_root must return a root or error")
|
|
2552
|
+
|
|
2553
|
+
namespace_issues = validate_sml_tag_prefixes(xml)
|
|
2554
|
+
root_name = xml_local_name(root.tag)
|
|
2555
|
+
sxsd_issues = validate_sxsd_document(xml, root)
|
|
2556
|
+
iconpark_issues = validate_iconpark_icon_types(root)
|
|
2557
|
+
top_level_issues = [
|
|
2558
|
+
normalize_issue(issue, None, {})
|
|
2559
|
+
for issue in [
|
|
2560
|
+
*namespace_issues,
|
|
2561
|
+
*[
|
|
2562
|
+
issue
|
|
2563
|
+
for issue in sxsd_issues
|
|
2564
|
+
if not is_slide_scoped_sxsd_issue(issue, root_name)
|
|
2565
|
+
],
|
|
2566
|
+
*iconpark_issues,
|
|
2567
|
+
]
|
|
2568
|
+
]
|
|
2569
|
+
if any(issue["level"] == "error" for issue in top_level_issues):
|
|
2570
|
+
return build_result(
|
|
2571
|
+
source_path,
|
|
2572
|
+
{"width": 960, "height": 540},
|
|
2573
|
+
top_level_issues,
|
|
2574
|
+
[],
|
|
2575
|
+
)
|
|
2576
|
+
|
|
2577
|
+
presentation = parse_presentation(root)
|
|
2578
|
+
slide_roots = presentation["slide_roots"]
|
|
2579
|
+
slides: list[dict[str, Any]] = []
|
|
2580
|
+
for index, slide_xml in enumerate(presentation["slides"]):
|
|
2581
|
+
slide_number = index + 1
|
|
2582
|
+
slide_root = slide_roots[index]
|
|
2583
|
+
slide_sxsd_issues = [
|
|
2584
|
+
normalize_issue(issue, slide_number, {})
|
|
2585
|
+
for issue in validate_sxsd_document(slide_xml, slide_root)
|
|
2586
|
+
]
|
|
2587
|
+
slide_sxsd_errors = [
|
|
2588
|
+
issue for issue in slide_sxsd_issues if issue["level"] == "error"
|
|
2589
|
+
]
|
|
2590
|
+
if slide_sxsd_errors:
|
|
2591
|
+
slide_sxsd_warnings = [
|
|
2592
|
+
issue for issue in slide_sxsd_issues if issue["level"] == "warning"
|
|
2593
|
+
]
|
|
2594
|
+
slides.append(
|
|
2595
|
+
{
|
|
2596
|
+
"slide_number": slide_number,
|
|
2597
|
+
"status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
|
|
2598
|
+
"element_count": 0,
|
|
2599
|
+
"errors": slide_sxsd_errors,
|
|
2600
|
+
"warnings": slide_sxsd_warnings,
|
|
2601
|
+
"infos": [],
|
|
2602
|
+
"issues": slide_sxsd_issues,
|
|
2603
|
+
}
|
|
2604
|
+
)
|
|
2605
|
+
continue
|
|
2606
|
+
|
|
2607
|
+
geometry = lint_slide(
|
|
2608
|
+
slide_xml,
|
|
2609
|
+
slide_number,
|
|
2610
|
+
presentation["width"],
|
|
2611
|
+
presentation["height"],
|
|
2612
|
+
)
|
|
2613
|
+
density_elements = extract_density_elements(slide_xml)
|
|
2614
|
+
extra_elements = [
|
|
2615
|
+
element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
|
|
2616
|
+
]
|
|
2617
|
+
elements_by_id = {
|
|
2618
|
+
element["id"]: element for element in [*density_elements, *extra_elements]
|
|
2619
|
+
}
|
|
2620
|
+
# geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
|
|
2621
|
+
# decided with inside lint_slide; prefer them so measurement/related_objects stay consistent
|
|
2622
|
+
# with whatever actually triggered the issue, instead of density_elements' separate re-parse.
|
|
2623
|
+
elements_by_id.update({element["id"]: element for element in geometry["elements"]})
|
|
2624
|
+
extra_overflow_issues = detect_elements_out_of_canvas(
|
|
2625
|
+
extra_elements,
|
|
2626
|
+
presentation["width"],
|
|
2627
|
+
presentation["height"],
|
|
2628
|
+
)
|
|
2629
|
+
raw_issues = [
|
|
2630
|
+
*geometry["issues"],
|
|
2631
|
+
*extra_overflow_issues,
|
|
2632
|
+
*detect_blank_slide(
|
|
2633
|
+
density_elements,
|
|
2634
|
+
slide_number,
|
|
2635
|
+
presentation["width"],
|
|
2636
|
+
presentation["height"],
|
|
2637
|
+
),
|
|
2638
|
+
*detect_sparse_container_content(
|
|
2639
|
+
density_elements,
|
|
2640
|
+
slide_number,
|
|
2641
|
+
presentation["width"],
|
|
2642
|
+
presentation["height"],
|
|
2643
|
+
),
|
|
2644
|
+
*detect_sparse_slide_content(
|
|
2645
|
+
density_elements,
|
|
2646
|
+
slide_number,
|
|
2647
|
+
presentation["width"],
|
|
2648
|
+
presentation["height"],
|
|
2649
|
+
),
|
|
2650
|
+
]
|
|
2651
|
+
issues = [
|
|
2652
|
+
*slide_sxsd_issues,
|
|
2653
|
+
*[
|
|
2654
|
+
normalize_issue(issue, slide_number, elements_by_id)
|
|
2655
|
+
for issue in raw_issues
|
|
2656
|
+
],
|
|
2657
|
+
]
|
|
2658
|
+
errors = [issue for issue in issues if issue["level"] == "error"]
|
|
2659
|
+
warnings = [issue for issue in issues if issue["level"] == "warning"]
|
|
2660
|
+
infos = [issue for issue in issues if issue["level"] == "info"]
|
|
2661
|
+
slides.append(
|
|
2662
|
+
{
|
|
2663
|
+
"slide_number": slide_number,
|
|
2664
|
+
"status": slide_status(errors, warnings),
|
|
2665
|
+
"element_count": len(elements_by_id),
|
|
2666
|
+
"errors": errors,
|
|
2667
|
+
"warnings": warnings,
|
|
2668
|
+
"infos": infos,
|
|
2669
|
+
"issues": issues,
|
|
2670
|
+
}
|
|
2671
|
+
)
|
|
2672
|
+
|
|
2673
|
+
return build_result(
|
|
2674
|
+
source_path,
|
|
2675
|
+
{"width": presentation["width"], "height": presentation["height"]},
|
|
2676
|
+
top_level_issues,
|
|
2677
|
+
slides,
|
|
2678
|
+
)
|
|
2679
|
+
|
|
2680
|
+
|
|
1196
2681
|
def print_usage() -> None:
|
|
1197
2682
|
print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
|
|
1198
2683
|
|
|
@@ -1215,6 +2700,6 @@ def run_cli(argv: list[str] | None = None) -> None:
|
|
|
1215
2700
|
if __name__ == "__main__":
|
|
1216
2701
|
try:
|
|
1217
2702
|
run_cli()
|
|
1218
|
-
except
|
|
2703
|
+
except XmlLayoutLintError as error:
|
|
1219
2704
|
print(f"xml-text-overlap-lint error: {error}", file=sys.stderr)
|
|
1220
2705
|
raise SystemExit(1) from error
|